From 5f45afb8ab04d934fc0601a202f95267ebc20059 Mon Sep 17 00:00:00 2001 From: Malte Wechter Date: Wed, 3 Jun 2026 16:20:51 +0200 Subject: [PATCH 0001/1352] scripts: generate_rust_analyzer.py: pass cfg to macros crate The configuration passed to rust-analyzer for the `macros` create is different from the configuration used to build the crate. Update rust-analyzer configuration for the `macros` crate to reflect the settings used to compile the crate. Without this change, rust-analyzer does not understand conditional compilation gated by configuration redicates based on the `CONFIG_*`configuration values in the macros crate. [ Tamir: reworded subject in keeping with convention. ] Fixes: 36174d16f3ec ("rust: kunit: support KUnit-mapped `assert!` macros in `#[test]`s") Signed-off-by: Malte Wechter Link: https://patch.msgid.link/20260603-rust-analyzer-macro-v3-1-9f7fdb4908e5@gmail.com Signed-off-by: Tamir Duberstein --- scripts/generate_rust_analyzer.py | 1 + 1 file changed, 1 insertion(+) diff --git a/scripts/generate_rust_analyzer.py b/scripts/generate_rust_analyzer.py index d5f9a0ca742c42..69990a96522e34 100755 --- a/scripts/generate_rust_analyzer.py +++ b/scripts/generate_rust_analyzer.py @@ -238,6 +238,7 @@ def append_sysroot_crate( "macros", srctree / "rust" / "macros" / "lib.rs", [std, proc_macro, proc_macro2, quote, syn], + cfg=generated_cfg, ) build_error = append_crate( From 0ba1ee052b3f9917ae6ad0c29dc520827f2bbf55 Mon Sep 17 00:00:00 2001 From: Mehmet Koseoglu Date: Sat, 29 Aug 2026 19:10:13 +0300 Subject: [PATCH 0002/1352] rust: cpufreq: reject NULL from cpufreq_cpu_get() MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit cpufreq_cpu_get() returns either a referenced policy or NULL. PolicyCpu::from_cpu() passed its return value to from_err_ptr(), which rejects ERR_PTR values but accepts NULL. If the lookup fails, Policy::from_raw_mut() therefore constructs a mutable reference from NULL. Dropping the resulting PolicyCpu then passes the invalid pointer to cpufreq_cpu_put(), causing an oops in kobject_put(). Reject NULL with NonNull before constructing the Policy reference. Return ENODEV instead. A KUnit negative-control run reproduced the oops with the original conversion. The same test passed with this change. The reproducer is available on request. Fixes: 6ebdd7c93177 ("rust: cpufreq: Extend abstractions for policy and driver ops") Cc: stable@vger.kernel.org Assisted-by: LLM Signed-off-by: Mehmet Koseoglu Reviewed-by: Onur Özkan Signed-off-by: Viresh Kumar --- rust/kernel/cpufreq.rs | 18 ++++++++++++++---- 1 file changed, 14 insertions(+), 4 deletions(-) diff --git a/rust/kernel/cpufreq.rs b/rust/kernel/cpufreq.rs index affa2b9490efc9..4b992ea9a0f01d 100644 --- a/rust/kernel/cpufreq.rs +++ b/rust/kernel/cpufreq.rs @@ -14,7 +14,13 @@ use crate::{ cpumask, device::{Bound, Device}, devres, - error::{code::*, from_err_ptr, from_result, to_result, Result, VTABLE_DEFAULT_ERROR}, + error::{ + code::*, + from_result, + to_result, + Result, + VTABLE_DEFAULT_ERROR, // + }, ffi::{c_char, c_ulong}, prelude::*, types::ForeignOwnable, @@ -29,7 +35,10 @@ use core::{ marker::PhantomData, ops::{Deref, DerefMut}, pin::Pin, - ptr, + ptr::{ + self, + NonNull, // + }, }; use macros::vtable; @@ -687,12 +696,13 @@ struct PolicyCpu<'a>(&'a mut Policy); impl<'a> PolicyCpu<'a> { fn from_cpu(cpu: CpuId) -> Result { // SAFETY: It is safe to call `cpufreq_cpu_get` for any valid CPU. - let ptr = from_err_ptr(unsafe { bindings::cpufreq_cpu_get(u32::from(cpu)) })?; + let ptr = + NonNull::new(unsafe { bindings::cpufreq_cpu_get(u32::from(cpu)) }).ok_or(ENODEV)?; Ok(Self( // SAFETY: The `ptr` is guaranteed to be valid and remains valid for the lifetime of // the returned reference. - unsafe { Policy::from_raw_mut(ptr) }, + unsafe { Policy::from_raw_mut(ptr.as_ptr()) }, )) } } From 9fa5ba6c746af674fd938968e6f6307c60786ceb Mon Sep 17 00:00:00 2001 From: Xueqin Luo Date: Mon, 31 Aug 2026 13:44:38 +0800 Subject: [PATCH 0003/1352] cpufreq: sparc-us2e: fix frequency table index copy-paste error In us2e_freq_cpu_init(), the last two frequency writes both use index [2] instead of [3] and [4] respectively, and the terminator uses [3] instead of [5]. This is a copy-paste error where the index was not incremented, causing the divider-6 and divider-8 entries to overwrite the already-written divider-4 entry. As a result, only three frequency steps (div 1, 2, and 4) are actually available to the cpufreq core, while the intended dividers 6 and 8 are silently lost. The struct us2e_freq_percpu_info::table[6] has room for 5 entries plus a terminator, matching the 5 hardware dividers. Fix the indices so all five frequency steps are correctly populated: table[0]=div1, table[1]=div2, table[2]=div4, table[3]=div6, table[4]=div8, table[5]=TABLE_END. Fixes: 1da177e4c3f4 ("Linux-2.6.12-rc2") Signed-off-by: Xueqin Luo Signed-off-by: Viresh Kumar --- drivers/cpufreq/sparc-us2e-cpufreq.c | 11 +++++------ 1 file changed, 5 insertions(+), 6 deletions(-) diff --git a/drivers/cpufreq/sparc-us2e-cpufreq.c b/drivers/cpufreq/sparc-us2e-cpufreq.c index a68706406b8805..5cda391ad03bbf 100644 --- a/drivers/cpufreq/sparc-us2e-cpufreq.c +++ b/drivers/cpufreq/sparc-us2e-cpufreq.c @@ -282,12 +282,11 @@ static int us2e_freq_cpu_init(struct cpufreq_policy *policy) table[1].frequency = clock_tick / 2; table[2].driver_data = 2; table[2].frequency = clock_tick / 4; - table[2].driver_data = 3; - table[2].frequency = clock_tick / 6; - table[2].driver_data = 4; - table[2].frequency = clock_tick / 8; - table[2].driver_data = 5; - table[3].frequency = CPUFREQ_TABLE_END; + table[3].driver_data = 3; + table[3].frequency = clock_tick / 6; + table[4].driver_data = 4; + table[4].frequency = clock_tick / 8; + table[5].frequency = CPUFREQ_TABLE_END; policy->cpuinfo.transition_latency = 0; policy->cur = clock_tick; From c40bfa097eb58f3c3e0eb1384395c7528a41dca1 Mon Sep 17 00:00:00 2001 From: Xueqin Luo Date: Mon, 31 Aug 2026 15:30:08 +0800 Subject: [PATCH 0004/1352] cpufreq: tegra194: fix double-pointer error in get_cpu_ndiv ndiv is already a u64 pointer, passing &ndiv to smp_call_function_single() results in a u64** being written to instead of the caller's u64 variable, so the caller always reads back an uninitialized ndiv. Drop the spurious '&'. Fixes: 0839ed1fd7ac ("cpufreq: tegra194: add soc data to support multiple soc") Signed-off-by: Xueqin Luo Reviewed-by: Sumit Gupta Signed-off-by: Viresh Kumar --- drivers/cpufreq/tegra194-cpufreq.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/cpufreq/tegra194-cpufreq.c b/drivers/cpufreq/tegra194-cpufreq.c index c6375e14d445d7..1eb96e9f3310db 100644 --- a/drivers/cpufreq/tegra194-cpufreq.c +++ b/drivers/cpufreq/tegra194-cpufreq.c @@ -367,7 +367,7 @@ static void tegra194_get_cpu_ndiv_sysreg(void *ndiv) static int tegra194_get_cpu_ndiv(u32 cpu, u32 cpuid, u32 clusterid, u64 *ndiv) { - return smp_call_function_single(cpu, tegra194_get_cpu_ndiv_sysreg, &ndiv, true); + return smp_call_function_single(cpu, tegra194_get_cpu_ndiv_sysreg, ndiv, true); } static void tegra194_set_cpu_ndiv_sysreg(void *data) From 40bad7f0aaf0043457a1d54aeb2197fd1b7d729f Mon Sep 17 00:00:00 2001 From: Xueqin Luo Date: Mon, 31 Aug 2026 17:11:59 +0800 Subject: [PATCH 0005/1352] cpufreq: sti: avoid NULL dereference in dev_err() ddata.cpu is NULL-checked, then immediately passed to dev_err(). Replace with pr_err() to avoid dereferencing NULL. Fixes: ab0ea257fc58 ("cpufreq: st: Provide runtime initialised driver for ST's platforms") Signed-off-by: Xueqin Luo Signed-off-by: Viresh Kumar --- drivers/cpufreq/sti-cpufreq.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/drivers/cpufreq/sti-cpufreq.c b/drivers/cpufreq/sti-cpufreq.c index b15b3142b5fec3..144ac0f1d85d79 100644 --- a/drivers/cpufreq/sti-cpufreq.c +++ b/drivers/cpufreq/sti-cpufreq.c @@ -263,7 +263,7 @@ static int __init sti_cpufreq_init(void) ddata.cpu = get_cpu_device(0); if (!ddata.cpu) { - dev_err(ddata.cpu, "Failed to get device for CPU0\n"); + pr_err("Failed to get device for CPU0\n"); goto skip_voltage_scaling; } @@ -281,7 +281,7 @@ static int __init sti_cpufreq_init(void) goto register_cpufreq_dt; skip_voltage_scaling: - dev_err(ddata.cpu, "Not doing voltage scaling\n"); + pr_err("Not doing voltage scaling\n"); register_cpufreq_dt: platform_device_register_simple("cpufreq-dt", -1, NULL, 0); From 95ac22dd4bf13373f90bb6d51bcd1fd1d8f5db54 Mon Sep 17 00:00:00 2001 From: Rickey Bartlett Date: Mon, 24 Aug 2026 21:32:20 -0700 Subject: [PATCH 0006/1352] PCI: vmd: Flush initiator posted writes before demuxing interrupts on Meteor Lake Meteor Lake VMD (8086:7d0b) is affected by erratum MTL016: the VMD can signal its MSI before the posted writes carrying a child device's DMA data have landed in memory. vmd_irq() then demuxes to the child handler while the child's completion queue is not yet coherent, so the handler observes no completion and returns without consuming it. The I/O is only recovered when the block layer timeout expires and polls the queue: nvme nvme0: I/O tag 253 (50fd) QID 1 timeout, completion polled The practical effect is therefore not a lost I/O but a 30 second stall of the entire storage stack, repeated under any sustained read load. This was originally diagnosed and fixed by Kai-Heng Feng in September 2024 [1]. That patch used udelay(4); Keith Busch objected that the delay is merely a side effect of the read, and that flushing the pending device-to-host writes is what the erratum actually requires. Kai-Heng agreed to respin with a dummy register read. The thread then stalled on an open question from Manivannan Sadhasivam [2]: whether the read must target the child device (the "MSI initiator" named by the erratum) rather than the VMD, and whether the workaround belongs in the NVMe driver instead. No revision followed, and the erratum has remained unmitigated in mainline since. This implements the flush read Keith asked for, and answers the open question empirically: the read MUST complete at the initiating child device. A read of the VMD's own config BAR was tried first and does not help - it terminates at the VMD and never traverses the downstream link, so it does not order against the child's posted writes (measured: timeout rate unchanged). Reading the initiator's config space does order correctly: per PCIe ordering rules the read completion cannot pass the device's earlier posted writes, so returning from the read guarantees the completion queue entry is visible. The initiator's config address is captured per-IRQ at MSI allocation time from the requesting device, so the hot path adds one config read only on affected parts, and only for vectors owned by a child device. The read is serialized with cfg_lock like all other VMD config access. No NVMe driver change is needed. Measured on a Dell Pro Max 14 MC14250 (Meteor Lake, VMD 8086:7d0b, KIOXIA BG6 512GB, 7.0.0-30-generic). The drive is healthy (46C, 0 media errors, 2% used) and ASPM is disabled on the link with all L1 substates off, so neither ASPM nor APST is involved. Dropping caches and reading 3000 shared libraries, measuring /proc/pressure/io "full" (every task on the system blocked on I/O), with nvme_core.io_timeout=5: unpatched: round 1 wall 37.4s full I/O stall 32.6s timeouts 0 round 2 wall 30.6s full I/O stall 28.7s timeouts 1 round 3 wall 30.3s full I/O stall 27.4s timeouts 1 VMD-BAR read: round 1 wall 44.8s full I/O stall 38.6s timeouts 8 (insufficient) round 2 wall 234.8s full I/O stall 207.9s timeouts 37 round 3 wall 3.6s full I/O stall 0.8s timeouts 0 this patch: round 1 wall 3.6s full I/O stall 1.3s timeouts 0 round 2 wall 3.6s full I/O stall 1.3s timeouts 0 round 3 wall 3.6s full I/O stall 1.4s timeouts 0 Before any workaround, 208 seconds of total-system I/O stall accumulated in the first 13 minutes of uptime, roughly 27% of wall clock. Userspace experiences this as GUI applications taking 30-60+ seconds to start while throughput between stalls looks entirely normal (1.9 GB/s QD1) - which is what makes the fault easy to misattribute to ASPM or to the drive. [1] https://lore.kernel.org/all/20240909082657.19660-1-kai.heng.feng@canonical.com/ [2] https://lkml.iu.edu/hypermail/linux/kernel/2409.1/08047.html Reported-by: Kai-Heng Feng Suggested-by: Keith Busch Assisted-by: LLM Signed-off-by: Rickey Bartlett Signed-off-by: Manivannan Sadhasivam Link: https://lore.kernel.org/all/20240909082657.19660-1-kai.heng.feng@canonical.com/ Link: https://lkml.iu.edu/hypermail/linux/kernel/2409.1/08047.html Link: https://patch.msgid.link/20260825043220.9047-1-subtexel@gmail.com --- drivers/pci/controller/vmd.c | 52 ++++++++++++++++++++++++++++++++++-- 1 file changed, 50 insertions(+), 2 deletions(-) diff --git a/drivers/pci/controller/vmd.c b/drivers/pci/controller/vmd.c index 241023ecf67741..e45ef8cb16e47f 100644 --- a/drivers/pci/controller/vmd.c +++ b/drivers/pci/controller/vmd.c @@ -92,6 +92,22 @@ enum vmd_features { * referred to as MEMBAR2 or MSI-X BAR. */ VMD_FEAT_USE_BIOS_INFO = (1 << 6), + + /* + * Meteor Lake VMD (device ID 0x7d0b) is affected by erratum MTL016: + * the VMD may signal its MSI before the posted writes that carry the + * child device's DMA data have landed in memory. The demuxed handler + * then runs against a not-yet-coherent completion queue and misses the + * completion entirely, so the I/O is only recovered when the block + * layer timeout fires and polls the queue ("timeout, completion + * polled"). Intel's documented workaround is to issue a dummy read + * to the MSI initiator (the child device) before handling the + * interrupt: the read completion cannot pass the device's earlier + * posted writes, so it pulls them into memory per PCIe ordering + * rules. A read that terminates at the VMD itself does not order + * against the child's writes and is not sufficient (measured). + */ + VMD_FEAT_INTERRUPT_QUIRK = (1 << 7), }; #define VMD_BIOS_PM_QUIRK_LTR 0x1003 /* 3145728 ns */ @@ -114,6 +130,9 @@ static DEFINE_RAW_SPINLOCK(list_lock); * @irq: back pointer to parent. * @enabled: true if driver enabled IRQ * @virq: the virtual IRQ value provided to the requesting driver. + * @flush_addr: config space address of the initiating device, read before + * demuxing to flush its posted writes (MTL016); NULL if the + * VMD is not affected. * * Every MSI/MSI-X IRQ requested for a device in a VMD domain will be mapped to * a VMD IRQ using this structure. @@ -123,8 +142,14 @@ struct vmd_irq { struct vmd_irq_list *irq; bool enabled; unsigned int virq; + void __iomem *flush_addr; }; +struct vmd_dev; + +static void __iomem *vmd_cfg_addr(struct vmd_dev *vmd, struct pci_bus *bus, + unsigned int devfn, int reg, int len); + /** * struct vmd_irq_list - list of driver requested IRQs mapping to a VMD vector * @irq_list: the list of irq's the VMD one demuxes to. @@ -132,12 +157,15 @@ struct vmd_irq { * @count: number of child IRQs assigned to this vector; used to track * sharing. * @virq: The underlying VMD Linux interrupt number + * @vmd: back pointer to the owning VMD device; serializes the MTL016 + * flush read against other config space access. */ struct vmd_irq_list { struct list_head irq_list; struct srcu_struct srcu; unsigned int count; unsigned int virq; + struct vmd_dev *vmd; }; struct vmd_dev { @@ -295,6 +323,13 @@ static int vmd_msi_alloc(struct irq_domain *domain, unsigned int virq, INIT_LIST_HEAD(&vmdirq->node); vmdirq->irq = vmd_next_irq(vmd, desc); vmdirq->virq = virq + i; + if (vmd->features & VMD_FEAT_INTERRUPT_QUIRK) { + struct pci_dev *pdev = msi_desc_to_pci_dev(desc); + + vmdirq->flush_addr = vmd_cfg_addr(vmd, pdev->bus, + pdev->devfn, + PCI_VENDOR_ID, 2); + } irq_domain_set_info(domain, virq + i, vmdirq->irq->virq, &vmd_msi_controller, vmdirq, @@ -756,8 +791,20 @@ static irqreturn_t vmd_irq(int irq, void *data) int idx; idx = srcu_read_lock(&irqs->srcu); - list_for_each_entry_rcu(vmdirq, &irqs->irq_list, node) + list_for_each_entry_rcu(vmdirq, &irqs->irq_list, node) { + /* + * MTL016: the MSI may have outrun the initiating device's + * posted writes (e.g. its NVMe completion entry). A read + * that completes at the initiator flushes them, so the + * demuxed handler observes a coherent completion queue. + * The value is discarded; only the ordering matters. + */ + if (vmdirq->flush_addr) { + guard(raw_spinlock)(&irqs->vmd->cfg_lock); + readw(vmdirq->flush_addr); + } generic_handle_irq(vmdirq->virq); + } srcu_read_unlock(&irqs->srcu, idx); return IRQ_HANDLED; @@ -788,6 +835,7 @@ static int vmd_alloc_irqs(struct vmd_dev *vmd) return err; INIT_LIST_HEAD(&vmd->irqs[i].irq_list); + vmd->irqs[i].vmd = vmd; vmd->irqs[i].virq = pci_irq_vector(dev, i); err = devm_request_irq(&dev->dev, vmd->irqs[i].virq, vmd_irq, IRQF_NO_THREAD, @@ -1250,7 +1298,7 @@ static const struct pci_device_id vmd_ids[] = { {PCI_VDEVICE(INTEL, 0xa77f), .driver_data = VMD_FEATS_CLIENT,}, {PCI_VDEVICE(INTEL, 0x7d0b), - .driver_data = VMD_FEATS_CLIENT,}, + .driver_data = VMD_FEATS_CLIENT | VMD_FEAT_INTERRUPT_QUIRK,}, {PCI_VDEVICE(INTEL, 0xad0b), .driver_data = VMD_FEATS_CLIENT,}, {PCI_VDEVICE(INTEL, PCI_DEVICE_ID_INTEL_VMD_9A0B), From a500a1b148a85ffb02ad1be7ec2603f6b311a446 Mon Sep 17 00:00:00 2001 From: Frank Zhang Date: Fri, 4 Sep 2026 15:56:35 +0800 Subject: [PATCH 0007/1352] rust: rcpufreq_dt: add module alias Add the missing `platform:cpufreq-dt` module alias so that the driver can be automatically loaded when the platform device is registered. Fixes: 06149d8f2216 ("cpufreq: Add Rust-based cpufreq-dt driver") Cc: stable@vger.kernel.org Signed-off-by: Frank Zhang Signed-off-by: Viresh Kumar --- drivers/cpufreq/rcpufreq_dt.rs | 1 + 1 file changed, 1 insertion(+) diff --git a/drivers/cpufreq/rcpufreq_dt.rs b/drivers/cpufreq/rcpufreq_dt.rs index d7ead60bf8c244..2b1f18b7a1e777 100644 --- a/drivers/cpufreq/rcpufreq_dt.rs +++ b/drivers/cpufreq/rcpufreq_dt.rs @@ -225,4 +225,5 @@ module_platform_driver! { authors: ["Viresh Kumar "], description: "Generic CPUFreq DT driver", license: "GPL v2", + alias: ["platform:cpufreq-dt"], } From d82d896f00e7bf697b61d5df0555941afa8d0657 Mon Sep 17 00:00:00 2001 From: Sumeet Pawnikar Date: Sun, 6 Sep 2026 11:47:52 +0530 Subject: [PATCH 0008/1352] cpufreq: Use %pe to print error pointers symbolically Replace PTR_ERR() and %ld with %pe and pass the original pointer directly to pr_err() and pr_warn(). The %pe format specifier prints a symbolic error name (e.g. -ENOMEM) when CONFIG_SYMBOLIC_ERRNAME is enabled, otherwise it falls back gracefully and prints the raw integer value. This makes messages more readable without any functional change. Signed-off-by: Sumeet Pawnikar Signed-off-by: Viresh Kumar --- drivers/cpufreq/bmips-cpufreq.c | 4 ++-- drivers/cpufreq/cppc_cpufreq.c | 4 ++-- drivers/cpufreq/qoriq-cpufreq.c | 3 +-- drivers/cpufreq/s3c64xx-cpufreq.c | 5 ++--- 4 files changed, 7 insertions(+), 9 deletions(-) diff --git a/drivers/cpufreq/bmips-cpufreq.c b/drivers/cpufreq/bmips-cpufreq.c index a8e35bc75fb2b3..389fff5e6f6529 100644 --- a/drivers/cpufreq/bmips-cpufreq.c +++ b/drivers/cpufreq/bmips-cpufreq.c @@ -132,8 +132,8 @@ static int bmips_cpufreq_init(struct cpufreq_policy *policy) freq_table = bmips_cpufreq_get_freq_table(policy); if (IS_ERR(freq_table)) { - pr_err("%s: couldn't determine frequency table (%ld).\n", - BMIPS_CPUFREQ_NAME, PTR_ERR(freq_table)); + pr_err("%s: couldn't determine frequency table (%pe).\n", + BMIPS_CPUFREQ_NAME, freq_table); return PTR_ERR(freq_table); } diff --git a/drivers/cpufreq/cppc_cpufreq.c b/drivers/cpufreq/cppc_cpufreq.c index 80893844353ca1..f767898ebfb571 100644 --- a/drivers/cpufreq/cppc_cpufreq.c +++ b/drivers/cpufreq/cppc_cpufreq.c @@ -230,8 +230,8 @@ static void cppc_fie_kworker_init(void) kworker_fie = kthread_run_worker(0, "cppc_fie"); if (IS_ERR(kworker_fie)) { - pr_warn("%s: failed to create kworker_fie: %ld\n", __func__, - PTR_ERR(kworker_fie)); + pr_warn("%s: failed to create kworker_fie: %pe\n", __func__, + kworker_fie); fie_disabled = FIE_DISABLED; kworker_fie = NULL; return; diff --git a/drivers/cpufreq/qoriq-cpufreq.c b/drivers/cpufreq/qoriq-cpufreq.c index 42edb41ad45950..a21b6f01218ac5 100644 --- a/drivers/cpufreq/qoriq-cpufreq.c +++ b/drivers/cpufreq/qoriq-cpufreq.c @@ -57,8 +57,7 @@ static u32 get_bus_freq(void) /* get platform freq by its clock name */ pltclk = clk_get(NULL, "cg-pll0-div1"); if (IS_ERR(pltclk)) { - pr_err("%s: can't get bus frequency %ld\n", - __func__, PTR_ERR(pltclk)); + pr_err("%s: can't get bus frequency %pe\n", __func__, pltclk); return PTR_ERR(pltclk); } diff --git a/drivers/cpufreq/s3c64xx-cpufreq.c b/drivers/cpufreq/s3c64xx-cpufreq.c index 9cef7152807626..9a01592425ee6e 100644 --- a/drivers/cpufreq/s3c64xx-cpufreq.c +++ b/drivers/cpufreq/s3c64xx-cpufreq.c @@ -152,15 +152,14 @@ static int s3c64xx_cpufreq_driver_init(struct cpufreq_policy *policy) policy->clk = clk_get(NULL, "armclk"); if (IS_ERR(policy->clk)) { - pr_err("Unable to obtain ARMCLK: %ld\n", - PTR_ERR(policy->clk)); + pr_err("Unable to obtain ARMCLK: %pe\n", policy->clk); return PTR_ERR(policy->clk); } #ifdef CONFIG_REGULATOR vddarm = regulator_get(NULL, "vddarm"); if (IS_ERR(vddarm)) { - pr_err("Failed to obtain VDDARM: %ld\n", PTR_ERR(vddarm)); + pr_err("Failed to obtain VDDARM: %pe\n", vddarm); pr_err("Only frequency scaling available\n"); vddarm = NULL; } else { From 3018008dc1c3ca1d8c3f0a2362e15a1f47e66616 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Grzelak?= Date: Fri, 4 Sep 2026 14:31:41 +0200 Subject: [PATCH 0009/1352] drm/i915/bios: search for VBT #57 by default MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Start searching for Vswing / Preemphasis Override Block during VBT parsing at init_bdb_blocks(). Check for failure since pre-ICL GOPs do not contain the block. Check also if VBT version is appropriately up-to-date. v6->v7 - parse VBT#57 before blocks dependent on child device list (Jani) - remove debug message (Suraj) v3->v4 - add Bspec (Suraj) Bspec: 32063 Signed-off-by: Michał Grzelak Reviewed-by: Suraj Kandpal Acked-by: Jani Nikula Signed-off-by: Suraj Kandpal Link: https://patch.msgid.link/20260904123148.2165596-2-michal.grzelak@intel.com --- drivers/gpu/drm/i915/display/intel_bios.c | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/drivers/gpu/drm/i915/display/intel_bios.c b/drivers/gpu/drm/i915/display/intel_bios.c index 97cbae2e547e2d..ece4df4a9d62a5 100644 --- a/drivers/gpu/drm/i915/display/intel_bios.c +++ b/drivers/gpu/drm/i915/display/intel_bios.c @@ -200,6 +200,8 @@ static const struct { .min_size = sizeof(struct bdb_mipi_sequence) }, { .section_id = BDB_COMPRESSION_PARAMETERS, .min_size = sizeof(struct bdb_compression_parameters), }, + { .section_id = BDB_VSWING_PREEMPH, + .min_size = sizeof(struct bdb_vswing_preemph), }, { .section_id = BDB_GENERIC_DTD, .min_size = sizeof(struct bdb_generic_dtd), }, }; @@ -2203,6 +2205,21 @@ parse_compression_parameters(struct intel_display *display) } } +static void +parse_vswing_preemph_override(struct intel_display *display) +{ + const struct bdb_vswing_preemph *block; + + if (display->vbt.version < 218) + return; + + block = bdb_find_section(display, BDB_VSWING_PREEMPH); + + /* pre-ICL GOPs don't have VBT #57 */ + if (!block) + return; +} + static u8 translate_iboost(struct intel_display *display, u8 val) { static const u8 mapping[] = { 1, 3, 7 }; /* See VBT spec */ @@ -3291,6 +3308,7 @@ void intel_bios_init(struct intel_display *display) parse_general_features(display); parse_general_definitions(display); parse_driver_features(display); + parse_vswing_preemph_override(display); /* Depends on child device list */ parse_compression_parameters(display); From c76c03dab828709469558a6bdd55336ef9fd9420 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Grzelak?= Date: Fri, 4 Sep 2026 14:31:42 +0200 Subject: [PATCH 0010/1352] drm/i915/bios: store VBT #57's metadata in intel_vbt_data MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Store tables, number of tables, number of rows and number of columns in intel_vbt_data when search for the VBT #57 has succeeded. Structurize all VS/PE-O relevant metadata inside anonymous struct named as vspeo. Presence of C20 or newer PHY causes each table to contain 16 rows. Each table contains 10 rows in case C20 PHY is absent. Use display version to determine number of rows since there is no helper in intel_bios.c to check presence of any C20+ PHY. pre-MTL platforms should have 10 rows while MTL+ should have 16 rows. v5->v6 - add Bspec (Suraj) v3->v4 - remove unnecessary init of VS/PE-O metadata (Suraj) - add helper for computing number of rows (Suraj) - fix num_rows's type (Jani, Suraj) - declare num_rows (Suraj) Bspec: 68963 Signed-off-by: Michał Grzelak Reviewed-by: Suraj Kandpal Acked-by: Jani Nikula Signed-off-by: Suraj Kandpal Link: https://patch.msgid.link/20260904123148.2165596-3-michal.grzelak@intel.com --- drivers/gpu/drm/i915/display/intel_bios.c | 10 ++++++++++ drivers/gpu/drm/i915/display/intel_display_core.h | 7 +++++++ 2 files changed, 17 insertions(+) diff --git a/drivers/gpu/drm/i915/display/intel_bios.c b/drivers/gpu/drm/i915/display/intel_bios.c index ece4df4a9d62a5..06a8dd7581bb8a 100644 --- a/drivers/gpu/drm/i915/display/intel_bios.c +++ b/drivers/gpu/drm/i915/display/intel_bios.c @@ -2205,6 +2205,11 @@ parse_compression_parameters(struct intel_display *display) } } +static int vswing_preemph_num_rows(struct intel_display *display) +{ + return DISPLAY_VER(display) >= 14 ? 16 : 10; +} + static void parse_vswing_preemph_override(struct intel_display *display) { @@ -2218,6 +2223,11 @@ parse_vswing_preemph_override(struct intel_display *display) /* pre-ICL GOPs don't have VBT #57 */ if (!block) return; + + display->vbt.vspeo.tables = block->tables; + display->vbt.vspeo.num_tables = block->num_tables; + display->vbt.vspeo.num_columns = block->num_columns; + display->vbt.vspeo.num_rows = vswing_preemph_num_rows(display); } static u8 translate_iboost(struct intel_display *display, u8 val) diff --git a/drivers/gpu/drm/i915/display/intel_display_core.h b/drivers/gpu/drm/i915/display/intel_display_core.h index 7e988b7b1fe7a5..c80b4a2f74f9e9 100644 --- a/drivers/gpu/drm/i915/display/intel_display_core.h +++ b/drivers/gpu/drm/i915/display/intel_display_core.h @@ -244,6 +244,13 @@ struct intel_vbt_data { struct list_head display_devices; struct list_head bdb_blocks; + struct { + const u32 *tables; + int num_tables; + int num_columns; + int num_rows; + } vspeo; + struct sdvo_device_mapping { u8 initialized; u8 dvo_port; From 1b15117625045baa107970a9a1ffb5638ab05414 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Grzelak?= Date: Fri, 4 Sep 2026 14:31:43 +0200 Subject: [PATCH 0011/1352] drm/i915/bios: print VS/PE-O port info MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Issue a debug message when port asks to override default Vswing / Preemphasis tables. Add helper intel_bios_encoder_requests_vspeo() to check if port requests for overriding default VS/PE tables. v6->v7 - expand VS/PE-O acronym in debug logging (Jani) v3->v4 - change debug message when requesting VS/PE-O (Suraj) Signed-off-by: Michał Grzelak Reviewed-by: Suraj Kandpal Acked-by: Jani Nikula Signed-off-by: Suraj Kandpal Link: https://patch.msgid.link/20260904123148.2165596-4-michal.grzelak@intel.com --- drivers/gpu/drm/i915/display/intel_bios.c | 10 ++++++++++ drivers/gpu/drm/i915/display/intel_bios.h | 1 + 2 files changed, 11 insertions(+) diff --git a/drivers/gpu/drm/i915/display/intel_bios.c b/drivers/gpu/drm/i915/display/intel_bios.c index 06a8dd7581bb8a..9610b794bc147e 100644 --- a/drivers/gpu/drm/i915/display/intel_bios.c +++ b/drivers/gpu/drm/i915/display/intel_bios.c @@ -2801,6 +2801,11 @@ static void print_ddi_port(const struct intel_bios_encoder_data *devdata) "Port %c supports dynamic DDI allocation in TCSS\n", port_name(port)); + if (intel_bios_encoder_requests_vspeo(devdata)) + drm_dbg_kms(display->drm, + "Port %c requests vswing/pre-emphasis override\n", + port_name(port)); + hdmi_level_shift = intel_bios_hdmi_level_shift(devdata); if (hdmi_level_shift >= 0) { drm_dbg_kms(display->drm, @@ -3829,6 +3834,11 @@ int intel_bios_hdmi_ddc_pin(const struct intel_bios_encoder_data *devdata) return map_ddc_pin(devdata->display, devdata->child.ddc_pin); } +bool intel_bios_encoder_requests_vspeo(const struct intel_bios_encoder_data *devdata) +{ + return devdata->display->vbt.version >= 218 && devdata->child.use_vbt_vswing; +} + bool intel_bios_encoder_supports_typec_usb(const struct intel_bios_encoder_data *devdata) { return devdata->display->vbt.version >= 195 && devdata->child.dp_usb_type_c; diff --git a/drivers/gpu/drm/i915/display/intel_bios.h b/drivers/gpu/drm/i915/display/intel_bios.h index 75dff27b42289d..7a50a272cd27d7 100644 --- a/drivers/gpu/drm/i915/display/intel_bios.h +++ b/drivers/gpu/drm/i915/display/intel_bios.h @@ -73,6 +73,7 @@ bool intel_bios_get_dsc_params(struct intel_encoder *encoder, const struct intel_bios_encoder_data * intel_bios_encoder_data_lookup(struct intel_display *display, enum port port); +bool intel_bios_encoder_requests_vspeo(const struct intel_bios_encoder_data *devdata); bool intel_bios_encoder_supports_dvi(const struct intel_bios_encoder_data *devdata); bool intel_bios_encoder_supports_hdmi(const struct intel_bios_encoder_data *devdata); bool intel_bios_encoder_supports_dp(const struct intel_bios_encoder_data *devdata); From a138a81d7805a025a778035c0850642381d1fdf9 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Grzelak?= Date: Fri, 4 Sep 2026 14:31:44 +0200 Subject: [PATCH 0012/1352] drm/i915/bios: de/allocate VS/PE-O buffers for each port MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Every devdata needs VS/PE-O dedicated buffers since each port can request an override. Add intel_ddi_buf_trans{,_entry} pointers into intel_bios_encoder_data. Allocate struct intel_ddi_buf_trans{,_entry} for the port if VS/PE-O was requested and is supported. Keep NULL in vspeo if any allocation failed or VS/PE-O was not requested. It will be used later for checking if override should actually take place. Note that we theoretically could store intel_ddi_buf_trans_entry inside `entries` field of newly allocated intel_ddi_buf_trans. However it will be impossible to overwrite the buffer during intel_ddi_get_buf_trans() without discarding const qualifier of `entries` field. This would involve either void casting or deconstifying entries field and in turn all predefined tables as well. Thus add a separate non-const qualified field into intel_bios_encoder_data for the buffer, which after overwriting will be promoted to be const qualified. Deallocate the buffer as well as entries if requested. v11->v12 - set vspeo->num_entries once (Sashiko) - free allocated vspeo->entries (Sashiko) v9->v10 - add separate non-const field for `entries` caching - cache `entries` into const field after data is overwritten (Jani) v4->v5 - set devdata->vspeo->num_entries in intel_bios.c Signed-off-by: Michał Grzelak Reviewed-by: Suraj Kandpal Acked-by: Jani Nikula Signed-off-by: Suraj Kandpal Link: https://patch.msgid.link/20260904123148.2165596-5-michal.grzelak@intel.com --- drivers/gpu/drm/i915/display/intel_bios.c | 31 +++++++++++++++++++++++ 1 file changed, 31 insertions(+) diff --git a/drivers/gpu/drm/i915/display/intel_bios.c b/drivers/gpu/drm/i915/display/intel_bios.c index 9610b794bc147e..1a09f7933e4916 100644 --- a/drivers/gpu/drm/i915/display/intel_bios.c +++ b/drivers/gpu/drm/i915/display/intel_bios.c @@ -34,6 +34,7 @@ #include #include +#include "intel_ddi_buf_trans.h" #include "intel_display.h" #include "intel_display_core.h" #include "intel_display_rpm.h" @@ -72,6 +73,8 @@ struct intel_bios_encoder_data { struct intel_display *display; + struct intel_ddi_buf_trans *vspeo; + union intel_ddi_buf_trans_entry *entries; struct child_device_config child; struct dsc_compression_parameters_entry *dsc; struct list_head node; @@ -2648,6 +2651,30 @@ static void sanitize_device_type(struct intel_bios_encoder_data *devdata, devdata->child.device_type |= DEVICE_TYPE_NOT_HDMI_OUTPUT; } +static void allocate_vswing_preemph_override(struct intel_bios_encoder_data *devdata) +{ + int num_rows = devdata->display->vbt.vspeo.num_rows; + union intel_ddi_buf_trans_entry *entries; + struct intel_ddi_buf_trans *vspeo; + + if (!intel_bios_encoder_requests_vspeo(devdata)) + return; + + vspeo = kzalloc_obj(*vspeo); + if (!vspeo) + return; + + entries = kzalloc_objs(*entries, num_rows); + if (!entries) { + kfree(vspeo); + return; + } + + devdata->entries = entries; + devdata->vspeo = vspeo; + devdata->vspeo->num_entries = num_rows; +} + static void sanitize_hdmi_level_shift(struct intel_bios_encoder_data *devdata, enum port port) { @@ -2866,6 +2893,7 @@ static void parse_ddi_port(struct intel_bios_encoder_data *devdata) sanitize_dedicated_external(devdata, port); sanitize_device_type(devdata, port); sanitize_hdmi_level_shift(devdata, port); + allocate_vswing_preemph_override(devdata); } static bool has_ddi_port_info(struct intel_display *display) @@ -3403,6 +3431,9 @@ void intel_bios_driver_remove(struct intel_display *display) list_for_each_entry_safe(devdata, nd, &display->vbt.display_devices, node) { list_del(&devdata->node); + + kfree(devdata->entries); + kfree(devdata->vspeo); kfree(devdata->dsc); kfree(devdata); } From e084812de907c63b0b943b3955bf629df62f6745 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Grzelak?= Date: Fri, 4 Sep 2026 14:31:45 +0200 Subject: [PATCH 0013/1352] drm/i915/buf_trans: add vfunc for VS/PE-O MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Choosing correct table for Vswing / Pre-emphasis Override is platform specific. It also requires different checks that are already used for choosing predefined tables. Add new get_buf_trans_override() vfunc into intel_encoder returning deparsed table from VBT#57. In next patches, set it inside already present if-ladder from intel_ddi_buf_trans_init() instead of duplicating it. Note that get_buf_trans() cannot be overwritten since there are cases when we need to rollback although VS/PE-O was requested, eg. DP is not connected or feature is not yet implemented for the platform. Assume that vfunc returns NULL on rollback and return predefined tables. Suggested-by: Jani Nikula Signed-off-by: Michał Grzelak Reviewed-by: Suraj Kandpal Acked-by: Jani Nikula Signed-off-by: Suraj Kandpal Link: https://patch.msgid.link/20260904123148.2165596-6-michal.grzelak@intel.com --- drivers/gpu/drm/i915/display/intel_ddi_buf_trans.c | 8 ++++++++ drivers/gpu/drm/i915/display/intel_display_types.h | 3 +++ 2 files changed, 11 insertions(+) diff --git a/drivers/gpu/drm/i915/display/intel_ddi_buf_trans.c b/drivers/gpu/drm/i915/display/intel_ddi_buf_trans.c index 4cd1e4d76c7afa..f31283a0331b78 100644 --- a/drivers/gpu/drm/i915/display/intel_ddi_buf_trans.c +++ b/drivers/gpu/drm/i915/display/intel_ddi_buf_trans.c @@ -1857,5 +1857,13 @@ const struct intel_ddi_buf_trans *intel_ddi_buf_trans_get(struct intel_encoder * const struct intel_crtc_state *crtc_state, int *n_entries) { + if (encoder->get_buf_trans_override) { + const struct intel_ddi_buf_trans *override; + + override = encoder->get_buf_trans_override(encoder, crtc_state, n_entries); + if (override) + return intel_get_buf_trans(override, n_entries); + } + return encoder->get_buf_trans(encoder, crtc_state, n_entries); } diff --git a/drivers/gpu/drm/i915/display/intel_display_types.h b/drivers/gpu/drm/i915/display/intel_display_types.h index 5f0fe18c0614e4..9016be52c7eac8 100644 --- a/drivers/gpu/drm/i915/display/intel_display_types.h +++ b/drivers/gpu/drm/i915/display/intel_display_types.h @@ -289,6 +289,9 @@ struct intel_encoder { */ enum icl_port_dpll_id (*port_pll_type)(struct intel_encoder *encoder, const struct intel_crtc_state *crtc_state); + const struct intel_ddi_buf_trans *(*get_buf_trans_override)(struct intel_encoder *encoder, + const struct intel_crtc_state *crtc_state, + int *n_entries); const struct intel_ddi_buf_trans *(*get_buf_trans)(struct intel_encoder *encoder, const struct intel_crtc_state *crtc_state, int *n_entries); From e8fe5008ae3da28c06ccdaf63d0621c0c50f39dc Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Grzelak?= Date: Fri, 4 Sep 2026 14:31:46 +0200 Subject: [PATCH 0014/1352] drm/i915: override Snps's VS/PE when requested MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add accessor functions for Snps to read requested table from VBT #57. Parse the requested table and transform data into port's buffer. Actual data is fully stored in the lowest byte although each entry is 4 bytes wide. Thus convert u32 into u8 and store the data. For C20, use 6th table if encoder supports DP 2.0 or higher. Otherwise use 5th table for DP. For C20, tables 1-4 are not used at all and are most likely to be zeroed. 5th table is used for any mode below DP 2.0 (exclusive). 6th table is used for any mode above DP 2.0 (inclusive). For C10, use 2nd table for external DP if encoder supports any mode beyond or including HBR2. Use 1st table if external DP encoder supports anything lower than HBR2. For eDP, use 4th table if encoder supports HBR3. Otherwise use 3rd table for eDP. For C10, 1st table is used for external DP with modes below HBR2 (exclusive). 2nd table is used for external DP with modes higher than HBR2 (inclusive). 3rd table is used for eDP with modes lower than HBR3 (exclusive). 4th table is used for eDP with modes higher than HBR3 (inclusive). Indices for other tables have not yet been observed to be used as of now. There are no changes to intel_ddi_dp_level() since selection of correct row of intel_ddi_buf_trans_entry is same as when no override request has been done. v11->v12 - don't set vspeo->num_entries per PHY - don't refer to 1st table as fallback for non-DP for C10 (Sashiko) v10->v11 - remove no-longer-relevant check for NULL devdata (Jani) - initialize local variables at declaration block (Jani) - branch with 'else` instead of initializing twice (Jani) - use blank line before 'return` (Jani) v9->v10 - call dedicated VS/PE-O vfunc - drop deconstifying default tables (Suraj, Jani) - cache `entries` into const field after data is overwritten (Jani) v8->v9 - init vspeo before using it - deconstify intel_ddi_buf_trans_entry v7->v8 - remove comments (Suraj) - add check for LT (Suraj) v6->v7 - handle VS/PE-O's VBT details in intel_bios_* functions (Jani) - remove vspeo's cast to (void *) (Jani) - check devdata->vspeo if VS/PE-O was requested - call encoder->get_buf_trans() once (Jani) - return NULL from intel_bios_get_* when using default (Jani) - validate VS/PE-O in intel_bios.c (Jani) - inline mtl_{c10,c20}_get_vspeo_buf_trans() - remove temporarily LT v4->v5 - blend index computation with table parsing - remove enums entirely - change funcs prefix from snps_ to mtl_ (Suraj) - add spaces around operators (Suraj) - remove spaces after type casting (Suraj) - remove INTEL_DISPLAY_STATE_WARN (Suraj) v3->v4 - stick to solely changing VBT data into current structures (Jani) - move iterator declaration to declaration block (Suraj) v2->v3 - remove unnecessary braces from if block (Suraj) - return -EINVAL instead of -1 (Suraj) Signed-off-by: Michał Grzelak Reviewed-by: Suraj Kandpal Acked-by: Jani Nikula Signed-off-by: Suraj Kandpal Link: https://patch.msgid.link/20260904123148.2165596-7-michal.grzelak@intel.com --- drivers/gpu/drm/i915/display/intel_bios.c | 91 +++++++++++++++++++ drivers/gpu/drm/i915/display/intel_bios.h | 7 ++ .../drm/i915/display/intel_ddi_buf_trans.c | 37 +++++++- 3 files changed, 133 insertions(+), 2 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_bios.c b/drivers/gpu/drm/i915/display/intel_bios.c index 1a09f7933e4916..be7554c483ad2e 100644 --- a/drivers/gpu/drm/i915/display/intel_bios.c +++ b/drivers/gpu/drm/i915/display/intel_bios.c @@ -3880,6 +3880,97 @@ bool intel_bios_encoder_supports_tbt(const struct intel_bios_encoder_data *devda return devdata->display->vbt.version >= 209 && devdata->child.tbt; } +static bool +validate_vspeo(const struct intel_bios_encoder_data *devdata, bool has_dp) +{ + struct intel_ddi_buf_trans *vspeo; + + vspeo = devdata->vspeo; + if (!vspeo) + return false; + + if (!has_dp) + return false; + + return true; +} + +const struct intel_ddi_buf_trans * +intel_bios_get_c20_vspeo(const struct intel_bios_encoder_data *devdata, + bool has_dp, bool is_uhbr) +{ + struct intel_display *display = devdata->display; + union intel_ddi_buf_trans_entry *entries = devdata->entries; + struct intel_ddi_buf_trans *vspeo = devdata->vspeo; + const u32 *tables = display->vbt.vspeo.tables; + int num_columns = display->vbt.vspeo.num_columns; + int num_rows = display->vbt.vspeo.num_rows; + int idx = is_uhbr ? 5 : 4; + size_t offset = 0; + int level; + + if (!validate_vspeo(devdata, has_dp)) + return NULL; + + offset += idx * num_rows * num_columns; + + for (level = 0; level < num_rows; level++) { + u32 vswing = tables[offset]; + u32 pre_cursor = tables[offset + 1]; + u32 post_cursor = tables[offset + 2]; + + entries[level].snps.vswing = vswing; + entries[level].snps.pre_cursor = pre_cursor; + entries[level].snps.post_cursor = post_cursor; + + offset += num_columns; + } + + vspeo->entries = entries; + + return vspeo; +} + +const struct intel_ddi_buf_trans * +intel_bios_get_c10_vspeo(const struct intel_bios_encoder_data *devdata, + bool has_dp, int port_clock, bool has_edp) +{ + struct intel_display *display = devdata->display; + union intel_ddi_buf_trans_entry *entries = devdata->entries; + struct intel_ddi_buf_trans *vspeo = devdata->vspeo; + const u32 *tables = display->vbt.vspeo.tables; + int num_columns = display->vbt.vspeo.num_columns; + int num_rows = display->vbt.vspeo.num_rows; + size_t offset = 0; + int level, idx; + + if (!validate_vspeo(devdata, has_dp)) + return NULL; + + if (has_edp) + idx = port_clock > 540000 ? 3 : 2; + else + idx = port_clock > 270000 ? 1 : 0; + + offset += idx * num_rows * num_columns; + + for (level = 0; level < num_rows; level++) { + u32 vswing = tables[offset]; + u32 pre_cursor = tables[offset + 1]; + u32 post_cursor = tables[offset + 2]; + + entries[level].snps.vswing = vswing; + entries[level].snps.pre_cursor = pre_cursor; + entries[level].snps.post_cursor = post_cursor; + + offset += num_columns; + } + + vspeo->entries = entries; + + return vspeo; +} + bool intel_bios_encoder_is_dedicated_external(const struct intel_bios_encoder_data *devdata) { return devdata->display->vbt.version >= 264 && diff --git a/drivers/gpu/drm/i915/display/intel_bios.h b/drivers/gpu/drm/i915/display/intel_bios.h index 7a50a272cd27d7..49acf8c405e2bc 100644 --- a/drivers/gpu/drm/i915/display/intel_bios.h +++ b/drivers/gpu/drm/i915/display/intel_bios.h @@ -73,6 +73,13 @@ bool intel_bios_get_dsc_params(struct intel_encoder *encoder, const struct intel_bios_encoder_data * intel_bios_encoder_data_lookup(struct intel_display *display, enum port port); +const struct intel_ddi_buf_trans * +intel_bios_get_c20_vspeo(const struct intel_bios_encoder_data *devdata, + bool has_dp, bool is_uhbr); +const struct intel_ddi_buf_trans * +intel_bios_get_c10_vspeo(const struct intel_bios_encoder_data *devdata, + bool has_dp, int port_clock, bool has_edp); + bool intel_bios_encoder_requests_vspeo(const struct intel_bios_encoder_data *devdata); bool intel_bios_encoder_supports_dvi(const struct intel_bios_encoder_data *devdata); bool intel_bios_encoder_supports_hdmi(const struct intel_bios_encoder_data *devdata); diff --git a/drivers/gpu/drm/i915/display/intel_ddi_buf_trans.c b/drivers/gpu/drm/i915/display/intel_ddi_buf_trans.c index f31283a0331b78..92c0d0f933ab44 100644 --- a/drivers/gpu/drm/i915/display/intel_ddi_buf_trans.c +++ b/drivers/gpu/drm/i915/display/intel_ddi_buf_trans.c @@ -1784,6 +1784,36 @@ xe3plpd_get_lt_buf_trans(struct intel_encoder *encoder, return intel_get_buf_trans(&xe3plpd_lt_trans_dp14, n_entries); } +static const struct intel_ddi_buf_trans * +mtl_get_c10_buf_trans_override(struct intel_encoder *encoder, + const struct intel_crtc_state *crtc_state, + int *n_entries) +{ + const struct intel_bios_encoder_data *devdata = encoder->devdata; + bool has_edp, has_dp; + int port_clock; + + has_edp = intel_crtc_has_type(crtc_state, INTEL_OUTPUT_EDP); + has_dp = intel_crtc_has_dp_encoder(crtc_state); + port_clock = crtc_state->port_clock; + + return intel_bios_get_c10_vspeo(devdata, has_dp, port_clock, has_edp); +} + +static const struct intel_ddi_buf_trans * +mtl_get_c20_buf_trans_override(struct intel_encoder *encoder, + const struct intel_crtc_state *crtc_state, + int *n_entries) +{ + const struct intel_bios_encoder_data *devdata = encoder->devdata; + bool has_dp, is_uhbr; + + has_dp = intel_crtc_has_dp_encoder(crtc_state); + is_uhbr = intel_dp_is_uhbr(crtc_state); + + return intel_bios_get_c20_vspeo(devdata, has_dp, is_uhbr); +} + void intel_ddi_buf_trans_init(struct intel_encoder *encoder) { struct intel_display *display = to_intel_display(encoder); @@ -1791,10 +1821,13 @@ void intel_ddi_buf_trans_init(struct intel_encoder *encoder) if (HAS_LT_PHY(display)) { encoder->get_buf_trans = xe3plpd_get_lt_buf_trans; } else if (DISPLAY_VER(display) >= 14) { - if (intel_encoder_is_c10phy(encoder)) + if (intel_encoder_is_c10phy(encoder)) { encoder->get_buf_trans = mtl_get_c10_buf_trans; - else + encoder->get_buf_trans_override = mtl_get_c10_buf_trans_override; + } else { encoder->get_buf_trans = mtl_get_c20_buf_trans; + encoder->get_buf_trans_override = mtl_get_c20_buf_trans_override; + } } else if (display->platform.dg2) { encoder->get_buf_trans = dg2_get_snps_buf_trans; } else if (display->platform.alderlake_p) { From 7fdf86e530c7f6655aad0612a8c90ab1559984bc Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Grzelak?= Date: Mon, 7 Sep 2026 14:27:42 +0200 Subject: [PATCH 0015/1352] drm/i915: override Combo's VS/PE when requested MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add accessor function for Combo to read requested table from VBT #57. Parse the requested table and transform data into port's buffer. Actual data is fully stored in the lowest byte although each entry is 4 bytes wide. Thus convert u32 into u8 and store the data. For EHL, in cases when eDP encoder uses low vswing, choose 3rd table if encoder supports HBR3. Otherwise use 2nd table for eDP using low vswing. In cases when eDP encoder does not use low vswing, choose 2nd table if encoder supports mode higher or including HBR2. Otherwise use 1st table for eDP not using low vswing. For external DP follow same path and use same indices as in eDP without low vswing case. For JSL, always use 1st table for external DP. For eDPs not using low vswing use 1st table as well. In cases when eDP encoder uses low vswing, choose 1st table if encoder supports HBR3. When encoder supports HBR2 choose 3rd table. When encoder supports modes lower than HBR2 choose 2nd table. There are no changes to intel_ddi_dp_level() since selection of correct row of intel_ddi_buf_trans_entry is same as when no override request has been done. Looking from other OSes, in case when encoder does not support DP we could theoretically use 1st table. However, as of now, use default tables. v11->v12 - don't set vspeo->num_entries per PHY/platform - check for low vswing eDP for EHL (Sashiko) - reverse order of indices for JSL (Sashiko) v10->v11 - initialize local variables at declaration block (Jani) - branch with 'else` instead of initializing twice (Jani) v9->v10 - call dedicated VS/PE-O vfunc - drop deconstifying default tables (Suraj, Jani) - cache `entries` into const field after data is overwritten (Jani) v8->v9 - deconstify intel_ddi_buf_trans_entry v6->v7 - handle VS/PE-O's VBT details in intel_bios_* functions (Jani) - remove vspeo's cast to (void *) (Jani) - call encoder->get_buf_trans() once (Jani) - return NULL from intel_bios_get_* when using default (Jani) - validate VS/PE-O in intel_bios.c (Jani) - check devdata->vspeo if VS/PE-O was requested - inline {jsl,ehl}_combo_get_vspeo_buf_trans() - remove temporarily LT v4->v5 - blend index computation with table parsing - remove enums entirely - add spaces around operators (Suraj) - remove spaces after type casting (Suraj) - remove INTEL_DISPLAY_STATE_WARN (Suraj) Signed-off-by: Michał Grzelak Reviewed-by: Suraj Kandpal Acked-by: Jani Nikula Signed-off-by: Suraj Kandpal Link: https://patch.msgid.link/20260907122742.2512901-1-michal.grzelak@intel.com --- drivers/gpu/drm/i915/display/intel_bios.c | 93 +++++++++++++++++++ drivers/gpu/drm/i915/display/intel_bios.h | 6 ++ .../drm/i915/display/intel_ddi_buf_trans.c | 45 ++++++++- 3 files changed, 140 insertions(+), 4 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_bios.c b/drivers/gpu/drm/i915/display/intel_bios.c index be7554c483ad2e..11d2cbb78eabf3 100644 --- a/drivers/gpu/drm/i915/display/intel_bios.c +++ b/drivers/gpu/drm/i915/display/intel_bios.c @@ -3971,6 +3971,99 @@ intel_bios_get_c10_vspeo(const struct intel_bios_encoder_data *devdata, return vspeo; } +const struct intel_ddi_buf_trans * +intel_bios_get_ehl_combo_vspeo(const struct intel_bios_encoder_data *devdata, + bool has_dp, int port_clock, bool low_vswing_edp) +{ + struct intel_display *display = devdata->display; + union intel_ddi_buf_trans_entry *entries = devdata->entries; + struct intel_ddi_buf_trans *vspeo = devdata->vspeo; + const u32 *tables = display->vbt.vspeo.tables; + int num_columns = display->vbt.vspeo.num_columns; + int num_rows = display->vbt.vspeo.num_rows; + size_t offset = 0; + int level, idx; + + if (!validate_vspeo(devdata, has_dp)) + return NULL; + + if (low_vswing_edp) + idx = port_clock > 540000 ? 2 : 1; + else + idx = port_clock > 270000 ? 1 : 0; + + offset += idx * num_rows * num_columns; + + for (level = 0; level < num_rows; level++) { + u32 dw2_swing_sel = tables[offset]; + u32 dw7_n_scalar = tables[offset + 1]; + u32 dw4_cursor_coeff = tables[offset + 2]; + u32 dw4_post_cursor_2 = tables[offset + 3]; + u32 dw4_post_cursor_1 = tables[offset + 4]; + + entries[level].icl.dw2_swing_sel = dw2_swing_sel; + entries[level].icl.dw7_n_scalar = dw7_n_scalar; + entries[level].icl.dw4_cursor_coeff = dw4_cursor_coeff; + entries[level].icl.dw4_post_cursor_2 = dw4_post_cursor_2; + entries[level].icl.dw4_post_cursor_1 = dw4_post_cursor_1; + + offset += num_columns; + } + + vspeo->entries = entries; + + return vspeo; +} + +const struct intel_ddi_buf_trans * +intel_bios_get_jsl_combo_vspeo(const struct intel_bios_encoder_data *devdata, + bool has_dp, int port_clock, bool low_vswing_edp) +{ + struct intel_display *display = devdata->display; + union intel_ddi_buf_trans_entry *entries = devdata->entries; + struct intel_ddi_buf_trans *vspeo = devdata->vspeo; + const u32 *tables = display->vbt.vspeo.tables; + int num_columns = display->vbt.vspeo.num_columns; + int num_rows = display->vbt.vspeo.num_rows; + size_t offset = 0; + int idx = 0; + int level; + + if (!validate_vspeo(devdata, has_dp)) + return NULL; + + if (low_vswing_edp) { + if (port_clock > 540000) + idx = 0; + else if (port_clock > 270000) + idx = 2; + else + idx = 1; + } + + offset += idx * num_rows * num_columns; + + for (level = 0; level < num_rows; level++) { + u32 dw2_swing_sel = tables[offset]; + u32 dw7_n_scalar = tables[offset + 1]; + u32 dw4_cursor_coeff = tables[offset + 2]; + u32 dw4_post_cursor_2 = tables[offset + 3]; + u32 dw4_post_cursor_1 = tables[offset + 4]; + + entries[level].icl.dw2_swing_sel = dw2_swing_sel; + entries[level].icl.dw7_n_scalar = dw7_n_scalar; + entries[level].icl.dw4_cursor_coeff = dw4_cursor_coeff; + entries[level].icl.dw4_post_cursor_2 = dw4_post_cursor_2; + entries[level].icl.dw4_post_cursor_1 = dw4_post_cursor_1; + + offset += num_columns; + } + + vspeo->entries = entries; + + return vspeo; +} + bool intel_bios_encoder_is_dedicated_external(const struct intel_bios_encoder_data *devdata) { return devdata->display->vbt.version >= 264 && diff --git a/drivers/gpu/drm/i915/display/intel_bios.h b/drivers/gpu/drm/i915/display/intel_bios.h index 49acf8c405e2bc..196b8b3f434f54 100644 --- a/drivers/gpu/drm/i915/display/intel_bios.h +++ b/drivers/gpu/drm/i915/display/intel_bios.h @@ -79,6 +79,12 @@ intel_bios_get_c20_vspeo(const struct intel_bios_encoder_data *devdata, const struct intel_ddi_buf_trans * intel_bios_get_c10_vspeo(const struct intel_bios_encoder_data *devdata, bool has_dp, int port_clock, bool has_edp); +const struct intel_ddi_buf_trans * +intel_bios_get_ehl_combo_vspeo(const struct intel_bios_encoder_data *devdata, + bool has_dp, int port_clock, bool low_vswing_edp); +const struct intel_ddi_buf_trans * +intel_bios_get_jsl_combo_vspeo(const struct intel_bios_encoder_data *devdata, + bool has_dp, int port_clock, bool low_vswing_edp); bool intel_bios_encoder_requests_vspeo(const struct intel_bios_encoder_data *devdata); bool intel_bios_encoder_supports_dvi(const struct intel_bios_encoder_data *devdata); diff --git a/drivers/gpu/drm/i915/display/intel_ddi_buf_trans.c b/drivers/gpu/drm/i915/display/intel_ddi_buf_trans.c index 92c0d0f933ab44..cc5912178d9cdb 100644 --- a/drivers/gpu/drm/i915/display/intel_ddi_buf_trans.c +++ b/drivers/gpu/drm/i915/display/intel_ddi_buf_trans.c @@ -1784,6 +1784,40 @@ xe3plpd_get_lt_buf_trans(struct intel_encoder *encoder, return intel_get_buf_trans(&xe3plpd_lt_trans_dp14, n_entries); } +static const struct intel_ddi_buf_trans * +jsl_get_combo_buf_trans_override(struct intel_encoder *encoder, + const struct intel_crtc_state *crtc_state, + int *n_entries) +{ + const struct intel_bios_encoder_data *devdata = encoder->devdata; + bool has_edp, has_dp; + int port_clock; + + has_edp = intel_crtc_has_type(crtc_state, INTEL_OUTPUT_EDP); + has_dp = intel_crtc_has_dp_encoder(crtc_state); + port_clock = crtc_state->port_clock; + + return intel_bios_get_jsl_combo_vspeo(devdata, has_dp, port_clock, + has_edp && use_edp_low_vswing(encoder)); +} + +static const struct intel_ddi_buf_trans * +ehl_get_combo_buf_trans_override(struct intel_encoder *encoder, + const struct intel_crtc_state *crtc_state, + int *n_entries) +{ + const struct intel_bios_encoder_data *devdata = encoder->devdata; + bool has_edp, has_dp; + int port_clock; + + has_edp = intel_crtc_has_type(crtc_state, INTEL_OUTPUT_EDP); + has_dp = intel_crtc_has_dp_encoder(crtc_state); + port_clock = crtc_state->port_clock; + + return intel_bios_get_ehl_combo_vspeo(devdata, has_dp, port_clock, + has_edp && use_edp_low_vswing(encoder)); +} + static const struct intel_ddi_buf_trans * mtl_get_c10_buf_trans_override(struct intel_encoder *encoder, const struct intel_crtc_state *crtc_state, @@ -1847,14 +1881,17 @@ void intel_ddi_buf_trans_init(struct intel_encoder *encoder) else encoder->get_buf_trans = tgl_get_dkl_buf_trans; } else if (DISPLAY_VER(display) == 11) { - if (display->platform.jasperlake) + if (display->platform.jasperlake) { encoder->get_buf_trans = jsl_get_combo_buf_trans; - else if (display->platform.elkhartlake) + encoder->get_buf_trans_override = jsl_get_combo_buf_trans_override; + } else if (display->platform.elkhartlake) { encoder->get_buf_trans = ehl_get_combo_buf_trans; - else if (intel_encoder_is_combo(encoder)) + encoder->get_buf_trans_override = ehl_get_combo_buf_trans_override; + } else if (intel_encoder_is_combo(encoder)) { encoder->get_buf_trans = icl_get_combo_buf_trans; - else + } else { encoder->get_buf_trans = icl_get_mg_buf_trans; + } } else if (display->platform.geminilake || display->platform.broxton) { encoder->get_buf_trans = bxt_get_buf_trans; } else if (display->platform.cometlake_ulx || From 4bda5543b3c6697ca435e89aeea655eea9087f96 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Grzelak?= Date: Fri, 4 Sep 2026 14:31:48 +0200 Subject: [PATCH 0016/1352] drm/i915/bios: remove VS/PE-O warning MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit There is not much use of warning when port asks to override default VS/PE since it is already logged. Remove drm_WARN() and child_device from print_ddi_port() since drm_WARN() was the only user of it. Signed-off-by: Michał Grzelak Reviewed-by: Suraj Kandpal Acked-by: Jani Nikula Signed-off-by: Suraj Kandpal Link: https://patch.msgid.link/20260904123148.2165596-9-michal.grzelak@intel.com --- drivers/gpu/drm/i915/display/intel_bios.c | 9 --------- 1 file changed, 9 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_bios.c b/drivers/gpu/drm/i915/display/intel_bios.c index 11d2cbb78eabf3..7b6a6f8c3cf196 100644 --- a/drivers/gpu/drm/i915/display/intel_bios.c +++ b/drivers/gpu/drm/i915/display/intel_bios.c @@ -2791,7 +2791,6 @@ static bool is_port_valid(struct intel_display *display, enum port port) static void print_ddi_port(const struct intel_bios_encoder_data *devdata) { struct intel_display *display = devdata->display; - const struct child_device_config *child = &devdata->child; bool is_dvi, is_hdmi, is_dp, is_edp, is_dsi, is_crt, supports_typec_usb, supports_tbt; int dp_boost_level, dp_max_link_rate, hdmi_boost_level, hdmi_level_shift, max_tmds_clock; enum port port; @@ -2864,14 +2863,6 @@ static void print_ddi_port(const struct intel_bios_encoder_data *devdata) drm_dbg_kms(display->drm, "Port %c VBT DP max link rate: %d\n", port_name(port), dp_max_link_rate); - - /* - * FIXME need to implement support for VBT - * vswing/preemph tables should this ever trigger. - */ - drm_WARN(display->drm, child->use_vbt_vswing, - "Port %c asks to use VBT vswing/preemph tables\n", - port_name(port)); } static void parse_ddi_port(struct intel_bios_encoder_data *devdata) From 9977e9d84f46d4f12ad35fbbc0ec4638554bce87 Mon Sep 17 00:00:00 2001 From: Thorsten Blum Date: Sun, 23 Aug 2026 22:50:28 +0200 Subject: [PATCH 0017/1352] drm/i915: Fix memory leak in query_perf_config_list() When krealloc() fails, free the original oa_config_ids before returning to avoid a memory leak. Fixes: 4f6ccc74a85c ("drm/i915: add support for perf configuration queries") Signed-off-by: Thorsten Blum Cc: # v5.5+ Reviewed-by: Andi Shyti Signed-off-by: Andi Shyti Link: https://patch.msgid.link/20260823205028.178597-2-thorsten.blum@linux.dev --- drivers/gpu/drm/i915/i915_query.c | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/drivers/gpu/drm/i915/i915_query.c b/drivers/gpu/drm/i915/i915_query.c index 0c55fb6e9727d6..11157fb14db313 100644 --- a/drivers/gpu/drm/i915/i915_query.c +++ b/drivers/gpu/drm/i915/i915_query.c @@ -403,8 +403,10 @@ static int query_perf_config_list(struct drm_i915_private *i915, ids = krealloc(oa_config_ids, n_configs * sizeof(*oa_config_ids), GFP_KERNEL); - if (!ids) + if (!ids) { + kfree(oa_config_ids); return -ENOMEM; + } alloc = fetch_and_zero(&n_configs); From d33cf26315d4e667886cc6199845758e79e1bdfa Mon Sep 17 00:00:00 2001 From: Zenghui Yu Date: Sat, 20 Jun 2026 21:14:25 +0800 Subject: [PATCH 0018/1352] cachefiles: Fix path of the "debug" module parameter in help paragraph The correct path of the "debug" module parameter should be /sys/module/cachefiles/parameters/debug. Fix it. Signed-off-by: Zenghui Yu Link: https://patch.msgid.link/20260620131425.8210-1-zenghui.yu@linux.dev Signed-off-by: Christian Brauner (Amutable) --- fs/cachefiles/Kconfig | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/cachefiles/Kconfig b/fs/cachefiles/Kconfig index afb25b6af5aa17..c9c168c7e07268 100644 --- a/fs/cachefiles/Kconfig +++ b/fs/cachefiles/Kconfig @@ -17,7 +17,7 @@ config CACHEFILES_DEBUG help This permits debugging to be dynamically enabled in the filesystem caching on files module. If this is set, the debugging output may be - enabled by setting bits in /sys/modules/cachefiles/parameter/debug or + enabled by setting bits in /sys/module/cachefiles/parameters/debug or by including a debugging specifier in /etc/cachefilesd.conf. config CACHEFILES_ERROR_INJECTION From 01309cd35349aedc9c459a21f3bfbf7e4a904db8 Mon Sep 17 00:00:00 2001 From: David Howells Date: Wed, 9 Sep 2026 08:20:55 +0100 Subject: [PATCH 0019/1352] netfs: Use uoff_t instead of unsigned long long and loff_t Use uoff_t instead of unsigned long long and loff_t for file positions that can't be negative. Signed-off-by: David Howells Link: https://patch.msgid.link/20260909072105.1663687-2-dhowells@redhat.com Reviewed-by: Paulo Alcantara cc: netfs@lists.linux.dev cc: linux-fsdevel@vger.kernel.org Signed-off-by: Christian Brauner (Amutable) --- fs/afs/file.c | 6 +-- fs/afs/internal.h | 2 +- fs/cachefiles/interface.c | 10 ++-- fs/cachefiles/internal.h | 4 +- fs/cachefiles/io.c | 24 +++++----- fs/ceph/addr.c | 4 +- fs/netfs/buffered_read.c | 22 ++++----- fs/netfs/buffered_write.c | 10 ++-- fs/netfs/direct_read.c | 2 +- fs/netfs/direct_write.c | 8 ++-- fs/netfs/fscache_cookie.c | 8 ++-- fs/netfs/fscache_io.c | 8 ++-- fs/netfs/internal.h | 10 ++-- fs/netfs/iterator.c | 2 +- fs/netfs/misc.c | 8 ++-- fs/netfs/objects.c | 2 +- fs/netfs/read_collect.c | 4 +- fs/netfs/read_pgpriv2.c | 8 ++-- fs/netfs/read_retry.c | 2 +- fs/netfs/write_collect.c | 10 ++-- fs/netfs/write_issue.c | 10 ++-- fs/netfs/write_retry.c | 2 +- include/linux/fscache-cache.h | 2 +- include/linux/fscache.h | 36 +++++++-------- include/linux/netfs.h | 74 ++++++++++++++--------------- include/trace/events/cachefiles.h | 30 ++++++------ include/trace/events/fscache.h | 10 ++-- include/trace/events/netfs.h | 77 +++++++++++++++---------------- 28 files changed, 197 insertions(+), 198 deletions(-) diff --git a/fs/afs/file.c b/fs/afs/file.c index 0467742bfeee34..bdb4ca0ef8da73 100644 --- a/fs/afs/file.c +++ b/fs/afs/file.c @@ -413,7 +413,7 @@ static int afs_init_request(struct netfs_io_request *rreq, struct file *file) return 0; } -static int afs_check_write_begin(struct file *file, loff_t pos, unsigned len, +static int afs_check_write_begin(struct file *file, uoff_t pos, unsigned len, struct folio **foliop, void **_fsdata) { struct afs_vnode *vnode = AFS_FS_I(file_inode(file)); @@ -434,7 +434,7 @@ static void afs_free_request(struct netfs_io_request *rreq) * Also, estimate the number of 512 bytes blocks used, rounded up to nearest 1K * for consistency with other AFS clients. */ -void afs_set_i_size(struct afs_vnode *vnode, loff_t new_i_size) +void afs_set_i_size(struct afs_vnode *vnode, uoff_t new_i_size) { struct inode *inode = &vnode->netfs.inode; loff_t i_size; @@ -451,7 +451,7 @@ void afs_set_i_size(struct afs_vnode *vnode, loff_t new_i_size) fscache_update_cookie(afs_vnode_cache(vnode), NULL, &new_i_size); } -static void afs_update_i_size(struct inode *inode, loff_t new_i_size) +static void afs_update_i_size(struct inode *inode, uoff_t new_i_size) { afs_set_i_size(AFS_FS_I(inode), new_i_size); } diff --git a/fs/afs/internal.h b/fs/afs/internal.h index 330654ed16ecea..d58273c00fbc7d 100644 --- a/fs/afs/internal.h +++ b/fs/afs/internal.h @@ -1171,7 +1171,7 @@ extern int afs_open(struct inode *, struct file *); extern int afs_release(struct inode *, struct file *); void afs_fetch_data_async_rx(struct work_struct *work); void afs_fetch_data_immediate_cancel(struct afs_call *call); -void afs_set_i_size(struct afs_vnode *vnode, loff_t new_i_size); +void afs_set_i_size(struct afs_vnode *vnode, uoff_t new_i_size); /* * flock.c diff --git a/fs/cachefiles/interface.c b/fs/cachefiles/interface.c index 50a000310a8c54..a160d5c3e74c0e 100644 --- a/fs/cachefiles/interface.c +++ b/fs/cachefiles/interface.c @@ -111,7 +111,7 @@ static int cachefiles_adjust_size(struct cachefiles_object *object) struct iattr newattrs; struct file *file = object->file; uint64_t ni_size; - loff_t oi_size; + uoff_t oi_size; int ret; ni_size = object->cookie->object_size; @@ -225,11 +225,11 @@ static bool cachefiles_lookup_cookie(struct fscache_cookie *cookie) * any unused granules. */ static bool cachefiles_shorten_object(struct cachefiles_object *object, - struct file *file, loff_t new_size) + struct file *file, uoff_t new_size) { struct cachefiles_cache *cache = object->volume->cache; struct inode *inode = file_inode(file); - loff_t i_size, dio_size; + uoff_t i_size, dio_size; int ret; dio_size = round_up(new_size, CACHEFILES_DIO_BLOCK_SIZE); @@ -271,14 +271,14 @@ static bool cachefiles_shorten_object(struct cachefiles_object *object, * Resize the backing object. */ static void cachefiles_resize_cookie(struct netfs_cache_resources *cres, - loff_t new_size) + uoff_t new_size) { struct cachefiles_object *object = cachefiles_cres_object(cres); struct cachefiles_cache *cache = object->volume->cache; struct fscache_cookie *cookie = object->cookie; const struct cred *saved_cred; struct file *file = cachefiles_cres_file(cres); - loff_t old_size = cookie->object_size; + uoff_t old_size = cookie->object_size; _enter("%llu->%llu", old_size, new_size); diff --git a/fs/cachefiles/internal.h b/fs/cachefiles/internal.h index c93324e0f98c76..60bd801ada04f3 100644 --- a/fs/cachefiles/internal.h +++ b/fs/cachefiles/internal.h @@ -203,11 +203,11 @@ extern bool cachefiles_begin_operation(struct netfs_cache_resources *cres, enum fscache_want_state want_state); extern int __cachefiles_prepare_write(struct cachefiles_object *object, struct file *file, - loff_t *_start, size_t *_len, size_t upper_len, + uoff_t *_start, size_t *_len, size_t upper_len, bool no_space_allocated_yet); extern int __cachefiles_write(struct cachefiles_object *object, struct file *file, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, netfs_io_terminated_t term_func, void *term_func_priv); diff --git a/fs/cachefiles/io.c b/fs/cachefiles/io.c index 9540ec25b3cb6c..7de8069d15b6b5 100644 --- a/fs/cachefiles/io.c +++ b/fs/cachefiles/io.c @@ -19,7 +19,7 @@ struct cachefiles_kiocb { struct kiocb iocb; refcount_t ki_refcnt; - loff_t start; + uoff_t start; union { size_t skipped; size_t len; @@ -73,7 +73,7 @@ static void cachefiles_read_complete(struct kiocb *iocb, long ret) * Initiate a read from the cache. */ static int cachefiles_read(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, enum netfs_read_from_hole read_hole, netfs_io_terminated_t term_func, @@ -197,8 +197,8 @@ static int cachefiles_read(struct netfs_cache_resources *cres, * of data starts and how long it is. */ static int cachefiles_query_occupancy(struct netfs_cache_resources *cres, - loff_t start, size_t len, size_t granularity, - loff_t *_data_start, size_t *_data_len) + uoff_t start, size_t len, size_t granularity, + uoff_t *_data_start, size_t *_data_len) { struct cachefiles_object *object; struct file *file; @@ -280,7 +280,7 @@ static void cachefiles_write_complete(struct kiocb *iocb, long ret) */ int __cachefiles_write(struct cachefiles_object *object, struct file *file, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, netfs_io_terminated_t term_func, void *term_func_priv) @@ -357,7 +357,7 @@ int __cachefiles_write(struct cachefiles_object *object, } static int cachefiles_write(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, netfs_io_terminated_t term_func, void *term_func_priv) @@ -377,7 +377,7 @@ static int cachefiles_write(struct netfs_cache_resources *cres, static inline enum netfs_io_source cachefiles_do_prepare_read(struct netfs_cache_resources *cres, - loff_t start, size_t *_len, loff_t i_size, + uoff_t start, size_t *_len, loff_t i_size, unsigned long *_flags, ino_t netfs_ino) { enum cachefiles_prepare_read_trace why; @@ -483,7 +483,7 @@ cachefiles_do_prepare_read(struct netfs_cache_resources *cres, * boundary as appropriate. */ static enum netfs_io_source cachefiles_prepare_read(struct netfs_io_subrequest *subreq, - unsigned long long i_size) + uoff_t i_size) { return cachefiles_do_prepare_read(&subreq->rreq->cache_resources, subreq->start, &subreq->len, i_size, @@ -495,7 +495,7 @@ static enum netfs_io_source cachefiles_prepare_read(struct netfs_io_subrequest * */ int __cachefiles_prepare_write(struct cachefiles_object *object, struct file *file, - loff_t *_start, size_t *_len, size_t upper_len, + uoff_t *_start, size_t *_len, size_t upper_len, bool no_space_allocated_yet) { struct cachefiles_cache *cache = object->volume->cache; @@ -577,8 +577,8 @@ int __cachefiles_prepare_write(struct cachefiles_object *object, } static int cachefiles_prepare_write(struct netfs_cache_resources *cres, - loff_t *_start, size_t *_len, size_t upper_len, - loff_t i_size, bool no_space_allocated_yet) + uoff_t *_start, size_t *_len, size_t upper_len, + uoff_t i_size, bool no_space_allocated_yet) { struct cachefiles_object *object = cachefiles_cres_object(cres); struct cachefiles_cache *cache = object->volume->cache; @@ -628,7 +628,7 @@ static void cachefiles_issue_write(struct netfs_io_subrequest *subreq) struct netfs_io_stream *stream = &wreq->io_streams[subreq->stream_nr]; const struct cred *saved_cred; size_t off, pre, post, len = subreq->len; - loff_t start = subreq->start; + uoff_t start = subreq->start; int ret; _enter("W=%x[%x] %llx-%llx", diff --git a/fs/ceph/addr.c b/fs/ceph/addr.c index e598b2d424ec16..42b55ce30a327b 100644 --- a/fs/ceph/addr.c +++ b/fs/ceph/addr.c @@ -65,7 +65,7 @@ (CONGESTION_ON_THRESH(congestion_kb) - \ (CONGESTION_ON_THRESH(congestion_kb) >> 2)) -static int ceph_netfs_check_write_begin(struct file *file, loff_t pos, unsigned int len, +static int ceph_netfs_check_write_begin(struct file *file, uoff_t pos, unsigned int len, struct folio **foliop, void **_fsdata); static inline struct ceph_snap_context *page_snap_context(struct page *page) @@ -1868,7 +1868,7 @@ ceph_find_incompatible(struct folio *folio) return NULL; } -static int ceph_netfs_check_write_begin(struct file *file, loff_t pos, unsigned int len, +static int ceph_netfs_check_write_begin(struct file *file, uoff_t pos, unsigned int len, struct folio **foliop, void **_fsdata) { struct inode *inode = file_inode(file); diff --git a/fs/netfs/buffered_read.c b/fs/netfs/buffered_read.c index 424df70a5c30f1..61cf82b4b60d38 100644 --- a/fs/netfs/buffered_read.c +++ b/fs/netfs/buffered_read.c @@ -10,9 +10,9 @@ #include "internal.h" static void netfs_cache_expand_readahead(struct netfs_io_request *rreq, - unsigned long long *_start, - unsigned long long *_len, - unsigned long long i_size) + uoff_t *_start, + uoff_t *_len, + uoff_t i_size) { struct netfs_cache_resources *cres = &rreq->cache_resources; @@ -139,7 +139,7 @@ static ssize_t netfs_prepare_read_iterator(struct netfs_io_subrequest *subreq) static enum netfs_io_source netfs_cache_prepare_read(struct netfs_io_request *rreq, struct netfs_io_subrequest *subreq, - loff_t i_size) + uoff_t i_size) { struct netfs_cache_resources *cres = &rreq->cache_resources; enum netfs_io_source source; @@ -269,9 +269,9 @@ static void netfs_mark_copy_to_cache(struct netfs_io_request *rreq, static void netfs_read_to_pagecache(struct netfs_io_request *rreq) { struct folio_queue *fq = rreq->buffer.tail; - unsigned long long start = rreq->start; unsigned int offset = 0; ssize_t size = rreq->len; + uoff_t start = rreq->start; int ret = 0, slot = 0; do { @@ -293,8 +293,8 @@ static void netfs_read_to_pagecache(struct netfs_io_request *rreq) source = netfs_cache_prepare_read(rreq, subreq, rreq->i_size); subreq->source = source; if (source == NETFS_DOWNLOAD_FROM_SERVER) { - unsigned long long zero_point = netfs_read_zero_point(rreq->inode); - unsigned long long zp = umin(zero_point, rreq->i_size); + uoff_t zero_point = netfs_read_zero_point(rreq->inode); + uoff_t zp = umin(zero_point, rreq->i_size); size_t len = subreq->len; if (unlikely(rreq->origin == NETFS_READ_SINGLE)) @@ -642,11 +642,11 @@ EXPORT_SYMBOL(netfs_read_folio); * If any of these criteria are met, then zero out the unwritten parts * of the folio and return true. Otherwise, return false. */ -static bool netfs_skip_folio_read(struct folio *folio, loff_t pos, size_t len, +static bool netfs_skip_folio_read(struct folio *folio, uoff_t pos, size_t len, bool always_fill) { struct inode *inode = folio_inode(folio); - loff_t i_size = i_size_read(inode); + uoff_t i_size = i_size_read(inode); size_t offset = offset_in_folio(folio, pos); size_t plen = folio_size(folio); @@ -711,7 +711,7 @@ static bool netfs_skip_folio_read(struct folio *folio, loff_t pos, size_t len, */ int netfs_write_begin(struct netfs_inode *ctx, struct file *file, struct address_space *mapping, - loff_t pos, unsigned int len, struct folio **_folio, + uoff_t pos, unsigned int len, struct folio **_folio, void **_fsdata) { struct netfs_io_request *rreq; @@ -807,7 +807,7 @@ int netfs_prefetch_for_write(struct file *file, struct folio *folio, struct netfs_io_request *rreq; struct address_space *mapping = folio->mapping; struct netfs_inode *ctx = netfs_inode(mapping->host); - unsigned long long start = folio_pos(folio); + uoff_t start = folio_pos(folio); size_t flen = folio_size(folio); int ret; diff --git a/fs/netfs/buffered_write.c b/fs/netfs/buffered_write.c index 2cdb68e6b16fff..df496873e4f445 100644 --- a/fs/netfs/buffered_write.c +++ b/fs/netfs/buffered_write.c @@ -17,7 +17,7 @@ * as possible to hold as much of the remaining length as possible in one go. */ static struct folio *netfs_grab_folio_for_write(struct address_space *mapping, - loff_t pos, size_t part) + uoff_t pos, size_t part) { pgoff_t index = pos / PAGE_SIZE; fgf_t fgp_flags = FGP_WRITEBEGIN; @@ -35,9 +35,9 @@ static struct folio *netfs_grab_folio_for_write(struct address_space *mapping, * the values actually are. */ void netfs_update_i_size(struct netfs_inode *ctx, struct inode *inode, - loff_t pos, size_t copied) + uoff_t pos, size_t copied) { - loff_t i_size, end = pos + copied; + uoff_t i_size, end = pos + copied; blkcnt_t add; size_t gap; @@ -102,7 +102,7 @@ ssize_t netfs_perform_write(struct kiocb *iocb, struct iov_iter *iter, struct folio *folio = NULL, *writethrough = NULL; unsigned int bdp_flags = (iocb->ki_flags & IOCB_NOWAIT) ? BDP_ASYNC : 0; ssize_t written = 0, ret, ret2; - loff_t pos = iocb->ki_pos; + uoff_t pos = iocb->ki_pos; size_t max_chunk = mapping_max_folio_size(mapping); bool maybe_trouble = false; @@ -134,7 +134,7 @@ ssize_t netfs_perform_write(struct kiocb *iocb, struct iov_iter *iter, enum netfs_folio_trace trace; struct netfs_folio *finfo; struct netfs_group *group; - unsigned long long fpos; + uoff_t fpos; size_t flen; size_t offset; /* Offset into pagecache folio */ size_t part; /* Bytes to write to folio */ diff --git a/fs/netfs/direct_read.c b/fs/netfs/direct_read.c index 6a8fb0d55e040e..aa10af5171a860 100644 --- a/fs/netfs/direct_read.c +++ b/fs/netfs/direct_read.c @@ -47,8 +47,8 @@ static void netfs_prepare_dio_read_iterator(struct netfs_io_subrequest *subreq) */ static void netfs_dispatch_unbuffered_reads(struct netfs_io_request *rreq) { - unsigned long long start = rreq->start; ssize_t size = rreq->len; + uoff_t start = rreq->start; int ret; do { diff --git a/fs/netfs/direct_write.c b/fs/netfs/direct_write.c index 2361277416c730..32200c10d2a45c 100644 --- a/fs/netfs/direct_write.c +++ b/fs/netfs/direct_write.c @@ -225,9 +225,9 @@ ssize_t netfs_unbuffered_write_iter_locked(struct kiocb *iocb, struct iov_iter * struct netfs_group *netfs_group) { struct netfs_io_request *wreq; - unsigned long long start = iocb->ki_pos; - unsigned long long end = start + iov_iter_count(iter); ssize_t ret, n; + uoff_t start = iocb->ki_pos; + uoff_t end = start + iov_iter_count(iter); size_t len = iov_iter_count(iter); bool async = !is_sync_kiocb(iocb); @@ -336,8 +336,8 @@ ssize_t netfs_unbuffered_write_iter(struct kiocb *iocb, struct iov_iter *from) struct inode *inode = mapping->host; struct netfs_inode *ictx = netfs_inode(inode); ssize_t ret; - loff_t pos = iocb->ki_pos; - unsigned long long end = pos + iov_iter_count(from) - 1; + uoff_t pos = iocb->ki_pos; + uoff_t end = pos + iov_iter_count(from) - 1; _enter("%llx,%zx,%llx", pos, iov_iter_count(from), i_size_read(inode)); diff --git a/fs/netfs/fscache_cookie.c b/fs/netfs/fscache_cookie.c index 3d56fc73435ffb..5a226f9cbdeaa0 100644 --- a/fs/netfs/fscache_cookie.c +++ b/fs/netfs/fscache_cookie.c @@ -327,7 +327,7 @@ static struct fscache_cookie *fscache_alloc_cookie( u8 advice, const void *index_key, size_t index_key_len, const void *aux_data, size_t aux_data_len, - loff_t object_size) + uoff_t object_size) { struct fscache_cookie *cookie; @@ -452,7 +452,7 @@ struct fscache_cookie *__fscache_acquire_cookie( u8 advice, const void *index_key, size_t index_key_len, const void *aux_data, size_t aux_data_len, - loff_t object_size) + uoff_t object_size) { struct fscache_cookie *cookie; @@ -663,7 +663,7 @@ static void fscache_unuse_cookie_locked(struct fscache_cookie *cookie) * Stop using the cookie for I/O. */ void __fscache_unuse_cookie(struct fscache_cookie *cookie, - const void *aux_data, const loff_t *object_size) + const void *aux_data, const uoff_t *object_size) { unsigned int debug_id = cookie->debug_id; unsigned int r = refcount_read(&cookie->ref); @@ -1049,7 +1049,7 @@ static void fscache_perform_invalidation(struct fscache_cookie *cookie) * Invalidate an object. */ void __fscache_invalidate(struct fscache_cookie *cookie, - const void *aux_data, loff_t new_size, + const void *aux_data, uoff_t new_size, unsigned int flags) { bool is_caching; diff --git a/fs/netfs/fscache_io.c b/fs/netfs/fscache_io.c index 37f05b4d34699d..8bca63721eeb03 100644 --- a/fs/netfs/fscache_io.c +++ b/fs/netfs/fscache_io.c @@ -162,7 +162,7 @@ EXPORT_SYMBOL(__fscache_begin_write_operation); struct fscache_write_request { struct netfs_cache_resources cache_resources; struct address_space *mapping; - loff_t start; + uoff_t start; size_t len; bool set_bits; bool using_pgpriv2; @@ -171,7 +171,7 @@ struct fscache_write_request { }; void __fscache_clear_page_bits(struct address_space *mapping, - loff_t start, size_t len) + uoff_t start, size_t len) { pgoff_t first = start / PAGE_SIZE; pgoff_t last = (start + len - 1) / PAGE_SIZE; @@ -208,7 +208,7 @@ static void fscache_wreq_done(void *priv, ssize_t transferred_or_error) void __fscache_write_to_cache(struct fscache_cookie *cookie, struct address_space *mapping, - loff_t start, size_t len, loff_t i_size, + uoff_t start, size_t len, uoff_t i_size, netfs_io_terminated_t term_func, void *term_func_priv, bool using_pgpriv2, bool cond) @@ -267,7 +267,7 @@ EXPORT_SYMBOL(__fscache_write_to_cache); /* * Change the size of a backing object. */ -void __fscache_resize_cookie(struct fscache_cookie *cookie, loff_t new_size) +void __fscache_resize_cookie(struct fscache_cookie *cookie, uoff_t new_size) { struct netfs_cache_resources cres; diff --git a/fs/netfs/internal.h b/fs/netfs/internal.h index c79c8e69d60ca7..9bd7ad10cc0cff 100644 --- a/fs/netfs/internal.h +++ b/fs/netfs/internal.h @@ -33,7 +33,7 @@ int netfs_prefetch_for_write(struct file *file, struct folio *folio, * buffered_write.c */ void netfs_update_i_size(struct netfs_inode *ctx, struct inode *inode, - loff_t pos, size_t copied); + uoff_t pos, size_t copied); /* * main.c @@ -86,7 +86,7 @@ void netfs_wait_for_put_ra_refs(struct netfs_io_request *rreq); */ struct netfs_io_request *netfs_alloc_request(struct address_space *mapping, struct file *file, - loff_t start, size_t len, + uoff_t start, size_t len, enum netfs_io_origin origin); void netfs_get_request(struct netfs_io_request *rreq, enum netfs_rreq_ref_trace what); void netfs_clear_subrequests(struct netfs_io_request *rreq); @@ -203,11 +203,11 @@ void netfs_write_collection_worker(struct work_struct *work); */ struct netfs_io_request *netfs_create_write_req(struct address_space *mapping, struct file *file, - loff_t start, + uoff_t start, enum netfs_io_origin origin); void netfs_prepare_write(struct netfs_io_request *wreq, struct netfs_io_stream *stream, - loff_t start); + uoff_t start); void netfs_reissue_write(struct netfs_io_stream *stream, struct netfs_io_subrequest *subreq, struct iov_iter *source); @@ -215,7 +215,7 @@ void netfs_issue_write(struct netfs_io_request *wreq, struct netfs_io_stream *stream); size_t netfs_advance_write(struct netfs_io_request *wreq, struct netfs_io_stream *stream, - loff_t start, size_t len, bool to_eof); + uoff_t start, size_t len, bool to_eof); struct netfs_io_request *netfs_begin_writethrough(struct kiocb *iocb, size_t len); int netfs_advance_writethrough(struct netfs_io_request *wreq, struct writeback_control *wbc, struct folio *folio, size_t copied, bool to_page_end, diff --git a/fs/netfs/iterator.c b/fs/netfs/iterator.c index b375567e0520ed..eb1efb17f53aaa 100644 --- a/fs/netfs/iterator.c +++ b/fs/netfs/iterator.c @@ -209,7 +209,7 @@ static size_t netfs_limit_xarray(const struct iov_iter *iter, size_t start_offse { struct folio *folio; unsigned int nsegs = 0; - loff_t pos = iter->xarray_start + iter->iov_offset; + uoff_t pos = iter->xarray_start + iter->iov_offset; pgoff_t index = pos / PAGE_SIZE; size_t span = 0, n = iter->count; diff --git a/fs/netfs/misc.c b/fs/netfs/misc.c index f5c1c463f4ff7a..eafc4edae6a06e 100644 --- a/fs/netfs/misc.c +++ b/fs/netfs/misc.c @@ -193,7 +193,7 @@ void netfs_clear_inode_writeback(struct inode *inode, const void *aux) struct fscache_cookie *cookie = netfs_i_cookie(netfs_inode(inode)); if (inode_state_read_once(inode) & I_PINNING_NETFS_WB) { - loff_t i_size = i_size_read(inode); + uoff_t i_size = i_size_read(inode); fscache_unuse_cookie(cookie, aux, &i_size); } } @@ -218,8 +218,8 @@ void netfs_invalidate_folio(struct folio *folio, size_t offset, size_t length) _enter("{%lx},%zx,%zx", folio->index, offset, length); if (offset == 0 && length == flen) { - unsigned long long i_size, remote_i_size, zero_point; - unsigned long long fpos = folio_pos(folio), end; + uoff_t i_size, remote_i_size, zero_point; + uoff_t fpos = folio_pos(folio), end; netfs_read_sizes(inode, &i_size, &remote_i_size, &zero_point); end = umin(fpos + flen, i_size); @@ -305,7 +305,7 @@ bool netfs_release_folio(struct folio *folio, gfp_t gfp) { struct inode *inode = folio_inode(folio); struct netfs_inode *ctx = netfs_inode(inode); - unsigned long long i_size, remote_i_size, zero_point, end; + uoff_t i_size, remote_i_size, zero_point, end; if (folio_test_dirty(folio)) return false; diff --git a/fs/netfs/objects.c b/fs/netfs/objects.c index 7f6a3e912602ee..3460aa1c4af1e7 100644 --- a/fs/netfs/objects.c +++ b/fs/netfs/objects.c @@ -16,7 +16,7 @@ static void netfs_free_request(struct work_struct *work); */ struct netfs_io_request *netfs_alloc_request(struct address_space *mapping, struct file *file, - loff_t start, size_t len, + uoff_t start, size_t len, enum netfs_io_origin origin) { static atomic_t debug_ids; diff --git a/fs/netfs/read_collect.c b/fs/netfs/read_collect.c index 5cf22087d2439d..6576a9b8671d30 100644 --- a/fs/netfs/read_collect.c +++ b/fs/netfs/read_collect.c @@ -153,8 +153,8 @@ static void netfs_read_unlock_folios(struct netfs_io_request *rreq, unsigned int *notes) { struct folio_queue *folioq = rreq->buffer.tail; - unsigned long long collected_to = rreq->collected_to; unsigned int slot = rreq->buffer.first_tail_slot; + uoff_t collected_to = rreq->collected_to; if (rreq->cleaned_to >= rreq->collected_to) return; @@ -179,7 +179,7 @@ static void netfs_read_unlock_folios(struct netfs_io_request *rreq, for (;;) { struct folio *folio; - unsigned long long fpos, fend; + uoff_t fpos, fend; size_t fsize; folio = folioq_folio(folioq, slot); diff --git a/fs/netfs/read_pgpriv2.c b/fs/netfs/read_pgpriv2.c index a4b7bb88cbdb60..4d50585f69a608 100644 --- a/fs/netfs/read_pgpriv2.c +++ b/fs/netfs/read_pgpriv2.c @@ -20,7 +20,7 @@ static void netfs_pgpriv2_copy_folio(struct netfs_io_request *creq, struct folio { struct netfs_io_stream *cache = &creq->io_streams[1]; size_t fsize = folio_size(folio), flen = fsize; - loff_t fpos = folio_pos(folio), i_size; + uoff_t fpos = folio_pos(folio), i_size; bool to_eof = false; _enter(""); @@ -175,8 +175,8 @@ void netfs_pgpriv2_end_copy_to_cache(struct netfs_io_request *rreq) bool netfs_pgpriv2_unlock_copied_folios(struct netfs_io_request *creq) { struct folio_queue *folioq = creq->buffer.tail; - unsigned long long collected_to = creq->collected_to; unsigned int slot = creq->buffer.first_tail_slot; + uoff_t collected_to = creq->collected_to; bool made_progress = false; if (slot >= folioq_nr_slots(folioq)) { @@ -186,7 +186,7 @@ bool netfs_pgpriv2_unlock_copied_folios(struct netfs_io_request *creq) for (;;) { struct folio *folio; - unsigned long long fpos, fend; + uoff_t fpos, fend; size_t fsize, flen; folio = folioq_folio(folioq, slot); @@ -199,7 +199,7 @@ bool netfs_pgpriv2_unlock_copied_folios(struct netfs_io_request *creq) fsize = folio_size(folio); flen = fsize; - fend = min_t(unsigned long long, fpos + flen, creq->i_size); + fend = min_t(uoff_t, fpos + flen, creq->i_size); trace_netfs_collect_folio(creq, folio, fend, collected_to); diff --git a/fs/netfs/read_retry.c b/fs/netfs/read_retry.c index 4f6a36c6e214f7..46810d29f0c0e3 100644 --- a/fs/netfs/read_retry.c +++ b/fs/netfs/read_retry.c @@ -75,7 +75,7 @@ static void netfs_retry_read_subrequests(struct netfs_io_request *rreq) do { struct netfs_io_subrequest *from, *to, *tmp; struct iov_iter source; - unsigned long long start, len; + uoff_t start, len; size_t part; bool boundary = false, subreq_superfluous = false; diff --git a/fs/netfs/write_collect.c b/fs/netfs/write_collect.c index 210eb8f3958d77..100a5038c61e39 100644 --- a/fs/netfs/write_collect.c +++ b/fs/netfs/write_collect.c @@ -67,7 +67,7 @@ int netfs_folio_written_back(struct folio *folio) /* Streaming writes cannot be redirtied whilst under writeback, * so discard the streaming record. */ - unsigned long long fend; + uoff_t fend; fend = folio_pos(folio) + finfo->dirty_offset + finfo->dirty_len; spin_lock(&ictx->inode.i_lock); @@ -115,8 +115,8 @@ static void netfs_writeback_unlock_folios(struct netfs_io_request *wreq, unsigned int *notes) { struct folio_queue *folioq = wreq->buffer.tail; - unsigned long long collected_to = wreq->collected_to; unsigned int slot = wreq->buffer.first_tail_slot; + uoff_t collected_to = wreq->collected_to; if (WARN_ON_ONCE(!folioq)) { pr_err("[!] Writeback unlock found empty rolling buffer!\n"); @@ -140,7 +140,7 @@ static void netfs_writeback_unlock_folios(struct netfs_io_request *wreq, for (;;) { struct folio *folio; struct netfs_folio *finfo; - unsigned long long fpos, fend; + uoff_t fpos, fend; size_t fsize, flen; folio = folioq_folio(folioq, slot); @@ -154,7 +154,7 @@ static void netfs_writeback_unlock_folios(struct netfs_io_request *wreq, finfo = netfs_folio_info(folio); flen = finfo ? finfo->dirty_offset + finfo->dirty_len : fsize; - fend = min_t(unsigned long long, fpos + flen, wreq->i_size); + fend = min_t(uoff_t, fpos + flen, wreq->i_size); trace_netfs_collect_folio(wreq, folio, fend, collected_to); @@ -201,8 +201,8 @@ static void netfs_collect_write_results(struct netfs_io_request *wreq) { struct netfs_io_subrequest *front, *remove; struct netfs_io_stream *stream; - unsigned long long collected_to, issued_to; unsigned int notes; + uoff_t collected_to, issued_to; int s; _enter("%llx-%llx", wreq->start, wreq->start + wreq->len); diff --git a/fs/netfs/write_issue.c b/fs/netfs/write_issue.c index 851f6f93ad45ab..3b33ad5b69c39c 100644 --- a/fs/netfs/write_issue.c +++ b/fs/netfs/write_issue.c @@ -89,7 +89,7 @@ static void netfs_kill_dirty_pages(struct address_space *mapping, */ struct netfs_io_request *netfs_create_write_req(struct address_space *mapping, struct file *file, - loff_t start, + uoff_t start, enum netfs_io_origin origin) { struct netfs_io_request *wreq; @@ -156,7 +156,7 @@ EXPORT_SYMBOL(netfs_prepare_write_failed); */ void netfs_prepare_write(struct netfs_io_request *wreq, struct netfs_io_stream *stream, - loff_t start) + uoff_t start) { struct netfs_io_subrequest *subreq; struct iov_iter *wreq_iter = &wreq->buffer.iter; @@ -279,7 +279,7 @@ void netfs_issue_write(struct netfs_io_request *wreq, */ size_t netfs_advance_write(struct netfs_io_request *wreq, struct netfs_io_stream *stream, - loff_t start, size_t len, bool to_eof) + uoff_t start, size_t len, bool to_eof) { struct netfs_io_subrequest *subreq = stream->construct; size_t part; @@ -330,7 +330,7 @@ static int netfs_write_folio(struct netfs_io_request *wreq, struct netfs_folio *finfo; size_t iter_off = 0; size_t fsize = folio_size(folio), flen = fsize, foff = 0; - loff_t fpos = folio_pos(folio), i_size; + uoff_t fpos = folio_pos(folio), i_size; bool to_eof = false, streamw = false; bool debug = false; @@ -721,7 +721,7 @@ static int netfs_write_folio_single(struct netfs_io_request *wreq, struct netfs_io_stream *stream; size_t iter_off = 0; size_t fsize = folio_size(folio), flen; - loff_t fpos = folio_pos(folio); + uoff_t fpos = folio_pos(folio); ssize_t ret; bool to_eof = false; bool no_debug = false; diff --git a/fs/netfs/write_retry.c b/fs/netfs/write_retry.c index 058bc7a166a59f..6cd584242af282 100644 --- a/fs/netfs/write_retry.c +++ b/fs/netfs/write_retry.c @@ -55,7 +55,7 @@ static void netfs_retry_write_stream(struct netfs_io_request *wreq, do { struct netfs_io_subrequest *subreq = NULL, *from, *to, *tmp; struct iov_iter source; - unsigned long long start, len; + uoff_t start, len; size_t part; bool boundary = false; diff --git a/include/linux/fscache-cache.h b/include/linux/fscache-cache.h index 4c91a019972b6a..ee524c863fa9be 100644 --- a/include/linux/fscache-cache.h +++ b/include/linux/fscache-cache.h @@ -67,7 +67,7 @@ struct fscache_cache_ops { /* Change the size of a data object */ void (*resize_cookie)(struct netfs_cache_resources *cres, - loff_t new_size); + uoff_t new_size); /* Invalidate an object */ bool (*invalidate_cookie)(struct fscache_cookie *cookie); diff --git a/include/linux/fscache.h b/include/linux/fscache.h index 58fdb9605425dc..e19fca38382b9b 100644 --- a/include/linux/fscache.h +++ b/include/linux/fscache.h @@ -112,7 +112,7 @@ struct fscache_cookie { struct list_head proc_link; /* Link in proc list */ struct list_head commit_link; /* Link in commit queue */ struct work_struct work; /* Commit/relinq/withdraw work */ - loff_t object_size; /* Size of the netfs object */ + uoff_t object_size; /* Size of the netfs object */ unsigned long unused_at; /* Time at which unused (jiffies) */ unsigned long flags; #define FSCACHE_COOKIE_RELINQUISHED 0 /* T if cookie has been relinquished */ @@ -163,22 +163,22 @@ extern struct fscache_cookie *__fscache_acquire_cookie( u8, const void *, size_t, const void *, size_t, - loff_t); + uoff_t); extern void __fscache_use_cookie(struct fscache_cookie *, bool); -extern void __fscache_unuse_cookie(struct fscache_cookie *, const void *, const loff_t *); +extern void __fscache_unuse_cookie(struct fscache_cookie *, const void *, const uoff_t *); extern void __fscache_relinquish_cookie(struct fscache_cookie *, bool); -extern void __fscache_resize_cookie(struct fscache_cookie *, loff_t); -extern void __fscache_invalidate(struct fscache_cookie *, const void *, loff_t, unsigned int); +extern void __fscache_resize_cookie(struct fscache_cookie *, uoff_t); +extern void __fscache_invalidate(struct fscache_cookie *, const void *, uoff_t, unsigned int); extern int __fscache_begin_read_operation(struct netfs_cache_resources *, struct fscache_cookie *); extern int __fscache_begin_write_operation(struct netfs_cache_resources *, struct fscache_cookie *); void __fscache_write_to_cache(struct fscache_cookie *cookie, struct address_space *mapping, - loff_t start, size_t len, loff_t i_size, + uoff_t start, size_t len, uoff_t i_size, netfs_io_terminated_t term_func, void *term_func_priv, bool using_pgpriv2, bool cond); -extern void __fscache_clear_page_bits(struct address_space *, loff_t, size_t); +extern void __fscache_clear_page_bits(struct address_space *, uoff_t, size_t); /** * fscache_acquire_volume - Register a volume as desiring caching services @@ -249,7 +249,7 @@ struct fscache_cookie *fscache_acquire_cookie(struct fscache_volume *volume, size_t index_key_len, const void *aux_data, size_t aux_data_len, - loff_t object_size) + uoff_t object_size) { if (!fscache_volume_valid(volume)) return NULL; @@ -286,7 +286,7 @@ static inline void fscache_use_cookie(struct fscache_cookie *cookie, */ static inline void fscache_unuse_cookie(struct fscache_cookie *cookie, const void *aux_data, - const loff_t *object_size) + const uoff_t *object_size) { if (fscache_cookie_valid(cookie)) __fscache_unuse_cookie(cookie, aux_data, object_size); @@ -327,7 +327,7 @@ static inline void *fscache_get_aux(struct fscache_cookie *cookie) */ static inline void fscache_update_aux(struct fscache_cookie *cookie, - const void *aux_data, const loff_t *object_size) + const void *aux_data, const uoff_t *object_size) { void *p = fscache_get_aux(cookie); @@ -343,7 +343,7 @@ extern atomic_t fscache_n_updates; static inline void __fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data, - const loff_t *object_size) + const uoff_t *object_size) { #ifdef CONFIG_FSCACHE_STATS atomic_inc(&fscache_n_updates); @@ -369,7 +369,7 @@ void __fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data */ static inline void fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data, - const loff_t *object_size) + const uoff_t *object_size) { if (fscache_cookie_enabled(cookie)) __fscache_update_cookie(cookie, aux_data, object_size); @@ -386,7 +386,7 @@ void fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data, * description. */ static inline -void fscache_resize_cookie(struct fscache_cookie *cookie, loff_t new_size) +void fscache_resize_cookie(struct fscache_cookie *cookie, uoff_t new_size) { if (fscache_cookie_enabled(cookie)) __fscache_resize_cookie(cookie, new_size); @@ -413,7 +413,7 @@ void fscache_resize_cookie(struct fscache_cookie *cookie, loff_t new_size) */ static inline void fscache_invalidate(struct fscache_cookie *cookie, - const void *aux_data, loff_t size, unsigned int flags) + const void *aux_data, uoff_t size, unsigned int flags) { if (fscache_cookie_enabled(cookie)) __fscache_invalidate(cookie, aux_data, size, flags); @@ -502,7 +502,7 @@ static inline void fscache_end_operation(struct netfs_cache_resources *cres) */ static inline int fscache_read(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, enum netfs_read_from_hole read_hole, netfs_io_terminated_t term_func, @@ -561,7 +561,7 @@ int fscache_begin_write_operation(struct netfs_cache_resources *cres, */ static inline int fscache_write(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, netfs_io_terminated_t term_func, void *term_func_priv) @@ -581,7 +581,7 @@ int fscache_write(struct netfs_cache_resources *cres, * waiting. */ static inline void fscache_clear_page_bits(struct address_space *mapping, - loff_t start, size_t len, + uoff_t start, size_t len, bool caching) { if (caching) @@ -615,7 +615,7 @@ static inline void fscache_clear_page_bits(struct address_space *mapping, */ static inline void fscache_write_to_cache(struct fscache_cookie *cookie, struct address_space *mapping, - loff_t start, size_t len, loff_t i_size, + uoff_t start, size_t len, uoff_t i_size, netfs_io_terminated_t term_func, void *term_func_priv, bool using_pgpriv2, bool caching) diff --git a/include/linux/netfs.h b/include/linux/netfs.h index b4dd32863dd45f..e239d104f1a5f2 100644 --- a/include/linux/netfs.h +++ b/include/linux/netfs.h @@ -62,8 +62,8 @@ struct netfs_inode { struct fscache_cookie *cache; #endif struct list_head wb_queue; /* Queue of processes wanting to do writeback */ - loff_t _remote_i_size; /* Size of the remote file */ - loff_t _zero_point; /* Size after which we assume there's no data + uoff_t _remote_i_size; /* Size of the remote file */ + uoff_t _zero_point; /* Size after which we assume there's no data * on the server */ spinlock_t lock; /* Lock covering wb_queue */ atomic_t io_count; /* Number of outstanding reqs */ @@ -142,7 +142,7 @@ struct netfs_io_stream { void (*issue_write)(struct netfs_io_subrequest *subreq); /* Collection tracking */ struct list_head subrequests; /* Contributory I/O operations */ - unsigned long long collected_to; /* Position we've collected results to */ + uoff_t collected_to; /* Position we've collected results to */ size_t transferred; /* The amount transferred from this stream */ unsigned short error; /* Aggregate error for the stream */ enum netfs_io_source source; /* Where to read from/write to */ @@ -177,7 +177,7 @@ struct netfs_io_subrequest { struct work_struct work; struct list_head rreq_link; /* Link in rreq->subrequests */ struct iov_iter io_iter; /* Iterator for this subrequest */ - unsigned long long start; /* Where to start the I/O */ + uoff_t start; /* Where to start the I/O */ size_t len; /* Size of the I/O */ size_t transferred; /* Amount of data transferred */ refcount_t ref; @@ -243,17 +243,17 @@ struct netfs_io_request { void *netfs_priv; /* Private data for the netfs */ void *netfs_priv2; /* Private data for the netfs */ struct bio_vec *direct_bv; /* DIO buffer list (when handling iovec-iter) */ - unsigned long long submitted; /* Amount submitted for I/O so far */ - unsigned long long len; /* Length of the request */ + uoff_t submitted; /* Amount submitted for I/O so far */ + uoff_t len; /* Length of the request */ size_t transferred; /* Amount to be indicated as transferred */ size_t progress_at; /* Report read progress when hit this much read */ long error; /* 0 or error that occurred */ - unsigned long long i_size; /* Size of the file */ - unsigned long long start; /* Start position */ + uoff_t i_size; /* Size of the file */ + uoff_t start; /* Start position */ atomic64_t issued_to; /* Write issuer folio cursor */ - unsigned long long collected_to; /* Point we've collected to */ - unsigned long long cleaned_to; /* Position we've cleaned folios to */ - unsigned long long abandon_to; /* Position to abandon folios to */ + uoff_t collected_to; /* Point we've collected to */ + uoff_t cleaned_to; /* Position we've cleaned folios to */ + uoff_t abandon_to; /* Position to abandon folios to */ const struct folio *no_unlock_folio; /* Don't unlock this folio after read */ gfp_t gfp; /* GFP flags to use */ unsigned int direct_bv_count; /* Number of elements in direct_bv[] */ @@ -299,12 +299,12 @@ struct netfs_request_ops { int (*prepare_read)(struct netfs_io_subrequest *subreq); void (*issue_read)(struct netfs_io_subrequest *subreq); bool (*is_still_valid)(struct netfs_io_request *rreq); - int (*check_write_begin)(struct file *file, loff_t pos, unsigned len, + int (*check_write_begin)(struct file *file, uoff_t pos, unsigned len, struct folio **foliop, void **_fsdata); void (*done)(struct netfs_io_request *rreq); /* Modification handling */ - void (*update_i_size)(struct inode *inode, loff_t i_size); + void (*update_i_size)(struct inode *inode, uoff_t i_size); void (*post_modify)(struct inode *inode); /* Write request handling */ @@ -332,7 +332,7 @@ struct netfs_cache_ops { /* Read data from the cache */ int (*read)(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, enum netfs_read_from_hole read_hole, netfs_io_terminated_t term_func, @@ -340,7 +340,7 @@ struct netfs_cache_ops { /* Write data to the cache */ int (*write)(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, netfs_io_terminated_t term_func, void *term_func_priv); @@ -350,15 +350,15 @@ struct netfs_cache_ops { /* Expand readahead request */ void (*expand_readahead)(struct netfs_cache_resources *cres, - unsigned long long *_start, - unsigned long long *_len, - unsigned long long i_size); + uoff_t *_start, + uoff_t *_len, + uoff_t i_size); /* Prepare a read operation, shortening it to a cached/uncached * boundary as appropriate. */ enum netfs_io_source (*prepare_read)(struct netfs_io_subrequest *subreq, - unsigned long long i_size); + uoff_t i_size); /* Prepare a write subrequest, working out if we're allowed to do it * and finding out the maximum amount of data to gather before @@ -371,15 +371,15 @@ struct netfs_cache_ops { * actually do. */ int (*prepare_write)(struct netfs_cache_resources *cres, - loff_t *_start, size_t *_len, size_t upper_len, - loff_t i_size, bool no_space_allocated_yet); + uoff_t *_start, size_t *_len, size_t upper_len, + uoff_t i_size, bool no_space_allocated_yet); /* Query the occupancy of the cache in a region, returning where the * next chunk of data starts and how long it is. */ int (*query_occupancy)(struct netfs_cache_resources *cres, - loff_t start, size_t len, size_t granularity, - loff_t *_data_start, size_t *_data_len); + uoff_t start, size_t len, size_t granularity, + uoff_t *_data_start, size_t *_data_len); }; /* High-level read API. */ @@ -410,7 +410,7 @@ struct readahead_control; void netfs_readahead(struct readahead_control *); int netfs_read_folio(struct file *, struct folio *); int netfs_write_begin(struct netfs_inode *, struct file *, - struct address_space *, loff_t pos, unsigned int len, + struct address_space *, uoff_t pos, unsigned int len, struct folio **, void **fsdata); int netfs_writepages(struct address_space *mapping, struct writeback_control *wbc); @@ -488,10 +488,10 @@ static inline struct netfs_inode *netfs_inode(struct inode *inode) * cmpxchg8b without the need of the lock prefix). For SMP compiles and 64bit * archs it makes no difference if preempt is enabled or not. */ -static inline unsigned long long netfs_read_remote_i_size(const struct inode *inode) +static inline uoff_t netfs_read_remote_i_size(const struct inode *inode) { const struct netfs_inode *ictx = container_of(inode, struct netfs_inode, inode); - unsigned long long remote_i_size; + uoff_t remote_i_size; #if BITS_PER_LONG==32 && defined(CONFIG_SMP) unsigned int seq; @@ -526,7 +526,7 @@ static inline unsigned long long netfs_read_remote_i_size(const struct inode *in * spinning forever. */ static inline void netfs_write_remote_i_size(struct inode *inode, - unsigned long long remote_i_size) + uoff_t remote_i_size) { struct netfs_inode *ictx = netfs_inode(inode); @@ -563,10 +563,10 @@ static inline void netfs_write_remote_i_size(struct inode *inode, * cmpxchg8b without the need of the lock prefix). For SMP compiles and 64bit * archs it makes no difference if preempt is enabled or not. */ -static inline unsigned long long netfs_read_zero_point(const struct inode *inode) +static inline uoff_t netfs_read_zero_point(const struct inode *inode) { struct netfs_inode *ictx = container_of(inode, struct netfs_inode, inode); - unsigned long long zero_point; + uoff_t zero_point; #if BITS_PER_LONG==32 && defined(CONFIG_SMP) unsigned int seq; @@ -601,7 +601,7 @@ static inline unsigned long long netfs_read_zero_point(const struct inode *inode * forever. */ static inline void netfs_write_zero_point(struct inode *inode, - unsigned long long zero_point) + uoff_t zero_point) { struct netfs_inode *ictx = netfs_inode(inode); @@ -642,9 +642,9 @@ static inline void netfs_write_zero_point(struct inode *inode, * archs it makes no difference if preempt is enabled or not. */ static inline void netfs_read_sizes(const struct inode *inode, - unsigned long long *i_size, - unsigned long long *remote_i_size, - unsigned long long *zero_point) + uoff_t *i_size, + uoff_t *remote_i_size, + uoff_t *zero_point) { const struct netfs_inode *ictx = container_of(inode, struct netfs_inode, inode); #if BITS_PER_LONG==32 && defined(CONFIG_SMP) @@ -690,9 +690,9 @@ static inline void netfs_read_sizes(const struct inode *inode, * forever. */ static inline void netfs_write_sizes(struct inode *inode, - unsigned long long i_size, - unsigned long long remote_i_size, - unsigned long long zero_point) + uoff_t i_size, + uoff_t remote_i_size, + uoff_t zero_point) { struct netfs_inode *ictx = netfs_inode(inode); @@ -760,7 +760,7 @@ static inline void netfs_inode_init(struct netfs_inode *ctx, * Inform the netfs lib that a file got resized so that it can adjust its state. */ static inline void netfs_resize_file(struct netfs_inode *ictx, - unsigned long long new_i_size, + uoff_t new_i_size, bool changed_on_server) { #if BITS_PER_LONG==32 && defined(CONFIG_SMP) diff --git a/include/trace/events/cachefiles.h b/include/trace/events/cachefiles.h index e3101410e8b2dc..1938d51a945940 100644 --- a/include/trace/events/cachefiles.h +++ b/include/trace/events/cachefiles.h @@ -449,7 +449,7 @@ TRACE_EVENT(cachefiles_vol_coherency, TRACE_EVENT(cachefiles_prep_read, TP_PROTO(struct cachefiles_object *obj, - loff_t start, + uoff_t start, size_t len, unsigned short flags, enum netfs_io_source source, @@ -464,7 +464,7 @@ TRACE_EVENT(cachefiles_prep_read, __field(enum netfs_io_source, source) __field(enum cachefiles_prepare_read_trace, why) __field(size_t, len) - __field(loff_t, start) + __field(uoff_t, start) __field(unsigned int, netfs_inode) __field(unsigned int, cache_inode) ), @@ -492,16 +492,16 @@ TRACE_EVENT(cachefiles_prep_read, TRACE_EVENT(cachefiles_read, TP_PROTO(struct cachefiles_object *obj, struct inode *backer, - loff_t start, + uoff_t start, size_t len), TP_ARGS(obj, backer, start, len), TP_STRUCT__entry( - __field(unsigned int, obj) - __field(unsigned int, backer) - __field(size_t, len) - __field(loff_t, start) + __field(unsigned int, obj) + __field(unsigned int, backer) + __field(size_t, len) + __field(uoff_t, start) ), TP_fast_assign( @@ -521,16 +521,16 @@ TRACE_EVENT(cachefiles_read, TRACE_EVENT(cachefiles_write, TP_PROTO(struct cachefiles_object *obj, struct inode *backer, - loff_t start, + uoff_t start, size_t len), TP_ARGS(obj, backer, start, len), TP_STRUCT__entry( - __field(unsigned int, obj) - __field(unsigned int, backer) - __field(size_t, len) - __field(loff_t, start) + __field(unsigned int, obj) + __field(unsigned int, backer) + __field(size_t, len) + __field(uoff_t, start) ), TP_fast_assign( @@ -549,7 +549,7 @@ TRACE_EVENT(cachefiles_write, TRACE_EVENT(cachefiles_trunc, TP_PROTO(struct cachefiles_object *obj, struct inode *backer, - loff_t from, loff_t to, enum cachefiles_trunc_trace why), + uoff_t from, uoff_t to, enum cachefiles_trunc_trace why), TP_ARGS(obj, backer, from, to, why), @@ -557,8 +557,8 @@ TRACE_EVENT(cachefiles_trunc, __field(unsigned int, obj) __field(unsigned int, backer) __field(enum cachefiles_trunc_trace, why) - __field(loff_t, from) - __field(loff_t, to) + __field(uoff_t, from) + __field(uoff_t, to) ), TP_fast_assign( diff --git a/include/trace/events/fscache.h b/include/trace/events/fscache.h index f1a73aa83fbbfb..8735d428ebd9ba 100644 --- a/include/trace/events/fscache.h +++ b/include/trace/events/fscache.h @@ -460,13 +460,13 @@ TRACE_EVENT(fscache_relinquish, ); TRACE_EVENT(fscache_invalidate, - TP_PROTO(struct fscache_cookie *cookie, loff_t new_size), + TP_PROTO(struct fscache_cookie *cookie, uoff_t new_size), TP_ARGS(cookie, new_size), TP_STRUCT__entry( __field(unsigned int, cookie ) - __field(loff_t, new_size ) + __field(uoff_t, new_size ) ), TP_fast_assign( @@ -479,14 +479,14 @@ TRACE_EVENT(fscache_invalidate, ); TRACE_EVENT(fscache_resize, - TP_PROTO(struct fscache_cookie *cookie, loff_t new_size), + TP_PROTO(struct fscache_cookie *cookie, uoff_t new_size), TP_ARGS(cookie, new_size), TP_STRUCT__entry( __field(unsigned int, cookie ) - __field(loff_t, old_size ) - __field(loff_t, new_size ) + __field(uoff_t, old_size ) + __field(uoff_t, new_size ) ), TP_fast_assign( diff --git a/include/trace/events/netfs.h b/include/trace/events/netfs.h index 3fec3e8f91c859..2010c878b0dacc 100644 --- a/include/trace/events/netfs.h +++ b/include/trace/events/netfs.h @@ -301,7 +301,7 @@ netfs_folioq_traces; TRACE_EVENT(netfs_read, TP_PROTO(struct netfs_io_request *rreq, - loff_t start, size_t len, + uoff_t start, size_t len, enum netfs_read_trace what), TP_ARGS(rreq, start, len, what), @@ -309,8 +309,8 @@ TRACE_EVENT(netfs_read, TP_STRUCT__entry( __field(unsigned int, rreq) __field(unsigned int, cookie) - __field(loff_t, i_size) - __field(loff_t, start) + __field(uoff_t, i_size) + __field(uoff_t, start) __field(size_t, len) __field(enum netfs_read_trace, what) __field(u64, netfs_inode) @@ -377,7 +377,7 @@ TRACE_EVENT(netfs_sreq, __field(u8, slot) __field(size_t, len) __field(size_t, transferred) - __field(loff_t, start) + __field(uoff_t, start) ), TP_fast_assign( @@ -418,7 +418,7 @@ TRACE_EVENT(netfs_failure, __field(enum netfs_failure, what) __field(size_t, len) __field(size_t, transferred) - __field(loff_t, start) + __field(uoff_t, start) ), TP_fast_assign( @@ -524,10 +524,10 @@ TRACE_EVENT(netfs_write_iter, TP_ARGS(iocb, from), TP_STRUCT__entry( - __field(unsigned long long, start) - __field(size_t, len) - __field(unsigned int, flags) - __field(unsigned int, ino) + __field(uoff_t, start) + __field(size_t, len) + __field(unsigned int, flags) + __field(unsigned int, ino) ), TP_fast_assign( @@ -552,8 +552,8 @@ TRACE_EVENT(netfs_write, __field(unsigned int, cookie) __field(unsigned int, ino) __field(enum netfs_write_trace, what) - __field(unsigned long long, start) - __field(unsigned long long, len) + __field(uoff_t, start) + __field(uoff_t, len) ), TP_fast_assign( @@ -582,10 +582,10 @@ TRACE_EVENT(netfs_copy2cache, TP_ARGS(rreq, creq), TP_STRUCT__entry( - __field(unsigned int, rreq) - __field(unsigned int, creq) - __field(unsigned int, cookie) - __field(unsigned int, ino) + __field(unsigned int, rreq) + __field(unsigned int, creq) + __field(unsigned int, cookie) + __field(unsigned int, ino) ), TP_fast_assign( @@ -610,10 +610,10 @@ TRACE_EVENT(netfs_collect, TP_ARGS(wreq), TP_STRUCT__entry( - __field(unsigned int, wreq) - __field(unsigned int, len) - __field(unsigned long long, transferred) - __field(unsigned long long, start) + __field(unsigned int, wreq) + __field(unsigned int, len) + __field(uoff_t, transferred) + __field(uoff_t, start) ), TP_fast_assign( @@ -636,12 +636,12 @@ TRACE_EVENT(netfs_collect_sreq, TP_ARGS(wreq, subreq), TP_STRUCT__entry( - __field(unsigned int, wreq) - __field(unsigned int, subreq) - __field(unsigned int, stream) - __field(unsigned int, len) - __field(unsigned int, transferred) - __field(unsigned long long, start) + __field(unsigned int, wreq) + __field(unsigned int, subreq) + __field(unsigned int, stream) + __field(unsigned int, len) + __field(unsigned int, transferred) + __field(uoff_t, start) ), TP_fast_assign( @@ -661,17 +661,16 @@ TRACE_EVENT(netfs_collect_sreq, TRACE_EVENT(netfs_collect_folio, TP_PROTO(const struct netfs_io_request *wreq, const struct folio *folio, - unsigned long long fend, - unsigned long long collected_to), + uoff_t fend, uoff_t collected_to), TP_ARGS(wreq, folio, fend, collected_to), TP_STRUCT__entry( __field(unsigned int, wreq) __field(unsigned long, index) - __field(unsigned long long, fend) - __field(unsigned long long, cleaned_to) - __field(unsigned long long, collected_to) + __field(uoff_t, fend) + __field(uoff_t, cleaned_to) + __field(uoff_t, collected_to) ), TP_fast_assign( @@ -684,13 +683,13 @@ TRACE_EVENT(netfs_collect_folio, TP_printk("R=%08x ix=%05lx r=%llx-%llx t=%llx/%llx", __entry->wreq, __entry->index, - (unsigned long long)__entry->index * PAGE_SIZE, __entry->fend, + (uoff_t)__entry->index * PAGE_SIZE, __entry->fend, __entry->cleaned_to, __entry->collected_to) ); TRACE_EVENT(netfs_collect_state, TP_PROTO(const struct netfs_io_request *wreq, - unsigned long long collected_to, + uoff_t collected_to, unsigned int notes), TP_ARGS(wreq, collected_to, notes), @@ -698,8 +697,8 @@ TRACE_EVENT(netfs_collect_state, TP_STRUCT__entry( __field(unsigned int, wreq) __field(unsigned int, notes) - __field(unsigned long long, collected_to) - __field(unsigned long long, cleaned_to) + __field(uoff_t, collected_to) + __field(uoff_t, cleaned_to) ), TP_fast_assign( @@ -718,7 +717,7 @@ TRACE_EVENT(netfs_collect_state, TRACE_EVENT(netfs_collect_gap, TP_PROTO(const struct netfs_io_request *wreq, const struct netfs_io_stream *stream, - unsigned long long jump_to, char type), + uoff_t jump_to, char type), TP_ARGS(wreq, stream, jump_to, type), @@ -726,8 +725,8 @@ TRACE_EVENT(netfs_collect_gap, __field(unsigned int, wreq) __field(unsigned char, stream) __field(unsigned char, type) - __field(unsigned long long, from) - __field(unsigned long long, to) + __field(uoff_t, from) + __field(uoff_t, to) ), TP_fast_assign( @@ -752,8 +751,8 @@ TRACE_EVENT(netfs_collect_stream, TP_STRUCT__entry( __field(unsigned int, wreq) __field(unsigned char, stream) - __field(unsigned long long, collected_to) - __field(unsigned long long, issued_to) + __field(uoff_t, collected_to) + __field(uoff_t, issued_to) ), TP_fast_assign( From 1f88d469ae24e08740b73732303e01091952c964 Mon Sep 17 00:00:00 2001 From: David Howells Date: Wed, 9 Sep 2026 08:20:56 +0100 Subject: [PATCH 0020/1352] netfs: Remove the writethrough code Remove the netfs writethrough code as it's very tricky to get the locking right and it will probably deadlock if used in conjunction with Ceph snapshots because it excludes writeback for the duration, but to flush out old snapshots, it does a synchronous flush that invokes writeback. Instead, O_SYNC writes do a flush after performing the write - which is already there as the callers of netfs_perform_write() all call generic_write_sync(). Signed-off-by: David Howells Link: https://patch.msgid.link/20260909072105.1663687-3-dhowells@redhat.com Reviewed-by: Paulo Alcantara cc: netfs@lists.linux.dev cc: linux-fsdevel@vger.kernel.org Signed-off-by: Christian Brauner (Amutable) --- fs/9p/vfs_addr.c | 1 - fs/afs/file.c | 1 - fs/netfs/buffered_write.c | 56 ++----------------- fs/netfs/internal.h | 7 --- fs/netfs/main.c | 1 - fs/netfs/stats.c | 4 +- fs/netfs/write_collect.c | 2 - fs/netfs/write_issue.c | 104 +---------------------------------- include/linux/netfs.h | 1 - include/trace/events/netfs.h | 8 +-- 10 files changed, 9 insertions(+), 176 deletions(-) diff --git a/fs/9p/vfs_addr.c b/fs/9p/vfs_addr.c index 1ac0b3dcc0778e..2129fcb0f65c14 100644 --- a/fs/9p/vfs_addr.c +++ b/fs/9p/vfs_addr.c @@ -124,7 +124,6 @@ static int v9fs_init_request(struct netfs_io_request *rreq, struct file *file) struct p9_fid *fid; struct dentry *dentry; bool writing = (rreq->origin == NETFS_READ_FOR_WRITE || - rreq->origin == NETFS_WRITETHROUGH || rreq->origin == NETFS_UNBUFFERED_WRITE || rreq->origin == NETFS_DIO_WRITE); diff --git a/fs/afs/file.c b/fs/afs/file.c index bdb4ca0ef8da73..b7f461b9d5f506 100644 --- a/fs/afs/file.c +++ b/fs/afs/file.c @@ -400,7 +400,6 @@ static int afs_init_request(struct netfs_io_request *rreq, struct file *file) } break; case NETFS_WRITEBACK: - case NETFS_WRITETHROUGH: case NETFS_UNBUFFERED_WRITE: case NETFS_DIO_WRITE: if (S_ISREG(rreq->inode->i_mode)) diff --git a/fs/netfs/buffered_write.c b/fs/netfs/buffered_write.c index df496873e4f445..ead22980075fea 100644 --- a/fs/netfs/buffered_write.c +++ b/fs/netfs/buffered_write.c @@ -91,44 +91,14 @@ ssize_t netfs_perform_write(struct kiocb *iocb, struct iov_iter *iter, struct inode *inode = file_inode(file); struct address_space *mapping = inode->i_mapping; struct netfs_inode *ctx = netfs_inode(inode); - struct writeback_control wbc = { - .sync_mode = WB_SYNC_NONE, - .for_sync = true, - .nr_to_write = LONG_MAX, - .range_start = iocb->ki_pos, - .range_end = iocb->ki_pos + iter->count, - }; - struct netfs_io_request *wreq = NULL; - struct folio *folio = NULL, *writethrough = NULL; + struct folio *folio = NULL; unsigned int bdp_flags = (iocb->ki_flags & IOCB_NOWAIT) ? BDP_ASYNC : 0; - ssize_t written = 0, ret, ret2; + ssize_t written = 0, ret; uoff_t pos = iocb->ki_pos; size_t max_chunk = mapping_max_folio_size(mapping); bool maybe_trouble = false; - if (unlikely(iocb->ki_flags & (IOCB_DSYNC | IOCB_SYNC)) - ) { - wbc_attach_fdatawrite_inode(&wbc, mapping->host); - - ret = filemap_write_and_wait_range(mapping, pos, pos + iter->count); - if (ret < 0) { - wbc_detach_inode(&wbc); - goto out; - } - - wreq = netfs_begin_writethrough(iocb, iter->count); - if (IS_ERR(wreq)) { - wbc_detach_inode(&wbc); - ret = PTR_ERR(wreq); - wreq = NULL; - goto out; - } - if (!is_sync_kiocb(iocb)) - wreq->iocb = iocb; - netfs_stat(&netfs_n_wh_writethrough); - } else { - netfs_stat(&netfs_n_wh_buffered_write); - } + netfs_stat(&netfs_n_wh_buffered_write); do { enum netfs_folio_trace trace; @@ -390,15 +360,8 @@ ssize_t netfs_perform_write(struct kiocb *iocb, struct iov_iter *iter, pos += copied; written += copied; - if (likely(!wreq)) { - folio_mark_dirty(folio); - folio_unlock(folio); - } else { - netfs_advance_writethrough(wreq, &wbc, folio, copied, - offset + copied == flen, - &writethrough); - /* Folio unlocked */ - } + folio_mark_dirty(folio); + folio_unlock(folio); retry: folio_put(folio); folio = NULL; @@ -420,15 +383,6 @@ ssize_t netfs_perform_write(struct kiocb *iocb, struct iov_iter *iter, ctx->ops->post_modify(inode); } - if (unlikely(wreq)) { - ret2 = netfs_end_writethrough(wreq, &wbc, writethrough); - wbc_detach_inode(&wbc); - if (ret2 == -EIOCBQUEUED) - return ret2; - if (ret == 0 && ret2 < 0) - ret = ret2; - } - iocb->ki_pos += written; _leave(" = %zd [%zd]", written, ret); return written ? written : ret; diff --git a/fs/netfs/internal.h b/fs/netfs/internal.h index 9bd7ad10cc0cff..a4c834e32214f7 100644 --- a/fs/netfs/internal.h +++ b/fs/netfs/internal.h @@ -157,7 +157,6 @@ extern atomic_t netfs_n_rh_write_zskip; extern atomic_t netfs_n_rh_retry_read_req; extern atomic_t netfs_n_rh_retry_read_subreq; extern atomic_t netfs_n_wh_buffered_write; -extern atomic_t netfs_n_wh_writethrough; extern atomic_t netfs_n_wh_dio_write; extern atomic_t netfs_n_wh_writepages; extern atomic_t netfs_n_wh_copy_to_cache; @@ -216,12 +215,6 @@ void netfs_issue_write(struct netfs_io_request *wreq, size_t netfs_advance_write(struct netfs_io_request *wreq, struct netfs_io_stream *stream, uoff_t start, size_t len, bool to_eof); -struct netfs_io_request *netfs_begin_writethrough(struct kiocb *iocb, size_t len); -int netfs_advance_writethrough(struct netfs_io_request *wreq, struct writeback_control *wbc, - struct folio *folio, size_t copied, bool to_page_end, - struct folio **writethrough_cache); -ssize_t netfs_end_writethrough(struct netfs_io_request *wreq, struct writeback_control *wbc, - struct folio *writethrough_cache); /* * write_retry.c diff --git a/fs/netfs/main.c b/fs/netfs/main.c index 927badf3989db4..609e22e8f76a06 100644 --- a/fs/netfs/main.c +++ b/fs/netfs/main.c @@ -44,7 +44,6 @@ static const char *netfs_origins[nr__netfs_io_origin] = { [NETFS_DIO_READ] = "DR", [NETFS_WRITEBACK] = "WB", [NETFS_WRITEBACK_SINGLE] = "W1", - [NETFS_WRITETHROUGH] = "WT", [NETFS_UNBUFFERED_WRITE] = "UW", [NETFS_DIO_WRITE] = "DW", [NETFS_PGPRIV2_COPY_TO_CACHE] = "2C", diff --git a/fs/netfs/stats.c b/fs/netfs/stats.c index ab6b916addc443..9a607c4e62dda5 100644 --- a/fs/netfs/stats.c +++ b/fs/netfs/stats.c @@ -32,7 +32,6 @@ atomic_t netfs_n_rh_write_zskip; atomic_t netfs_n_rh_retry_read_req; atomic_t netfs_n_rh_retry_read_subreq; atomic_t netfs_n_wh_buffered_write; -atomic_t netfs_n_wh_writethrough; atomic_t netfs_n_wh_dio_write; atomic_t netfs_n_wh_writepages; atomic_t netfs_n_wh_copy_to_cache; @@ -58,9 +57,8 @@ int netfs_stats_show(struct seq_file *m, void *v) atomic_read(&netfs_n_rh_read_single), atomic_read(&netfs_n_rh_write_begin), atomic_read(&netfs_n_rh_write_zskip)); - seq_printf(m, "Writes : BW=%u WT=%u DW=%u WP=%u 2C=%u\n", + seq_printf(m, "Writes : BW=%u DW=%u WP=%u 2C=%u\n", atomic_read(&netfs_n_wh_buffered_write), - atomic_read(&netfs_n_wh_writethrough), atomic_read(&netfs_n_wh_dio_write), atomic_read(&netfs_n_wh_writepages), atomic_read(&netfs_n_wh_copy_to_cache)); diff --git a/fs/netfs/write_collect.c b/fs/netfs/write_collect.c index 100a5038c61e39..244a68e04624bc 100644 --- a/fs/netfs/write_collect.c +++ b/fs/netfs/write_collect.c @@ -214,7 +214,6 @@ static void netfs_collect_write_results(struct netfs_io_request *wreq) smp_rmb(); collected_to = ULLONG_MAX; if (wreq->origin == NETFS_WRITEBACK || - wreq->origin == NETFS_WRITETHROUGH || wreq->origin == NETFS_PGPRIV2_COPY_TO_CACHE) notes = NEED_UNLOCK; else @@ -411,7 +410,6 @@ bool netfs_write_collection(struct netfs_io_request *wreq) switch (wreq->origin) { case NETFS_WRITEBACK: case NETFS_WRITEBACK_SINGLE: - case NETFS_WRITETHROUGH: netfs_wb_end(ictx); break; default: diff --git a/fs/netfs/write_issue.c b/fs/netfs/write_issue.c index 3b33ad5b69c39c..5d130df2ff0d06 100644 --- a/fs/netfs/write_issue.c +++ b/fs/netfs/write_issue.c @@ -96,7 +96,6 @@ struct netfs_io_request *netfs_create_write_req(struct address_space *mapping, struct netfs_inode *ictx; bool is_cacheable = (origin == NETFS_WRITEBACK || origin == NETFS_WRITEBACK_SINGLE || - origin == NETFS_WRITETHROUGH || origin == NETFS_PGPRIV2_COPY_TO_CACHE); wreq = netfs_alloc_request(mapping, file, start, 0, origin); @@ -367,11 +366,7 @@ static int netfs_write_folio(struct netfs_io_request *wreq, streamw = true; } - if (wreq->origin == NETFS_WRITETHROUGH) { - to_eof = false; - if (flen > i_size - fpos) - flen = i_size - fpos; - } else if (flen > i_size - fpos) { + if (flen > i_size - fpos) { flen = i_size - fpos; if (!streamw) folio_zero_segment(folio, flen, fsize); @@ -613,103 +608,6 @@ int netfs_writepages(struct address_space *mapping, } EXPORT_SYMBOL(netfs_writepages); -/* - * Begin a write operation for writing through the pagecache. - */ -struct netfs_io_request *netfs_begin_writethrough(struct kiocb *iocb, size_t len) -{ - struct netfs_io_request *wreq = NULL; - struct netfs_inode *ictx = netfs_inode(file_inode(iocb->ki_filp)); - - netfs_wb_begin(ictx, false); - - wreq = netfs_create_write_req(iocb->ki_filp->f_mapping, iocb->ki_filp, - iocb->ki_pos, NETFS_WRITETHROUGH); - if (IS_ERR(wreq)) { - netfs_wb_end(ictx); - return wreq; - } - - wreq->io_streams[0].avail = true; - __set_bit(NETFS_RREQ_OFFLOAD_COLLECTION, &wreq->flags); - trace_netfs_write(wreq, netfs_write_trace_writethrough); - return wreq; -} - -/* - * Advance the state of the write operation used when writing through the - * pagecache. Data has been copied into the pagecache that we need to append - * to the request. If we've added more than wsize then we need to create a new - * subrequest. - */ -int netfs_advance_writethrough(struct netfs_io_request *wreq, struct writeback_control *wbc, - struct folio *folio, size_t copied, bool to_page_end, - struct folio **writethrough_cache) -{ - int ret; - - _enter("R=%x ic=%zu ws=%u cp=%zu tp=%u", - wreq->debug_id, wreq->buffer.iter.count, wreq->wsize, copied, to_page_end); - - /* The folio is locked. */ - - if (*writethrough_cache != folio) { - if (*writethrough_cache) { - /* Did the folio get moved? */ - folio_put(*writethrough_cache); - *writethrough_cache = NULL; - } - /* We can make multiple writes to the folio... */ - if (wreq->len == 0) - trace_netfs_folio(folio, netfs_folio_trace_wthru); - else - trace_netfs_folio(folio, netfs_folio_trace_wthru_plus); - *writethrough_cache = folio; - folio_get(folio); - } - - wreq->len += copied; - - if (!to_page_end) { - folio_mark_dirty(folio); - folio_unlock(folio); - return 0; - } - - ret = netfs_write_folio(wreq, wbc, folio); - folio_put(*writethrough_cache); - *writethrough_cache = NULL; - wreq->submitted = wreq->len; - return ret; -} - -/* - * End a write operation used when writing through the pagecache. - */ -ssize_t netfs_end_writethrough(struct netfs_io_request *wreq, struct writeback_control *wbc, - struct folio *writethrough_cache) -{ - ssize_t ret; - - _enter("R=%x", wreq->debug_id); - - if (writethrough_cache) { - folio_lock(writethrough_cache); - netfs_write_folio(wreq, wbc, writethrough_cache); - folio_put(writethrough_cache); - wreq->submitted = wreq->len; - } - - netfs_end_issue_write(wreq); - - if (wreq->iocb) - ret = -EIOCBQUEUED; - else - ret = netfs_wait_for_write(wreq); - netfs_put_request(wreq, netfs_rreq_trace_put_return); - return ret; -} - /* * Write some of a pending folio data back to the server and/or the cache. */ diff --git a/include/linux/netfs.h b/include/linux/netfs.h index e239d104f1a5f2..a3ef0e983a867c 100644 --- a/include/linux/netfs.h +++ b/include/linux/netfs.h @@ -208,7 +208,6 @@ enum netfs_io_origin { NETFS_DIO_READ, /* This is a direct I/O read */ NETFS_WRITEBACK, /* This write was triggered by writepages */ NETFS_WRITEBACK_SINGLE, /* This monolithic write was triggered by writepages */ - NETFS_WRITETHROUGH, /* This write was made by netfs_perform_write() */ NETFS_UNBUFFERED_WRITE, /* This is an unbuffered write */ NETFS_DIO_WRITE, /* This is a direct I/O write */ NETFS_PGPRIV2_COPY_TO_CACHE, /* [DEPRECATED] This is writing read data to the cache */ diff --git a/include/trace/events/netfs.h b/include/trace/events/netfs.h index 2010c878b0dacc..ef1185993ba45e 100644 --- a/include/trace/events/netfs.h +++ b/include/trace/events/netfs.h @@ -30,8 +30,7 @@ EM(netfs_write_trace_dio_write, "DIO-WRITE") \ EM(netfs_write_trace_unbuffered_write, "UNB-WRITE") \ EM(netfs_write_trace_writeback, "WRITEBACK") \ - EM(netfs_write_trace_writeback_single, "WB-SINGLE") \ - E_(netfs_write_trace_writethrough, "WRITETHRU") + E_(netfs_write_trace_writeback_single, "WB-SINGLE") #define netfs_rreq_origins \ EM(NETFS_READAHEAD, "RA") \ @@ -43,7 +42,6 @@ EM(NETFS_DIO_READ, "DR") \ EM(NETFS_WRITEBACK, "WB") \ EM(NETFS_WRITEBACK_SINGLE, "W1") \ - EM(NETFS_WRITETHROUGH, "WT") \ EM(NETFS_UNBUFFERED_WRITE, "UW") \ EM(NETFS_DIO_WRITE, "DW") \ E_(NETFS_PGPRIV2_COPY_TO_CACHE, "2C") @@ -223,9 +221,7 @@ EM(netfs_folio_trace_sched_copy, "sched-copy") \ EM(netfs_folio_trace_store, "store") \ EM(netfs_folio_trace_store_copy, "store-copy") \ - EM(netfs_folio_trace_store_plus, "store+") \ - EM(netfs_folio_trace_wthru, "wthru") \ - E_(netfs_folio_trace_wthru_plus, "wthru+") + E_(netfs_folio_trace_store_plus, "store+") #define netfs_collect_contig_traces \ EM(netfs_contig_trace_collect, "Collect") \ From 36b9fb2a98de830f7acceec9aecb029eca9e737e Mon Sep 17 00:00:00 2001 From: David Howells Date: Wed, 9 Sep 2026 08:20:57 +0100 Subject: [PATCH 0021/1352] netfs: trace: Change the "clear" folio traces to "endwb" Change the "clear" folio traces to "endwb" as it's more obvious what it means. Signed-off-by: David Howells Link: https://patch.msgid.link/20260909072105.1663687-4-dhowells@redhat.com Reviewed-by: Paulo Alcantara cc: netfs@lists.linux.dev cc: linux-fsdevel@vger.kernel.org Signed-off-by: Christian Brauner (Amutable) --- fs/netfs/write_collect.c | 8 ++++---- include/trace/events/netfs.h | 8 ++++---- 2 files changed, 8 insertions(+), 8 deletions(-) diff --git a/fs/netfs/write_collect.c b/fs/netfs/write_collect.c index 244a68e04624bc..6114bdf27ce088 100644 --- a/fs/netfs/write_collect.c +++ b/fs/netfs/write_collect.c @@ -56,7 +56,7 @@ static void netfs_dump_request(const struct netfs_io_request *rreq) */ int netfs_folio_written_back(struct folio *folio) { - enum netfs_folio_trace why = netfs_folio_trace_clear; + enum netfs_folio_trace why = netfs_folio_trace_endwb; struct inode *inode = folio_inode(folio); struct netfs_inode *ictx = netfs_inode(inode); struct netfs_folio *finfo; @@ -79,13 +79,13 @@ int netfs_folio_written_back(struct folio *folio) group = finfo->netfs_group; gcount++; kfree(finfo); - why = netfs_folio_trace_clear_s; + why = netfs_folio_trace_endwb_s; goto end_wb; } if ((group = netfs_folio_group(folio))) { if (group == NETFS_FOLIO_COPY_TO_CACHE) { - why = netfs_folio_trace_clear_cc; + why = netfs_folio_trace_endwb_cc; folio_detach_private(folio); goto end_wb; } @@ -98,7 +98,7 @@ int netfs_folio_written_back(struct folio *folio) if (!folio_test_dirty(folio)) { folio_detach_private(folio); gcount++; - why = netfs_folio_trace_clear_g; + why = netfs_folio_trace_endwb_g; } } diff --git a/include/trace/events/netfs.h b/include/trace/events/netfs.h index ef1185993ba45e..2491a0bc417904 100644 --- a/include/trace/events/netfs.h +++ b/include/trace/events/netfs.h @@ -192,11 +192,11 @@ EM(netfs_folio_trace_alloc_buffer, "alloc-buf") \ EM(netfs_folio_trace_cancel_copy, "cancel-copy") \ EM(netfs_folio_trace_cancel_store, "cancel-store") \ - EM(netfs_folio_trace_clear, "clear") \ - EM(netfs_folio_trace_clear_cc, "clear-cc") \ - EM(netfs_folio_trace_clear_g, "clear-g") \ - EM(netfs_folio_trace_clear_s, "clear-s") \ EM(netfs_folio_trace_end_copy, "end-copy") \ + EM(netfs_folio_trace_endwb, "endwb") \ + EM(netfs_folio_trace_endwb_cc, "endwb-cc") \ + EM(netfs_folio_trace_endwb_g, "endwb-g") \ + EM(netfs_folio_trace_endwb_s, "endwb-s") \ EM(netfs_folio_trace_filled_gaps, "filled-gaps") \ EM(netfs_folio_trace_invalidate_all, "inval-all") \ EM(netfs_folio_trace_invalidate_front, "inval-front") \ From 7c027ce5150952f54414745efb533d59aaee7b5a Mon Sep 17 00:00:00 2001 From: David Howells Date: Wed, 9 Sep 2026 08:20:58 +0100 Subject: [PATCH 0022/1352] netfs: trace: Rejig a couple of the tracepoints Rejig the following tracepoints: (1) Change netfs_folio to show the pfn. (2) Change netfs_collect_folio to show a folio index range rather than file position range and don't show the cleaned_to or collected_to points. Signed-off-by: David Howells Link: https://patch.msgid.link/20260909072105.1663687-5-dhowells@redhat.com Reviewed-by: Paulo Alcantara cc: netfs@lists.linux.dev cc: linux-fsdevel@vger.kernel.org Signed-off-by: Christian Brauner (Amutable) --- fs/netfs/read_collect.c | 2 +- fs/netfs/read_pgpriv2.c | 2 +- fs/netfs/write_collect.c | 2 +- include/trace/events/netfs.h | 23 ++++++++++------------- 4 files changed, 13 insertions(+), 16 deletions(-) diff --git a/fs/netfs/read_collect.c b/fs/netfs/read_collect.c index 6576a9b8671d30..61f2664de6d3a0 100644 --- a/fs/netfs/read_collect.c +++ b/fs/netfs/read_collect.c @@ -192,7 +192,7 @@ static void netfs_read_unlock_folios(struct netfs_io_request *rreq, fpos = folio_pos(folio); fend = fpos + fsize; - trace_netfs_collect_folio(rreq, folio, fend, collected_to); + trace_netfs_collect_folio(rreq, folio); /* Unlock any folio we've transferred all of. */ if (collected_to < fend) diff --git a/fs/netfs/read_pgpriv2.c b/fs/netfs/read_pgpriv2.c index 4d50585f69a608..bf4d9d6877d7b2 100644 --- a/fs/netfs/read_pgpriv2.c +++ b/fs/netfs/read_pgpriv2.c @@ -201,7 +201,7 @@ bool netfs_pgpriv2_unlock_copied_folios(struct netfs_io_request *creq) fend = min_t(uoff_t, fpos + flen, creq->i_size); - trace_netfs_collect_folio(creq, folio, fend, collected_to); + trace_netfs_collect_folio(creq, folio); /* Unlock any folio we've transferred all of. */ if (collected_to < fend) diff --git a/fs/netfs/write_collect.c b/fs/netfs/write_collect.c index 6114bdf27ce088..7194182b975cf2 100644 --- a/fs/netfs/write_collect.c +++ b/fs/netfs/write_collect.c @@ -156,7 +156,7 @@ static void netfs_writeback_unlock_folios(struct netfs_io_request *wreq, fend = min_t(uoff_t, fpos + flen, wreq->i_size); - trace_netfs_collect_folio(wreq, folio, fend, collected_to); + trace_netfs_collect_folio(wreq, folio); /* Unlock any folio we've transferred all of. */ if (collected_to < fend) diff --git a/include/trace/events/netfs.h b/include/trace/events/netfs.h index 2491a0bc417904..21a661cd1ca472 100644 --- a/include/trace/events/netfs.h +++ b/include/trace/events/netfs.h @@ -497,6 +497,7 @@ TRACE_EVENT(netfs_folio, TP_STRUCT__entry( __field(u64, ino) __field(pgoff_t, index) + __field(unsigned long, pfn) __field(unsigned int, nr) __field(enum netfs_folio_trace, why) ), @@ -507,9 +508,11 @@ TRACE_EVENT(netfs_folio, __entry->why = why; __entry->index = folio->index; __entry->nr = folio_nr_pages(folio); + __entry->pfn = folio_pfn(folio); ), - TP_printk("i=%05llx ix=%05lx-%05lx %s", + TP_printk("p=%lx i=%05llx ix=%05lx-%05lx %s", + __entry->pfn, __entry->ino, __entry->index, __entry->index + __entry->nr - 1, __print_symbolic(__entry->why, netfs_folio_traces)) ); @@ -656,31 +659,25 @@ TRACE_EVENT(netfs_collect_sreq, TRACE_EVENT(netfs_collect_folio, TP_PROTO(const struct netfs_io_request *wreq, - const struct folio *folio, - uoff_t fend, uoff_t collected_to), + const struct folio *folio), - TP_ARGS(wreq, folio, fend, collected_to), + TP_ARGS(wreq, folio), TP_STRUCT__entry( __field(unsigned int, wreq) __field(unsigned long, index) - __field(uoff_t, fend) - __field(uoff_t, cleaned_to) - __field(uoff_t, collected_to) + __field(unsigned int, nr) ), TP_fast_assign( __entry->wreq = wreq->debug_id; __entry->index = folio->index; - __entry->fend = fend; - __entry->cleaned_to = wreq->cleaned_to; - __entry->collected_to = collected_to; + __entry->nr = folio_nr_pages(folio); ), - TP_printk("R=%08x ix=%05lx r=%llx-%llx t=%llx/%llx", + TP_printk("R=%08x ix=%05lx-%05lx", __entry->wreq, __entry->index, - (uoff_t)__entry->index * PAGE_SIZE, __entry->fend, - __entry->cleaned_to, __entry->collected_to) + __entry->index + __entry->nr - 1) ); TRACE_EVENT(netfs_collect_state, From 5cc2c1f53841a2ead03b895cc45833a9b27e9143 Mon Sep 17 00:00:00 2001 From: David Howells Date: Wed, 9 Sep 2026 08:20:59 +0100 Subject: [PATCH 0023/1352] netfs: Add the cache object ID to netfs_read/write tracepoints Add the cache object debug ID to netfs_read/write tracepoints to make debugging easier as there's now a direct cross-reference with the cachefiles tracepoints that only log that debug ID. Signed-off-by: David Howells Link: https://patch.msgid.link/20260909072105.1663687-6-dhowells@redhat.com Reviewed-by: Paulo Alcantara cc: netfs@lists.linux.dev cc: linux-fsdevel@vger.kernel.org Signed-off-by: Christian Brauner (Amutable) --- fs/cachefiles/io.c | 1 + fs/netfs/fscache_io.c | 2 +- include/linux/netfs.h | 3 ++- include/trace/events/netfs.h | 27 +++++++++++++++------------ 4 files changed, 19 insertions(+), 14 deletions(-) diff --git a/fs/cachefiles/io.c b/fs/cachefiles/io.c index 7de8069d15b6b5..e61e885784d695 100644 --- a/fs/cachefiles/io.c +++ b/fs/cachefiles/io.c @@ -721,6 +721,7 @@ bool cachefiles_begin_operation(struct netfs_cache_resources *cres, if (!cachefiles_cres_file(cres)) { cres->ops = &cachefiles_netfs_cache_ops; + cres->object_id = object->debug_id; if (object->file) { spin_lock(&object->lock); if (!cres->cache_priv2 && object->file) diff --git a/fs/netfs/fscache_io.c b/fs/netfs/fscache_io.c index 8bca63721eeb03..056a2bae5d99a3 100644 --- a/fs/netfs/fscache_io.c +++ b/fs/netfs/fscache_io.c @@ -79,7 +79,7 @@ static int fscache_begin_operation(struct netfs_cache_resources *cres, cres->ops = NULL; cres->cache_priv = cookie; cres->cache_priv2 = NULL; - cres->debug_id = cookie->debug_id; + cres->cookie_id = cookie->debug_id; cres->inval_counter = cookie->inval_counter; if (!fscache_begin_cookie_access(cookie, why)) { diff --git a/include/linux/netfs.h b/include/linux/netfs.h index a3ef0e983a867c..f2b3e61c1891f3 100644 --- a/include/linux/netfs.h +++ b/include/linux/netfs.h @@ -161,7 +161,8 @@ struct netfs_cache_resources { const struct netfs_cache_ops *ops; void *cache_priv; void *cache_priv2; - unsigned int debug_id; /* Cookie debug ID */ + unsigned int cookie_id; /* Cache cookie debug ID */ + unsigned int object_id; /* Cache object debug ID */ unsigned int inval_counter; /* object->inval_counter at begin_op */ }; diff --git a/include/trace/events/netfs.h b/include/trace/events/netfs.h index 21a661cd1ca472..c9ec4da97f0bdd 100644 --- a/include/trace/events/netfs.h +++ b/include/trace/events/netfs.h @@ -305,6 +305,7 @@ TRACE_EVENT(netfs_read, TP_STRUCT__entry( __field(unsigned int, rreq) __field(unsigned int, cookie) + __field(unsigned int, object) __field(uoff_t, i_size) __field(uoff_t, start) __field(size_t, len) @@ -314,7 +315,8 @@ TRACE_EVENT(netfs_read, TP_fast_assign( __entry->rreq = rreq->debug_id; - __entry->cookie = rreq->cache_resources.debug_id; + __entry->cookie = rreq->cache_resources.cookie_id; + __entry->object = rreq->cache_resources.object_id; __entry->i_size = rreq->i_size; __entry->start = start; __entry->len = len; @@ -322,10 +324,10 @@ TRACE_EVENT(netfs_read, __entry->netfs_inode = rreq->inode->i_ino; ), - TP_printk("R=%08x %s c=%08x ni=%llx s=%llx l=%zx sz=%llx", + TP_printk("R=%08x %s c=%08x o=%08x ni=%llx s=%llx l=%zx sz=%llx", __entry->rreq, __print_symbolic(__entry->what, netfs_read_traces), - __entry->cookie, + __entry->cookie, __entry->object, __entry->netfs_inode, __entry->start, __entry->len, __entry->i_size) ); @@ -549,6 +551,7 @@ TRACE_EVENT(netfs_write, TP_STRUCT__entry( __field(unsigned int, wreq) __field(unsigned int, cookie) + __field(unsigned int, object) __field(unsigned int, ino) __field(enum netfs_write_trace, what) __field(uoff_t, start) @@ -556,20 +559,19 @@ TRACE_EVENT(netfs_write, ), TP_fast_assign( - struct netfs_inode *__ctx = netfs_inode(wreq->inode); - struct fscache_cookie *__cookie = netfs_i_cookie(__ctx); __entry->wreq = wreq->debug_id; - __entry->cookie = __cookie ? __cookie->debug_id : 0; + __entry->cookie = wreq->cache_resources.cookie_id; + __entry->object = wreq->cache_resources.object_id; __entry->ino = wreq->inode->i_ino; __entry->what = what; __entry->start = wreq->start; __entry->len = wreq->len; ), - TP_printk("R=%08x %s c=%08x i=%x by=%llx-%llx", + TP_printk("R=%08x %s c=%08x o=%08x i=%x by=%llx-%llx", __entry->wreq, __print_symbolic(__entry->what, netfs_write_traces), - __entry->cookie, + __entry->cookie, __entry->object, __entry->ino, __entry->start, __entry->start + __entry->len - 1) ); @@ -584,22 +586,23 @@ TRACE_EVENT(netfs_copy2cache, __field(unsigned int, rreq) __field(unsigned int, creq) __field(unsigned int, cookie) + __field(unsigned int, object) __field(unsigned int, ino) ), TP_fast_assign( - struct netfs_inode *__ctx = netfs_inode(rreq->inode); - struct fscache_cookie *__cookie = netfs_i_cookie(__ctx); __entry->rreq = rreq->debug_id; __entry->creq = creq->debug_id; - __entry->cookie = __cookie ? __cookie->debug_id : 0; + __entry->cookie = rreq->cache_resources.cookie_id; + __entry->object = rreq->cache_resources.object_id; __entry->ino = rreq->inode->i_ino; ), - TP_printk("R=%08x CR=%08x c=%08x i=%x ", + TP_printk("R=%08x CR=%08x c=%08x o=%08x i=%x ", __entry->rreq, __entry->creq, __entry->cookie, + __entry->object, __entry->ino) ); From bdfd7248dadb83a5a00c3050066eb00b0022bc21 Mon Sep 17 00:00:00 2001 From: David Howells Date: Wed, 9 Sep 2026 08:21:00 +0100 Subject: [PATCH 0024/1352] netfs: Make deprecated PG_private_2 support opt-in Make the deprecated PG_private_2 support opt-in, requiring it to be selected by the filesystems that might want to use it. Signed-off-by: David Howells Link: https://patch.msgid.link/20260909072105.1663687-7-dhowells@redhat.com Reviewed-by: Paulo Alcantara cc: Trond Myklebust cc: Anna Schumaker cc: Ilya Dryomov cc: Alex Markuze cc: Viacheslav Dubeyko cc: netfs@lists.linux.dev cc: linux-nfs@vger.kernel.org cc: ceph-devel@vger.kernel.org cc: linux-fsdevel@vger.kernel.org Signed-off-by: Christian Brauner (Amutable) --- fs/ceph/Kconfig | 1 + fs/netfs/Kconfig | 3 +++ fs/netfs/Makefile | 2 +- fs/netfs/buffered_read.c | 2 +- fs/netfs/internal.h | 28 ++++++++++++++++++++++++++++ fs/netfs/read_collect.c | 4 ++-- fs/nfs/Kconfig | 1 + include/linux/netfs.h | 2 ++ 8 files changed, 39 insertions(+), 4 deletions(-) diff --git a/fs/ceph/Kconfig b/fs/ceph/Kconfig index 3d64a316ca31d3..aa6ccd7794d233 100644 --- a/fs/ceph/Kconfig +++ b/fs/ceph/Kconfig @@ -4,6 +4,7 @@ config CEPH_FS depends on INET select CEPH_LIB select NETFS_SUPPORT + select NETFS_PGPRIV2 select FS_ENCRYPTION_ALGS if FS_ENCRYPTION default n help diff --git a/fs/netfs/Kconfig b/fs/netfs/Kconfig index 7701c037c3283f..d0e7b0971fa3f0 100644 --- a/fs/netfs/Kconfig +++ b/fs/netfs/Kconfig @@ -22,6 +22,9 @@ config NETFS_STATS between CPUs. On the other hand, the stats are very useful for debugging purposes. Saying 'Y' here is recommended. +config NETFS_PGPRIV2 + bool + config NETFS_DEBUG bool "Enable dynamic debugging netfslib and FS-Cache" depends on NETFS_SUPPORT diff --git a/fs/netfs/Makefile b/fs/netfs/Makefile index b43188d64bd87e..54834cde7e5686 100644 --- a/fs/netfs/Makefile +++ b/fs/netfs/Makefile @@ -11,7 +11,6 @@ netfs-y := \ misc.o \ objects.o \ read_collect.o \ - read_pgpriv2.o \ read_retry.o \ read_single.o \ rolling_buffer.o \ @@ -19,6 +18,7 @@ netfs-y := \ write_issue.o \ write_retry.o +netfs-$(CONFIG_NETFS_PGPRIV2) += read_pgpriv2.o netfs-$(CONFIG_NETFS_STATS) += stats.o netfs-$(CONFIG_FSCACHE) += \ diff --git a/fs/netfs/buffered_read.c b/fs/netfs/buffered_read.c index 61cf82b4b60d38..54287a8ef0f8b7 100644 --- a/fs/netfs/buffered_read.c +++ b/fs/netfs/buffered_read.c @@ -242,7 +242,7 @@ static void netfs_mark_copy_to_cache(struct netfs_io_request *rreq, if (overlap > 0 && copy) { folio = folioq_folio(*fq, *slot); - if (unlikely(test_bit(NETFS_RREQ_USE_PGPRIV2, &rreq->flags))) { + if (netfs_using_pgpriv2(rreq)) { if (!folio_test_private_2(folio)) folio_start_private_2(folio); } else { diff --git a/fs/netfs/internal.h b/fs/netfs/internal.h index a4c834e32214f7..3aebe4a4f7b0c8 100644 --- a/fs/netfs/internal.h +++ b/fs/netfs/internal.h @@ -120,9 +120,37 @@ void netfs_cache_read_terminated(void *priv, ssize_t transferred_or_error); /* * read_pgpriv2.c */ +#ifdef CONFIG_NETFS_PGPRIV2 +int netfs_prepare_pgpriv2_write_buffer(struct netfs_io_subrequest *subreq, + unsigned int max_segs); void netfs_pgpriv2_copy_to_cache(struct netfs_io_request *rreq, struct folio *folio); void netfs_pgpriv2_end_copy_to_cache(struct netfs_io_request *rreq); bool netfs_pgpriv2_unlock_copied_folios(struct netfs_io_request *wreq); +static inline bool netfs_using_pgpriv2(const struct netfs_io_request *rreq) +{ + return unlikely(test_bit(NETFS_RREQ_USE_PGPRIV2, &rreq->flags)); +} +#else +static inline int netfs_prepare_pgpriv2_write_buffer(struct netfs_io_subrequest *subreq, + unsigned int max_segs) +{ + return -EIO; +} +static inline void netfs_pgpriv2_copy_to_cache(struct netfs_io_request *rreq, struct folio *folio) +{ +} +static inline void netfs_pgpriv2_end_copy_to_cache(struct netfs_io_request *rreq) +{ +} +static inline bool netfs_pgpriv2_unlock_copied_folios(struct netfs_io_request *wreq) +{ + return true; +} +static inline bool netfs_using_pgpriv2(const struct netfs_io_request *rreq) +{ + return false; +} +#endif /* * read_retry.c diff --git a/fs/netfs/read_collect.c b/fs/netfs/read_collect.c index 61f2664de6d3a0..01ea1ddae04b39 100644 --- a/fs/netfs/read_collect.c +++ b/fs/netfs/read_collect.c @@ -38,7 +38,7 @@ static void netfs_clear_unread(struct netfs_io_subrequest *subreq) */ void netfs_cancel_copy_to_cache(struct netfs_io_request *rreq, struct folio *folio) { - if (!test_bit(NETFS_RREQ_USE_PGPRIV2, &rreq->flags)) { + if (!netfs_using_pgpriv2(rreq)) { if (folio_get_private(folio) == NETFS_FOLIO_COPY_TO_CACHE) { folio_detach_private(folio); trace_netfs_folio(folio, netfs_folio_trace_cancel_copy); @@ -81,7 +81,7 @@ static void netfs_unlock_read_folio(struct netfs_io_request *rreq, if (unlikely(test_bit(NETFS_RREQ_CANCEL_CACHING, &rreq->flags))) netfs_cancel_copy_to_cache(rreq, folio); - if (!test_bit(NETFS_RREQ_USE_PGPRIV2, &rreq->flags)) { + if (!netfs_using_pgpriv2(rreq)) { if (netfs_folio_group(folio) == NETFS_FOLIO_COPY_TO_CACHE) { trace_netfs_folio(folio, netfs_folio_trace_sched_copy); folio_mark_dirty(folio); diff --git a/fs/nfs/Kconfig b/fs/nfs/Kconfig index 6bb30543eff00f..e7862f35b72c8c 100644 --- a/fs/nfs/Kconfig +++ b/fs/nfs/Kconfig @@ -174,6 +174,7 @@ config NFS_FSCACHE bool "Provide NFS client caching support" depends on NFS_FS select NETFS_SUPPORT + select NETFS_PGPRIV2 select FSCACHE help Say Y here if you want NFS data to be cached locally on disc through diff --git a/include/linux/netfs.h b/include/linux/netfs.h index f2b3e61c1891f3..71fdd6ef43a7d2 100644 --- a/include/linux/netfs.h +++ b/include/linux/netfs.h @@ -279,8 +279,10 @@ struct netfs_io_request { #define NETFS_RREQ_UPLOAD_TO_SERVER 11 /* Need to write to the server */ #define NETFS_RREQ_USE_IO_ITER 12 /* Use ->io_iter rather than ->i_pages */ #define NETFS_RREQ_NEED_PUT_RA_REFS 17 /* Need to put the folio refs RA gave us */ +#ifdef CONFIG_NETFS_PGPRIV2 #define NETFS_RREQ_USE_PGPRIV2 31 /* [DEPRECATED] Use PG_private_2 to mark * write to cache on read */ +#endif const struct netfs_request_ops *netfs_ops; }; From 39c8461a809ef88161d9ea6bf82783714723e258 Mon Sep 17 00:00:00 2001 From: David Howells Date: Wed, 9 Sep 2026 08:21:01 +0100 Subject: [PATCH 0025/1352] netfs: Add some functions to wrap the all-queued handling Add some helper functions to wrap the handling of the NETFS_RREQ_ALL_QUEUED flag and to insert the appropriate barriers. Also add an smb_mb__after_atomic() after the set_bit() to make sure stuff after the set_bit() in the same thread doesn't get ordered before. Signed-off-by: David Howells Link: https://patch.msgid.link/20260909072105.1663687-8-dhowells@redhat.com Reviewed-by: Paulo Alcantara cc: netfs@lists.linux.dev cc: linux-fsdevel@vger.kernel.org Signed-off-by: Christian Brauner (Amutable) --- fs/netfs/buffered_read.c | 9 +++------ fs/netfs/direct_read.c | 9 +++------ fs/netfs/internal.h | 20 ++++++++++++++++++++ fs/netfs/misc.c | 2 +- fs/netfs/read_collect.c | 4 +--- fs/netfs/read_pgpriv2.c | 3 +-- fs/netfs/read_single.c | 9 +++------ fs/netfs/write_collect.c | 3 +-- fs/netfs/write_issue.c | 6 ++---- include/trace/events/netfs.h | 1 + 10 files changed, 36 insertions(+), 30 deletions(-) diff --git a/fs/netfs/buffered_read.c b/fs/netfs/buffered_read.c index 54287a8ef0f8b7..887d45f3745a34 100644 --- a/fs/netfs/buffered_read.c +++ b/fs/netfs/buffered_read.c @@ -355,10 +355,8 @@ static void netfs_read_to_pagecache(struct netfs_io_request *rreq) } start += slice; size -= slice; - if (size <= 0) { - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags); - } + if (size <= 0) + netfs_all_subreqs_queued(rreq); if (fq) { /* See if the cache indicated this should be cached. */ @@ -378,8 +376,7 @@ static void netfs_read_to_pagecache(struct netfs_io_request *rreq) } while (size > 0); if (unlikely(size > 0)) { - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags); + netfs_all_subreqs_queued(rreq); netfs_wake_collector(rreq); } diff --git a/fs/netfs/direct_read.c b/fs/netfs/direct_read.c index aa10af5171a860..5405e108b7a33e 100644 --- a/fs/netfs/direct_read.c +++ b/fs/netfs/direct_read.c @@ -84,10 +84,8 @@ static void netfs_dispatch_unbuffered_reads(struct netfs_io_request *rreq) size -= slice; start += slice; rreq->submitted += slice; - if (size <= 0) { - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags); - } + if (size <= 0) + netfs_all_subreqs_queued(rreq); rreq->netfs_ops->issue_read(subreq); @@ -99,8 +97,7 @@ static void netfs_dispatch_unbuffered_reads(struct netfs_io_request *rreq) } while (size > 0); if (unlikely(size > 0)) { - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags); + netfs_all_subreqs_queued(rreq); netfs_wake_collector(rreq); } } diff --git a/fs/netfs/internal.h b/fs/netfs/internal.h index 3aebe4a4f7b0c8..d579c5d7960947 100644 --- a/fs/netfs/internal.h +++ b/fs/netfs/internal.h @@ -341,6 +341,26 @@ static inline bool netfs_check_subreq_in_progress(const struct netfs_io_subreque return test_bit_acquire(NETFS_SREQ_IN_PROGRESS, &subreq->flags); } +/* + * Indicate that we've generated and queued all the subrequests we're going to. + */ +static inline void netfs_all_subreqs_queued(struct netfs_io_request *rreq) +{ + smp_wmb(); /* Write lists before ALL_QUEUED. */ + set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags); + smp_mb__after_atomic(); + trace_netfs_rreq(rreq, netfs_rreq_trace_all_queued); +} + +/* + * Query if all subrequests are queued. + */ +static inline bool netfs_are_all_subreqs_queued(const struct netfs_io_request *rreq) +{ + /* Read lists after ALL_QUEUED. */ + return test_bit_acquire(NETFS_RREQ_ALL_QUEUED, &rreq->flags); +} + /* * fscache-cache.c */ diff --git a/fs/netfs/misc.c b/fs/netfs/misc.c index eafc4edae6a06e..a3cd76d584b8b9 100644 --- a/fs/netfs/misc.c +++ b/fs/netfs/misc.c @@ -424,7 +424,7 @@ static int netfs_collect_in_app(struct netfs_io_request *rreq, need_collect = true; break; } - if (subreq || !test_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags)) + if (subreq || !netfs_are_all_subreqs_queued(rreq)) done = false; } diff --git a/fs/netfs/read_collect.c b/fs/netfs/read_collect.c index 01ea1ddae04b39..cb97dedad170f6 100644 --- a/fs/netfs/read_collect.c +++ b/fs/netfs/read_collect.c @@ -462,10 +462,8 @@ bool netfs_read_collection(struct netfs_io_request *rreq) /* We're done when the app thread has finished posting subreqs and the * queue is empty. */ - if (!test_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags)) + if (!netfs_are_all_subreqs_queued(rreq)) return false; - smp_rmb(); /* Read ALL_QUEUED before subreq lists. */ - if (!list_empty(&stream->subrequests)) return false; diff --git a/fs/netfs/read_pgpriv2.c b/fs/netfs/read_pgpriv2.c index bf4d9d6877d7b2..5280b606fda410 100644 --- a/fs/netfs/read_pgpriv2.c +++ b/fs/netfs/read_pgpriv2.c @@ -158,8 +158,7 @@ void netfs_pgpriv2_end_copy_to_cache(struct netfs_io_request *rreq) return; netfs_issue_write(creq, &creq->io_streams[1]); - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &creq->flags); + netfs_all_subreqs_queued(creq); trace_netfs_rreq(rreq, netfs_rreq_trace_end_copy_to_cache); if (list_empty_careful(&creq->io_streams[1].subrequests)) netfs_wake_collector(creq); diff --git a/fs/netfs/read_single.c b/fs/netfs/read_single.c index de67ac41548d1a..ccb5fc809d993f 100644 --- a/fs/netfs/read_single.c +++ b/fs/netfs/read_single.c @@ -113,14 +113,12 @@ static int netfs_single_dispatch_read(struct netfs_io_request *rreq) goto cancel; } - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags); + netfs_all_subreqs_queued(rreq); rreq->netfs_ops->issue_read(subreq); rreq->submitted += subreq->len; break; case NETFS_READ_FROM_CACHE: - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags); + netfs_all_subreqs_queued(rreq); trace_netfs_sreq(subreq, netfs_sreq_trace_submit); netfs_single_read_cache(rreq, subreq); rreq->submitted += subreq->len; @@ -136,8 +134,7 @@ static int netfs_single_dispatch_read(struct netfs_io_request *rreq) return ret; cancel: netfs_cancel_read(subreq, ret); - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags); + netfs_all_subreqs_queued(rreq); netfs_wake_collector(rreq); return ret; } diff --git a/fs/netfs/write_collect.c b/fs/netfs/write_collect.c index 7194182b975cf2..6d99d4a6f7808c 100644 --- a/fs/netfs/write_collect.c +++ b/fs/netfs/write_collect.c @@ -371,9 +371,8 @@ bool netfs_write_collection(struct netfs_io_request *wreq) /* We're done when the app thread has finished posting subreqs and all * the queues in all the streams are empty. */ - if (!test_bit(NETFS_RREQ_ALL_QUEUED, &wreq->flags)) + if (!netfs_are_all_subreqs_queued(wreq)) return false; - smp_rmb(); /* Read ALL_QUEUED before lists. */ transferred = LONG_MAX; for (s = 0; s < NR_IO_STREAMS; s++) { diff --git a/fs/netfs/write_issue.c b/fs/netfs/write_issue.c index 5d130df2ff0d06..9a528da4bf9fe5 100644 --- a/fs/netfs/write_issue.c +++ b/fs/netfs/write_issue.c @@ -520,8 +520,7 @@ static void netfs_end_issue_write(struct netfs_io_request *wreq) { bool needs_poke = true; - smp_wmb(); /* Write subreq lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &wreq->flags); + netfs_all_subreqs_queued(wreq); for (int s = 0; s < NR_IO_STREAMS; s++) { struct netfs_io_stream *stream = &wreq->io_streams[s]; @@ -789,8 +788,7 @@ int netfs_writeback_single(struct address_space *mapping, stop: for (int s = 0; s < NR_IO_STREAMS; s++) netfs_issue_write(wreq, &wreq->io_streams[s]); - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &wreq->flags); + netfs_all_subreqs_queued(wreq); netfs_wake_collector(wreq); diff --git a/include/trace/events/netfs.h b/include/trace/events/netfs.h index c9ec4da97f0bdd..303309be253f2a 100644 --- a/include/trace/events/netfs.h +++ b/include/trace/events/netfs.h @@ -47,6 +47,7 @@ E_(NETFS_PGPRIV2_COPY_TO_CACHE, "2C") #define netfs_rreq_traces \ + EM(netfs_rreq_trace_all_queued, "ALL-Q ") \ EM(netfs_rreq_trace_assess, "ASSESS ") \ EM(netfs_rreq_trace_collect, "COLLECT") \ EM(netfs_rreq_trace_complete, "COMPLET") \ From db208f7581b84fbd8c3d9daca5b8b7416f2cc517 Mon Sep 17 00:00:00 2001 From: David Howells Date: Wed, 9 Sep 2026 08:21:02 +0100 Subject: [PATCH 0026/1352] netfs: Set subrequest->source at alloc before trace emission Set subrequest->source in netfs_alloc_subrequest() before we emit the trace line indicating we allocated the subrequest. Note that this requires the allocation of the subreq in netfs_read_to_pagecache() to be pushed to after the decision about what sort of subreq it should be. Signed-off-by: David Howells Link: https://patch.msgid.link/20260909072105.1663687-9-dhowells@redhat.com Reviewed-by: Paulo Alcantara cc: Paulo Alcantara cc: netfs@lists.linux.dev cc: linux-fsdevel@vger.kernel.org cc: linux-mm@kvack.org Signed-off-by: Christian Brauner (Amutable) --- fs/netfs/buffered_read.c | 4 ++-- fs/netfs/direct_read.c | 3 +-- fs/netfs/internal.h | 3 ++- fs/netfs/objects.c | 4 +++- fs/netfs/read_retry.c | 3 +-- fs/netfs/read_single.c | 3 +-- fs/netfs/write_issue.c | 3 +-- fs/netfs/write_retry.c | 3 +-- 8 files changed, 12 insertions(+), 14 deletions(-) diff --git a/fs/netfs/buffered_read.c b/fs/netfs/buffered_read.c index 887d45f3745a34..68496e1a171f53 100644 --- a/fs/netfs/buffered_read.c +++ b/fs/netfs/buffered_read.c @@ -276,10 +276,10 @@ static void netfs_read_to_pagecache(struct netfs_io_request *rreq) do { struct netfs_io_subrequest *subreq; - enum netfs_io_source source = NETFS_SOURCE_UNKNOWN; + enum netfs_io_source source; ssize_t slice; - subreq = netfs_alloc_subrequest(rreq); + subreq = netfs_alloc_subrequest(rreq, NETFS_SOURCE_UNKNOWN); if (!subreq) { ret = -ENOMEM; break; diff --git a/fs/netfs/direct_read.c b/fs/netfs/direct_read.c index 5405e108b7a33e..8c15f307972383 100644 --- a/fs/netfs/direct_read.c +++ b/fs/netfs/direct_read.c @@ -55,7 +55,7 @@ static void netfs_dispatch_unbuffered_reads(struct netfs_io_request *rreq) struct netfs_io_subrequest *subreq; ssize_t slice; - subreq = netfs_alloc_subrequest(rreq); + subreq = netfs_alloc_subrequest(rreq, NETFS_DOWNLOAD_FROM_SERVER); if (!subreq) { /* Stash the error in the request if there's not * already an error set. @@ -64,7 +64,6 @@ static void netfs_dispatch_unbuffered_reads(struct netfs_io_request *rreq) break; } - subreq->source = NETFS_DOWNLOAD_FROM_SERVER; subreq->start = start; subreq->len = size; diff --git a/fs/netfs/internal.h b/fs/netfs/internal.h index d579c5d7960947..4891e6e3c5db68 100644 --- a/fs/netfs/internal.h +++ b/fs/netfs/internal.h @@ -92,7 +92,8 @@ void netfs_get_request(struct netfs_io_request *rreq, enum netfs_rreq_ref_trace void netfs_clear_subrequests(struct netfs_io_request *rreq); void netfs_put_request(struct netfs_io_request *rreq, enum netfs_rreq_ref_trace what); void netfs_put_failed_request(struct netfs_io_request *rreq); -struct netfs_io_subrequest *netfs_alloc_subrequest(struct netfs_io_request *rreq); +struct netfs_io_subrequest *netfs_alloc_subrequest(struct netfs_io_request *rreq, + enum netfs_io_source source); static inline void netfs_see_request(struct netfs_io_request *rreq, enum netfs_rreq_ref_trace what) diff --git a/fs/netfs/objects.c b/fs/netfs/objects.c index 3460aa1c4af1e7..9c6ea718692a7d 100644 --- a/fs/netfs/objects.c +++ b/fs/netfs/objects.c @@ -207,7 +207,8 @@ void netfs_put_failed_request(struct netfs_io_request *rreq) /* * Allocate and partially initialise an I/O request structure. */ -struct netfs_io_subrequest *netfs_alloc_subrequest(struct netfs_io_request *rreq) +struct netfs_io_subrequest *netfs_alloc_subrequest(struct netfs_io_request *rreq, + enum netfs_io_source source) { struct netfs_io_subrequest *subreq; mempool_t *mempool = rreq->netfs_ops->subrequest_pool ?: &netfs_subrequest_pool; @@ -224,6 +225,7 @@ struct netfs_io_subrequest *netfs_alloc_subrequest(struct netfs_io_request *rreq INIT_WORK(&subreq->work, NULL); INIT_LIST_HEAD(&subreq->rreq_link); refcount_set(&subreq->ref, 2); + subreq->source = source; subreq->rreq = rreq; subreq->debug_index = atomic_inc_return(&rreq->subreq_counter); netfs_get_request(rreq, netfs_rreq_trace_get_subreq); diff --git a/fs/netfs/read_retry.c b/fs/netfs/read_retry.c index 46810d29f0c0e3..396a05432a7e92 100644 --- a/fs/netfs/read_retry.c +++ b/fs/netfs/read_retry.c @@ -195,12 +195,11 @@ static void netfs_retry_read_subrequests(struct netfs_io_request *rreq) * and insert them after. */ do { - subreq = netfs_alloc_subrequest(rreq); + subreq = netfs_alloc_subrequest(rreq, NETFS_DOWNLOAD_FROM_SERVER); if (!subreq) { subreq = to; goto abandon_after; } - subreq->source = NETFS_DOWNLOAD_FROM_SERVER; subreq->start = start; subreq->len = len; subreq->stream_nr = stream->stream_nr; diff --git a/fs/netfs/read_single.c b/fs/netfs/read_single.c index ccb5fc809d993f..81295c055cedbc 100644 --- a/fs/netfs/read_single.c +++ b/fs/netfs/read_single.c @@ -92,11 +92,10 @@ static int netfs_single_dispatch_read(struct netfs_io_request *rreq) struct netfs_io_subrequest *subreq; int ret = 0; - subreq = netfs_alloc_subrequest(rreq); + subreq = netfs_alloc_subrequest(rreq, NETFS_SOURCE_UNKNOWN); if (!subreq) return -ENOMEM; - subreq->source = NETFS_SOURCE_UNKNOWN; subreq->start = 0; subreq->len = rreq->len; subreq->io_iter = rreq->buffer.iter; diff --git a/fs/netfs/write_issue.c b/fs/netfs/write_issue.c index 9a528da4bf9fe5..e7cb496f63246a 100644 --- a/fs/netfs/write_issue.c +++ b/fs/netfs/write_issue.c @@ -168,10 +168,9 @@ void netfs_prepare_write(struct netfs_io_request *wreq, wreq_iter->folioq_slot >= folioq_nr_slots(wreq_iter->folioq)) rolling_buffer_make_space(&wreq->buffer, wreq->gfp); - subreq = netfs_alloc_subrequest(wreq); + subreq = netfs_alloc_subrequest(wreq, stream->source); if (!subreq) return; - subreq->source = stream->source; subreq->start = start; subreq->stream_nr = stream->stream_nr; subreq->io_iter = *wreq_iter; diff --git a/fs/netfs/write_retry.c b/fs/netfs/write_retry.c index 6cd584242af282..d7d5348496fc8d 100644 --- a/fs/netfs/write_retry.c +++ b/fs/netfs/write_retry.c @@ -149,8 +149,7 @@ static void netfs_retry_write_stream(struct netfs_io_request *wreq, * and insert them after. */ do { - subreq = netfs_alloc_subrequest(wreq); - subreq->source = to->source; + subreq = netfs_alloc_subrequest(wreq, stream->source); subreq->start = start; subreq->stream_nr = to->stream_nr; subreq->retry_count = 1; From 8dbe5c47ca4254be32e8a7bed16ef911de3101fe Mon Sep 17 00:00:00 2001 From: Uma Shankar Date: Wed, 9 Sep 2026 13:38:07 +0530 Subject: [PATCH 0027/1352] drm/i915/display: Enable periodic AS SDP skip frames When Panel Replay is active the transcoder timing generator runs at the panel's maximum refresh rate. To drive the panel down to its minimum refresh rate the Adaptive-Sync SDP (AS SDP) only needs to reach the panel once per minimum-rate frame, so transmitting it on every (maximum-rate) frame is redundant and shows up as repeated SDPs on the link. Xe3p_LPD adds a HW skip-frame counter in PR_ALPM_CTL that lets the source send a single AS SDP and then suppress it for a programmed number of frames. Program this counter so that one AS SDP is followed by (max_vrefresh / min_vrefresh - 1) idle frames, i.e. one AS SDP per slowest panel frame, allowing the link to be driven down to as low as 1Hz when the hardware supports it. The maximum and minimum refresh rates come from the panel's adaptive-sync monitor range, so the skip count is a function of the sink's capabilities and independent of the current content/flip rate. If the panel does not advertise a usable range the skip counter is left at zero, i.e. the feature is a no-op and AS SDP continues to be sent on every frame. Periodic AS SDP drives the panel down to its minimum refresh rate on its own, so it is only programmed when VRR is not actively driving the refresh rate. The skip-frame mechanism relies on the AS SDP still being transmitted (just less often) while Panel Replay is active, so when a non-zero skip-frame count is programmed both PR_ALPM_CTL_AS_SDP_TRANSMISSION_IN_ACTIVE_DISABLE and PR_ALPM_CTL_USE_DC3CO_IDLE_PROTOCOL are left cleared. The previous behaviour (honouring disable_as_sdp_when_pr_active and the DC3CO idle protocol) is retained for the non skip-frame case. v3: Fixed Sashiko review findings v2: Decoupled CMMRR dependency and using sink refresh rate range for skip frame claculations. This addresses Dibin's review feedback as well. Assisted-by: Claude:claude-opus-4-8 Signed-off-by: Uma Shankar Reviewed-by: Dibin Moolakadan Subrahmanian Tested-by: Naladala Ramanaidu Link: https://patch.msgid.link/20260909080810.2202879-2-uma.shankar@intel.com --- drivers/gpu/drm/i915/display/intel_alpm.c | 61 +++++++++++++++++-- drivers/gpu/drm/i915/display/intel_psr_regs.h | 2 + 2 files changed, 58 insertions(+), 5 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_alpm.c b/drivers/gpu/drm/i915/display/intel_alpm.c index f1383764b7028f..a6838e0fd3aa08 100644 --- a/drivers/gpu/drm/i915/display/intel_alpm.c +++ b/drivers/gpu/drm/i915/display/intel_alpm.c @@ -375,6 +375,35 @@ static u32 get_pr_alpm_as_sdp_transmission_time(const struct intel_crtc_state *c } } +/* + * Periodic Adaptive-Sync SDP skip frames. + * + * While Panel Replay is active the transcoder timing generator runs at the + * panel's maximum refresh rate. To drive the panel down to its minimum + * refresh rate the Adaptive-Sync SDP only needs to reach the panel once per + * minimum-rate frame, so transmitting it on every (maximum-rate) frame is + * unnecessary and shows up as repeated SDPs on the link. Program the HW skip + * counter so that a single AS SDP is followed by + * (max_vrefresh / min_vrefresh - 1) idle frames, i.e. one AS SDP per slowest + * panel frame. + * + * The maximum and minimum refresh rates come from the panel's adaptive-sync + * monitor range, so this is independent of the current content/flip rate. + */ +static u32 intel_pr_as_sdp_skip_frames(struct intel_dp *intel_dp) +{ + const struct drm_display_info *info = + &intel_dp->attached_connector->base.display_info; + int max_vrefresh = info->monitor_range.max_vfreq; + int min_vrefresh = info->monitor_range.min_vfreq; + + if (min_vrefresh <= 0 || max_vrefresh <= min_vrefresh) + return 0; + + return min_t(u32, max_vrefresh / min_vrefresh - 1, + REG_FIELD_MAX(PR_ALPM_CTL_AS_SDP_SKIP_FRAMES_MASK)); +} + static void lnl_alpm_configure(struct intel_dp *intel_dp, const struct intel_crtc_state *crtc_state) { @@ -399,16 +428,38 @@ static void lnl_alpm_configure(struct intel_dp *intel_dp, if (intel_dp->as_sdp_supported) { u32 pr_alpm_ctl = get_pr_alpm_as_sdp_transmission_time(crtc_state); + u32 skip_frames = 0; + + /* + * AS SDP skip frames field only exists on Xe3p_LPD+, and + * periodic AS SDP is a Panel Replay feature that is only + * used when VRR is not actively driving the refresh rate. + */ + if (DISPLAY_VER(display) >= 35 && crtc_state->has_panel_replay && + !crtc_state->vrr.enable) + skip_frames = intel_pr_as_sdp_skip_frames(intel_dp); if (crtc_state->link_off_after_as_sdp_when_pr_active) pr_alpm_ctl |= PR_ALPM_CTL_ALLOW_LINK_OFF_BETWEEN_AS_SDP_AND_SU; - if (crtc_state->disable_as_sdp_when_pr_active) - pr_alpm_ctl |= PR_ALPM_CTL_AS_SDP_TRANSMISSION_IN_ACTIVE_DISABLE; - if (intel_display_power_dc3co_allowed(display)) - pr_alpm_ctl |= PR_ALPM_CTL_USE_DC3CO_IDLE_PROTOCOL; - else + /* + * Skip frames needs the AS SDP to keep flowing during PR + * active, so it is mutually exclusive with disabling AS SDP + * transmission in active and with the DC3CO idle protocol. + */ + if (skip_frames) { + pr_alpm_ctl |= PR_ALPM_CTL_AS_SDP_SKIP_FRAMES(skip_frames); + pr_alpm_ctl &= ~PR_ALPM_CTL_AS_SDP_TRANSMISSION_IN_ACTIVE_DISABLE; pr_alpm_ctl &= ~PR_ALPM_CTL_USE_DC3CO_IDLE_PROTOCOL; + } else { + pr_alpm_ctl &= ~PR_ALPM_CTL_AS_SDP_SKIP_FRAMES_MASK; + + if (crtc_state->disable_as_sdp_when_pr_active) + pr_alpm_ctl |= PR_ALPM_CTL_AS_SDP_TRANSMISSION_IN_ACTIVE_DISABLE; + + if (intel_display_power_dc3co_allowed(display)) + pr_alpm_ctl |= PR_ALPM_CTL_USE_DC3CO_IDLE_PROTOCOL; + } intel_de_write(display, PR_ALPM_CTL(display, cpu_transcoder), pr_alpm_ctl); diff --git a/drivers/gpu/drm/i915/display/intel_psr_regs.h b/drivers/gpu/drm/i915/display/intel_psr_regs.h index 16a9e3af198d07..bb577e7e3bbda4 100644 --- a/drivers/gpu/drm/i915/display/intel_psr_regs.h +++ b/drivers/gpu/drm/i915/display/intel_psr_regs.h @@ -276,6 +276,8 @@ #define PR_ALPM_CTL_ADAPTIVE_SYNC_SDP_POSITION_T1_OR_T2 REG_FIELD_PREP(PR_ALPM_CTL_ADAPTIVE_SYNC_SDP_POSITION_MASK, 0) #define PR_ALPM_CTL_ADAPTIVE_SYNC_SDP_POSITION_T1 REG_FIELD_PREP(PR_ALPM_CTL_ADAPTIVE_SYNC_SDP_POSITION_MASK, 1) #define PR_ALPM_CTL_ADAPTIVE_SYNC_SDP_POSITION_T2 REG_FIELD_PREP(PR_ALPM_CTL_ADAPTIVE_SYNC_SDP_POSITION_MASK, 2) +#define PR_ALPM_CTL_AS_SDP_SKIP_FRAMES_MASK REG_GENMASK(27, 16) +#define PR_ALPM_CTL_AS_SDP_SKIP_FRAMES(frames) REG_FIELD_PREP(PR_ALPM_CTL_AS_SDP_SKIP_FRAMES_MASK, (frames)) #define _ALPM_CTL_A 0x60950 #define ALPM_CTL(dev_priv, tran) _MMIO_TRANS2(dev_priv, tran, _ALPM_CTL_A) From dc0fd724e13a313b64773200f84137bf81167a9c Mon Sep 17 00:00:00 2001 From: Uma Shankar Date: Wed, 9 Sep 2026 13:38:08 +0530 Subject: [PATCH 0028/1352] drm/i915/display: Force disable DC3co when AS SDP skip frames is enabled Periodic AS SDP (skip frames) relies on the AS SDP still being transmitted while Panel Replay is active. DC3co uses the idle protocol which suppresses AS SDP transmission entirely, so the two are mutually exclusive: leaving DC3co enabled while skip frames is programmed breaks the periodic AS SDP and the panel never sees the slower refresh. Add intel_alpm_pr_as_sdp_skip_frames_enabled() as the single predicate for "skip frames will be programmed" (mirroring the gating in lnl_alpm_configure(), including that it only applies when VRR is not active) and use it in intel_display_power_dc3co_compute() to force the DC3co trigger to NONE. This drops the pipe onto the DC_STATE_EN_UPTO_DC6 target instead of DC3co whenever skip frames is active, without touching the DC state module parameter or the allowed DC mask, and only for the skip-frame case. v2: Fixed Sashiko review findings Assisted-by: Claude:claude-opus-4-8 Signed-off-by: Uma Shankar Reviewed-by: Dibin Moolakadan Subrahmanian Tested-by: Naladala Ramanaidu Link: https://patch.msgid.link/20260909080810.2202879-3-uma.shankar@intel.com --- drivers/gpu/drm/i915/display/intel_alpm.c | 23 +++++++++++++++++++ drivers/gpu/drm/i915/display/intel_alpm.h | 2 ++ .../drm/i915/display/intel_display_power.c | 9 ++++++++ 3 files changed, 34 insertions(+) diff --git a/drivers/gpu/drm/i915/display/intel_alpm.c b/drivers/gpu/drm/i915/display/intel_alpm.c index a6838e0fd3aa08..0a33a89975bc63 100644 --- a/drivers/gpu/drm/i915/display/intel_alpm.c +++ b/drivers/gpu/drm/i915/display/intel_alpm.c @@ -404,6 +404,29 @@ static u32 intel_pr_as_sdp_skip_frames(struct intel_dp *intel_dp) REG_FIELD_MAX(PR_ALPM_CTL_AS_SDP_SKIP_FRAMES_MASK)); } +/* + * Whether periodic AS SDP transmission (AS SDP skip frames) will be programmed + * for this Panel Replay config. The skip counter needs the AS SDP to keep + * flowing during PR active, which is incompatible with DC3co, so this is used + * to keep DC3co disabled while skip frames is enabled. + */ +bool intel_alpm_pr_as_sdp_skip_frames_enabled(struct intel_dp *intel_dp, + const struct intel_crtc_state *crtc_state) +{ + struct intel_display *display = to_intel_display(intel_dp); + + /* + * The AS SDP skip frames field only exists on Xe3p_LPD+. Periodic AS SDP + * drives the panel down to its minimum refresh rate on its own, so it is + * only used when VRR is not actively driving the refresh rate. + */ + if (DISPLAY_VER(display) < 35 || !intel_dp->as_sdp_supported || + !crtc_state->has_panel_replay || crtc_state->vrr.enable) + return false; + + return intel_pr_as_sdp_skip_frames(intel_dp) > 0; +} + static void lnl_alpm_configure(struct intel_dp *intel_dp, const struct intel_crtc_state *crtc_state) { diff --git a/drivers/gpu/drm/i915/display/intel_alpm.h b/drivers/gpu/drm/i915/display/intel_alpm.h index 1cf70668ab1ba2..328920027f1c0a 100644 --- a/drivers/gpu/drm/i915/display/intel_alpm.h +++ b/drivers/gpu/drm/i915/display/intel_alpm.h @@ -34,6 +34,8 @@ bool intel_alpm_aux_wake_supported(struct intel_dp *intel_dp); bool intel_alpm_aux_less_wake_supported(struct intel_dp *intel_dp); bool intel_alpm_is_alpm_aux_less(struct intel_dp *intel_dp, const struct intel_crtc_state *crtc_state); +bool intel_alpm_pr_as_sdp_skip_frames_enabled(struct intel_dp *intel_dp, + const struct intel_crtc_state *crtc_state); void intel_alpm_disable(struct intel_dp *intel_dp); bool intel_alpm_get_error(struct intel_dp *intel_dp); void intel_alpm_lobf_compute_config_late(struct intel_dp *intel_dp, diff --git a/drivers/gpu/drm/i915/display/intel_display_power.c b/drivers/gpu/drm/i915/display/intel_display_power.c index 0ebec6e0c24004..1b60ce2dd00c4e 100644 --- a/drivers/gpu/drm/i915/display/intel_display_power.c +++ b/drivers/gpu/drm/i915/display/intel_display_power.c @@ -10,6 +10,7 @@ #include #include +#include "intel_alpm.h" #include "intel_backlight_regs.h" #include "intel_cdclk.h" #include "intel_clock_gating.h" @@ -488,6 +489,14 @@ void intel_display_power_dc3co_compute(struct intel_atomic_state *state) if (crtc_state->has_sel_update) trigger |= DC3CO_TRIGGER_PSR2; + /* + * Periodic AS SDP (skip frames) needs the AS SDP to keep flowing during + * PR active, which is incompatible with DC3co. Keep DC3co disabled while + * skip frames is enabled. + */ + if (intel_alpm_pr_as_sdp_skip_frames_enabled(intel_dp, crtc_state)) + trigger = DC3CO_TRIGGER_NONE; + done: intel_display_power_dc3co_update(display, trigger); } From eddf77ff1a237f394364259108ff38d21be938cb Mon Sep 17 00:00:00 2001 From: Uma Shankar Date: Wed, 9 Sep 2026 13:38:09 +0530 Subject: [PATCH 0029/1352] drm/i915/display: Reprogram AS SDP skip frames on seamless VRR transitions The AS SDP skip-frame count is written to PR_ALPM_CTL only from lnl_alpm_configure(), which runs from intel_psr_enable_locked() on a Panel Replay disabled->enabled transition. VRR, however, can be enabled and disabled seamlessly - without a modeset and without cycling Panel Replay (intel_crtc_vrr_enabling()/disabling() in the pipe update path). As a result, when a panel comes up with VRR off the non-zero skip count is programmed, and when VRR is later turned on seamlessly PR stays enabled, lnl_alpm_configure() is not re-invoked, and the stale skip count is left in the register. This also leaves the coupled AS SDP transmission / DC3CO idle-protocol bits inconsistent with the DC3co state, which is recomputed on every commit. Factor the PR_ALPM_CTL AS SDP programming out of lnl_alpm_configure() into intel_alpm_configure_pr_as_sdp() and expose intel_alpm_pr_as_sdp_update(), which recomputes those fields for the current VRR state. Call it from the seamless VRR enable and disable sites (non-modeset only; a modeset re-runs PR enable anyway) so the skip counter always matches whether VRR is actively driving the refresh rate. v2: Fixed Sashiko review comments Assisted-by: Claude:claude-opus-4-8 Signed-off-by: Uma Shankar Reviewed-by: Dibin Moolakadan Subrahmanian Tested-by: Naladala Ramanaidu Link: https://patch.msgid.link/20260909080810.2202879-4-uma.shankar@intel.com --- drivers/gpu/drm/i915/display/intel_alpm.c | 119 +++++++++++++------ drivers/gpu/drm/i915/display/intel_alpm.h | 1 + drivers/gpu/drm/i915/display/intel_display.c | 18 +++ 3 files changed, 100 insertions(+), 38 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_alpm.c b/drivers/gpu/drm/i915/display/intel_alpm.c index 0a33a89975bc63..c784e77f610b89 100644 --- a/drivers/gpu/drm/i915/display/intel_alpm.c +++ b/drivers/gpu/drm/i915/display/intel_alpm.c @@ -427,6 +427,85 @@ bool intel_alpm_pr_as_sdp_skip_frames_enabled(struct intel_dp *intel_dp, return intel_pr_as_sdp_skip_frames(intel_dp) > 0; } +/* + * Program the AS SDP portion of PR_ALPM_CTL: the transmission position, the + * skip-frame counter and the coupled AS SDP transmission / DC3CO idle-protocol + * bits. This is a full recompute of those fields (the base value is built from + * scratch, not read back), so it can be called both at PR enable time and when + * VRR is toggled seamlessly - which changes whether periodic AS SDP is used. + * + * Caller must hold intel_dp->alpm.lock. + */ +static void intel_alpm_configure_pr_as_sdp(struct intel_dp *intel_dp, + const struct intel_crtc_state *crtc_state) +{ + struct intel_display *display = to_intel_display(intel_dp); + enum transcoder cpu_transcoder = crtc_state->cpu_transcoder; + u32 pr_alpm_ctl = get_pr_alpm_as_sdp_transmission_time(crtc_state); + u32 skip_frames = 0; + + if (intel_alpm_pr_as_sdp_skip_frames_enabled(intel_dp, crtc_state)) + skip_frames = intel_pr_as_sdp_skip_frames(intel_dp); + + if (crtc_state->link_off_after_as_sdp_when_pr_active) + pr_alpm_ctl |= PR_ALPM_CTL_ALLOW_LINK_OFF_BETWEEN_AS_SDP_AND_SU; + + /* + * Skip frames needs the AS SDP to keep flowing during PR active, so it + * is mutually exclusive with disabling AS SDP transmission in active and + * with the DC3CO idle protocol. + */ + if (skip_frames) { + pr_alpm_ctl |= PR_ALPM_CTL_AS_SDP_SKIP_FRAMES(skip_frames); + pr_alpm_ctl &= ~PR_ALPM_CTL_AS_SDP_TRANSMISSION_IN_ACTIVE_DISABLE; + pr_alpm_ctl &= ~PR_ALPM_CTL_USE_DC3CO_IDLE_PROTOCOL; + } else { + pr_alpm_ctl &= ~PR_ALPM_CTL_AS_SDP_SKIP_FRAMES_MASK; + + if (crtc_state->disable_as_sdp_when_pr_active) + pr_alpm_ctl |= PR_ALPM_CTL_AS_SDP_TRANSMISSION_IN_ACTIVE_DISABLE; + + if (intel_display_power_dc3co_allowed(display)) + pr_alpm_ctl |= PR_ALPM_CTL_USE_DC3CO_IDLE_PROTOCOL; + } + + intel_de_write(display, PR_ALPM_CTL(display, cpu_transcoder), pr_alpm_ctl); +} + +/* + * VRR can be enabled or disabled seamlessly, i.e. without a modeset and without + * cycling Panel Replay, so the AS SDP skip-frame programming done at PR enable + * time would otherwise go stale across a VRR toggle. Reprogram it here so the + * skip counter (and the coupled bits) matches the new VRR state. + */ +void intel_alpm_pr_as_sdp_update(const struct intel_crtc_state *crtc_state) +{ + struct intel_display *display = to_intel_display(crtc_state); + struct intel_encoder *encoder; + + /* AS SDP skip frames field only exists on Xe3p_LPD+ */ + if (DISPLAY_VER(display) < 35) + return; + + for_each_intel_encoder_mask(display->drm, encoder, + crtc_state->uapi.encoder_mask) { + struct intel_dp *intel_dp; + + if (!intel_encoder_is_dp(encoder)) + continue; + + intel_dp = enc_to_intel_dp(encoder); + + if (!intel_dp->as_sdp_supported || + !intel_alpm_is_alpm_aux_less(intel_dp, crtc_state)) + continue; + + mutex_lock(&intel_dp->alpm.lock); + intel_alpm_configure_pr_as_sdp(intel_dp, crtc_state); + mutex_unlock(&intel_dp->alpm.lock); + } +} + static void lnl_alpm_configure(struct intel_dp *intel_dp, const struct intel_crtc_state *crtc_state) { @@ -449,44 +528,8 @@ static void lnl_alpm_configure(struct intel_dp *intel_dp, ALPM_CTL_AUX_LESS_SLEEP_HOLD_TIME_50_SYMBOLS | ALPM_CTL_AUX_LESS_WAKE_TIME(crtc_state->alpm_state.aux_less_wake_lines); - if (intel_dp->as_sdp_supported) { - u32 pr_alpm_ctl = get_pr_alpm_as_sdp_transmission_time(crtc_state); - u32 skip_frames = 0; - - /* - * AS SDP skip frames field only exists on Xe3p_LPD+, and - * periodic AS SDP is a Panel Replay feature that is only - * used when VRR is not actively driving the refresh rate. - */ - if (DISPLAY_VER(display) >= 35 && crtc_state->has_panel_replay && - !crtc_state->vrr.enable) - skip_frames = intel_pr_as_sdp_skip_frames(intel_dp); - - if (crtc_state->link_off_after_as_sdp_when_pr_active) - pr_alpm_ctl |= PR_ALPM_CTL_ALLOW_LINK_OFF_BETWEEN_AS_SDP_AND_SU; - - /* - * Skip frames needs the AS SDP to keep flowing during PR - * active, so it is mutually exclusive with disabling AS SDP - * transmission in active and with the DC3CO idle protocol. - */ - if (skip_frames) { - pr_alpm_ctl |= PR_ALPM_CTL_AS_SDP_SKIP_FRAMES(skip_frames); - pr_alpm_ctl &= ~PR_ALPM_CTL_AS_SDP_TRANSMISSION_IN_ACTIVE_DISABLE; - pr_alpm_ctl &= ~PR_ALPM_CTL_USE_DC3CO_IDLE_PROTOCOL; - } else { - pr_alpm_ctl &= ~PR_ALPM_CTL_AS_SDP_SKIP_FRAMES_MASK; - - if (crtc_state->disable_as_sdp_when_pr_active) - pr_alpm_ctl |= PR_ALPM_CTL_AS_SDP_TRANSMISSION_IN_ACTIVE_DISABLE; - - if (intel_display_power_dc3co_allowed(display)) - pr_alpm_ctl |= PR_ALPM_CTL_USE_DC3CO_IDLE_PROTOCOL; - } - - intel_de_write(display, PR_ALPM_CTL(display, cpu_transcoder), - pr_alpm_ctl); - } + if (intel_dp->as_sdp_supported) + intel_alpm_configure_pr_as_sdp(intel_dp, crtc_state); } else { alpm_ctl = ALPM_CTL_EXTENDED_FAST_WAKE_ENABLE | diff --git a/drivers/gpu/drm/i915/display/intel_alpm.h b/drivers/gpu/drm/i915/display/intel_alpm.h index 328920027f1c0a..f8f605d94f96d8 100644 --- a/drivers/gpu/drm/i915/display/intel_alpm.h +++ b/drivers/gpu/drm/i915/display/intel_alpm.h @@ -36,6 +36,7 @@ bool intel_alpm_is_alpm_aux_less(struct intel_dp *intel_dp, const struct intel_crtc_state *crtc_state); bool intel_alpm_pr_as_sdp_skip_frames_enabled(struct intel_dp *intel_dp, const struct intel_crtc_state *crtc_state); +void intel_alpm_pr_as_sdp_update(const struct intel_crtc_state *crtc_state); void intel_alpm_disable(struct intel_dp *intel_dp); bool intel_alpm_get_error(struct intel_dp *intel_dp); void intel_alpm_lobf_compute_config_late(struct intel_dp *intel_dp, diff --git a/drivers/gpu/drm/i915/display/intel_display.c b/drivers/gpu/drm/i915/display/intel_display.c index fc30a455bed365..9151ea6c15ab84 100644 --- a/drivers/gpu/drm/i915/display/intel_display.c +++ b/drivers/gpu/drm/i915/display/intel_display.c @@ -1232,6 +1232,14 @@ static void intel_pre_plane_update(struct intel_atomic_state *state, intel_vrr_disable(old_crtc_state); intel_vrr_dcb_reset(old_crtc_state, crtc); intel_crtc_update_active_timings(old_crtc_state, false); + + /* + * VRR is being disabled seamlessly (no modeset, Panel Replay + * stays enabled), so re-apply the AS SDP skip-frame programming + * for the new (VRR off) state. + */ + if (!intel_crtc_needs_modeset(new_crtc_state)) + intel_alpm_pr_as_sdp_update(new_crtc_state); } if (audio_disabling(old_crtc_state, new_crtc_state)) @@ -6971,6 +6979,16 @@ static void intel_update_crtc(struct intel_atomic_state *state, intel_crtc_update_active_timings(new_crtc_state, new_crtc_state->vrr.enable); + /* + * VRR is being enabled seamlessly (no modeset, Panel Replay stays + * enabled), so re-apply the AS SDP skip-frame programming for the new + * (VRR on) state. Done here, outside the vblank-evasion critical section + * (which runs with interrupts disabled), because it takes alpm.lock. + */ + if (intel_crtc_vrr_enabling(state, crtc) && + !intel_crtc_needs_modeset(new_crtc_state)) + intel_alpm_pr_as_sdp_update(new_crtc_state); + if (new_crtc_state->vrr.dc_balance.enable) intel_vrr_dcb_increment_flip_count(new_crtc_state, crtc); From 9ed39b7d74b43825dbe3932d41d75294b0295064 Mon Sep 17 00:00:00 2001 From: Uma Shankar Date: Wed, 9 Sep 2026 13:38:10 +0530 Subject: [PATCH 0030/1352] drm/i915/display: Gate periodic AS SDP skip frames behind a debugfs knob Periodic AS SDP (skip frames) drives a Panel Replay panel down toward its minimum refresh rate. It is a new, panel- and platform-sensitive behaviour, so keep it opt-in rather than enabling it unconditionally. Expose it as a per-device debugfs knob, enable_periodic_assdp, rather than a module parameter. The behaviour is panel-specific, so the correct granularity is per-device, not per-module. The knob is added through the intel_display_params infrastructure (shared by i915 and xe) with a debugfs entry only. It defaults to false (feature disabled); write 1 to the debugfs file to enable periodic AS SDP at runtime. Gate the feature at its single choke point, intel_pr_as_sdp_skip_frames(): returning a zero skip count when the knob is off makes both the PR_ALPM_CTL programming (intel_alpm_configure_pr_as_sdp()) and the DC3co force-disable predicate (intel_alpm_pr_as_sdp_skip_frames_enabled()) a no-op, so AS SDP continues to be sent on every frame as before. v2: Switch to debugfs entry and drop module param Assisted-by: Claude:claude-opus-4-8 Signed-off-by: Uma Shankar Reviewed-by: Dibin Moolakadan Subrahmanian Tested-by: Naladala Ramanaidu Link: https://patch.msgid.link/20260909080810.2202879-5-uma.shankar@intel.com --- drivers/gpu/drm/i915/display/intel_alpm.c | 5 +++++ drivers/gpu/drm/i915/display/intel_display_params.h | 5 +++++ 2 files changed, 10 insertions(+) diff --git a/drivers/gpu/drm/i915/display/intel_alpm.c b/drivers/gpu/drm/i915/display/intel_alpm.c index c784e77f610b89..10943539bc7c0b 100644 --- a/drivers/gpu/drm/i915/display/intel_alpm.c +++ b/drivers/gpu/drm/i915/display/intel_alpm.c @@ -392,11 +392,16 @@ static u32 get_pr_alpm_as_sdp_transmission_time(const struct intel_crtc_state *c */ static u32 intel_pr_as_sdp_skip_frames(struct intel_dp *intel_dp) { + struct intel_display *display = to_intel_display(intel_dp); const struct drm_display_info *info = &intel_dp->attached_connector->base.display_info; int max_vrefresh = info->monitor_range.max_vfreq; int min_vrefresh = info->monitor_range.min_vfreq; + /* Off by default; gated by the enable_periodic_assdp debugfs knob. */ + if (!display->params.enable_periodic_assdp) + return 0; + if (min_vrefresh <= 0 || max_vrefresh <= min_vrefresh) return 0; diff --git a/drivers/gpu/drm/i915/display/intel_display_params.h b/drivers/gpu/drm/i915/display/intel_display_params.h index b95ecf728daab7..ba01aeaf894407 100644 --- a/drivers/gpu/drm/i915/display/intel_display_params.h +++ b/drivers/gpu/drm/i915/display/intel_display_params.h @@ -50,6 +50,11 @@ struct drm_printer; param(bool, psr_safest_params, false, 0400) \ param(bool, enable_psr2_sel_fetch, true, 0400) \ param(int, enable_dmc_wl, -1, 0400) \ + /* + * Debugfs-only knob (per-device): no matching module_param is registered + * in intel_display_params.c on purpose. Runtime-toggle via debugfs. + */ \ + param(bool, enable_periodic_assdp, false, 0600) \ #define MEMBER(T, member, ...) T member; struct intel_display_params { From fe05cb9b9fb0ecc10409c4c6133257214b6cd8c8 Mon Sep 17 00:00:00 2001 From: Luca Coelho Date: Tue, 8 Sep 2026 13:06:51 +0300 Subject: [PATCH 0031/1352] drm/i915/display: check configuration index before shifting The calc_allowed_config_filter() function passes the return value of iter_pos_to_idx() directly to BIT(), but the helper can return -1 for an invalid iterator. The iterator already rejects negative indices before doing a configuration, so this should not matter in normal flows. In any case, for robustness, check the index explicitly and warn if it is negative, avoiding an undefined shift. Fixes: 39e30bdf2f92 ("drm/i915/dp_link_caps: Add link configuration iterator") Reviewed-by: Imre Deak Link: https://patch.msgid.link/20260908100659.113555-1-luciano.coelho@intel.com Signed-off-by: Luca Coelho --- drivers/gpu/drm/i915/display/intel_dp_link_caps.c | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/drivers/gpu/drm/i915/display/intel_dp_link_caps.c b/drivers/gpu/drm/i915/display/intel_dp_link_caps.c index 7b6cc6055da82a..98657aa4d3d580 100644 --- a/drivers/gpu/drm/i915/display/intel_dp_link_caps.c +++ b/drivers/gpu/drm/i915/display/intel_dp_link_caps.c @@ -426,12 +426,15 @@ calc_allowed_config_filter(struct intel_dp_link_caps *link_caps, const struct intel_dp_link_config *forced_params) { struct intel_dp_link_caps_filter allowed_configs = INTEL_DP_LINK_CAPS_FILTER_NONE; + struct intel_display *display = to_intel_display(link_caps->dp); struct intel_dp_link_caps_order order = bw_desc_config_order(); struct intel_dp_link_caps_iter iter; struct intel_dp_link_config config; iter_start(&iter, link_caps, order, enabled_configs); for_each_dp_link_config(&iter, &config) { + int config_idx; + if (forced_params->rate && forced_params->rate != config.rate) continue; @@ -446,7 +449,11 @@ calc_allowed_config_filter(struct intel_dp_link_caps *link_caps, if (config.lane_count > max_limits->lane_count) continue; - allowed_configs.config_mask |= BIT(iter_pos_to_idx(link_caps, order, iter.pos)); + config_idx = iter_pos_to_idx(link_caps, order, iter.pos); + if (drm_WARN_ON(display->drm, config_idx < 0)) + continue; + + allowed_configs.config_mask |= BIT(config_idx); } intel_dp_link_caps_iter_end(&iter); From 3aef9c94c02b3e66cfacd70492b6604947010d35 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ville=20Syrj=C3=A4l=C3=A4?= Date: Thu, 3 Sep 2026 16:01:16 +0300 Subject: [PATCH 0032/1352] drm/i915: Perform full wedge on display reset MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit We lost the proper display reset deadlock handling in commit d59cf7bb73f3 ("drm/i915/display: Use dma_fence interfaces instead of i915_sw_fence"). Currently the only thing that eventually breaks the deadlock is the 10 second fence timeout, which is very slow. I tried to essentially restore the previous mechanism via a custom dma_fence in https://lore.kernel.org/intel-gfx/20260408233458.22666-6-ville.syrjala@linux.intel.com/ but Christian didn't want it. The ideal solution would be to allow the reset time modesets to proceed ahead of any already queued atomic commits, but that is quite involved since we need to be able to track the already committed (to the hardware) atomic states in addition to the userspace queued atomic states. Years ago I did implement something like that in https://lore.kernel.org/intel-gfx/20170629134948.5614-1-ville.syrjala@linux.intel.com/ but Sima didn't want it. In order to get rid of the dependency on the timeout, and make things faster, let's just effectively revert the remainders of commit 9db529aac938 ("drm/i915: More surgically unbreak the modeset vs reset deadlock"). The upside is that the reset is fast again, but the downside is that we now do a full wedge on all display resets, which will also kill innocent batches. But perhaps no one really cares since this is currently only needed for old pre-g4x hardware. But if anyone has plans on using eg. FLR as a backup GPU reset on new hardware then we probably need to come up with something better... Signed-off-by: Ville Syrjälä Link: https://patch.msgid.link/20260903130116.19089-1-ville.syrjala@linux.intel.com Acked-by: Jani Nikula --- drivers/gpu/drm/i915/gt/intel_reset.c | 8 +------- drivers/gpu/drm/i915/i915_dpt.c | 2 -- drivers/gpu/drm/i915/i915_drv.h | 2 -- drivers/gpu/drm/i915/i915_fb_pin.c | 6 ------ drivers/gpu/drm/i915/i915_overlay.c | 5 ----- 5 files changed, 1 insertion(+), 22 deletions(-) diff --git a/drivers/gpu/drm/i915/gt/intel_reset.c b/drivers/gpu/drm/i915/gt/intel_reset.c index b2cf672564dd9c..0968932c0b4fdf 100644 --- a/drivers/gpu/drm/i915/gt/intel_reset.c +++ b/drivers/gpu/drm/i915/gt/intel_reset.c @@ -1433,13 +1433,7 @@ static void intel_gt_reset_global(struct intel_gt *gt, need_display_reset; if (reset_display) { - if (atomic_read(&i915->pending_fb_pin)) { - drm_dbg_kms(&i915->drm, - "Modeset potentially stuck, unbreaking through wedging\n"); - - intel_gt_set_wedged(gt); - } - + intel_gt_set_wedged(gt); intel_display_reset_prepare(display); } diff --git a/drivers/gpu/drm/i915/i915_dpt.c b/drivers/gpu/drm/i915/i915_dpt.c index e01dc4de178893..85d872c0f38782 100644 --- a/drivers/gpu/drm/i915/i915_dpt.c +++ b/drivers/gpu/drm/i915/i915_dpt.c @@ -139,7 +139,6 @@ struct i915_vma *i915_dpt_pin_to_ggtt(struct intel_dpt *dpt, unsigned int alignm pin_flags |= PIN_MAPPABLE; wakeref = intel_runtime_pm_get(&i915->runtime_pm); - atomic_inc(&i915->pending_fb_pin); for_i915_gem_ww(&ww, err, true) { err = i915_gem_object_lock(dpt->obj, &ww); @@ -169,7 +168,6 @@ struct i915_vma *i915_dpt_pin_to_ggtt(struct intel_dpt *dpt, unsigned int alignm dpt->obj->mm.dirty = true; - atomic_dec(&i915->pending_fb_pin); intel_runtime_pm_put(&i915->runtime_pm, wakeref); return err ? ERR_PTR(err) : vma; diff --git a/drivers/gpu/drm/i915/i915_drv.h b/drivers/gpu/drm/i915/i915_drv.h index 844ed79e72114e..dafee3dcd1c521 100644 --- a/drivers/gpu/drm/i915/i915_drv.h +++ b/drivers/gpu/drm/i915/i915_drv.h @@ -315,8 +315,6 @@ struct drm_i915_private { /* The TTM device structure. */ struct ttm_device bdev; - atomic_t pending_fb_pin; - I915_SELFTEST_DECLARE(struct i915_selftest_stash selftest;) /* diff --git a/drivers/gpu/drm/i915/i915_fb_pin.c b/drivers/gpu/drm/i915/i915_fb_pin.c index 1034cb767e9f9e..63be136f1a0524 100644 --- a/drivers/gpu/drm/i915/i915_fb_pin.c +++ b/drivers/gpu/drm/i915/i915_fb_pin.c @@ -35,8 +35,6 @@ intel_fb_pin_to_dpt(struct drm_gem_object *_obj, struct intel_dpt *dpt, if (WARN_ON(!i915_gem_object_is_framebuffer(obj))) return ERR_PTR(-EINVAL); - atomic_inc(&i915->pending_fb_pin); - for_i915_gem_ww(&ww, ret, true) { ret = i915_gem_object_lock(obj, &ww); if (ret) @@ -98,7 +96,6 @@ intel_fb_pin_to_dpt(struct drm_gem_object *_obj, struct intel_dpt *dpt, */ drm_WARN_ON(&i915->drm, i915_dpt_offset(vma)); err: - atomic_dec(&i915->pending_fb_pin); return vma; } @@ -132,8 +129,6 @@ intel_fb_pin_to_ggtt(struct drm_gem_object *_obj, */ wakeref = intel_runtime_pm_get(&i915->runtime_pm); - atomic_inc(&i915->pending_fb_pin); - pinctl = 0; /* PIN_MAPPABLE limits the address to GMADR size */ if (pin_params->needs_low_address) @@ -206,7 +201,6 @@ intel_fb_pin_to_ggtt(struct drm_gem_object *_obj, if (ret) vma = ERR_PTR(ret); - atomic_dec(&i915->pending_fb_pin); intel_runtime_pm_put(&i915->runtime_pm, wakeref); return vma; } diff --git a/drivers/gpu/drm/i915/i915_overlay.c b/drivers/gpu/drm/i915/i915_overlay.c index 6de550a1775671..c1a7920c63247f 100644 --- a/drivers/gpu/drm/i915/i915_overlay.c +++ b/drivers/gpu/drm/i915/i915_overlay.c @@ -354,14 +354,11 @@ static struct i915_vma *i915_overlay_pin_fb(struct drm_device *drm, struct drm_gem_object *obj, u32 *offset) { - struct drm_i915_private *i915 = to_i915(drm); struct drm_i915_gem_object *new_bo = to_intel_bo(obj); struct i915_gem_ww_ctx ww; struct i915_vma *vma; int ret; - atomic_inc(&i915->pending_fb_pin); - i915_gem_ww_ctx_init(&ww, true); retry: ret = i915_gem_object_lock(new_bo, &ww); @@ -377,8 +374,6 @@ static struct i915_vma *i915_overlay_pin_fb(struct drm_device *drm, } i915_gem_ww_ctx_fini(&ww); - atomic_dec(&i915->pending_fb_pin); - if (ret) return ERR_PTR(ret); From 290153d46d1aef83a655623b909072fe82201e7a Mon Sep 17 00:00:00 2001 From: Jose Ignacio Tornos Martinez Date: Tue, 21 Jul 2026 10:13:01 +0200 Subject: [PATCH 0033/1352] PCI: Add device-specific reset for Qualcomm devices Some Qualcomm PCIe devices (WCN6855/WCN7850 WLAN cards, SDX62/SDX65 modems) lack working reset methods for VFIO passthrough scenarios. These devices have no FLR capability, advertise NoSoftRst+ (blocking PM reset), and have broken bus reset. The problem manifests in VFIO passthrough scenarios: - WCN6855 (17cb:1103) and WCN7850 (17cb:1107) WLAN devices: Normal VM operation works fine, including clean shutdown/reboot. However, when the VM terminates uncleanly (crash, force-off), VFIO attempts to reset the device before it can be assigned to another VM. Without a working reset method, the device remains in an undefined state, preventing reuse. - SDX62/SDX65 (17cb:0308) 5G modems: Never successfully initialize even on first VM assignment without proper reset capability. Add device-specific reset methods using BAR-space hardware reset registers that exist in these devices: - WCN6855/WCN7850 WLAN devices use SoC global reset via BAR0 (sequence from ath11k/ath12k driver: ath11k_pci_soc_global_reset(), ath11k_pci_sw_reset(), ath11k_mhi_set_mhictrl_reset()): - Write/clear reset bit at offset 0x3008 - Wait for PCIe link recovery (up to 5 seconds) - Clear MHI controller SYSERR status at offset 0x38 - SDX62/SDX65 modem devices use MHI SoC reset via BAR0 (sequence from MHI driver: mhi_soc_reset(), mhi_pci_reset_prepare()): - Write reset request to offset 0xb0 - Wait 2 seconds for reset completion These are true hardware reset mechanisms (not power management or firmware error recovery), providing proper device reset for VFIO scenarios. Testing was performed on desktop platforms with M.2 WLAN and modem cards using M.2-to-PCIe adapters, including extensive force-reset cycling to verify stability. Signed-off-by: Jose Ignacio Tornos Martinez Signed-off-by: Bjorn Helgaas Reviewed-by: Manivannan Sadhasivam Link: https://patch.msgid.link/20260721081301.205374-1-jtornosm@redhat.com --- drivers/pci/quirks.c | 116 +++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 116 insertions(+) diff --git a/drivers/pci/quirks.c b/drivers/pci/quirks.c index de9bbccda21fd1..e800dd517614e7 100644 --- a/drivers/pci/quirks.c +++ b/drivers/pci/quirks.c @@ -22,6 +22,7 @@ #include /* isa_dma_bridge_buggy */ #include #include +#include #include #include #include @@ -4230,6 +4231,118 @@ static int reset_hinic_vf_dev(struct pci_dev *pdev, bool probe) return 0; } +#define QUALCOMM_WLAN_PCIE_SOC_GLOBAL_RESET 0x3008 +#define QUALCOMM_WLAN_PCIE_SOC_GLOBAL_RESET_V BIT(0) +#define QUALCOMM_WLAN_MHICTRL 0x38 +#define QUALCOMM_WLAN_MHICTRL_RESET_MASK 0x2 + +/* + * Qualcomm WLAN device-specific reset using SoC global reset via BAR0 + * registers. + */ +static int reset_qualcomm_wlan(struct pci_dev *pdev, bool probe) +{ + void __iomem *bar; + u32 val; + u16 cmd; + int ret; + + if (probe) + return 0; + + if (pdev->current_state != PCI_D0) + return -EINVAL; + + pci_read_config_word(pdev, PCI_COMMAND, &cmd); + pci_write_config_word(pdev, PCI_COMMAND, cmd | PCI_COMMAND_MEMORY); + + bar = pci_iomap(pdev, 0, 0); + if (!bar) { + pci_write_config_word(pdev, PCI_COMMAND, cmd); + return -ENODEV; + } + + val = ioread32(bar + QUALCOMM_WLAN_PCIE_SOC_GLOBAL_RESET); + if (PCI_POSSIBLE_ERROR(val)) { + ret = -ENODEV; + goto out_restore; + } + val |= QUALCOMM_WLAN_PCIE_SOC_GLOBAL_RESET_V; + iowrite32(val, bar + QUALCOMM_WLAN_PCIE_SOC_GLOBAL_RESET); + ioread32(bar + QUALCOMM_WLAN_PCIE_SOC_GLOBAL_RESET); + + msleep(10); + + val &= ~QUALCOMM_WLAN_PCIE_SOC_GLOBAL_RESET_V; + iowrite32(val, bar + QUALCOMM_WLAN_PCIE_SOC_GLOBAL_RESET); + ioread32(bar + QUALCOMM_WLAN_PCIE_SOC_GLOBAL_RESET); + + msleep(10); + + ret = read_poll_timeout(ioread32, val, + !PCI_POSSIBLE_ERROR(val), + 20 * USEC_PER_MSEC, + 5 * USEC_PER_SEC, false, + bar + QUALCOMM_WLAN_PCIE_SOC_GLOBAL_RESET); + if (ret) { + pci_err(pdev, "PCIe link failed to recover after reset\n"); + goto out_restore; + } + + /* After SOC_GLOBAL_RESET, MHISTATUS may still have SYSERR bit set + * and thus need to set MHICTRL_RESET to clear SYSERR. + */ + iowrite32(QUALCOMM_WLAN_MHICTRL_RESET_MASK, bar + QUALCOMM_WLAN_MHICTRL); + ioread32(bar + QUALCOMM_WLAN_MHICTRL); + + msleep(10); + +out_restore: + pci_iounmap(pdev, bar); + pci_write_config_word(pdev, PCI_COMMAND, cmd); + + return ret; +} + +#define MHI_SOC_RESET_REQ_OFFSET 0xb0 +#define MHI_SOC_RESET_REQ BIT(0) + +/* + * Qualcomm modem device-specific reset using MHI SoC reset via BAR0 + * register. + */ +static int reset_qualcomm_modem(struct pci_dev *pdev, bool probe) +{ + void __iomem *bar; + u16 cmd; + + if (probe) + return 0; + + if (pdev->current_state != PCI_D0) + return -EINVAL; + + pci_read_config_word(pdev, PCI_COMMAND, &cmd); + pci_write_config_word(pdev, PCI_COMMAND, cmd | PCI_COMMAND_MEMORY); + + bar = pci_iomap(pdev, 0, 0); + if (!bar) { + pci_write_config_word(pdev, PCI_COMMAND, cmd); + return -ENODEV; + } + + iowrite32(MHI_SOC_RESET_REQ, bar + MHI_SOC_RESET_REQ_OFFSET); + ioread32(bar + MHI_SOC_RESET_REQ_OFFSET); + + /* Be sure device reset has been executed */ + msleep(2000); + + pci_iounmap(pdev, bar); + pci_write_config_word(pdev, PCI_COMMAND, cmd); + + return 0; +} + static const struct pci_dev_reset_methods pci_dev_reset_methods[] = { { PCI_VENDOR_ID_INTEL, PCI_DEVICE_ID_INTEL_82599_SFP_VF, reset_intel_82599_sfp_virtfn }, @@ -4245,6 +4358,9 @@ static const struct pci_dev_reset_methods pci_dev_reset_methods[] = { reset_chelsio_generic_dev }, { PCI_VENDOR_ID_HUAWEI, PCI_DEVICE_ID_HINIC_VF, reset_hinic_vf_dev }, + { PCI_VENDOR_ID_QCOM, 0x0308, reset_qualcomm_modem }, /* SDX62/SDX65 modems */ + { PCI_VENDOR_ID_QCOM, 0x1103, reset_qualcomm_wlan }, /* WCN6855 WLAN */ + { PCI_VENDOR_ID_QCOM, 0x1107, reset_qualcomm_wlan }, /* WCN7850 WLAN */ { 0 } }; From d393529394167e0f5f706657eebe84d8529ce4fc Mon Sep 17 00:00:00 2001 From: Nemesa Garg Date: Wed, 9 Sep 2026 16:33:31 +0530 Subject: [PATCH 0034/1352] Revert "drm/i915/display: Clear SEL_FETCH_PLANE_CTL on plane disable" MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit This reverts commit 7f1172a2ac0d7e50850785e2e65789c8aac8411a. This commit replaced the crtc_state->enable_psr2_sel_fetch guard in icl_plane_disable_sel_fetch_arm() and i9xx_cursor_disable_sel_fetch_arm() with HAS_PSR2_SEL_FETCH(). This is a display version check and says nothing about the pipe, so every plane and cursor disable on a display 12+ platform started writing SEL_FETCH_PLANE_CTL() / SEL_FETCH_CUR_CTL(), including on pipes that do not implement them. It shows up as an unclaimed register access on pipes driving HDMI where selective fetch was never enabled. The stale selective fetch enable bit that commit addressed is handled in the next patch. Fixes: 7f1172a2ac0d ("drm/i915/display: Clear SEL_FETCH_PLANE_CTL on plane disable") Closes: https://gitlab.freedesktop.org/drm/i915/kernel/-/work_items/16876 Cc: stable@vger.kernel.org Signed-off-by: Nemesa Garg Reviewed-by: Jouni Högander Signed-off-by: Suraj Kandpal Link: https://patch.msgid.link/20260909110332.3528029-2-nemesa.garg@intel.com --- drivers/gpu/drm/i915/display/intel_cursor.c | 15 +++++---------- .../gpu/drm/i915/display/skl_universal_plane.c | 15 +++++---------- 2 files changed, 10 insertions(+), 20 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_cursor.c b/drivers/gpu/drm/i915/display/intel_cursor.c index cce041e1da51d3..90173040d82532 100644 --- a/drivers/gpu/drm/i915/display/intel_cursor.c +++ b/drivers/gpu/drm/i915/display/intel_cursor.c @@ -531,18 +531,13 @@ static int i9xx_check_cursor(struct intel_crtc_state *crtc_state, } static void i9xx_cursor_disable_sel_fetch_arm(struct intel_dsb *dsb, - struct intel_plane *plane) + struct intel_plane *plane, + const struct intel_crtc_state *crtc_state) { struct intel_display *display = to_intel_display(plane); enum pipe pipe = plane->pipe; - /* - * Clear this whenever the hardware has selective fetch, not just when - * the current state uses it. The cursor may have been enabled with - * selective fetch earlier and had its enable bit orphaned when the - * feature was switched off. - */ - if (!HAS_PSR2_SEL_FETCH(display)) + if (!crtc_state->enable_psr2_sel_fetch) return; intel_de_write_dsb(display, dsb, SEL_FETCH_CUR_CTL(pipe), 0); @@ -592,7 +587,7 @@ static void i9xx_cursor_update_sel_fetch_arm(struct intel_dsb *dsb, if (crtc_state->enable_psr2_su_region_et) wa_16021440873(dsb, plane, crtc_state, plane_state); else - i9xx_cursor_disable_sel_fetch_arm(dsb, plane); + i9xx_cursor_disable_sel_fetch_arm(dsb, plane, crtc_state); } } @@ -701,7 +696,7 @@ static void i9xx_cursor_update_arm(struct intel_dsb *dsb, if (plane_state) i9xx_cursor_update_sel_fetch_arm(dsb, plane, crtc_state, plane_state); else - i9xx_cursor_disable_sel_fetch_arm(dsb, plane); + i9xx_cursor_disable_sel_fetch_arm(dsb, plane, crtc_state); if (plane->cursor.base != base || plane->cursor.size != fbc_ctl || diff --git a/drivers/gpu/drm/i915/display/skl_universal_plane.c b/drivers/gpu/drm/i915/display/skl_universal_plane.c index 5cda1ab90e40f8..07a68329335219 100644 --- a/drivers/gpu/drm/i915/display/skl_universal_plane.c +++ b/drivers/gpu/drm/i915/display/skl_universal_plane.c @@ -879,18 +879,13 @@ skl_plane_disable_arm(struct intel_dsb *dsb, } static void icl_plane_disable_sel_fetch_arm(struct intel_dsb *dsb, - struct intel_plane *plane) + struct intel_plane *plane, + const struct intel_crtc_state *crtc_state) { struct intel_display *display = to_intel_display(plane); enum pipe pipe = plane->pipe; - /* - * Clear this whenever the hardware has selective fetch, not just when - * the current state uses it. The plane may have been enabled with - * selective fetch earlier and had its enable bit orphaned when the - * feature was switched off. - */ - if (!HAS_PSR2_SEL_FETCH(display)) + if (!crtc_state->enable_psr2_sel_fetch) return; intel_de_write_dsb(display, dsb, SEL_FETCH_PLANE_CTL(pipe, plane->id), 0); @@ -926,7 +921,7 @@ icl_plane_disable_arm(struct intel_dsb *dsb, skl_write_plane_wm(dsb, plane, crtc_state); - icl_plane_disable_sel_fetch_arm(dsb, plane); + icl_plane_disable_sel_fetch_arm(dsb, plane, crtc_state); if (plane_has_normalizer(plane)) intel_de_write_dsb(display, dsb, @@ -1646,7 +1641,7 @@ static void icl_plane_update_sel_fetch_arm(struct intel_dsb *dsb, intel_de_write_dsb(display, dsb, SEL_FETCH_PLANE_CTL(pipe, plane->id), SEL_FETCH_PLANE_CTL_ENABLE); else - icl_plane_disable_sel_fetch_arm(dsb, plane); + icl_plane_disable_sel_fetch_arm(dsb, plane, crtc_state); } static void From a4c0e7f80429eda6990960971aebd4e4b9533cc6 Mon Sep 17 00:00:00 2001 From: Nemesa Garg Date: Wed, 9 Sep 2026 16:33:32 +0530 Subject: [PATCH 0035/1352] drm/i915/psr: Clear stale sel fetch enable bits on sel fetch disable MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Selective fetch is dropped while pipe CRC is active, and the planes keep their SEL_FETCH_PLANE_CTL / SEL_FETCH_CUR_CTL enable bit set in hardware over that. A plane disabled while selective fetch is off never gets the bit cleared, as the disable path is guarded by enable_psr2_sel_fetch. Once selective fetch comes back the hardware resumes fetching for a plane that is no longer enabled and keeps its DDB range reserved. Clear the bits as selective fetch is turned off instead. Atomic check has both the old and the new crtc state, so record the transition there and let the plane and cursor arm paths write the registers to 0 for that commit. v2: Drop the old_crtc_state->hw.active check. [Jouni] Fixes: b1f5279b5981 ("drm/i915/psr: Move plane sel fetch configuration into plane source files") Closes: https://gitlab.freedesktop.org/drm/xe/kernel/-/work_items/8739 Assisted-by: Copilot:Claude-Opus-5 Signed-off-by: Nemesa Garg Reviewed-by: Jouni Högander Signed-off-by: Suraj Kandpal Link: https://patch.msgid.link/20260909110332.3528029-3-nemesa.garg@intel.com --- drivers/gpu/drm/i915/display/intel_cursor.c | 7 +++++-- .../gpu/drm/i915/display/intel_display_types.h | 2 ++ drivers/gpu/drm/i915/display/intel_psr.c | 15 +++++++++++++++ .../gpu/drm/i915/display/skl_universal_plane.c | 9 ++++----- 4 files changed, 26 insertions(+), 7 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_cursor.c b/drivers/gpu/drm/i915/display/intel_cursor.c index 90173040d82532..4afd275de86eae 100644 --- a/drivers/gpu/drm/i915/display/intel_cursor.c +++ b/drivers/gpu/drm/i915/display/intel_cursor.c @@ -537,7 +537,8 @@ static void i9xx_cursor_disable_sel_fetch_arm(struct intel_dsb *dsb, struct intel_display *display = to_intel_display(plane); enum pipe pipe = plane->pipe; - if (!crtc_state->enable_psr2_sel_fetch) + if (!crtc_state->enable_psr2_sel_fetch && + !crtc_state->clear_psr2_sel_fetch) return; intel_de_write_dsb(display, dsb, SEL_FETCH_CUR_CTL(pipe), 0); @@ -570,8 +571,10 @@ static void i9xx_cursor_update_sel_fetch_arm(struct intel_dsb *dsb, struct intel_display *display = to_intel_display(plane); enum pipe pipe = plane->pipe; - if (!crtc_state->enable_psr2_sel_fetch) + if (!crtc_state->enable_psr2_sel_fetch) { + i9xx_cursor_disable_sel_fetch_arm(dsb, plane, crtc_state); return; + } if (drm_rect_height(&plane_state->psr2_sel_fetch_area) > 0) { if (crtc_state->enable_psr2_su_region_et) { diff --git a/drivers/gpu/drm/i915/display/intel_display_types.h b/drivers/gpu/drm/i915/display/intel_display_types.h index 9016be52c7eac8..ec99ff10391cf1 100644 --- a/drivers/gpu/drm/i915/display/intel_display_types.h +++ b/drivers/gpu/drm/i915/display/intel_display_types.h @@ -1190,6 +1190,8 @@ struct intel_crtc_state { bool has_sel_update; bool enable_psr2_sel_fetch; bool enable_psr2_su_region_et; + /* Drop the stale selective fetch enable bits as selective fetch is turned off */ + bool clear_psr2_sel_fetch; bool req_psr2_sdp_prior_scanline; bool has_panel_replay; bool link_off_after_as_sdp_when_pr_active; diff --git a/drivers/gpu/drm/i915/display/intel_psr.c b/drivers/gpu/drm/i915/display/intel_psr.c index f490beb66629e0..872e253db1786b 100644 --- a/drivers/gpu/drm/i915/display/intel_psr.c +++ b/drivers/gpu/drm/i915/display/intel_psr.c @@ -2889,6 +2889,8 @@ int intel_psr2_sel_fetch_update(struct intel_atomic_state *state, struct intel_crtc *crtc) { struct intel_display *display = to_intel_display(state); + const struct intel_crtc_state *old_crtc_state = + intel_atomic_get_old_crtc_state(state, crtc); struct intel_crtc_state *crtc_state = intel_atomic_get_new_crtc_state(state, crtc); struct intel_plane_state *new_plane_state, *old_plane_state; struct intel_plane *plane; @@ -2901,6 +2903,19 @@ int intel_psr2_sel_fetch_update(struct intel_atomic_state *state, bool full_update = false, su_area_changed; int i, ret; + /* + * Selective fetch is not always usable, for instance it is dropped + * while pipe CRC is active. The planes keep their selective fetch + * enable bit set in hardware over that, and a plane disabled while + * selective fetch is off never gets the bit cleared. Once selective + * fetch comes back the hardware would resume fetching for a plane that + * is no longer enabled and keep its DDB range reserved, so have the + * plane update drop the bit for every plane of the pipe as selective + * fetch is turned off. + */ + crtc_state->clear_psr2_sel_fetch = old_crtc_state->enable_psr2_sel_fetch && + !crtc_state->enable_psr2_sel_fetch; + if (!crtc_state->enable_psr2_sel_fetch) return 0; diff --git a/drivers/gpu/drm/i915/display/skl_universal_plane.c b/drivers/gpu/drm/i915/display/skl_universal_plane.c index 07a68329335219..eb5ed981b40f6a 100644 --- a/drivers/gpu/drm/i915/display/skl_universal_plane.c +++ b/drivers/gpu/drm/i915/display/skl_universal_plane.c @@ -885,7 +885,8 @@ static void icl_plane_disable_sel_fetch_arm(struct intel_dsb *dsb, struct intel_display *display = to_intel_display(plane); enum pipe pipe = plane->pipe; - if (!crtc_state->enable_psr2_sel_fetch) + if (!crtc_state->enable_psr2_sel_fetch && + !crtc_state->clear_psr2_sel_fetch) return; intel_de_write_dsb(display, dsb, SEL_FETCH_PLANE_CTL(pipe, plane->id), 0); @@ -1634,10 +1635,8 @@ static void icl_plane_update_sel_fetch_arm(struct intel_dsb *dsb, struct intel_display *display = to_intel_display(plane); enum pipe pipe = plane->pipe; - if (!crtc_state->enable_psr2_sel_fetch) - return; - - if (drm_rect_height(&plane_state->psr2_sel_fetch_area) > 0) + if (crtc_state->enable_psr2_sel_fetch && + drm_rect_height(&plane_state->psr2_sel_fetch_area) > 0) intel_de_write_dsb(display, dsb, SEL_FETCH_PLANE_CTL(pipe, plane->id), SEL_FETCH_PLANE_CTL_ENABLE); else From 985bc8ee612328771308761822fbc4c62a21a0d8 Mon Sep 17 00:00:00 2001 From: Chaitanya Kumar Borah Date: Wed, 2 Sep 2026 13:24:09 +0530 Subject: [PATCH 0036/1352] drm/i915/color: Add CSC on SDR plane color pipeline Add the fixed-function CSC block to color pipeline in SDR planes as a DRM_COLOROP_FIXED_MATRIX colorop. v2: - s/DRM_COLOROP_FM_YCBCR2020_FULL_RGB_NC/ DRM_COLOROP_FM_YCBCR2020_NC_FULL_RGB - Inline icl_is_hdr_plane() instead of storing in local variable v3: - In preparation of a simple pipeline [YUV Full/Limited -> RGB] Make the Fixed Matrix ColorOp support Limited Range enums (DRM_COLOROP_FM_YCBCRXXX_LIMITED_RGB) too. - Therefore s/sdr_plane_pipeline/sdr_plane_yuv_pipeline Signed-off-by: Chaitanya Kumar Borah Reviewed-by: Uma Shankar Link: https://patch.msgid.link/20260902075417.656673-2-chaitanya.kumar.borah@intel.com --- .../drm/i915/display/intel_color_pipeline.c | 23 ++++++++++++++++++- .../drm/i915/display/intel_display_limits.h | 1 + 2 files changed, 23 insertions(+), 1 deletion(-) diff --git a/drivers/gpu/drm/i915/display/intel_color_pipeline.c b/drivers/gpu/drm/i915/display/intel_color_pipeline.c index 6cf8080ee80003..efd4375c433189 100644 --- a/drivers/gpu/drm/i915/display/intel_color_pipeline.c +++ b/drivers/gpu/drm/i915/display/intel_color_pipeline.c @@ -43,6 +43,18 @@ static const enum intel_color_block hdr_plane_pipeline[] = { INTEL_PLANE_CB_POST_CSC_LUT, }; +static const enum intel_color_block sdr_plane_yuv_pipeline[] = { + INTEL_PLANE_CB_CSC_FF, +}; + +static const u64 intel_plane_supported_csc_ff = + BIT(DRM_COLOROP_FM_YCBCR601_FULL_RGB) | + BIT(DRM_COLOROP_FM_YCBCR601_LIMITED_RGB) | + BIT(DRM_COLOROP_FM_YCBCR709_FULL_RGB) | + BIT(DRM_COLOROP_FM_YCBCR709_LIMITED_RGB) | + BIT(DRM_COLOROP_FM_YCBCR2020_NC_FULL_RGB) | + BIT(DRM_COLOROP_FM_YCBCR2020_NC_LIMITED_RGB); + static bool plane_has_3dlut(struct intel_display *display, enum pipe pipe, struct drm_plane *plane) { @@ -92,6 +104,12 @@ struct intel_colorop *intel_color_pipeline_plane_add_colorop(struct drm_plane *p DRM_COLOROP_LUT1D_INTERPOLATION_LINEAR, DRM_COLOROP_FLAG_ALLOW_BYPASS); break; + case INTEL_PLANE_CB_CSC_FF: + ret = drm_plane_colorop_fixed_matrix_init(dev, &colorop->base, plane, + &intel_colorop_funcs, + intel_plane_supported_csc_ff, + DRM_COLOROP_FLAG_ALLOW_BYPASS); + break; default: drm_err(plane->dev, "Invalid colorop id [%d]", id); ret = -EINVAL; @@ -126,9 +144,12 @@ int _intel_color_pipeline_plane_init(struct drm_plane *plane, struct drm_prop_en if (plane_has_3dlut(display, pipe, plane)) { pipeline = xe3plpd_primary_plane_pipeline; pipeline_len = ARRAY_SIZE(xe3plpd_primary_plane_pipeline); - } else { + } else if (icl_is_hdr_plane(display, to_intel_plane(plane)->id)) { pipeline = hdr_plane_pipeline; pipeline_len = ARRAY_SIZE(hdr_plane_pipeline); + } else { + pipeline = sdr_plane_yuv_pipeline; + pipeline_len = ARRAY_SIZE(sdr_plane_yuv_pipeline); } for (i = 0; i < pipeline_len; i++) { diff --git a/drivers/gpu/drm/i915/display/intel_display_limits.h b/drivers/gpu/drm/i915/display/intel_display_limits.h index ea89473c177f41..7ba7360c574e03 100644 --- a/drivers/gpu/drm/i915/display/intel_display_limits.h +++ b/drivers/gpu/drm/i915/display/intel_display_limits.h @@ -169,6 +169,7 @@ enum aux_ch { enum intel_color_block { INTEL_PLANE_CB_PRE_CSC_LUT, INTEL_PLANE_CB_CSC, + INTEL_PLANE_CB_CSC_FF, INTEL_PLANE_CB_POST_CSC_LUT, INTEL_PLANE_CB_3DLUT, From 0bc13827b3dac986124d81c0d3ef840f664d5308 Mon Sep 17 00:00:00 2001 From: Chaitanya Kumar Borah Date: Wed, 2 Sep 2026 13:24:10 +0530 Subject: [PATCH 0037/1352] drm/i915/display: extract glk_plane_color_ctl_input_csc helper Extract the input CSC and YUV range correction logic from glk_plane_color_ctl() into a dedicated glk_plane_color_ctl_input_csc() helper. No functional change. Signed-off-by: Chaitanya Kumar Borah Reviewed-by: Uma Shankar Link: https://patch.msgid.link/20260902075417.656673-3-chaitanya.kumar.borah@intel.com --- .../drm/i915/display/skl_universal_plane.c | 32 +++++++++++-------- 1 file changed, 19 insertions(+), 13 deletions(-) diff --git a/drivers/gpu/drm/i915/display/skl_universal_plane.c b/drivers/gpu/drm/i915/display/skl_universal_plane.c index eb5ed981b40f6a..e5384d2e5e8eb9 100644 --- a/drivers/gpu/drm/i915/display/skl_universal_plane.c +++ b/drivers/gpu/drm/i915/display/skl_universal_plane.c @@ -1241,37 +1241,43 @@ static u32 glk_plane_color_ctl_crtc(const struct intel_crtc_state *crtc_state) return plane_color_ctl; } -static u32 glk_plane_color_ctl(const struct intel_plane_state *plane_state) +static u32 glk_plane_color_ctl_input_csc(const struct intel_plane_state *plane_state) { struct intel_display *display = to_intel_display(plane_state); const struct drm_framebuffer *fb = plane_state->hw.fb; struct intel_plane *plane = to_intel_plane(plane_state->uapi.plane); - u32 plane_color_ctl = 0; - - plane_color_ctl |= PLANE_COLOR_PLANE_GAMMA_DISABLE; - plane_color_ctl |= glk_plane_color_ctl_alpha(plane_state); + u32 ctl = 0; if (fb->format->is_yuv && !icl_is_hdr_plane(display, plane->id)) { switch (plane_state->hw.color_encoding) { case DRM_COLOR_YCBCR_BT709: - plane_color_ctl |= PLANE_COLOR_CSC_MODE_YUV709_TO_RGB709; + ctl |= PLANE_COLOR_CSC_MODE_YUV709_TO_RGB709; break; case DRM_COLOR_YCBCR_BT2020: - plane_color_ctl |= - PLANE_COLOR_CSC_MODE_YUV2020_TO_RGB2020; + ctl |= PLANE_COLOR_CSC_MODE_YUV2020_TO_RGB2020; break; default: - plane_color_ctl |= - PLANE_COLOR_CSC_MODE_YUV601_TO_RGB601; + ctl |= PLANE_COLOR_CSC_MODE_YUV601_TO_RGB601; } if (plane_state->hw.color_range == DRM_COLOR_YCBCR_FULL_RANGE) - plane_color_ctl |= PLANE_COLOR_YUV_RANGE_CORRECTION_DISABLE; + ctl |= PLANE_COLOR_YUV_RANGE_CORRECTION_DISABLE; } else if (fb->format->is_yuv) { - plane_color_ctl |= PLANE_COLOR_INPUT_CSC_ENABLE; + ctl |= PLANE_COLOR_INPUT_CSC_ENABLE; if (plane_state->hw.color_range == DRM_COLOR_YCBCR_FULL_RANGE) - plane_color_ctl |= PLANE_COLOR_YUV_RANGE_CORRECTION_DISABLE; + ctl |= PLANE_COLOR_YUV_RANGE_CORRECTION_DISABLE; } + return ctl; +} + +static u32 glk_plane_color_ctl(const struct intel_plane_state *plane_state) +{ + u32 plane_color_ctl = 0; + + plane_color_ctl |= PLANE_COLOR_PLANE_GAMMA_DISABLE; + plane_color_ctl |= glk_plane_color_ctl_alpha(plane_state); + plane_color_ctl |= glk_plane_color_ctl_input_csc(plane_state); + if (plane_state->force_black) plane_color_ctl |= PLANE_COLOR_PLANE_CSC_ENABLE; From f4a41051c370987cabff3504ff9c1f6e0e66fe53 Mon Sep 17 00:00:00 2001 From: Chaitanya Kumar Borah Date: Wed, 2 Sep 2026 13:24:11 +0530 Subject: [PATCH 0038/1352] drm/i915/display: simplify glk_plane_color_ctl_input_csc Add early return for non-YUV formats and hoist the duplicated color_range check out of the if/else branches. No functional change. Signed-off-by: Chaitanya Kumar Borah Reviewed-by: Uma Shankar Link: https://patch.msgid.link/20260902075417.656673-4-chaitanya.kumar.borah@intel.com --- drivers/gpu/drm/i915/display/skl_universal_plane.c | 14 ++++++++------ 1 file changed, 8 insertions(+), 6 deletions(-) diff --git a/drivers/gpu/drm/i915/display/skl_universal_plane.c b/drivers/gpu/drm/i915/display/skl_universal_plane.c index e5384d2e5e8eb9..4aafd9ff3ac64a 100644 --- a/drivers/gpu/drm/i915/display/skl_universal_plane.c +++ b/drivers/gpu/drm/i915/display/skl_universal_plane.c @@ -1248,7 +1248,10 @@ static u32 glk_plane_color_ctl_input_csc(const struct intel_plane_state *plane_s struct intel_plane *plane = to_intel_plane(plane_state->uapi.plane); u32 ctl = 0; - if (fb->format->is_yuv && !icl_is_hdr_plane(display, plane->id)) { + if (!fb->format->is_yuv) + return 0; + + if (!icl_is_hdr_plane(display, plane->id)) { switch (plane_state->hw.color_encoding) { case DRM_COLOR_YCBCR_BT709: ctl |= PLANE_COLOR_CSC_MODE_YUV709_TO_RGB709; @@ -1259,14 +1262,13 @@ static u32 glk_plane_color_ctl_input_csc(const struct intel_plane_state *plane_s default: ctl |= PLANE_COLOR_CSC_MODE_YUV601_TO_RGB601; } - if (plane_state->hw.color_range == DRM_COLOR_YCBCR_FULL_RANGE) - ctl |= PLANE_COLOR_YUV_RANGE_CORRECTION_DISABLE; - } else if (fb->format->is_yuv) { + } else { ctl |= PLANE_COLOR_INPUT_CSC_ENABLE; - if (plane_state->hw.color_range == DRM_COLOR_YCBCR_FULL_RANGE) - ctl |= PLANE_COLOR_YUV_RANGE_CORRECTION_DISABLE; } + if (plane_state->hw.color_range == DRM_COLOR_YCBCR_FULL_RANGE) + ctl |= PLANE_COLOR_YUV_RANGE_CORRECTION_DISABLE; + return ctl; } From f1bc49ff6209f7256ddb1f3804b374e033ff8379 Mon Sep 17 00:00:00 2001 From: Chaitanya Kumar Borah Date: Wed, 2 Sep 2026 13:24:12 +0530 Subject: [PATCH 0039/1352] drm/i915/display: Program CSC on SDR planes based on Fixed Matrix Colorop When a color pipeline is active, program the SDR plane fixed-function CSC based on the Fixed Matrix Colorop's state. Re-use the existing plane state variables for color_range and color_encoding. Track the bypass state explicitly as a boolean since bypass is managed separately from the FIXED_MATRIX enum value in the colorop framework. Keep the programming based on color_encoding/color_range legacy properties intact. Signed-off-by: Chaitanya Kumar Borah Reviewed-by: Uma Shankar Link: https://patch.msgid.link/20260902075417.656673-5-chaitanya.kumar.borah@intel.com --- .../drm/i915/display/intel_display_types.h | 1 + drivers/gpu/drm/i915/display/intel_plane.c | 53 ++++++++++++++++++- .../drm/i915/display/skl_universal_plane.c | 4 +- 3 files changed, 55 insertions(+), 3 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_display_types.h b/drivers/gpu/drm/i915/display/intel_display_types.h index ec99ff10391cf1..a5f18ac8a7d06e 100644 --- a/drivers/gpu/drm/i915/display/intel_display_types.h +++ b/drivers/gpu/drm/i915/display/intel_display_types.h @@ -686,6 +686,7 @@ struct intel_plane_state { enum drm_color_range color_range; enum drm_scaling_filter scaling_filter; struct drm_property_blob *ctm, *degamma_lut, *gamma_lut, *lut_3d; + bool csc_ff_enable; } hw; struct i915_vma *ggtt_vma; diff --git a/drivers/gpu/drm/i915/display/intel_plane.c b/drivers/gpu/drm/i915/display/intel_plane.c index d0f99a87c42e8e..25ca049009efbc 100644 --- a/drivers/gpu/drm/i915/display/intel_plane.c +++ b/drivers/gpu/drm/i915/display/intel_plane.c @@ -462,6 +462,42 @@ intel_plane_colorop_replace_blob(struct intel_plane_state *plane_state, return false; } +static u32 +fixedmatrix_colorop_to_encoding(enum drm_colorop_fixed_matrix_type fm_type) +{ + switch (fm_type) { + case DRM_COLOROP_FM_YCBCR709_FULL_RGB: + case DRM_COLOROP_FM_YCBCR709_LIMITED_RGB: + return DRM_COLOR_YCBCR_BT709; + + case DRM_COLOROP_FM_YCBCR2020_NC_FULL_RGB: + case DRM_COLOROP_FM_YCBCR2020_NC_LIMITED_RGB: + return DRM_COLOR_YCBCR_BT2020; + + case DRM_COLOROP_FM_YCBCR601_FULL_RGB: + case DRM_COLOROP_FM_YCBCR601_LIMITED_RGB: + default: + return DRM_COLOR_YCBCR_BT601; + } +} + +static u32 +fixedmatrix_colorop_to_range(enum drm_colorop_fixed_matrix_type fm_type) +{ + switch (fm_type) { + case DRM_COLOROP_FM_YCBCR601_FULL_RGB: + case DRM_COLOROP_FM_YCBCR709_FULL_RGB: + case DRM_COLOROP_FM_YCBCR2020_NC_FULL_RGB: + return DRM_COLOR_YCBCR_FULL_RANGE; + + case DRM_COLOROP_FM_YCBCR601_LIMITED_RGB: + case DRM_COLOROP_FM_YCBCR709_LIMITED_RGB: + case DRM_COLOROP_FM_YCBCR2020_NC_LIMITED_RGB: + default: + return DRM_COLOR_YCBCR_LIMITED_RANGE; + } +} + static void intel_plane_color_copy_uapi_to_hw_state(struct intel_atomic_state *state, struct intel_plane_state *plane_state, @@ -474,6 +510,7 @@ intel_plane_color_copy_uapi_to_hw_state(struct intel_atomic_state *state, struct drm_property_blob *blob; struct intel_crtc_state *new_crtc_state = state ? intel_atomic_get_new_crtc_state(state, crtc) : NULL; + enum drm_colorop_fixed_matrix_type fm_type; bool changed = false; int i = 0; @@ -485,11 +522,23 @@ intel_plane_color_copy_uapi_to_hw_state(struct intel_atomic_state *state, while (iter_colorop) { for_each_new_colorop_in_state(&state->base, colorop, new_colorop_state, i) { if (new_colorop_state->colorop == iter_colorop) { - blob = new_colorop_state->bypass ? NULL : new_colorop_state->data; intel_colorop = to_intel_colorop(colorop); - changed |= intel_plane_colorop_replace_blob(plane_state, + if (intel_colorop->id == INTEL_PLANE_CB_CSC_FF) { + fm_type = new_colorop_state->fixed_matrix_type; + + plane_state->hw.csc_ff_enable = + !new_colorop_state->bypass; + plane_state->hw.color_encoding = + fixedmatrix_colorop_to_encoding(fm_type); + plane_state->hw.color_range = + fixedmatrix_colorop_to_range(fm_type); + } else { + blob = new_colorop_state->bypass ? + NULL : new_colorop_state->data; + changed |= intel_plane_colorop_replace_blob(plane_state, intel_colorop, blob); + } } } iter_colorop = iter_colorop->next; diff --git a/drivers/gpu/drm/i915/display/skl_universal_plane.c b/drivers/gpu/drm/i915/display/skl_universal_plane.c index 4aafd9ff3ac64a..07d9ab5a378641 100644 --- a/drivers/gpu/drm/i915/display/skl_universal_plane.c +++ b/drivers/gpu/drm/i915/display/skl_universal_plane.c @@ -1246,9 +1246,11 @@ static u32 glk_plane_color_ctl_input_csc(const struct intel_plane_state *plane_s struct intel_display *display = to_intel_display(plane_state); const struct drm_framebuffer *fb = plane_state->hw.fb; struct intel_plane *plane = to_intel_plane(plane_state->uapi.plane); + bool color_pipeline = !!plane_state->uapi.color_pipeline; + bool needs_csc = color_pipeline ? plane_state->hw.csc_ff_enable : fb->format->is_yuv; u32 ctl = 0; - if (!fb->format->is_yuv) + if (!needs_csc) return 0; if (!icl_is_hdr_plane(display, plane->id)) { From 8249ceb6bb4c05825517d13eca28a5b572350946 Mon Sep 17 00:00:00 2001 From: Chaitanya Kumar Borah Date: Wed, 2 Sep 2026 13:24:13 +0530 Subject: [PATCH 0040/1352] drm/i915/color: Add support for 1D LUT in SDR planes Extend the SDR plane color pipeline to post-CSC 1D LUT block. v2: - In preparation of a simple pipeline [YUV Full/Limited -> RGB] -> [1D LUT] Drop pre-CSC LUT from the pipeline as it has no use in a YUV -> RGB pipeline. This makes the pipeline simple since the block lies between the YUV range correct block and Fixed function CSC. It can be added back when [RGB709 -> RGB2020] capability is added. Then it can be used for linearization. Signed-off-by: Chaitanya Kumar Borah Reviewed-by: Uma Shankar Link: https://patch.msgid.link/20260902075417.656673-6-chaitanya.kumar.borah@intel.com --- drivers/gpu/drm/i915/display/intel_color_pipeline.c | 1 + 1 file changed, 1 insertion(+) diff --git a/drivers/gpu/drm/i915/display/intel_color_pipeline.c b/drivers/gpu/drm/i915/display/intel_color_pipeline.c index efd4375c433189..53e55ce0a5a35a 100644 --- a/drivers/gpu/drm/i915/display/intel_color_pipeline.c +++ b/drivers/gpu/drm/i915/display/intel_color_pipeline.c @@ -45,6 +45,7 @@ static const enum intel_color_block hdr_plane_pipeline[] = { static const enum intel_color_block sdr_plane_yuv_pipeline[] = { INTEL_PLANE_CB_CSC_FF, + INTEL_PLANE_CB_POST_CSC_LUT, }; static const u64 intel_plane_supported_csc_ff = From 2a8563b84c806ef4b2f919b09089e10a92da447c Mon Sep 17 00:00:00 2001 From: Pranay Samala Date: Wed, 2 Sep 2026 13:24:14 +0530 Subject: [PATCH 0041/1352] drm/i915/color: Extract HDR post-CSC LUT programming to helper function Move HDR plane post-CSC LUT programming to improve code organization. While at it, remove the segment 0 index register writes as it is not currently programmed. Signed-off-by: Pranay Samala Signed-off-by: Chaitanya Kumar Borah Reviewed-by: Uma Shankar Link: https://patch.msgid.link/20260902075417.656673-7-chaitanya.kumar.borah@intel.com --- drivers/gpu/drm/i915/display/intel_color.c | 35 ++++++++++++---------- 1 file changed, 20 insertions(+), 15 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_color.c b/drivers/gpu/drm/i915/display/intel_color.c index 87ced9f6ff408f..36c8688ed28e9f 100644 --- a/drivers/gpu/drm/i915/display/intel_color.c +++ b/drivers/gpu/drm/i915/display/intel_color.c @@ -3998,25 +3998,17 @@ xelpd_program_plane_pre_csc_lut(struct intel_dsb *dsb, } static void -xelpd_program_plane_post_csc_lut(struct intel_dsb *dsb, - const struct intel_plane_state *plane_state) +xelpd_load_hdr_post_csc_lut(struct intel_display *display, + struct intel_dsb *dsb, + enum pipe pipe, + enum plane_id plane, + const struct drm_color_lut32 *post_csc_lut) { - struct intel_display *display = to_intel_display(plane_state); - const struct drm_plane_state *state = &plane_state->uapi; - enum pipe pipe = to_intel_plane(state->plane)->pipe; - enum plane_id plane = to_intel_plane(state->plane)->id; - const struct drm_color_lut32 *post_csc_lut = plane_state->hw.gamma_lut->data; int i, lut_size = 32; u32 lut_val; - if (!icl_is_hdr_plane(display, plane)) - return; - intel_de_write_dsb(display, dsb, PLANE_POST_CSC_GAMC_INDEX_ENH(pipe, plane, 0), PLANE_PAL_PREC_AUTO_INCREMENT); - /* TODO: Add macro */ - intel_de_write_dsb(display, dsb, PLANE_POST_CSC_GAMC_SEG0_INDEX_ENH(pipe, plane, 0), - PLANE_PAL_PREC_AUTO_INCREMENT); for (i = 0; i < lut_size + 3; i++) { if (post_csc_lut) { @@ -4036,8 +4028,21 @@ xelpd_program_plane_post_csc_lut(struct intel_dsb *dsb, } intel_de_write_dsb(display, dsb, PLANE_POST_CSC_GAMC_INDEX_ENH(pipe, plane, 0), 0); - intel_de_write_dsb(display, dsb, - PLANE_POST_CSC_GAMC_SEG0_INDEX_ENH(pipe, plane, 0), 0); +} + +static void +xelpd_program_plane_post_csc_lut(struct intel_dsb *dsb, + const struct intel_plane_state *plane_state) +{ + struct intel_display *display = to_intel_display(plane_state); + const struct drm_plane_state *state = &plane_state->uapi; + enum pipe pipe = to_intel_plane(state->plane)->pipe; + enum plane_id plane = to_intel_plane(state->plane)->id; + const struct drm_color_lut32 *post_csc_lut = plane_state->hw.gamma_lut ? + plane_state->hw.gamma_lut->data : NULL; + + if (icl_is_hdr_plane(display, plane)) + xelpd_load_hdr_post_csc_lut(display, dsb, pipe, plane, post_csc_lut); } static void From d547b3237f014d50bd1a4ae558c9865c0a6c1c3d Mon Sep 17 00:00:00 2001 From: Pranay Samala Date: Wed, 2 Sep 2026 13:24:15 +0530 Subject: [PATCH 0042/1352] drm/i915/color: Program Plane Post CSC registers for SDR planes Implement plane post-CSC LUT support for SDR planes. v2: - Restructure loop to match HDR function pattern Assisted-by: Claude:claude-opus-4.6 Signed-off-by: Pranay Samala Co-developed-by: Chaitanya Kumar Borah Signed-off-by: Chaitanya Kumar Borah Reviewed-by: Uma Shankar Link: https://patch.msgid.link/20260902075417.656673-8-chaitanya.kumar.borah@intel.com --- drivers/gpu/drm/i915/display/intel_color.c | 41 ++++++++++++++++++++++ 1 file changed, 41 insertions(+) diff --git a/drivers/gpu/drm/i915/display/intel_color.c b/drivers/gpu/drm/i915/display/intel_color.c index 36c8688ed28e9f..f1df0f9ba76241 100644 --- a/drivers/gpu/drm/i915/display/intel_color.c +++ b/drivers/gpu/drm/i915/display/intel_color.c @@ -4030,6 +4030,45 @@ xelpd_load_hdr_post_csc_lut(struct intel_display *display, intel_de_write_dsb(display, dsb, PLANE_POST_CSC_GAMC_INDEX_ENH(pipe, plane, 0), 0); } +static void +xelpd_load_sdr_post_csc_lut(struct intel_display *display, + struct intel_dsb *dsb, + enum pipe pipe, + enum plane_id plane, + const struct drm_color_lut32 *post_csc_lut) +{ + int i, lut_size = 32; + u32 lut_val; + + /* + * First 3 planes are HDR, so reduce by 3 to get to the right + * SDR plane offset + */ + plane = plane - 3; + + intel_de_write_dsb(display, dsb, PLANE_POST_CSC_GAMC_INDEX(pipe, plane, 0), + PLANE_PAL_PREC_AUTO_INCREMENT); + + for (i = 0; i < lut_size + 3; i++) { + if (post_csc_lut) { + if (i < lut_size) + lut_val = drm_color_lut32_extract(post_csc_lut[i].green, 16); + /* else duplicate last lut_val */ + } else { + if (i < lut_size) + lut_val = (i * ((1 << 16) - 1)) / (lut_size - 1); + else + lut_val = 1 << 16; + } + + intel_de_write_dsb(display, dsb, + PLANE_POST_CSC_GAMC_DATA(pipe, plane, 0), + lut_val); + } + + intel_de_write_dsb(display, dsb, PLANE_POST_CSC_GAMC_INDEX(pipe, plane, 0), 0); +} + static void xelpd_program_plane_post_csc_lut(struct intel_dsb *dsb, const struct intel_plane_state *plane_state) @@ -4043,6 +4082,8 @@ xelpd_program_plane_post_csc_lut(struct intel_dsb *dsb, if (icl_is_hdr_plane(display, plane)) xelpd_load_hdr_post_csc_lut(display, dsb, pipe, plane, post_csc_lut); + else + xelpd_load_sdr_post_csc_lut(display, dsb, pipe, plane, post_csc_lut); } static void From 334a8c2290603205eaa56f20bfa40c395df6dce0 Mon Sep 17 00:00:00 2001 From: Chaitanya Kumar Borah Date: Wed, 2 Sep 2026 13:24:16 +0530 Subject: [PATCH 0043/1352] drm/i915/color: Add color pipeline support for SDR planes Now that everything is in place expose the SDR plane color pipeline to user-space. Signed-off-by: Chaitanya Kumar Borah Reviewed-by: Uma Shankar Link: https://patch.msgid.link/20260902075417.656673-9-chaitanya.kumar.borah@intel.com --- drivers/gpu/drm/i915/display/intel_color_pipeline.c | 6 ------ 1 file changed, 6 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_color_pipeline.c b/drivers/gpu/drm/i915/display/intel_color_pipeline.c index 53e55ce0a5a35a..38cfd6ed585d0e 100644 --- a/drivers/gpu/drm/i915/display/intel_color_pipeline.c +++ b/drivers/gpu/drm/i915/display/intel_color_pipeline.c @@ -177,17 +177,11 @@ int _intel_color_pipeline_plane_init(struct drm_plane *plane, struct drm_prop_en int intel_color_pipeline_plane_init(struct drm_plane *plane, enum pipe pipe) { - struct drm_device *dev = plane->dev; - struct intel_display *display = to_intel_display(dev); struct drm_prop_enum_list pipelines[MAX_COLOR_PIPELINES] = {}; int len = 0; int ret = 0; int i; - /* Currently expose pipeline only for HDR planes */ - if (!icl_is_hdr_plane(display, to_intel_plane(plane)->id)) - return 0; - /* Add pipeline consisting of transfer functions */ ret = _intel_color_pipeline_plane_init(plane, &pipelines[len], pipe); if (ret) From a4e6c6a761f6699a2a1a84c0fe491bdb71d6a2d9 Mon Sep 17 00:00:00 2001 From: Chaitanya Kumar Borah Date: Wed, 2 Sep 2026 13:24:17 +0530 Subject: [PATCH 0044/1352] drm/i915/color: Add YUV buffer support on HDR planes Add the INTEL_PLANE_CB_CSC_FF (fixed function CSC) color block as the first stage for the HDR planes. This enables YUV-to-RGB color space conversion on HDR planes via the color pipeline. Also in icl_program_input_csc(), account for the color pipeline programming, in addition to FB format. v2: - Increase MAX_COLOROP to 5 (Sashiko) Signed-off-by: Chaitanya Kumar Borah Reviewed-by: Uma Shankar Link: https://patch.msgid.link/20260902075417.656673-10-chaitanya.kumar.borah@intel.com --- drivers/gpu/drm/i915/display/intel_color_pipeline.c | 4 +++- drivers/gpu/drm/i915/display/skl_universal_plane.c | 3 ++- 2 files changed, 5 insertions(+), 2 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_color_pipeline.c b/drivers/gpu/drm/i915/display/intel_color_pipeline.c index 38cfd6ed585d0e..7fc31fe891a85d 100644 --- a/drivers/gpu/drm/i915/display/intel_color_pipeline.c +++ b/drivers/gpu/drm/i915/display/intel_color_pipeline.c @@ -12,7 +12,7 @@ #include "skl_universal_plane.h" #define MAX_COLOR_PIPELINES 1 -#define MAX_COLOROP 4 +#define MAX_COLOROP 5 #define PLANE_DEGAMMA_SIZE 128 #define PLANE_GAMMA_SIZE 32 @@ -31,6 +31,7 @@ static const struct drm_colorop_funcs intel_colorop_funcs = { * the pipeline totally unusable. */ static const enum intel_color_block xe3plpd_primary_plane_pipeline[] = { + INTEL_PLANE_CB_CSC_FF, INTEL_PLANE_CB_PRE_CSC_LUT, INTEL_PLANE_CB_CSC, INTEL_PLANE_CB_3DLUT, @@ -38,6 +39,7 @@ static const enum intel_color_block xe3plpd_primary_plane_pipeline[] = { }; static const enum intel_color_block hdr_plane_pipeline[] = { + INTEL_PLANE_CB_CSC_FF, INTEL_PLANE_CB_PRE_CSC_LUT, INTEL_PLANE_CB_CSC, INTEL_PLANE_CB_POST_CSC_LUT, diff --git a/drivers/gpu/drm/i915/display/skl_universal_plane.c b/drivers/gpu/drm/i915/display/skl_universal_plane.c index 07d9ab5a378641..2636eba2ba465a 100644 --- a/drivers/gpu/drm/i915/display/skl_universal_plane.c +++ b/drivers/gpu/drm/i915/display/skl_universal_plane.c @@ -1622,7 +1622,8 @@ icl_plane_update_noarm(struct intel_dsb *dsb, intel_de_write_dsb(display, dsb, PLANE_COLOR_CTL(pipe, plane_id), plane_color_ctl); - if (fb->format->is_yuv && icl_is_hdr_plane(display, plane_id)) + if (icl_is_hdr_plane(display, plane_id) && + (fb->format->is_yuv || plane_state->hw.csc_ff_enable)) icl_program_input_csc(dsb, plane, plane_state); skl_write_plane_wm(dsb, plane, crtc_state); From 820b255537d98d888eb36b44b8edc78c6f7ff3ab Mon Sep 17 00:00:00 2001 From: Lalit Shankar Chowdhury Date: Thu, 10 Sep 2026 15:58:45 +0000 Subject: [PATCH 0045/1352] drm/i915/fb: replace kmalloc_array with kmalloc_objs Replace the remaining kmalloc_array instance with kmalloc_objs. Signed-off-by: Lalit Shankar Chowdhury Link: https://patch.msgid.link/20260910155846.79513-1-lalitshankarch@gmail.com Signed-off-by: Jani Nikula --- drivers/gpu/drm/i915/display/intel_fb.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/gpu/drm/i915/display/intel_fb.c b/drivers/gpu/drm/i915/display/intel_fb.c index 1c0859d5f82986..c3e4c3f6d8cf01 100644 --- a/drivers/gpu/drm/i915/display/intel_fb.c +++ b/drivers/gpu/drm/i915/display/intel_fb.c @@ -596,7 +596,7 @@ u64 *intel_fb_plane_get_modifiers(struct intel_display *display, count++; } - list = kmalloc_array(count, sizeof(*list), GFP_KERNEL); + list = kmalloc_objs(*list, count); if (drm_WARN_ON(display->drm, !list)) return NULL; From 550b703fdbb2a2022faa75b4b11ab135241afbd9 Mon Sep 17 00:00:00 2001 From: Ankit Nautiyal Date: Mon, 7 Sep 2026 09:15:55 +0530 Subject: [PATCH 0046/1352] drm/i915/quirks: Limit eDP rate to HBR2 on HP Pavilion Plus 14-ew1 The eDP panel on the HP Pavilion Plus Laptop 14-ew1xxx advertises HBR3 while leaving the TPS4 support bit clear. The output however flickers, once link is trained with HBR3. Until commit 8c9006283e4b ("Revert "drm/i915/dp: Reject HBR3 when sink doesn't support TPS4"") such sinks were capped at HBR2 by the TPS4 check which incidentally kept this panel stable. That check was reverted because other panels legitimately need HBR3 without advertising TPS4, and the per-machine QUIRK_EDP_LIMIT_RATE_HBR2 was introduced to handle the affected machines instead. Add the machine to the list of devices that need the QUIRK_EDP_LIMIT_RATE_HBR2. Fixes: 8c9006283e4b ("Revert "drm/i915/dp: Reject HBR3 when sink doesn't support TPS4"") Reported-by: Annoy Cc Closes: https://gitlab.freedesktop.org/drm/i915/kernel/-/work_items/16743 Cc: # v6.18+ Tested-by: Annoy Cc Signed-off-by: Ankit Nautiyal Reviewed-by: Nemesa Garg Link: https://patch.msgid.link/20260907034555.2753846-1-ankit.k.nautiyal@intel.com --- drivers/gpu/drm/i915/display/intel_quirks.c | 3 +++ 1 file changed, 3 insertions(+) diff --git a/drivers/gpu/drm/i915/display/intel_quirks.c b/drivers/gpu/drm/i915/display/intel_quirks.c index 33245f44c0d501..7d7db774d8c7b9 100644 --- a/drivers/gpu/drm/i915/display/intel_quirks.c +++ b/drivers/gpu/drm/i915/display/intel_quirks.c @@ -257,6 +257,9 @@ static struct intel_quirk intel_quirks[] = { /* Dell XPS 13 7390 2-in-1 */ { 0x8a52, 0x1028, 0x08b0, quirk_edp_limit_rate_hbr2 }, + /* HP Pavilion Plus Laptop 14-ew1xxx */ + { 0x7d55, 0x103c, 0x8c31, quirk_edp_limit_rate_hbr2 }, + /* Xiaomi Book Pro 14 2026 */ { 0xb081, 0x1d72, 0x2424, quirk_disable_psr2 }, }; From 270681fbffbba2b6ccf5b7e3c34b8b563b36167f Mon Sep 17 00:00:00 2001 From: Imre Deak Date: Mon, 7 Sep 2026 20:44:12 +0300 Subject: [PATCH 0047/1352] drm/i915/dp_mst: Fix configuring FEC for a disconnected stream During an atomic commit after all the MST stream CRTC state is computed the driver ensures that the FEC is configured the same way (enabled or disabled) for all the streams on a given MST topology's link. drm_dp_mst_port_downstream_of_parent() used to determine if a stream is downstream of an MST port will return false if the whole topology is disconnected, since in that case it can't verify that the port/ parent_port passed to it is in the given MST topology. This is a problem during the above FEC configuration check, since intel_dp_mst_check_dsc_change()->get_pipes_downstream_of_mst_ports() will not return all the stream CRTCs/pipes for the topology as expected. Since passing parent_port==NULL to get_pipes_downstream_of_mst_port() is meant to return all the streams for the given topology (i.e. mst_mgr) skip checking if an MST port is downstream of a parent port in this case. This fixes a problem where the FEC configuration check explained above failed to ensure that all streams' FEC is configured the same way if the topology was disconnected, leading to a FEC state mismatch error. Cc: stable@vger.kernel.org # v6.10+ Closes: https://gitlab.freedesktop.org/drm/i915/kernel/-/work_items/16073 Closes: https://gitlab.freedesktop.org/drm/i915/kernel/-/work_items/16384 Reviewed-by: Luca Coelho Signed-off-by: Imre Deak Link: https://patch.msgid.link/20260907174413.741851-1-imre.deak@intel.com --- drivers/gpu/drm/i915/display/intel_dp_mst.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/drivers/gpu/drm/i915/display/intel_dp_mst.c b/drivers/gpu/drm/i915/display/intel_dp_mst.c index 6a869d0f6ffc05..20766b6c0bcfad 100644 --- a/drivers/gpu/drm/i915/display/intel_dp_mst.c +++ b/drivers/gpu/drm/i915/display/intel_dp_mst.c @@ -856,7 +856,8 @@ static u8 get_pipes_downstream_of_mst_port(struct intel_atomic_state *state, if (&connector->mst.dp->mst.mgr != mst_mgr) continue; - if (connector->mst.port != parent_port && + if (parent_port && + connector->mst.port != parent_port && !drm_dp_mst_port_downstream_of_parent(mst_mgr, connector->mst.port, parent_port)) From ee00f8fbb2b202002ab90834e02e9ba372773a36 Mon Sep 17 00:00:00 2001 From: Imre Deak Date: Mon, 7 Sep 2026 20:44:13 +0300 Subject: [PATCH 0048/1352] drm/i915/dp_mst: Fix configuring TUs for a disconnected stream During an atomic commit after all the MST stream CRTC state is computed the driver ensures that the sum of TUs of all the streams on a given MST topology link is within limits (63 for 8b10 and 64 for 128b132b). For a disconnected stream the DRM MST core's BW verification doesn't ensure this, because the topology state it uses for this is destroyed as soon as the stream (i.e. MST connector/port) is disconnected. The driver should keep the link state valid even for such disconnected streams, as userspace may disable them one-by-one only in a deferred way. Ensure the link's sum of TUs stays within limits in this case by simply reusing the maximum link BPP limit from the stream's (i.e. CRTC's) old state. The disconnection can happen either via the whole topology getting disconnected or via only the given stream's port getting disconnected. Check for both of these conditions separately, as a connector gets unregistered after a link disconnect event only in a deferred way. Cc: stable@vger.kernel.org # v6.10+ Link: https://gitlab.freedesktop.org/drm/i915/kernel/-/work_items/16073 Link: https://gitlab.freedesktop.org/drm/i915/kernel/-/work_items/16384 Reviewed-by: Luca Coelho Signed-off-by: Imre Deak Link: https://patch.msgid.link/20260907174413.741851-2-imre.deak@intel.com --- drivers/gpu/drm/i915/display/intel_dp_mst.c | 21 ++++++++++++++++++++ drivers/gpu/drm/i915/display/intel_dp_mst.h | 2 ++ drivers/gpu/drm/i915/display/intel_link_bw.c | 3 ++- 3 files changed, 25 insertions(+), 1 deletion(-) diff --git a/drivers/gpu/drm/i915/display/intel_dp_mst.c b/drivers/gpu/drm/i915/display/intel_dp_mst.c index 20766b6c0bcfad..0c362784afe4a2 100644 --- a/drivers/gpu/drm/i915/display/intel_dp_mst.c +++ b/drivers/gpu/drm/i915/display/intel_dp_mst.c @@ -2172,6 +2172,27 @@ bool intel_dp_mst_crtc_needs_modeset(struct intel_atomic_state *state, return false; } +bool intel_dp_mst_stream_disconnected(struct intel_atomic_state *state, + const struct intel_crtc *crtc) +{ + struct intel_connector *connector; + + connector = get_connector_in_state_for_crtc(state, crtc); + if (!connector) + return false; + + if (!connector->mst.dp) + return false; + + if (!connector->mst.dp->mst.mgr.mst_state) + return true; + + if (drm_connector_is_unregistered(&connector->base)) + return true; + + return false; +} + /** * intel_dp_mst_prepare_probe - Prepare an MST link for topology probing * @intel_dp: DP port object diff --git a/drivers/gpu/drm/i915/display/intel_dp_mst.h b/drivers/gpu/drm/i915/display/intel_dp_mst.h index ab09b487c6bb5e..8ce89242c05c95 100644 --- a/drivers/gpu/drm/i915/display/intel_dp_mst.h +++ b/drivers/gpu/drm/i915/display/intel_dp_mst.h @@ -28,6 +28,8 @@ int intel_dp_mst_atomic_check_link(struct intel_atomic_state *state, struct intel_link_bw_limits *limits); bool intel_dp_mst_crtc_needs_modeset(struct intel_atomic_state *state, struct intel_crtc *crtc); +bool intel_dp_mst_stream_disconnected(struct intel_atomic_state *state, + const struct intel_crtc *crtc); void intel_dp_mst_prepare_probe(struct intel_dp *intel_dp); bool intel_dp_mst_verify_dpcd_state(struct intel_dp *intel_dp); diff --git a/drivers/gpu/drm/i915/display/intel_link_bw.c b/drivers/gpu/drm/i915/display/intel_link_bw.c index b47474a3e9fec5..e71e76d6fd3e06 100644 --- a/drivers/gpu/drm/i915/display/intel_link_bw.c +++ b/drivers/gpu/drm/i915/display/intel_link_bw.c @@ -64,7 +64,8 @@ void intel_link_bw_init_limits(struct intel_atomic_state *state, intel_atomic_get_new_crtc_state(state, crtc); int forced_bpp_x16 = get_forced_link_bpp_x16(state, crtc); - if (state->base.duplicated && crtc_state) { + if ((state->base.duplicated && crtc_state) || + intel_dp_mst_stream_disconnected(state, crtc)) { limits->max_bpp_x16[pipe] = crtc_state->max_link_bpp_x16; if (intel_dsc_enabled_on_link(crtc_state)) limits->link_dsc_pipes |= BIT(pipe); From b9bb4134a9583a388704433eef22c82c49963585 Mon Sep 17 00:00:00 2001 From: Arnd Bergmann Date: Wed, 16 Sep 2026 13:10:49 +0200 Subject: [PATCH 0049/1352] soc: document merges Signed-off-by: Arnd Bergmann --- arch/arm/arm-soc-for-next-contents.txt | 22 ++++++++++++++++++++++ 1 file changed, 22 insertions(+) create mode 100644 arch/arm/arm-soc-for-next-contents.txt diff --git a/arch/arm/arm-soc-for-next-contents.txt b/arch/arm/arm-soc-for-next-contents.txt new file mode 100644 index 00000000000000..2f1c655335b588 --- /dev/null +++ b/arch/arm/arm-soc-for-next-contents.txt @@ -0,0 +1,22 @@ +soc/arm + +soc/dt + +soc/drivers + +soc/defconfig + +soc/late + +arm/fixes + (cfc1e9a543e3589ba200795b6e7fd8ef4314efdf) + git://git.kernel.org/pub/scm/linux/kernel/git/dinguyen/linux tags/socfpga_fix_for_v7.3 + socfpga/fix + git://git.kernel.org/pub/scm/linux/kernel/git/dinguyen/linux tags/socfpga_dts_fix_for_v7.3 + renesas/fixes + git://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel tags/renesas-fixes-for-v7.3-tag1 + scmi/fixes + git://git.kernel.org/pub/scm/linux/kernel/git/sudeep.holla/linux tags/scmi-ffa-fixes-7.3 + (406292fd75f95aa3010fec95b5beb5a8b7e3ba3a) + https://git.kernel.org/pub/scm/linux/kernel/git/amlogic/linux tags/amlogic-fixes-v7.3-rc + From 4ffdb772716e4279d63dfaaadf965da73e799aeb Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Tue, 15 Sep 2026 19:06:20 +0300 Subject: [PATCH 0050/1352] drm/i915/dp: use EXPORT_SYMBOL_IF_KUNIT() for kunit helpers Use EXPORT_SYMBOL_IF_KUNIT() instead of the regular EXPORT_SYMBOL() to export the symbols to the kunit namespace. Otherwise, the symbols get exported for all the kernel to see, and the corresponding MODULE_IMPORT_NS("EXPORTED_FOR_KUNIT_TESTING") in the tests is meaningless. Fixes: 2eb9982ff179 ("drm/i915/kunit: Export link training and caps funcs for testing") Cc: Imre Deak Reviewed-by: Imre Deak Link: https://patch.msgid.link/20260915160620.779372-1-jani.nikula@intel.com Signed-off-by: Jani Nikula --- drivers/gpu/drm/i915/display/intel_dp_link_caps.c | 6 ++++-- drivers/gpu/drm/i915/display/intel_dp_link_training.c | 4 ++-- 2 files changed, 6 insertions(+), 4 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_dp_link_caps.c b/drivers/gpu/drm/i915/display/intel_dp_link_caps.c index 98657aa4d3d580..abec3e2519ca67 100644 --- a/drivers/gpu/drm/i915/display/intel_dp_link_caps.c +++ b/drivers/gpu/drm/i915/display/intel_dp_link_caps.c @@ -3,6 +3,8 @@ * Copyright © 2026 Intel Corporation */ +#include + #include #include #include @@ -1302,14 +1304,14 @@ void intel_dp_link_caps_cleanup(struct intel_dp_link_caps *link_caps) const struct intel_dp_link_caps_test_ops i915_display_dp_link_caps_test_ops = { INTEL_DP_LINK_CAPS_TEST_OPS_INIT }; -EXPORT_SYMBOL(i915_display_dp_link_caps_test_ops); +EXPORT_SYMBOL_IF_KUNIT(i915_display_dp_link_caps_test_ops); #else const struct intel_dp_link_caps_test_ops intel_display_dp_link_caps_test_ops = { INTEL_DP_LINK_CAPS_TEST_OPS_INIT }; -EXPORT_SYMBOL(intel_display_dp_link_caps_test_ops); +EXPORT_SYMBOL_IF_KUNIT(intel_display_dp_link_caps_test_ops); #endif /* I915 */ diff --git a/drivers/gpu/drm/i915/display/intel_dp_link_training.c b/drivers/gpu/drm/i915/display/intel_dp_link_training.c index cb92cff906146c..9a692f4fdfee68 100644 --- a/drivers/gpu/drm/i915/display/intel_dp_link_training.c +++ b/drivers/gpu/drm/i915/display/intel_dp_link_training.c @@ -2825,14 +2825,14 @@ void intel_dp_link_training_cleanup(struct intel_dp_link_training *link_training const struct intel_dp_link_training_test_ops i915_display_dp_link_training_test_ops = { INTEL_DP_LINK_TRAINING_TEST_OPS_INIT }; -EXPORT_SYMBOL(i915_display_dp_link_training_test_ops); +EXPORT_SYMBOL_IF_KUNIT(i915_display_dp_link_training_test_ops); #else const struct intel_dp_link_training_test_ops intel_display_dp_link_training_test_ops = { INTEL_DP_LINK_TRAINING_TEST_OPS_INIT }; -EXPORT_SYMBOL(intel_display_dp_link_training_test_ops); +EXPORT_SYMBOL_IF_KUNIT(intel_display_dp_link_training_test_ops); #endif /* I915 */ From 41dae8ac5a3e1709ccba9dc5bac184826711c03f Mon Sep 17 00:00:00 2001 From: Vinod Govindapillai Date: Tue, 15 Sep 2026 12:25:13 +0300 Subject: [PATCH 0051/1352] drm/xe/pm: introduce PM PME support MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Introduce PME support for PME capable devices. Whether device is PME capable is assessed during PCI probe routine. And the whether PME is enabled for a specific context is assessed during the PM runtime suspend call if the device is PME capable. If the PME is enabled, HPDs can generate PME which in turn call the runtime resume call and do the wakeup routines. Till now the driver was relying on HPD polling to wakeup in case of any HPDs. HPD polling can be avoided in platforms with PME support and instead rely on this PCI PME for HPD induced wakeup. v2: access functions for xe.pme.enabled status and clear the pme. enabled in case of error in xe_pm_runtime_suspend() v3: use the local pme_enabled flag to clear the device wakeup incase of error v4: call the devm_device_init_wakeup() only if device is PME capable to avoid false wakeup capable reporting in platforms where PME is not supported. Squash the patch which handled the lockdep deadlock handling Bspec: 52979, 52980, 68857, 68867, 68970 Assisted-by: GitHub_Copilot:claude-opus-5 Signed-off-by: Vinod Govindapillai Reviewed-by: Jouni Högander Link: https://patch.msgid.link/20260915092518.639448-2-vinod.govindapillai@intel.com Signed-off-by: Rodrigo Vivi --- drivers/gpu/drm/xe/xe_device_types.h | 13 ++++++++ drivers/gpu/drm/xe/xe_pci.c | 22 +++++++++++++- drivers/gpu/drm/xe/xe_pm.c | 45 ++++++++++++++++++++++++++++ drivers/gpu/drm/xe/xe_pm.h | 2 ++ 4 files changed, 81 insertions(+), 1 deletion(-) diff --git a/drivers/gpu/drm/xe/xe_device_types.h b/drivers/gpu/drm/xe/xe_device_types.h index 180d450a6deb9c..a489a8d73991e4 100644 --- a/drivers/gpu/drm/xe/xe_device_types.h +++ b/drivers/gpu/drm/xe/xe_device_types.h @@ -453,6 +453,19 @@ struct xe_device { struct mutex lock; } d3cold; + /** @pme: Encapsulate pme related stuff */ + struct { + /** @pme.capable: Indicates if device is PME capable */ + bool capable; + + /** @pme.enabled: + * + * Indicates if PME is enabled - depends on user controllable + * sysfs interface as well + */ + bool enabled; + } pme; + /** @pm_notifier: Our PM notifier to perform actions in response to various PM events. */ struct notifier_block pm_notifier; /** @pm_block: Completion to block validating tasks on suspend / hibernate prepare */ diff --git a/drivers/gpu/drm/xe/xe_pci.c b/drivers/gpu/drm/xe/xe_pci.c index 1e04e8ef2611fe..b134d90c01a1c0 100644 --- a/drivers/gpu/drm/xe/xe_pci.c +++ b/drivers/gpu/drm/xe/xe_pci.c @@ -1380,6 +1380,8 @@ static int xe_pci_runtime_suspend(struct device *dev) { struct pci_dev *pdev = to_pci_dev(dev); struct xe_device *xe = pdev_to_xe_device(pdev); + unsigned int flags; + bool pme_enabled; int err; /* @@ -1391,9 +1393,25 @@ static int xe_pci_runtime_suspend(struct device *dev) xe_assert(xe, !IS_SRIOV_VF(xe)); xe_assert(xe, !pci_num_vf(pdev)); + flags = memalloc_noreclaim_save(); + pme_enabled = xe->pme.capable && !xe->d3cold.allowed && + pci_enable_wake(pdev, PCI_D3hot, true) == 0; + memalloc_noreclaim_restore(flags); + + xe_pm_update_pme_enabled(xe, pme_enabled); + err = xe_pm_runtime_suspend(xe); - if (err) + if (err) { + if (pme_enabled) { + flags = memalloc_noreclaim_save(); + pci_enable_wake(pdev, PCI_D3hot, false); + memalloc_noreclaim_restore(flags); + + xe_pm_update_pme_enabled(xe, false); + } + return err; + } pci_save_state(pdev); @@ -1422,6 +1440,8 @@ static int xe_pci_runtime_resume(struct device *dev) pci_restore_state(pdev); + xe_pm_update_pme_enabled(xe, false); + if (xe->d3cold.allowed) { err = pci_enable_device(pdev); if (err) diff --git a/drivers/gpu/drm/xe/xe_pm.c b/drivers/gpu/drm/xe/xe_pm.c index f517bf453b54ca..e71bb90c9b80b0 100644 --- a/drivers/gpu/drm/xe/xe_pm.c +++ b/drivers/gpu/drm/xe/xe_pm.c @@ -78,6 +78,8 @@ * management (RPS). */ +#define HAS_PM_PME_SUPPORT(xe) (GRAPHICS_VERx100(xe) >= 3500) + #ifdef CONFIG_LOCKDEP static struct lockdep_map xe_pm_runtime_d3cold_map = { .name = "xe_rpm_d3cold_map" @@ -384,6 +386,14 @@ int xe_pm_init_early(struct xe_device *xe) } ALLOW_ERROR_INJECTION(xe_pm_init_early, ERRNO); /* See xe_pci_probe() */ +static bool xe_pm_pci_pme_capable(struct xe_device *xe) +{ + struct pci_dev *pdev = to_pci_dev(xe->drm.dev); + + return HAS_PM_PME_SUPPORT(xe) ? + pci_pme_capable(pdev, PCI_D3hot) : false; +} + /** * xe_pm_probe() - Initialize Xe Power Management * @xe: the &xe_device instance @@ -397,6 +407,16 @@ int xe_pm_probe(struct xe_device *xe) xe->d3cold.capable = xe_pm_pci_d3cold_capable(xe); xe_dbg(xe, "d3cold: capable=%s\n", str_yes_no(xe->d3cold.capable)); + xe->pme.capable = xe_pm_pci_pme_capable(xe); + xe_dbg(xe, "pme: capable=%s\n", str_yes_no(xe->pme.capable)); + + if (xe->pme.capable) { + int err = devm_device_init_wakeup(xe->drm.dev); + + if (err) + return err; + } + return 0; } @@ -650,6 +670,7 @@ int xe_pm_runtime_suspend(struct xe_device *xe) return 0; out_resume: + xe_pm_update_pme_enabled(xe, false); xe_display_pm_runtime_resume(xe); xe_pxp_pm_resume(xe->pxp); out: @@ -993,6 +1014,30 @@ int xe_pm_set_vram_threshold(struct xe_device *xe, u32 threshold) return 0; } +/** + * xe_pm_pme_enabled - get the current status of PME enabled + * @xe: xe device instance + * + * Returns: True if PME is enabled, false otherwise. + */ +bool xe_pm_pme_enabled(struct xe_device *xe) +{ + return xe->pme.enabled; +} + +/** + * xe_pm_update_pme_enabled - Update the PME enabled state + * @xe: xe device instance + * @status: New PME enabled status + * + * Called during runtime suspend / resume. Status is set to True if PME is + * enabled during runtime_suspend. Cleared on runtime_resume. + */ +void xe_pm_update_pme_enabled(struct xe_device *xe, bool status) +{ + xe->pme.enabled = status; +} + /** * xe_pm_d3cold_allowed_toggle - Check conditions to toggle d3cold.allowed * @xe: xe device instance diff --git a/drivers/gpu/drm/xe/xe_pm.h b/drivers/gpu/drm/xe/xe_pm.h index 6d5ab09cb7696f..037c73ea434208 100644 --- a/drivers/gpu/drm/xe/xe_pm.h +++ b/drivers/gpu/drm/xe/xe_pm.h @@ -33,6 +33,8 @@ bool xe_pm_runtime_resume_and_get(struct xe_device *xe); void xe_pm_assert_unbounded_bridge(struct xe_device *xe); int xe_pm_set_vram_threshold(struct xe_device *xe, u32 threshold); void xe_pm_d3cold_allowed_toggle(struct xe_device *xe); +bool xe_pm_pme_enabled(struct xe_device *xe); +void xe_pm_update_pme_enabled(struct xe_device *xe, bool status); bool xe_rpm_reclaim_safe(const struct xe_device *xe); struct task_struct *xe_pm_read_callback_task(struct xe_device *xe); int xe_pm_block_on_suspend(struct xe_device *xe); From 90e07834704f29e63ecfa598d2167ed9cbfb0a98 Mon Sep 17 00:00:00 2001 From: Vinod Govindapillai Date: Tue, 15 Sep 2026 12:25:14 +0300 Subject: [PATCH 0052/1352] drm/i915: add pme_enabled() to the parent interface MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit PME capabiliy need to be assessed very early during runtime suspend routine to avoid resetting HPDs during IRQ resets. As display runtime pm routines are being executed in the independent intel display driver entry points, we need to have independent access to the pme status as well. Add a provision to query the optional pme_enabled to the parent interface so that it could be called independently based on xe/i915's pme_enabled() implementation. Assisted-by: GitHub_Copilot:claude-opus-5 Signed-off-by: Vinod Govindapillai Reviewed-by: Jouni Högander Link: https://patch.msgid.link/20260915092518.639448-3-vinod.govindapillai@intel.com Signed-off-by: Rodrigo Vivi --- drivers/gpu/drm/i915/display/intel_display_rpm.c | 7 +++++++ drivers/gpu/drm/i915/display/intel_display_rpm.h | 1 + include/drm/intel/display_parent_interface.h | 1 + 3 files changed, 9 insertions(+) diff --git a/drivers/gpu/drm/i915/display/intel_display_rpm.c b/drivers/gpu/drm/i915/display/intel_display_rpm.c index 0a331f89b4db5c..9927119a0fd19d 100644 --- a/drivers/gpu/drm/i915/display/intel_display_rpm.c +++ b/drivers/gpu/drm/i915/display/intel_display_rpm.c @@ -46,6 +46,13 @@ bool intel_display_rpm_suspended(struct intel_display *display) return display->parent->rpm->suspended(display->drm); } +bool intel_display_rpm_pme_enabled(struct intel_display *display) +{ + const struct intel_display_rpm_interface *rpm = display->parent->rpm; + + return rpm->pme_enabled && rpm->pme_enabled(display->drm); +} + void assert_display_rpm_held(struct intel_display *display) { display->parent->rpm->assert_held(display->drm); diff --git a/drivers/gpu/drm/i915/display/intel_display_rpm.h b/drivers/gpu/drm/i915/display/intel_display_rpm.h index 6ef48515f84bbd..5d9a1cdddfa71f 100644 --- a/drivers/gpu/drm/i915/display/intel_display_rpm.h +++ b/drivers/gpu/drm/i915/display/intel_display_rpm.h @@ -21,6 +21,7 @@ void intel_display_rpm_put(struct intel_display *display, struct ref_tracker *wa /* Only for special cases. */ bool intel_display_rpm_suspended(struct intel_display *display); +bool intel_display_rpm_pme_enabled(struct intel_display *display); void assert_display_rpm_held(struct intel_display *display); void intel_display_rpm_assert_block(struct intel_display *display); diff --git a/include/drm/intel/display_parent_interface.h b/include/drm/intel/display_parent_interface.h index 5e44c022d1aee2..f36134e89cb6d1 100644 --- a/include/drm/intel/display_parent_interface.h +++ b/include/drm/intel/display_parent_interface.h @@ -196,6 +196,7 @@ struct intel_display_rpm_interface { void (*put_unchecked)(const struct drm_device *drm); bool (*suspended)(const struct drm_device *drm); + bool (*pme_enabled)(const struct drm_device *drm); /* Optional */ void (*assert_held)(const struct drm_device *drm); void (*assert_block)(const struct drm_device *drm); void (*assert_unblock)(const struct drm_device *drm); From e468a2117f413bea664b096b21521c5faa9606ea Mon Sep 17 00:00:00 2001 From: Vinod Govindapillai Date: Tue, 15 Sep 2026 12:25:15 +0300 Subject: [PATCH 0053/1352] drm/i915/xe: plug the pme_enabed implementation for xe MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Plug the pme_enabled query for xe. It will return the current PME status for the device. For supported platforms, PME is enabled during runtime suspend calls if the device is PME is capable. v2: use the xe_pm_pme_enabled() Assisted-by: GitHub_Copilot:claude-opus-5 Signed-off-by: Vinod Govindapillai Reviewed-by: Jouni Högander Link: https://patch.msgid.link/20260915092518.639448-4-vinod.govindapillai@intel.com Signed-off-by: Rodrigo Vivi --- drivers/gpu/drm/xe/display/xe_display_rpm.c | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/drivers/gpu/drm/xe/display/xe_display_rpm.c b/drivers/gpu/drm/xe/display/xe_display_rpm.c index 548b503aa02427..57bcfb41bddab2 100644 --- a/drivers/gpu/drm/xe/display/xe_display_rpm.c +++ b/drivers/gpu/drm/xe/display/xe_display_rpm.c @@ -45,6 +45,11 @@ static bool xe_display_rpm_suspended(const struct drm_device *drm) return pm_runtime_suspended(xe->drm.dev); } +static bool xe_display_rpm_pme_enabled(const struct drm_device *drm) +{ + return xe_pm_pme_enabled(to_xe_device(drm)); +} + static void xe_display_rpm_assert_held(const struct drm_device *drm) { /* FIXME */ @@ -69,6 +74,7 @@ const struct intel_display_rpm_interface xe_display_rpm_interface = { .put_raw = xe_display_rpm_put, .put_unchecked = xe_display_rpm_put_unchecked, .suspended = xe_display_rpm_suspended, + .pme_enabled = xe_display_rpm_pme_enabled, .assert_held = xe_display_rpm_assert_held, .assert_block = xe_display_rpm_assert_block, .assert_unblock = xe_display_rpm_assert_unblock From 56aa1f0f9465f34542f7691a42b60614cbfac437 Mon Sep 17 00:00:00 2001 From: Vinod Govindapillai Date: Tue, 15 Sep 2026 12:25:16 +0300 Subject: [PATCH 0054/1352] drm/i915/irq: conditional HPD IRQ resets based on PME capability MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit If a device supports generating PME from HPDs, resetting HPD IRQs will be counter productive as HPDs itself will be lost. During suspend routines, all the IRQs are reset. So if the device is capable of generating PME rom HPDs, keep the HPD related IRQs from reset based on the PME capability of the device on a target power state. PME capability will be assessed and updated separately. v2: change keep_hpd to reset_hpd v3: use intel_display_rpm_pme_enabled() directly (JaniN) v4: remove the unused include file from the previous revision Bspec: 52979, 52980, 68857, 68867, 68970 Assisted-by: GitHub_Copilot:claude-opus-5 Signed-off-by: Vinod Govindapillai Reviewed-by: Jouni Högander Link: https://patch.msgid.link/20260915092518.639448-5-vinod.govindapillai@intel.com Signed-off-by: Rodrigo Vivi --- .../gpu/drm/i915/display/intel_display_irq.c | 18 +++++++++++------- 1 file changed, 11 insertions(+), 7 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_display_irq.c b/drivers/gpu/drm/i915/display/intel_display_irq.c index a59b75830bd137..8ce30112de9743 100644 --- a/drivers/gpu/drm/i915/display/intel_display_irq.c +++ b/drivers/gpu/drm/i915/display/intel_display_irq.c @@ -2217,8 +2217,10 @@ static void gen11_display_irq_reset(struct intel_display *display) enum pipe pipe; u32 trans_mask = BIT(TRANSCODER_A) | BIT(TRANSCODER_B) | BIT(TRANSCODER_C) | BIT(TRANSCODER_D); + bool reset_hpd = !intel_display_rpm_pme_enabled(display); - intel_de_write(display, GEN11_DISPLAY_INT_CTL, 0); + if (reset_hpd) + intel_de_write(display, GEN11_DISPLAY_INT_CTL, 0); if (DISPLAY_VER(display) >= 12) { enum transcoder trans; @@ -2250,13 +2252,15 @@ static void gen11_display_irq_reset(struct intel_display *display) irq_reset(display, GEN8_DE_PORT_IRQ_REGS); irq_reset(display, GEN8_DE_MISC_IRQ_REGS); - if (DISPLAY_VER(display) >= 14) - irq_reset(display, PICAINTERRUPT_IRQ_REGS); - else - irq_reset(display, GEN11_DE_HPD_IRQ_REGS); + if (reset_hpd) { + if (DISPLAY_VER(display) >= 14) + irq_reset(display, PICAINTERRUPT_IRQ_REGS); + else + irq_reset(display, GEN11_DE_HPD_IRQ_REGS); - if (INTEL_PCH_TYPE(display) >= PCH_ICP) - irq_reset(display, SDE_IRQ_REGS); + if (INTEL_PCH_TYPE(display) >= PCH_ICP) + irq_reset(display, SDE_IRQ_REGS); + } } void gen8_irq_power_well_post_enable(struct intel_display *display, From f6d6d0ba4e9f7d3c5e80b7fa98b903c213e00e69 Mon Sep 17 00:00:00 2001 From: Vinod Govindapillai Date: Tue, 15 Sep 2026 12:25:17 +0300 Subject: [PATCH 0055/1352] drm/i915/hotplug: avoid HPD polling if the device is PME capable MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit In supported devices, HPDs can generate PME and which in turn can invoke runtime resume calls for xe. No need to keep the HPD polling in such PME capable devices. v2: use intel_display_rpm_pme_enabled() directly (JaniN) Bspec: 52979, 52980, 68857, 68867, 68970 Assisted-by: GitHub_Copilot:claude-opus-5 Signed-off-by: Vinod Govindapillai Reviewed-by: Jouni Högander Link: https://patch.msgid.link/20260915092518.639448-6-vinod.govindapillai@intel.com Signed-off-by: Rodrigo Vivi --- drivers/gpu/drm/i915/display/intel_hotplug.c | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/drivers/gpu/drm/i915/display/intel_hotplug.c b/drivers/gpu/drm/i915/display/intel_hotplug.c index 970aa95ee344ab..d095eb13dc06a7 100644 --- a/drivers/gpu/drm/i915/display/intel_hotplug.c +++ b/drivers/gpu/drm/i915/display/intel_hotplug.c @@ -867,6 +867,11 @@ void intel_hpd_poll_enable(struct intel_display *display) if (!HAS_DISPLAY(display) || !intel_display_device_enabled(display)) return; + if (intel_display_rpm_pme_enabled(display)) { + drm_dbg_kms(display->drm, "PME wake capable device, skipping HPD polling.\n"); + return; + } + WRITE_ONCE(display->hotplug.poll_enabled, true); /* From 908f1be9c34392d858c5159851421a3b9b954a17 Mon Sep 17 00:00:00 2001 From: Arun R Murthy Date: Tue, 15 Sep 2026 09:53:00 +0530 Subject: [PATCH 0056/1352] drm/i915/dp: On DPCD init wake the DPRx for eDP Its observed that on AUX_CH failure, even if the retry is increased to 1000, it does not succeed. Either the command might be wrong or sink in an unknown/sleep state can cause this. So try waking the sink device. Before reading the DPCD caps wake the sink for eDP. v2: Use poll_timeout_us (Jani N) Add the reason, why this change is required (Ville) v3: Wake sink only for eDP Remove the dpcd probe set to true/false in wake_sink (Imre) v4: make edp_wake_sinc() static and dpcd_readb -> dpcd_reab_byte (Suraj) v5: Drop poll_timeout_us. drm_dp_dpcd_read_byte() already retries the AUX transaction internally (up to ~16ms), so the outer 6ms poll could never iterate and only added a pre-read delay on the hot path; read DP_SET_POWER once instead. Wake the sink before reading the DPCD caps, otherwise a sleeping sink fails the caps read and returns early before the wake ever runs. (Suraj, Jani, Sashiko) v6: Mask and write DP_SET_POWER (Sashiko) Closes: https://gitlab.freedesktop.org/drm/i915/kernel/-/issues/4391 Closes: https://gitlab.freedesktop.org/drm/i915/kernel/-/work_items/16654 Signed-off-by: Arun R Murthy Reviewed-by: Suraj Kandpal Signed-off-by: Suraj Kandpal Link: https://patch.msgid.link/20260915042300.2201757-1-arun.r.murthy@intel.com --- drivers/gpu/drm/i915/display/intel_dp.c | 38 +++++++++++++++++++++++++ 1 file changed, 38 insertions(+) diff --git a/drivers/gpu/drm/i915/display/intel_dp.c b/drivers/gpu/drm/i915/display/intel_dp.c index 0cd5e6b5034cfe..c44d584cc07a75 100644 --- a/drivers/gpu/drm/i915/display/intel_dp.c +++ b/drivers/gpu/drm/i915/display/intel_dp.c @@ -4779,6 +4779,38 @@ intel_edp_set_sink_rates(struct intel_dp *intel_dp) intel_edp_set_data_override_rates(intel_dp); } +static void intel_edp_wake_sink(struct intel_dp *intel_dp) +{ + u8 value = 0; + int ret; + + /* + * Read the current sink power state. drm_dp_dpcd_read_byte() already + * retries the AUX transaction internally, so a single read suffices. + * First commercial eDP panels are Ver1.0 or 1.1, on which DPCD + * DP_SET_POWER is supported. + */ + ret = drm_dp_dpcd_read_byte(&intel_dp->aux, DP_SET_POWER, &value); + + /* + * If the AUX read failed the sink may be asleep and not responding, + * or it read back D3; in either case wake it up to D0. + * In case of AUX read failure which is usually a POR case, the + * remaining bits of register 0x600 is set to '0' on POR. So a bare + * write should be fine. + */ + if (ret < 0 || value == DP_SET_POWER_D3) { + value &= ~DP_SET_POWER_MASK; + value |= DP_SET_POWER_D0; + drm_dp_dpcd_write_byte(&intel_dp->aux, DP_SET_POWER, + value); + /* After setting to D0 need a min of 1ms to wake (Spec DP2.1 sec 2.3.1.2) */ + fsleep(1000); + drm_dp_dpcd_write_byte(&intel_dp->aux, DP_SET_POWER, + value); + } +} + static bool intel_edp_init_dpcd(struct intel_dp *intel_dp, struct intel_connector *connector) { @@ -4789,6 +4821,12 @@ intel_edp_init_dpcd(struct intel_dp *intel_dp, struct intel_connector *connector /* this function is meant to be called only once */ drm_WARN_ON(display->drm, intel_dp->dpcd[DP_DPCD_REV] != 0); + /* + * Spec DP2.1 Section 3.5.2.16 page 966. + * Also if sink is asleep, this will wake the sink. + */ + intel_edp_wake_sink(intel_dp); + if (drm_dp_read_dpcd_caps(&intel_dp->aux, intel_dp->dpcd) != 0) return false; From c1070e44033a9add45fdf7519afedb2fd6826ce8 Mon Sep 17 00:00:00 2001 From: Aditya Prakash Srivastava Date: Mon, 14 Sep 2026 09:31:51 +0000 Subject: [PATCH 0057/1352] xfs: prevent close() from hanging on frozen filesystems When a file is closed, xfs_file_release() attempts to trim speculative post-EOF blocks. This requires allocating a transaction, which blocks indefinitely if the filesystem is frozen. Fix the hang by wrapping the preallocation cleanup block with sb_start_write_trylock() and xfs_ilock_nowait() to bypass the trim best-effort when the filesystem is frozen or locking fails. Suggested-by: Darrick J. Wong Signed-off-by: Aditya Prakash Srivastava Reviewed-by: Christoph Hellwig Reviewed-by: Darrick J. Wong Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_file.c | 22 +++++++++++++--------- 1 file changed, 13 insertions(+), 9 deletions(-) diff --git a/fs/xfs/xfs_file.c b/fs/xfs/xfs_file.c index d8202da15aca91..dd6d2e08faff9e 100644 --- a/fs/xfs/xfs_file.c +++ b/fs/xfs/xfs_file.c @@ -1872,17 +1872,21 @@ xfs_file_release( return 0; /* - * If we can't get the iolock just skip truncating the blocks past EOF - * because we could deadlock with the mmap_lock otherwise. We'll get - * another chance to drop them once the last reference to the inode is - * dropped, so we'll never leak blocks permanently. + * If we can't get the iolock or if the filesystem is frozen, just skip + * truncating the blocks past EOF because we could deadlock with the + * mmap_lock or hang the close() call. We'll get another chance to drop + * them once the last reference to the inode is dropped, so we'll never + * leak blocks permanently. */ if (!xfs_iflags_test(ip, XFS_EOFBLOCKS_RELEASED) && - xfs_ilock_nowait(ip, XFS_IOLOCK_EXCL)) { - if (xfs_can_free_eofblocks(ip) && - !xfs_iflags_test_and_set(ip, XFS_EOFBLOCKS_RELEASED)) - xfs_free_eofblocks(ip); - xfs_iunlock(ip, XFS_IOLOCK_EXCL); + sb_start_write_trylock(mp->m_super)) { + if (xfs_ilock_nowait(ip, XFS_IOLOCK_EXCL)) { + if (xfs_can_free_eofblocks(ip) && + !xfs_iflags_test_and_set(ip, XFS_EOFBLOCKS_RELEASED)) + xfs_free_eofblocks(ip); + xfs_iunlock(ip, XFS_IOLOCK_EXCL); + } + sb_end_write(mp->m_super); } return 0; From c9096d7595b940ed4aad4efadb502cabf273b6c3 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 9 Sep 2026 09:05:26 +0300 Subject: [PATCH 0058/1352] xfs: convert all !XFS_RT stubs to inline functions Avoid random compiler warnings about unused variables when CONFIG_XFS_RT is not set. Signed-off-by: Christoph Hellwig Reviewed-by: Carlos Maiolino Reviewed-by: Darrick J. Wong Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_rtalloc.h | 49 ++++++++++++++++++++++++++++++++++++++------ 1 file changed, 43 insertions(+), 6 deletions(-) diff --git a/fs/xfs/xfs_rtalloc.h b/fs/xfs/xfs_rtalloc.h index 78a690b489ed59..1d5108861acdca 100644 --- a/fs/xfs/xfs_rtalloc.h +++ b/fs/xfs/xfs_rtalloc.h @@ -46,10 +46,34 @@ int xfs_rtalloc_reinit_frextents(struct xfs_mount *mp); int xfs_growfs_check_rtgeom(const struct xfs_mount *mp, xfs_rfsblock_t dblocks, xfs_rfsblock_t rblocks, xfs_agblock_t rextsize); #else -# define xfs_growfs_rt(mp,in) (-ENOSYS) -# define xfs_rtalloc_reinit_frextents(m) (0) -# define xfs_rtmount_readsb(mp) (0) -# define xfs_rtmount_freesb(mp) ((void)0) +static inline int +xfs_growfs_rt( + struct xfs_mount *mp, + struct xfs_growfs_rt *in) +{ + return -ENOSYS; +} + +static inline int +xfs_rtalloc_reinit_frextents( + struct xfs_mount *mp) +{ + return 0; +} + +static inline int +xfs_rtmount_readsb( + struct xfs_mount *mp) +{ + return 0; +} + +static inline void +xfs_rtmount_freesb( + struct xfs_mount *mp) +{ +} + static inline int /* error */ xfs_rtmount_init( xfs_mount_t *mp) /* file system mount structure */ @@ -60,8 +84,21 @@ xfs_rtmount_init( xfs_warn(mp, "Not built with CONFIG_XFS_RT"); return -ENOSYS; } -# define xfs_rtmount_inodes(m) (((mp)->m_sb.sb_rblocks == 0)? 0 : (-ENOSYS)) -# define xfs_rtunmount_inodes(m) + +static inline int +xfs_rtmount_inodes( + struct xfs_mount *mp) +{ + if (mp->m_sb.sb_rblocks) + return -ENOSYS; + return 0; +} + +static inline void +xfs_rtunmount_inodes( + struct xfs_mount *mp) +{ +} static inline int xfs_growfs_check_rtgeom(const struct xfs_mount *mp, From 345a1f9929bda1f7047d1013e63fd1ca6c1f02c3 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:00 +0200 Subject: [PATCH 0059/1352] xfs: remove an outdated comment above xfs_file_ioctl The return code sign flipping at the method boundary is long gone in XFS, so remove this comment documenting an exception from it. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl.c | 6 ------ 1 file changed, 6 deletions(-) diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c index 96ca3e480cb9fe..4a8921f38a082f 100644 --- a/fs/xfs/xfs_ioctl.c +++ b/fs/xfs/xfs_ioctl.c @@ -1209,12 +1209,6 @@ xfs_ioctl_fs_counts( #define XFS_IOC_ALLOCSP64 _IOW ('X', 36, struct xfs_flock64) #define XFS_IOC_FREESP64 _IOW ('X', 37, struct xfs_flock64) -/* - * Note: some of the ioctl's return positive numbers as a - * byte count indicating success, such as readlink_by_handle. - * So we don't "sign flip" like most other routines. This means - * true errors need to be returned as a negative value. - */ long xfs_file_ioctl( struct file *filp, From da8c47c244985206a0dd975f82d90725f658240a Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:01 +0200 Subject: [PATCH 0060/1352] xfs: rename xfs_ioc_swapext to xfs_swapext The usual convention is that the ioc_ prefix is used for direct ioctl handlers that take a user pointer. The current xfs_ioc_swapext does not fit that pattern, so rename it. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl.c | 8 ++++---- fs/xfs/xfs_ioctl.h | 4 +--- fs/xfs/xfs_ioctl32.c | 2 +- 3 files changed, 6 insertions(+), 8 deletions(-) diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c index 4a8921f38a082f..314eeec0234171 100644 --- a/fs/xfs/xfs_ioctl.c +++ b/fs/xfs/xfs_ioctl.c @@ -962,10 +962,10 @@ xfs_ioc_getbmap( } int -xfs_ioc_swapext( - xfs_swapext_t *sxp) +xfs_swapext( + struct xfs_swapext *sxp) { - xfs_inode_t *ip, *tip; + struct xfs_inode *ip, *tip; /* Pull information for the target fd */ CLASS(fd, f)((int)sxp->sx_fdtarget); @@ -1341,7 +1341,7 @@ xfs_file_ioctl( error = mnt_want_write_file(filp); if (error) return error; - error = xfs_ioc_swapext(&sxp); + error = xfs_swapext(&sxp); mnt_drop_write_file(filp); return error; } diff --git a/fs/xfs/xfs_ioctl.h b/fs/xfs/xfs_ioctl.h index f5ed5cf9d3df65..e57d8f5148bf7f 100644 --- a/fs/xfs/xfs_ioctl.h +++ b/fs/xfs/xfs_ioctl.h @@ -10,9 +10,7 @@ struct xfs_bstat; struct xfs_ibulk; struct xfs_inogrp; -int -xfs_ioc_swapext( - xfs_swapext_t *sxp); +int xfs_swapext(struct xfs_swapext *sxp); extern int xfs_fileattr_get( diff --git a/fs/xfs/xfs_ioctl32.c b/fs/xfs/xfs_ioctl32.c index c66e192448a89d..250df2bbf21417 100644 --- a/fs/xfs/xfs_ioctl32.c +++ b/fs/xfs/xfs_ioctl32.c @@ -475,7 +475,7 @@ xfs_file_compat_ioctl( error = mnt_want_write_file(filp); if (error) return error; - error = xfs_ioc_swapext(&sxp); + error = xfs_swapext(&sxp); mnt_drop_write_file(filp); return error; } From 0018d49c3844c67ae74c0122b8f3d8136c2d38a9 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:02 +0200 Subject: [PATCH 0061/1352] xfs: rename xfs_ioctl_fs_counts to xfs_ioc_fs_counts Match the naming scheme of most other ioctl handlers. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c index 314eeec0234171..e0da8dfc9ae67d 100644 --- a/fs/xfs/xfs_ioctl.c +++ b/fs/xfs/xfs_ioctl.c @@ -1183,7 +1183,7 @@ xfs_ioctl_getset_resblocks( } static int -xfs_ioctl_fs_counts( +xfs_ioc_fs_counts( struct xfs_mount *mp, struct xfs_fsop_counts __user *uarg) { @@ -1347,7 +1347,7 @@ xfs_file_ioctl( } case XFS_IOC_FSCOUNTS: - return xfs_ioctl_fs_counts(mp, arg); + return xfs_ioc_fs_counts(mp, arg); case XFS_IOC_SET_RESBLKS: case XFS_IOC_GET_RESBLKS: From 5703154167aab81054cb4aee699d9143598a3802 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:03 +0200 Subject: [PATCH 0062/1352] xfs: rename xfs_ioctl_getset_resblocks to xfs_ioc_getset_resblocks Match the naming scheme of most other ioctl handlers. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c index e0da8dfc9ae67d..479947456b1f02 100644 --- a/fs/xfs/xfs_ioctl.c +++ b/fs/xfs/xfs_ioctl.c @@ -1144,7 +1144,7 @@ xfs_fs_eofblocks_from_user( } static int -xfs_ioctl_getset_resblocks( +xfs_ioc_getset_resblocks( struct file *filp, unsigned int cmd, void __user *arg) @@ -1351,7 +1351,7 @@ xfs_file_ioctl( case XFS_IOC_SET_RESBLKS: case XFS_IOC_GET_RESBLKS: - return xfs_ioctl_getset_resblocks(filp, cmd, arg); + return xfs_ioc_getset_resblocks(filp, cmd, arg); case XFS_IOC_FSGROWFSDATA: { struct xfs_growfs_data in; From 79417ad07f902f6402024c169585ec25eec80020 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:04 +0200 Subject: [PATCH 0063/1352] xfs: split out the handler for XFS_IOC_DIOINFO Split out a helper for XFS_IOC_DIOINFO to keep the stack variables out of xfs_file_ioctl and to clean up the main ioctl handler flow. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl.c | 47 +++++++++++++++++++++++++++------------------- 1 file changed, 28 insertions(+), 19 deletions(-) diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c index 479947456b1f02..48cee2bf6691ff 100644 --- a/fs/xfs/xfs_ioctl.c +++ b/fs/xfs/xfs_ioctl.c @@ -1200,6 +1200,32 @@ xfs_ioc_fs_counts( return 0; } +static int +xfs_ioc_dioinfo( + struct file *file, + void __user *arg) +{ + struct kstat st; + struct dioattr da; + int error; + + error = vfs_getattr(&file->f_path, &st, STATX_DIOALIGN, 0); + if (error) + return error; + + /* + * Some userspace directly feeds the return value to posix_memalign, + * which fails for values that are smaller than the pointer size. + * Round up the value to not break userspace. + */ + da.d_mem = roundup(st.dio_mem_align, sizeof(void *)); + da.d_miniosz = st.dio_offset_align; + da.d_maxiosz = INT_MAX & ~(da.d_miniosz - 1); + if (copy_to_user(arg, &da, sizeof(da))) + return -EFAULT; + return 0; +} + /* * These long-unused ioctls were removed from the official ioctl API in 5.17, * but retain these definitions so that we can log warnings about them. @@ -1238,26 +1264,9 @@ xfs_file_ioctl( "%s should use fallocate; XFS_IOC_{ALLOC,FREE}SP ioctl unsupported", current->comm); return -ENOTTY; - case XFS_IOC_DIOINFO: { - struct kstat st; - struct dioattr da; - - error = vfs_getattr(&filp->f_path, &st, STATX_DIOALIGN, 0); - if (error) - return error; - /* - * Some userspace directly feeds the return value to - * posix_memalign, which fails for values that are smaller than - * the pointer size. Round up the value to not break userspace. - */ - da.d_mem = roundup(st.dio_mem_align, sizeof(void *)); - da.d_miniosz = st.dio_offset_align; - da.d_maxiosz = INT_MAX & ~(da.d_miniosz - 1); - if (copy_to_user(arg, &da, sizeof(da))) - return -EFAULT; - return 0; - } + case XFS_IOC_DIOINFO: + return xfs_ioc_dioinfo(filp, arg); case XFS_IOC_FSBULKSTAT_SINGLE: case XFS_IOC_FSBULKSTAT: From 1300f918c972720129183493f1b49055af6f71e7 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:05 +0200 Subject: [PATCH 0064/1352] xfs: split out the handlers for XFS_IOC_.*HANDLE Split out helpers for XFS_IOC_.*HANDLE to keep the stack variables out of xfs_file_ioctl and to clean up the main ioctl handler flow. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl.c | 65 ++++++++++++++++++++++++++++++---------------- 1 file changed, 42 insertions(+), 23 deletions(-) diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c index 48cee2bf6691ff..a038df0196f70d 100644 --- a/fs/xfs/xfs_ioctl.c +++ b/fs/xfs/xfs_ioctl.c @@ -1226,6 +1226,42 @@ xfs_ioc_dioinfo( return 0; } +static int +xfs_ioc_find_handle( + unsigned int cmd, + void __user *arg) +{ + struct xfs_fsop_handlereq hreq; + + if (copy_from_user(&hreq, arg, sizeof(hreq))) + return -EFAULT; + return xfs_find_handle(cmd, &hreq); +} + +static int +xfs_ioc_open_by_handle( + struct file *file, + void __user *arg) +{ + struct xfs_fsop_handlereq hreq; + + if (copy_from_user(&hreq, arg, sizeof(hreq))) + return -EFAULT; + return xfs_open_by_handle(file, &hreq); +} + +static int +xfs_ioc_readlink_by_handle( + struct file *file, + void __user *arg) +{ + struct xfs_fsop_handlereq hreq; + + if (copy_from_user(&hreq, arg, sizeof(hreq))) + return -EFAULT; + return xfs_readlink_by_handle(file, &hreq); +} + /* * These long-unused ioctls were removed from the official ioctl API in 5.17, * but retain these definitions so that we can log warnings about them. @@ -1314,31 +1350,14 @@ xfs_file_ioctl( case XFS_IOC_FD_TO_HANDLE: case XFS_IOC_PATH_TO_HANDLE: - case XFS_IOC_PATH_TO_FSHANDLE: { - xfs_fsop_handlereq_t hreq; - - if (copy_from_user(&hreq, arg, sizeof(hreq))) - return -EFAULT; - return xfs_find_handle(cmd, &hreq); - } - case XFS_IOC_OPEN_BY_HANDLE: { - xfs_fsop_handlereq_t hreq; - - if (copy_from_user(&hreq, arg, sizeof(xfs_fsop_handlereq_t))) - return -EFAULT; - return xfs_open_by_handle(filp, &hreq); - } - - case XFS_IOC_READLINK_BY_HANDLE: { - xfs_fsop_handlereq_t hreq; - - if (copy_from_user(&hreq, arg, sizeof(xfs_fsop_handlereq_t))) - return -EFAULT; - return xfs_readlink_by_handle(filp, &hreq); - } + case XFS_IOC_PATH_TO_FSHANDLE: + return xfs_ioc_find_handle(cmd, arg); + case XFS_IOC_OPEN_BY_HANDLE: + return xfs_ioc_open_by_handle(filp, arg); + case XFS_IOC_READLINK_BY_HANDLE: + return xfs_ioc_readlink_by_handle(filp, arg); case XFS_IOC_ATTRLIST_BY_HANDLE: return xfs_attrlist_by_handle(filp, arg); - case XFS_IOC_ATTRMULTI_BY_HANDLE: return xfs_attrmulti_by_handle(filp, arg); From f7e07ea82396ea6e577bfc65d6cc7672021d7421 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:06 +0200 Subject: [PATCH 0065/1352] xfs: split out the handler for XFS_IOC_SWAPEXT Split out a helper for XFS_IOC_SWAPEXT to keep the stack variables out of xfs_file_ioctl and to clean up the main ioctl handler flow. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl.c | 33 +++++++++++++++++++++------------ 1 file changed, 21 insertions(+), 12 deletions(-) diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c index a038df0196f70d..ecee39e2ac1943 100644 --- a/fs/xfs/xfs_ioctl.c +++ b/fs/xfs/xfs_ioctl.c @@ -1262,6 +1262,25 @@ xfs_ioc_readlink_by_handle( return xfs_readlink_by_handle(file, &hreq); } +static int +xfs_ioc_swapext( + struct file *file, + void __user *arg) +{ + struct xfs_swapext sxp; + int error; + + if (copy_from_user(&sxp, arg, sizeof(sxp))) + return -EFAULT; + + error = mnt_want_write_file(file); + if (error) + return error; + error = xfs_swapext(&sxp); + mnt_drop_write_file(file); + return error; +} + /* * These long-unused ioctls were removed from the official ioctl API in 5.17, * but retain these definitions so that we can log warnings about them. @@ -1361,18 +1380,8 @@ xfs_file_ioctl( case XFS_IOC_ATTRMULTI_BY_HANDLE: return xfs_attrmulti_by_handle(filp, arg); - case XFS_IOC_SWAPEXT: { - struct xfs_swapext sxp; - - if (copy_from_user(&sxp, arg, sizeof(xfs_swapext_t))) - return -EFAULT; - error = mnt_want_write_file(filp); - if (error) - return error; - error = xfs_swapext(&sxp); - mnt_drop_write_file(filp); - return error; - } + case XFS_IOC_SWAPEXT: + return xfs_ioc_swapext(filp, arg); case XFS_IOC_FSCOUNTS: return xfs_ioc_fs_counts(mp, arg); From 71a7d9bdb8f4495fe37441bc0769cc91b40af227 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:07 +0200 Subject: [PATCH 0066/1352] xfs: split out the handlers for XFS_IOC_FSGROWFS* Split out helpers for XFS_IOC_FSGROWFS* to keep the stack variables out of xfs_file_ioctl and to clean up the main ioctl handler flow. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl.c | 107 ++++++++++++++++++++++++++++----------------- 1 file changed, 66 insertions(+), 41 deletions(-) diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c index ecee39e2ac1943..92bdf0e37f3fb8 100644 --- a/fs/xfs/xfs_ioctl.c +++ b/fs/xfs/xfs_ioctl.c @@ -1281,6 +1281,66 @@ xfs_ioc_swapext( return error; } +static int +xfs_ioc_growfs_data( + struct file *file, + struct xfs_mount *mp, + void __user *arg) +{ + struct xfs_growfs_data in; + int error; + + if (copy_from_user(&in, arg, sizeof(in))) + return -EFAULT; + + error = mnt_want_write_file(file); + if (error) + return error; + error = xfs_growfs_data(mp, &in); + mnt_drop_write_file(file); + return error; +} + +static int +xfs_ioc_growfs_log( + struct file *file, + struct xfs_mount *mp, + void __user *arg) +{ + struct xfs_growfs_log in; + int error; + + if (copy_from_user(&in, arg, sizeof(in))) + return -EFAULT; + + error = mnt_want_write_file(file); + if (error) + return error; + error = xfs_growfs_log(mp, &in); + mnt_drop_write_file(file); + return error; +} + +static int +xfs_ioc_growfs_rt( + struct file *file, + struct xfs_mount *mp, + void __user *arg) +{ + struct xfs_growfs_rt in; + int error; + + if (copy_from_user(&in, arg, sizeof(in))) + return -EFAULT; + + error = mnt_want_write_file(file); + if (error) + return error; + error = xfs_growfs_rt(mp, &in); + mnt_drop_write_file(file); + return error; +} + /* * These long-unused ioctls were removed from the official ioctl API in 5.17, * but retain these definitions so that we can log warnings about them. @@ -1390,47 +1450,12 @@ xfs_file_ioctl( case XFS_IOC_GET_RESBLKS: return xfs_ioc_getset_resblocks(filp, cmd, arg); - case XFS_IOC_FSGROWFSDATA: { - struct xfs_growfs_data in; - - if (copy_from_user(&in, arg, sizeof(in))) - return -EFAULT; - - error = mnt_want_write_file(filp); - if (error) - return error; - error = xfs_growfs_data(mp, &in); - mnt_drop_write_file(filp); - return error; - } - - case XFS_IOC_FSGROWFSLOG: { - struct xfs_growfs_log in; - - if (copy_from_user(&in, arg, sizeof(in))) - return -EFAULT; - - error = mnt_want_write_file(filp); - if (error) - return error; - error = xfs_growfs_log(mp, &in); - mnt_drop_write_file(filp); - return error; - } - - case XFS_IOC_FSGROWFSRT: { - xfs_growfs_rt_t in; - - if (copy_from_user(&in, arg, sizeof(in))) - return -EFAULT; - - error = mnt_want_write_file(filp); - if (error) - return error; - error = xfs_growfs_rt(mp, &in); - mnt_drop_write_file(filp); - return error; - } + case XFS_IOC_FSGROWFSDATA: + return xfs_ioc_growfs_data(filp, mp, arg); + case XFS_IOC_FSGROWFSLOG: + return xfs_ioc_growfs_log(filp, mp, arg); + case XFS_IOC_FSGROWFSRT: + return xfs_ioc_growfs_rt(filp, mp, arg); case XFS_IOC_GOINGDOWN: { uint32_t in; From 1e787c1e5777ba4448007437c86e6295e7242096 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:08 +0200 Subject: [PATCH 0067/1352] xfs: split out the handler for XFS_IOC_GOINGDOWN Split out a helper for XFS_IOC_GOINGDOWN to keep the stack variables out of xfs_file_ioctl and to clean up the main ioctl handler flow. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl.c | 28 ++++++++++++++++------------ 1 file changed, 16 insertions(+), 12 deletions(-) diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c index 92bdf0e37f3fb8..1580f8d1299ca5 100644 --- a/fs/xfs/xfs_ioctl.c +++ b/fs/xfs/xfs_ioctl.c @@ -1341,6 +1341,20 @@ xfs_ioc_growfs_rt( return error; } +static int +xfs_ioc_goingdown( + struct xfs_mount *mp, + uint32_t __user *arg) +{ + uint32_t in; + + if (!capable(CAP_SYS_ADMIN)) + return -EPERM; + if (get_user(in, arg)) + return -EFAULT; + return xfs_fs_goingdown(mp, in); +} + /* * These long-unused ioctls were removed from the official ioctl API in 5.17, * but retain these definitions so that we can log warnings about them. @@ -1457,18 +1471,8 @@ xfs_file_ioctl( case XFS_IOC_FSGROWFSRT: return xfs_ioc_growfs_rt(filp, mp, arg); - case XFS_IOC_GOINGDOWN: { - uint32_t in; - - if (!capable(CAP_SYS_ADMIN)) - return -EPERM; - - if (get_user(in, (uint32_t __user *)arg)) - return -EFAULT; - - return xfs_fs_goingdown(mp, in); - } - + case XFS_IOC_GOINGDOWN: + return xfs_ioc_goingdown(mp, arg); case XFS_IOC_ERROR_INJECTION: { xfs_error_injection_t in; From 809f7f37c83ce3bef0fcc6cc06e97fa27d98c13c Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:09 +0200 Subject: [PATCH 0068/1352] xfs: split out the handler for XFS_IOC_ERROR_INJECTION Split out a helper for XFS_IOC_ERROR_INJECTION to keep the stack variables out of xfs_file_ioctl and to clean up the main ioctl handler flow. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl.c | 29 ++++++++++++++++------------- 1 file changed, 16 insertions(+), 13 deletions(-) diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c index 1580f8d1299ca5..a226b33a74b517 100644 --- a/fs/xfs/xfs_ioctl.c +++ b/fs/xfs/xfs_ioctl.c @@ -1355,6 +1355,20 @@ xfs_ioc_goingdown( return xfs_fs_goingdown(mp, in); } +static int +xfs_ioc_error_injection( + struct xfs_mount *mp, + struct xfs_error_injection __user *arg) +{ + struct xfs_error_injection in; + + if (!capable(CAP_SYS_ADMIN)) + return -EPERM; + if (copy_from_user(&in, arg, sizeof(in))) + return -EFAULT; + return xfs_errortag_add(mp, in.errtag); +} + /* * These long-unused ioctls were removed from the official ioctl API in 5.17, * but retain these definitions so that we can log warnings about them. @@ -1473,22 +1487,11 @@ xfs_file_ioctl( case XFS_IOC_GOINGDOWN: return xfs_ioc_goingdown(mp, arg); - case XFS_IOC_ERROR_INJECTION: { - xfs_error_injection_t in; - - if (!capable(CAP_SYS_ADMIN)) - return -EPERM; - - if (copy_from_user(&in, arg, sizeof(in))) - return -EFAULT; - - return xfs_errortag_add(mp, in.errtag); - } - + case XFS_IOC_ERROR_INJECTION: + return xfs_ioc_error_injection(mp, arg); case XFS_IOC_ERROR_CLEARALL: if (!capable(CAP_SYS_ADMIN)) return -EPERM; - return xfs_errortag_clearall(mp); case XFS_IOC_FREE_EOFBLOCKS: { From 284fac26cced34cf9c3daf4471f21933b5d53ef9 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:10 +0200 Subject: [PATCH 0069/1352] xfs: split out the handler for XFS_IOC_FREE_EOFBLOCKS Split out a helper for XFS_IOC_FREE_EOFBLOCKS to keep the stack variables out of xfs_file_ioctl and to clean up the main ioctl handler flow. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl.c | 52 ++++++++++++++++++++++++++-------------------- 1 file changed, 29 insertions(+), 23 deletions(-) diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c index a226b33a74b517..10673cf7f2d6fd 100644 --- a/fs/xfs/xfs_ioctl.c +++ b/fs/xfs/xfs_ioctl.c @@ -1369,6 +1369,33 @@ xfs_ioc_error_injection( return xfs_errortag_add(mp, in.errtag); } +static int +xfs_ioc_free_eofblocks( + struct xfs_mount *mp, + struct xfs_fs_eofblocks __user *arg) +{ + struct xfs_fs_eofblocks eofb; + struct xfs_icwalk icw; + int error; + + if (!capable(CAP_SYS_ADMIN)) + return -EPERM; + if (xfs_is_readonly(mp)) + return -EROFS; + + if (copy_from_user(&eofb, arg, sizeof(eofb))) + return -EFAULT; + + error = xfs_fs_eofblocks_from_user(&eofb, &icw); + if (error) + return error; + + trace_xfs_ioc_free_eofblocks(mp, &icw, _RET_IP_); + + guard(super_write)(mp->m_super); + return xfs_blockgc_free_space(mp, &icw); +} + /* * These long-unused ioctls were removed from the official ioctl API in 5.17, * but retain these definitions so that we can log warnings about them. @@ -1388,7 +1415,6 @@ xfs_file_ioctl( struct xfs_inode *ip = XFS_I(inode); struct xfs_mount *mp = ip->i_mount; void __user *arg = (void __user *)p; - int error; trace_xfs_file_ioctl(ip); @@ -1494,28 +1520,8 @@ xfs_file_ioctl( return -EPERM; return xfs_errortag_clearall(mp); - case XFS_IOC_FREE_EOFBLOCKS: { - struct xfs_fs_eofblocks eofb; - struct xfs_icwalk icw; - - if (!capable(CAP_SYS_ADMIN)) - return -EPERM; - - if (xfs_is_readonly(mp)) - return -EROFS; - - if (copy_from_user(&eofb, arg, sizeof(eofb))) - return -EFAULT; - - error = xfs_fs_eofblocks_from_user(&eofb, &icw); - if (error) - return error; - - trace_xfs_ioc_free_eofblocks(mp, &icw, _RET_IP_); - - guard(super_write)(mp->m_super); - return xfs_blockgc_free_space(mp, &icw); - } + case XFS_IOC_FREE_EOFBLOCKS: + return xfs_ioc_free_eofblocks(mp, arg); case XFS_IOC_EXCHANGE_RANGE: return xfs_ioc_exchange_range(filp, arg); From d6040e8c186ac2be16ae969d4c6b1fad7d43577a Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:11 +0200 Subject: [PATCH 0070/1352] xfs: split out the handlers for XFS_IOC_FSGROWFS_*_32 Split out helpers for XFS_IOC_FSGROWFS_*_32 to keep the stack variables out of xfs_file_compat_ioctl and to clean up the main compat ioctl handler flow. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl32.c | 76 ++++++++++++++++++++++---------------------- 1 file changed, 38 insertions(+), 38 deletions(-) diff --git a/fs/xfs/xfs_ioctl32.c b/fs/xfs/xfs_ioctl32.c index 250df2bbf21417..2162eda017430f 100644 --- a/fs/xfs/xfs_ioctl32.c +++ b/fs/xfs/xfs_ioctl32.c @@ -44,26 +44,46 @@ xfs_compat_ioc_fsgeometry_v1( return 0; } -STATIC int -xfs_compat_growfs_data_copyin( - struct xfs_growfs_data *in, - compat_xfs_growfs_data_t __user *arg32) +static int +xfs_compat_ioc_growfs_data( + struct file *file, + struct xfs_mount *mp, + struct compat_xfs_growfs_data __user *arg32) { - if (get_user(in->newblocks, &arg32->newblocks) || - get_user(in->imaxpct, &arg32->imaxpct)) + struct xfs_growfs_data in = { }; + int error; + + if (get_user(in.newblocks, &arg32->newblocks) || + get_user(in.imaxpct, &arg32->imaxpct)) return -EFAULT; - return 0; + + error = mnt_want_write_file(file); + if (error) + return error; + error = xfs_growfs_data(mp, &in); + mnt_drop_write_file(file); + return error; } -STATIC int -xfs_compat_growfs_rt_copyin( - struct xfs_growfs_rt *in, - compat_xfs_growfs_rt_t __user *arg32) +static int +xfs_compat_ioc_growfs_rt( + struct file *file, + struct xfs_mount *mp, + struct compat_xfs_growfs_rt __user *arg32) { - if (get_user(in->newblocks, &arg32->newblocks) || - get_user(in->extsize, &arg32->extsize)) + struct xfs_growfs_rt in = {}; + int error; + + if (get_user(in.newblocks, &arg32->newblocks) || + get_user(in.extsize, &arg32->extsize)) return -EFAULT; - return 0; + + error = mnt_want_write_file(file); + if (error) + return error; + error = xfs_growfs_rt(mp, &in); + mnt_drop_write_file(file); + return error; } STATIC int @@ -434,30 +454,10 @@ xfs_file_compat_ioctl( #if defined(BROKEN_X86_ALIGNMENT) case XFS_IOC_FSGEOMETRY_V1_32: return xfs_compat_ioc_fsgeometry_v1(ip->i_mount, arg); - case XFS_IOC_FSGROWFSDATA_32: { - struct xfs_growfs_data in; - - if (xfs_compat_growfs_data_copyin(&in, arg)) - return -EFAULT; - error = mnt_want_write_file(filp); - if (error) - return error; - error = xfs_growfs_data(ip->i_mount, &in); - mnt_drop_write_file(filp); - return error; - } - case XFS_IOC_FSGROWFSRT_32: { - struct xfs_growfs_rt in; - - if (xfs_compat_growfs_rt_copyin(&in, arg)) - return -EFAULT; - error = mnt_want_write_file(filp); - if (error) - return error; - error = xfs_growfs_rt(ip->i_mount, &in); - mnt_drop_write_file(filp); - return error; - } + case XFS_IOC_FSGROWFSDATA_32: + return xfs_compat_ioc_growfs_data(filp, ip->i_mount, arg); + case XFS_IOC_FSGROWFSRT_32: + return xfs_compat_ioc_growfs_rt(filp, ip->i_mount, arg); #endif /* long changes size, but xfs only copiese out 32 bits */ case XFS_IOC_GETVERSION_32: From 3b26e116e828bfee07bc5443db172ef870bef160 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:12 +0200 Subject: [PATCH 0071/1352] xfs: cleanup XFS_IOC_GETVERSION_32 handling Move the cmd assignment into the function call, fix a spelling error in the comment and move the comment next to the call. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl32.c | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/fs/xfs/xfs_ioctl32.c b/fs/xfs/xfs_ioctl32.c index 2162eda017430f..ede890d4cbc38c 100644 --- a/fs/xfs/xfs_ioctl32.c +++ b/fs/xfs/xfs_ioctl32.c @@ -459,10 +459,9 @@ xfs_file_compat_ioctl( case XFS_IOC_FSGROWFSRT_32: return xfs_compat_ioc_growfs_rt(filp, ip->i_mount, arg); #endif - /* long changes size, but xfs only copiese out 32 bits */ case XFS_IOC_GETVERSION_32: - cmd = _NATIVE_IOC(cmd, long); - return xfs_file_ioctl(filp, cmd, p); + /* long changes size, but xfs only copies out 32 bits */ + return xfs_file_ioctl(filp, _NATIVE_IOC(cmd, long), p); case XFS_IOC_SWAPEXT_32: { struct xfs_swapext sxp; struct compat_xfs_swapext __user *sxu = arg; From 75331a4bd410edf3165c634689035dc01cb06645 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:13 +0200 Subject: [PATCH 0072/1352] xfs: split out the handlers for XFS_IOC_SWAPEXT_32 Split out a helper for XFS_IOC_SWAPEXT_32 to keep the stack variables out of xfs_file_compat_ioctl and to clean up the main compat ioctl handler flow. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl32.c | 40 +++++++++++++++++++++++----------------- 1 file changed, 23 insertions(+), 17 deletions(-) diff --git a/fs/xfs/xfs_ioctl32.c b/fs/xfs/xfs_ioctl32.c index ede890d4cbc38c..a6e3b35db6e20b 100644 --- a/fs/xfs/xfs_ioctl32.c +++ b/fs/xfs/xfs_ioctl32.c @@ -158,6 +158,27 @@ xfs_ioctl32_bstat_copyin( return 0; } +static int +xfs_compat_ioc_swapext( + struct file *file, + struct compat_xfs_swapext __user *sxu) +{ + struct xfs_swapext sxp; + int error; + + /* Bulk copy in up to the sx_stat field, then copy bstat */ + if (copy_from_user(&sxp, sxu, offsetof(struct xfs_swapext, sx_stat)) || + xfs_ioctl32_bstat_copyin(&sxp.sx_stat, &sxu->sx_stat)) + return -EFAULT; + + error = mnt_want_write_file(file); + if (error) + return error; + error = xfs_swapext(&sxp); + mnt_drop_write_file(file); + return error; +} + /* XFS_IOC_FSBULKSTAT and friends */ STATIC int @@ -446,7 +467,6 @@ xfs_file_compat_ioctl( struct inode *inode = file_inode(filp); struct xfs_inode *ip = XFS_I(inode); void __user *arg = compat_ptr(p); - int error; trace_xfs_file_compat_ioctl(ip); @@ -462,22 +482,8 @@ xfs_file_compat_ioctl( case XFS_IOC_GETVERSION_32: /* long changes size, but xfs only copies out 32 bits */ return xfs_file_ioctl(filp, _NATIVE_IOC(cmd, long), p); - case XFS_IOC_SWAPEXT_32: { - struct xfs_swapext sxp; - struct compat_xfs_swapext __user *sxu = arg; - - /* Bulk copy in up to the sx_stat field, then copy bstat */ - if (copy_from_user(&sxp, sxu, - offsetof(struct xfs_swapext, sx_stat)) || - xfs_ioctl32_bstat_copyin(&sxp.sx_stat, &sxu->sx_stat)) - return -EFAULT; - error = mnt_want_write_file(filp); - if (error) - return error; - error = xfs_swapext(&sxp); - mnt_drop_write_file(filp); - return error; - } + case XFS_IOC_SWAPEXT_32: + return xfs_compat_ioc_swapext(filp, arg); case XFS_IOC_FSBULKSTAT_32: case XFS_IOC_FSBULKSTAT_SINGLE_32: case XFS_IOC_FSINUMBERS_32: From 3724aef91a0f888e6b73b99cc55e0b1f8ae76857 Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:14 +0200 Subject: [PATCH 0073/1352] xfs: split out the handlers for XFS_IOC_*_BY_HANDLE_32 Split out helpers for XFS_IOC_*_BY_HANDLE_32 to keep the stack variables out of xfs_file_compat_ioctl and to clean up the main compat ioctl handler flow. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl32.c | 65 +++++++++++++++++++++++++++++--------------- 1 file changed, 43 insertions(+), 22 deletions(-) diff --git a/fs/xfs/xfs_ioctl32.c b/fs/xfs/xfs_ioctl32.c index a6e3b35db6e20b..688eb3300495d6 100644 --- a/fs/xfs/xfs_ioctl32.c +++ b/fs/xfs/xfs_ioctl32.c @@ -379,6 +379,43 @@ xfs_compat_handlereq_to_dentry( compat_ptr(hreq->ihandle), hreq->ihandlen); } +static int +xfs_compat_ioc_find_handle( + unsigned int cmd, + void __user *arg) +{ + struct xfs_fsop_handlereq hreq; + + if (xfs_compat_handlereq_copyin(&hreq, arg)) + return -EFAULT; + return xfs_find_handle(_NATIVE_IOC(cmd, struct xfs_fsop_handlereq), + &hreq); +} + +static int +xfs_compat_ioc_open_by_handle( + struct file *file, + void __user *arg) +{ + struct xfs_fsop_handlereq hreq; + + if (xfs_compat_handlereq_copyin(&hreq, arg)) + return -EFAULT; + return xfs_open_by_handle(file, &hreq); +} + +static int +xfs_compat_ioc_readlink_by_handle( + struct file *file, + void __user *arg) +{ + struct xfs_fsop_handlereq hreq; + + if (xfs_compat_handlereq_copyin(&hreq, arg)) + return -EFAULT; + return xfs_readlink_by_handle(file, &hreq); +} + STATIC int xfs_compat_attrlist_by_handle( struct file *parfilp, @@ -490,28 +527,12 @@ xfs_file_compat_ioctl( return xfs_compat_ioc_fsbulkstat(filp, cmd, arg); case XFS_IOC_FD_TO_HANDLE_32: case XFS_IOC_PATH_TO_HANDLE_32: - case XFS_IOC_PATH_TO_FSHANDLE_32: { - struct xfs_fsop_handlereq hreq; - - if (xfs_compat_handlereq_copyin(&hreq, arg)) - return -EFAULT; - cmd = _NATIVE_IOC(cmd, struct xfs_fsop_handlereq); - return xfs_find_handle(cmd, &hreq); - } - case XFS_IOC_OPEN_BY_HANDLE_32: { - struct xfs_fsop_handlereq hreq; - - if (xfs_compat_handlereq_copyin(&hreq, arg)) - return -EFAULT; - return xfs_open_by_handle(filp, &hreq); - } - case XFS_IOC_READLINK_BY_HANDLE_32: { - struct xfs_fsop_handlereq hreq; - - if (xfs_compat_handlereq_copyin(&hreq, arg)) - return -EFAULT; - return xfs_readlink_by_handle(filp, &hreq); - } + case XFS_IOC_PATH_TO_FSHANDLE_32: + return xfs_compat_ioc_find_handle(cmd, arg); + case XFS_IOC_OPEN_BY_HANDLE_32: + return xfs_compat_ioc_open_by_handle(filp, arg); + case XFS_IOC_READLINK_BY_HANDLE_32: + return xfs_compat_ioc_readlink_by_handle(filp, arg); case XFS_IOC_ATTRLIST_BY_HANDLE_32: return xfs_compat_attrlist_by_handle(filp, arg); case XFS_IOC_ATTRMULTI_BY_HANDLE_32: From fa93b3a499e17c01c464297e36e313ba8497939d Mon Sep 17 00:00:00 2001 From: Christoph Hellwig Date: Wed, 29 Jul 2026 15:03:15 +0200 Subject: [PATCH 0074/1352] xfs: remove an extra cast in xfs_file_compat_ioctl The ioctl argument is already available as an unsigned long in the p variable, so use that directly instead of casting back from p, which has the same value but was casted to a void pointer before. Signed-off-by: Christoph Hellwig Reviewed-by: "Darrick J. Wong" Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_ioctl32.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/xfs/xfs_ioctl32.c b/fs/xfs/xfs_ioctl32.c index 688eb3300495d6..f6875a2cc7065d 100644 --- a/fs/xfs/xfs_ioctl32.c +++ b/fs/xfs/xfs_ioctl32.c @@ -539,6 +539,6 @@ xfs_file_compat_ioctl( return xfs_compat_attrmulti_by_handle(filp, arg); default: /* try the native version */ - return xfs_file_ioctl(filp, cmd, (unsigned long)arg); + return xfs_file_ioctl(filp, cmd, p); } } From 9821d9dfca71e745fa2c7f87aa27e2c97753ad62 Mon Sep 17 00:00:00 2001 From: Eric Sandeen Date: Mon, 14 Sep 2026 17:28:47 -0500 Subject: [PATCH 0075/1352] xfs: remove unused args argument from xfs_attr_node_lookup() Signed-off-by: Eric Sandeen Reviewed-by: Carlos Maiolino Reviewed-by: Christoph Hellwig Reviewed-by: Darrick J. Wong Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_attr.c | 12 +++++------- 1 file changed, 5 insertions(+), 7 deletions(-) diff --git a/fs/xfs/libxfs/xfs_attr.c b/fs/xfs/libxfs/xfs_attr.c index b3f7b2c34ad766..8fb3adc3dd2d14 100644 --- a/fs/xfs/libxfs/xfs_attr.c +++ b/fs/xfs/libxfs/xfs_attr.c @@ -59,8 +59,7 @@ STATIC void xfs_attr_restore_rmt_blk(struct xfs_da_args *args); static int xfs_attr_node_try_addname(struct xfs_attr_intent *attr); STATIC int xfs_attr_node_addname_find_attr(struct xfs_attr_intent *attr); STATIC int xfs_attr_node_remove_attr(struct xfs_attr_intent *attr); -STATIC int xfs_attr_node_lookup(struct xfs_da_args *args, - struct xfs_da_state *state); +STATIC int xfs_attr_node_lookup(struct xfs_da_state *state); int xfs_inode_hasattr( @@ -709,7 +708,7 @@ int xfs_attr_node_removename_setup( int error; xfs_attr_item_init_da_state(attr); - error = xfs_attr_node_lookup(args, attr->xattri_da_state); + error = xfs_attr_node_lookup(attr->xattri_da_state); if (error != -EEXIST) goto out; error = 0; @@ -985,7 +984,7 @@ xfs_attr_lookup( } state = xfs_da_state_alloc(args); - error = xfs_attr_node_lookup(args, state); + error = xfs_attr_node_lookup(state); xfs_da_state_free(state); return error; } @@ -1386,7 +1385,6 @@ xfs_attr_leaf_get( /* Return EEXIST if attr is found, or ENOATTR if not. */ STATIC int xfs_attr_node_lookup( - struct xfs_da_args *args, struct xfs_da_state *state) { int retval, error; @@ -1417,7 +1415,7 @@ xfs_attr_node_addname_find_attr( * to where it should go. */ xfs_attr_item_init_da_state(attr); - error = xfs_attr_node_lookup(args, attr->xattri_da_state); + error = xfs_attr_node_lookup(attr->xattri_da_state); switch (error) { case -ENOATTR: if (args->op_flags & XFS_DA_OP_REPLACE) @@ -1588,7 +1586,7 @@ xfs_attr_node_get( * Search to see if name exists, and get back a pointer to it. */ state = xfs_da_state_alloc(args); - error = xfs_attr_node_lookup(args, state); + error = xfs_attr_node_lookup(state); if (error != -EEXIST) goto out_release; From 168757e7d7092076b263027d113bba2fba23303f Mon Sep 17 00:00:00 2001 From: Eric Sandeen Date: Mon, 14 Sep 2026 17:28:48 -0500 Subject: [PATCH 0076/1352] xfs: remove unused rsvd argument from xfs_bmap_add_attrfork() Signed-off-by: Eric Sandeen Reviewed-by: Carlos Maiolino Reviewed-by: Christoph Hellwig Reviewed-by: Darrick J. Wong Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_attr.c | 2 +- fs/xfs/libxfs/xfs_bmap.c | 3 +-- fs/xfs/libxfs/xfs_bmap.h | 2 +- fs/xfs/scrub/inode_repair.c | 2 +- fs/xfs/scrub/orphanage.c | 2 +- 5 files changed, 5 insertions(+), 6 deletions(-) diff --git a/fs/xfs/libxfs/xfs_attr.c b/fs/xfs/libxfs/xfs_attr.c index 8fb3adc3dd2d14..c4bb59633ce856 100644 --- a/fs/xfs/libxfs/xfs_attr.c +++ b/fs/xfs/libxfs/xfs_attr.c @@ -1013,7 +1013,7 @@ xfs_attr_add_fork( if (xfs_inode_has_attr_fork(ip)) goto trans_cancel; - error = xfs_bmap_add_attrfork(tp, ip, size, rsvd); + error = xfs_bmap_add_attrfork(tp, ip, size); if (error) goto trans_cancel; diff --git a/fs/xfs/libxfs/xfs_bmap.c b/fs/xfs/libxfs/xfs_bmap.c index d64defeda645ed..ca147dfaed05b4 100644 --- a/fs/xfs/libxfs/xfs_bmap.c +++ b/fs/xfs/libxfs/xfs_bmap.c @@ -1026,8 +1026,7 @@ int /* error code */ xfs_bmap_add_attrfork( struct xfs_trans *tp, struct xfs_inode *ip, /* incore inode pointer */ - int size, /* space new attribute needs */ - int rsvd) /* xact may use reserved blks */ + int size) /* space new attribute needs */ { struct xfs_mount *mp = tp->t_mountp; int logflags; /* logging flags */ diff --git a/fs/xfs/libxfs/xfs_bmap.h b/fs/xfs/libxfs/xfs_bmap.h index d5f2729305fada..60f6df4ce0579a 100644 --- a/fs/xfs/libxfs/xfs_bmap.h +++ b/fs/xfs/libxfs/xfs_bmap.h @@ -181,7 +181,7 @@ void xfs_trim_extent(struct xfs_bmbt_irec *irec, xfs_fileoff_t bno, xfs_filblks_t len); unsigned int xfs_bmap_compute_attr_offset(struct xfs_mount *mp); int xfs_bmap_add_attrfork(struct xfs_trans *tp, struct xfs_inode *ip, - int size, int rsvd); + int size); void xfs_bmap_local_to_extents_empty(struct xfs_trans *tp, struct xfs_inode *ip, int whichfork); int xfs_bmap_local_to_extents(struct xfs_trans *tp, struct xfs_inode *ip, diff --git a/fs/xfs/scrub/inode_repair.c b/fs/xfs/scrub/inode_repair.c index b87c2214623383..1e0434f3d61518 100644 --- a/fs/xfs/scrub/inode_repair.c +++ b/fs/xfs/scrub/inode_repair.c @@ -1949,7 +1949,7 @@ xrep_inode_pptr( return 0; return xfs_bmap_add_attrfork(sc->tp, ip, - sizeof(struct xfs_attr_sf_hdr), true); + sizeof(struct xfs_attr_sf_hdr)); } /* Fix COW extent size hint problems. */ diff --git a/fs/xfs/scrub/orphanage.c b/fs/xfs/scrub/orphanage.c index 21e31eeaa042f9..d8e1457f0d5e1a 100644 --- a/fs/xfs/scrub/orphanage.c +++ b/fs/xfs/scrub/orphanage.c @@ -550,7 +550,7 @@ xrep_adoption_move( if (!xfs_inode_has_attr_fork(sc->ip) && xfs_has_parent(sc->mp)) { int sf_size = xrep_adoption_attr_sizeof(adopt); - error = xfs_bmap_add_attrfork(sc->tp, sc->ip, sf_size, true); + error = xfs_bmap_add_attrfork(sc->tp, sc->ip, sf_size); if (error) return error; } From 0d498f3385e51a0a8eaac71aab6736b5a8cd4049 Mon Sep 17 00:00:00 2001 From: Eric Sandeen Date: Mon, 14 Sep 2026 17:28:49 -0500 Subject: [PATCH 0077/1352] xfs: remove unused mp and ops arguments from xfbtree_rec_bytes() Signed-off-by: Eric Sandeen Reviewed-by: Carlos Maiolino Reviewed-by: Christoph Hellwig Reviewed-by: Darrick J. Wong Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_btree_mem.c | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/fs/xfs/libxfs/xfs_btree_mem.c b/fs/xfs/libxfs/xfs_btree_mem.c index 1d83a4251ceeb6..3d5c7a014d4fff 100644 --- a/fs/xfs/libxfs/xfs_btree_mem.c +++ b/fs/xfs/libxfs/xfs_btree_mem.c @@ -73,9 +73,7 @@ xfbtree_destroy( /* Compute the number of bytes available for records. */ static inline unsigned int -xfbtree_rec_bytes( - struct xfs_mount *mp, - const struct xfs_btree_ops *ops) +xfbtree_rec_bytes(void) { return XMBUF_BLOCKSIZE - XFS_BTREE_LBLOCK_CRC_LEN; } @@ -118,7 +116,7 @@ xfbtree_init( const struct xfs_btree_ops *ops) { unsigned long long owner = xfbt->owner; - unsigned int blocklen = xfbtree_rec_bytes(mp, ops); + unsigned int blocklen = xfbtree_rec_bytes(); unsigned int keyptr_len; int error; From 4f4f0e78aab58e1bdaf97758e5801c6912f0397f Mon Sep 17 00:00:00 2001 From: Eric Sandeen Date: Mon, 14 Sep 2026 17:28:50 -0500 Subject: [PATCH 0078/1352] xfs: remove unused mp argument from xfs_exchmaps_check_forks() Signed-off-by: Eric Sandeen Reviewed-by: Carlos Maiolino Reviewed-by: Christoph Hellwig Reviewed-by: Darrick J. Wong Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_exchmaps.c | 1 - fs/xfs/libxfs/xfs_exchmaps.h | 3 +-- fs/xfs/xfs_exchrange.c | 2 +- 3 files changed, 2 insertions(+), 4 deletions(-) diff --git a/fs/xfs/libxfs/xfs_exchmaps.c b/fs/xfs/libxfs/xfs_exchmaps.c index 6a66b6075e0af4..ccb97da3765f22 100644 --- a/fs/xfs/libxfs/xfs_exchmaps.c +++ b/fs/xfs/libxfs/xfs_exchmaps.c @@ -135,7 +135,6 @@ xmi_has_postop_work(const struct xfs_exchmaps_intent *xmi) /* Check all mappings to make sure we can actually exchange them. */ int xfs_exchmaps_check_forks( - struct xfs_mount *mp, const struct xfs_exchmaps_req *req) { struct xfs_ifork *ifp1, *ifp2; diff --git a/fs/xfs/libxfs/xfs_exchmaps.h b/fs/xfs/libxfs/xfs_exchmaps.h index fa822dff202ade..055b8dbabd5bc1 100644 --- a/fs/xfs/libxfs/xfs_exchmaps.h +++ b/fs/xfs/libxfs/xfs_exchmaps.h @@ -115,8 +115,7 @@ void xfs_exchmaps_upgrade_extent_counts(struct xfs_trans *tp, int xfs_exchmaps_finish_one(struct xfs_trans *tp, struct xfs_exchmaps_intent *xmi); -int xfs_exchmaps_check_forks(struct xfs_mount *mp, - const struct xfs_exchmaps_req *req); +int xfs_exchmaps_check_forks(const struct xfs_exchmaps_req *req); void xfs_exchange_mappings(struct xfs_trans *tp, const struct xfs_exchmaps_req *req); diff --git a/fs/xfs/xfs_exchrange.c b/fs/xfs/xfs_exchrange.c index fafb4e3f065c75..a1001315d4a7bf 100644 --- a/fs/xfs/xfs_exchrange.c +++ b/fs/xfs/xfs_exchrange.c @@ -238,7 +238,7 @@ xfs_exchrange_mappings( trace_xfs_exchrange_before(ip2, 2); trace_xfs_exchrange_before(ip1, 1); - error = xfs_exchmaps_check_forks(mp, &req); + error = xfs_exchmaps_check_forks(&req); if (error) goto out_trans_cancel; From 383641abd3cc89e2b7422f59ebccb89d63404bfa Mon Sep 17 00:00:00 2001 From: Eric Sandeen Date: Mon, 14 Sep 2026 17:28:51 -0500 Subject: [PATCH 0079/1352] xfs: remove unused mp argument from xfs_inobt_rec_check_count() Signed-off-by: Eric Sandeen Reviewed-by: Carlos Maiolino Reviewed-by: Christoph Hellwig Reviewed-by: Darrick J. Wong Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_ialloc.c | 2 +- fs/xfs/libxfs/xfs_ialloc_btree.c | 1 - fs/xfs/libxfs/xfs_ialloc_btree.h | 5 ++--- 3 files changed, 3 insertions(+), 5 deletions(-) diff --git a/fs/xfs/libxfs/xfs_ialloc.c b/fs/xfs/libxfs/xfs_ialloc.c index 58dac4d505ba63..19b513b1169269 100644 --- a/fs/xfs/libxfs/xfs_ialloc.c +++ b/fs/xfs/libxfs/xfs_ialloc.c @@ -615,7 +615,7 @@ xfs_inobt_insert_sprec( trace_xfs_irec_merge_post(pag, nrec); - error = xfs_inobt_rec_check_count(mp, nrec); + error = xfs_inobt_rec_check_count(nrec); if (error) goto error; diff --git a/fs/xfs/libxfs/xfs_ialloc_btree.c b/fs/xfs/libxfs/xfs_ialloc_btree.c index 1376e8630449af..1f0bace2f144ee 100644 --- a/fs/xfs/libxfs/xfs_ialloc_btree.c +++ b/fs/xfs/libxfs/xfs_ialloc_btree.c @@ -687,7 +687,6 @@ xfs_inobt_irec_to_allocmask( */ int xfs_inobt_rec_check_count( - struct xfs_mount *mp, struct xfs_inobt_rec_incore *rec) { int inocount = 0; diff --git a/fs/xfs/libxfs/xfs_ialloc_btree.h b/fs/xfs/libxfs/xfs_ialloc_btree.h index 300edf5bc00949..e04c63c66f397f 100644 --- a/fs/xfs/libxfs/xfs_ialloc_btree.h +++ b/fs/xfs/libxfs/xfs_ialloc_btree.h @@ -57,10 +57,9 @@ unsigned int xfs_inobt_maxrecs(struct xfs_mount *mp, unsigned int blocklen, uint64_t xfs_inobt_irec_to_allocmask(const struct xfs_inobt_rec_incore *irec); #if defined(DEBUG) || defined(XFS_WARN) -int xfs_inobt_rec_check_count(struct xfs_mount *, - struct xfs_inobt_rec_incore *); +int xfs_inobt_rec_check_count(struct xfs_inobt_rec_incore *); #else -#define xfs_inobt_rec_check_count(mp, rec) 0 +#define xfs_inobt_rec_check_count(rec) 0 #endif /* DEBUG */ int xfs_finobt_calc_reserves(struct xfs_perag *perag, struct xfs_trans *tp, From 76a2796d4a671d71bcd4baf970accce61608c3f5 Mon Sep 17 00:00:00 2001 From: Eric Sandeen Date: Mon, 14 Sep 2026 17:28:52 -0500 Subject: [PATCH 0080/1352] xfs: remove unused mp argument from xfs_parent_finish() Signed-off-by: Eric Sandeen Reviewed-by: Carlos Maiolino Reviewed-by: Christoph Hellwig Reviewed-by: Darrick J. Wong Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_metadir.c | 2 +- fs/xfs/libxfs/xfs_parent.h | 1 - fs/xfs/xfs_inode.c | 18 +++++++++--------- fs/xfs/xfs_symlink.c | 4 ++-- 4 files changed, 12 insertions(+), 13 deletions(-) diff --git a/fs/xfs/libxfs/xfs_metadir.c b/fs/xfs/libxfs/xfs_metadir.c index 7c6b086b73db61..cc45a8a5aed973 100644 --- a/fs/xfs/libxfs/xfs_metadir.c +++ b/fs/xfs/libxfs/xfs_metadir.c @@ -163,7 +163,7 @@ xfs_metadir_teardown( trace_xfs_metadir_teardown(upd, error); if (upd->ppargs) { - xfs_parent_finish(upd->dp->i_mount, upd->ppargs); + xfs_parent_finish(upd->ppargs); upd->ppargs = NULL; } diff --git a/fs/xfs/libxfs/xfs_parent.h b/fs/xfs/libxfs/xfs_parent.h index 8eb4de9c5f1a9c..1dd3968a78b1ef 100644 --- a/fs/xfs/libxfs/xfs_parent.h +++ b/fs/xfs/libxfs/xfs_parent.h @@ -72,7 +72,6 @@ xfs_parent_start( /* Finish a parent pointer update by freeing the context object. */ static inline void xfs_parent_finish( - struct xfs_mount *mp, struct xfs_parent_args *ppargs) { if (ppargs) diff --git a/fs/xfs/xfs_inode.c b/fs/xfs/xfs_inode.c index 621513d7215eff..9165b161f0c662 100644 --- a/fs/xfs/xfs_inode.c +++ b/fs/xfs/xfs_inode.c @@ -755,7 +755,7 @@ xfs_create( *ipp = du.ip; xfs_iunlock(du.ip, XFS_ILOCK_EXCL); xfs_iunlock(dp, XFS_ILOCK_EXCL); - xfs_parent_finish(mp, du.ppargs); + xfs_parent_finish(du.ppargs); return 0; out_trans_cancel: @@ -772,7 +772,7 @@ xfs_create( xfs_irele(du.ip); } out_parent: - xfs_parent_finish(mp, du.ppargs); + xfs_parent_finish(du.ppargs); out_release_dquots: xfs_qm_dqrele(udqp); xfs_qm_dqrele(gdqp); @@ -972,7 +972,7 @@ xfs_link( error = xfs_trans_commit(tp); xfs_iunlock(tdp, XFS_ILOCK_EXCL); xfs_iunlock(sip, XFS_ILOCK_EXCL); - xfs_parent_finish(mp, du.ppargs); + xfs_parent_finish(du.ppargs); return error; error_return: @@ -980,7 +980,7 @@ xfs_link( xfs_iunlock(tdp, XFS_ILOCK_EXCL); xfs_iunlock(sip, XFS_ILOCK_EXCL); out_parent: - xfs_parent_finish(mp, du.ppargs); + xfs_parent_finish(du.ppargs); std_return: if (error == -ENOSPC && nospace_error) error = nospace_error; @@ -1985,7 +1985,7 @@ xfs_remove( xfs_iunlock(ip, XFS_ILOCK_EXCL); xfs_iunlock(dp, XFS_ILOCK_EXCL); - xfs_parent_finish(mp, du.ppargs); + xfs_parent_finish(du.ppargs); return 0; out_trans_cancel: @@ -1994,7 +1994,7 @@ xfs_remove( xfs_iunlock(ip, XFS_ILOCK_EXCL); xfs_iunlock(dp, XFS_ILOCK_EXCL); out_parent: - xfs_parent_finish(mp, du.ppargs); + xfs_parent_finish(du.ppargs); std_return: return error; } @@ -2357,11 +2357,11 @@ xfs_rename( out_unlock: xfs_iunlock_rename(inodes, num_inodes); out_tgt_ppargs: - xfs_parent_finish(mp, du_tgt.ppargs); + xfs_parent_finish(du_tgt.ppargs); out_wip_ppargs: - xfs_parent_finish(mp, du_wip.ppargs); + xfs_parent_finish(du_wip.ppargs); out_src_ppargs: - xfs_parent_finish(mp, du_src.ppargs); + xfs_parent_finish(du_src.ppargs); out_release_wip: if (du_wip.ip) xfs_irele(du_wip.ip); diff --git a/fs/xfs/xfs_symlink.c b/fs/xfs/xfs_symlink.c index 5585ac7f4d16bf..cc13819df6f258 100644 --- a/fs/xfs/xfs_symlink.c +++ b/fs/xfs/xfs_symlink.c @@ -219,7 +219,7 @@ xfs_symlink( *ipp = du.ip; xfs_iunlock(du.ip, XFS_ILOCK_EXCL); xfs_iunlock(dp, XFS_ILOCK_EXCL); - xfs_parent_finish(mp, du.ppargs); + xfs_parent_finish(du.ppargs); return 0; out_trans_cancel: @@ -236,7 +236,7 @@ xfs_symlink( xfs_irele(du.ip); } out_parent: - xfs_parent_finish(mp, du.ppargs); + xfs_parent_finish(du.ppargs); out_release_dquots: xfs_qm_dqrele(udqp); xfs_qm_dqrele(gdqp); From da86bd00b8532bf304dd32c51d0ddb4f57086184 Mon Sep 17 00:00:00 2001 From: Eric Sandeen Date: Mon, 14 Sep 2026 17:28:53 -0500 Subject: [PATCH 0081/1352] xfs: remove unused tp argument from xfs_rmap_update_hook() Signed-off-by: Eric Sandeen Reviewed-by: Carlos Maiolino Reviewed-by: Christoph Hellwig Reviewed-by: Darrick J. Wong Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_rmap.c | 9 ++++----- 1 file changed, 4 insertions(+), 5 deletions(-) diff --git a/fs/xfs/libxfs/xfs_rmap.c b/fs/xfs/libxfs/xfs_rmap.c index 34d218de21a956..37780a0526d8ea 100644 --- a/fs/xfs/libxfs/xfs_rmap.c +++ b/fs/xfs/libxfs/xfs_rmap.c @@ -904,7 +904,6 @@ xfs_rmap_hook_enable(void) /* Call downstream hooks for a reverse mapping update. */ static inline void xfs_rmap_update_hook( - struct xfs_trans *tp, struct xfs_group *xg, enum xfs_rmap_intent_type op, xfs_agblock_t startblock, @@ -952,7 +951,7 @@ xfs_rmap_hook_setup( xfs_hook_setup(&hook->rmap_hook, mod_fn); } #else -# define xfs_rmap_update_hook(t, p, o, s, b, u, oi) do { } while (0) +# define xfs_rmap_update_hook(p, o, s, b, u, oi) do { } while (0) #endif /* CONFIG_XFS_LIVE_HOOKS */ /* @@ -975,7 +974,7 @@ xfs_rmap_free( return 0; cur = xfs_rmapbt_init_cursor(mp, tp, agbp, pag); - xfs_rmap_update_hook(tp, pag_group(pag), XFS_RMAP_UNMAP, bno, len, + xfs_rmap_update_hook(pag_group(pag), XFS_RMAP_UNMAP, bno, len, false, oinfo); error = xfs_rmap_unmap(cur, bno, len, false, oinfo); @@ -1220,7 +1219,7 @@ xfs_rmap_alloc( return 0; cur = xfs_rmapbt_init_cursor(mp, tp, agbp, pag); - xfs_rmap_update_hook(tp, pag_group(pag), XFS_RMAP_MAP, bno, len, false, + xfs_rmap_update_hook(pag_group(pag), XFS_RMAP_MAP, bno, len, false, oinfo); error = xfs_rmap_map(cur, bno, len, false, oinfo); @@ -2721,7 +2720,7 @@ xfs_rmap_finish_one( if (error) return error; - xfs_rmap_update_hook(tp, ri->ri_group, ri->ri_type, bno, + xfs_rmap_update_hook(ri->ri_group, ri->ri_type, bno, ri->ri_bmap.br_blockcount, unwritten, &oinfo); return 0; } From 4e7d4c62ab0e2e68764e1501588d1e21b23b5645 Mon Sep 17 00:00:00 2001 From: Eric Sandeen Date: Mon, 14 Sep 2026 17:28:54 -0500 Subject: [PATCH 0082/1352] xfs: remove unused mp argument from xfs_rtrefcount_broot_space_calc() Signed-off-by: Eric Sandeen Reviewed-by: Carlos Maiolino Reviewed-by: Christoph Hellwig Reviewed-by: Darrick J. Wong Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_rtrefcount_btree.c | 6 +++--- fs/xfs/libxfs/xfs_rtrefcount_btree.h | 3 +-- fs/xfs/scrub/rtrefcount_repair.c | 3 +-- 3 files changed, 5 insertions(+), 7 deletions(-) diff --git a/fs/xfs/libxfs/xfs_rtrefcount_btree.c b/fs/xfs/libxfs/xfs_rtrefcount_btree.c index dcc89b8e149b05..7df33b5edfda56 100644 --- a/fs/xfs/libxfs/xfs_rtrefcount_btree.c +++ b/fs/xfs/libxfs/xfs_rtrefcount_btree.c @@ -311,7 +311,7 @@ xfs_rtrefcountbt_broot_realloc( unsigned int old_size = ifp->if_broot_bytes; const unsigned int level = cur->bc_nlevels - 1; - new_size = xfs_rtrefcount_broot_space_calc(mp, level, new_numrecs); + new_size = xfs_rtrefcount_broot_space_calc(level, new_numrecs); /* Handle the nop case quietly. */ if (new_size == old_size) @@ -661,7 +661,7 @@ xfs_iformat_rtrefcount( } broot = xfs_broot_alloc(xfs_ifork_ptr(ip, XFS_DATA_FORK), - xfs_rtrefcount_broot_space_calc(mp, level, numrecs)); + xfs_rtrefcount_broot_space_calc(level, numrecs)); if (broot) xfs_rtrefcountbt_from_disk(ip, dfp, dsize, broot); return 0; @@ -751,7 +751,7 @@ xfs_rtrefcountbt_create( /* Initialize the empty incore btree root. */ broot = xfs_broot_realloc(ifp, - xfs_rtrefcount_broot_space_calc(mp, 0, 0)); + xfs_rtrefcount_broot_space_calc(0, 0)); if (broot) xfs_btree_init_block(mp, broot, &xfs_rtrefcountbt_ops, 0, 0, I_INO(ip)); diff --git a/fs/xfs/libxfs/xfs_rtrefcount_btree.h b/fs/xfs/libxfs/xfs_rtrefcount_btree.h index a99b7a8aec8659..aeef004ffdc91e 100644 --- a/fs/xfs/libxfs/xfs_rtrefcount_btree.h +++ b/fs/xfs/libxfs/xfs_rtrefcount_btree.h @@ -129,7 +129,6 @@ xfs_rtrefcount_broot_ptr_addr( */ static inline size_t xfs_rtrefcount_broot_space_calc( - struct xfs_mount *mp, unsigned int level, unsigned int nrecs) { @@ -148,7 +147,7 @@ xfs_rtrefcount_broot_space_calc( static inline size_t xfs_rtrefcount_broot_space(struct xfs_mount *mp, struct xfs_rtrefcount_root *bb) { - return xfs_rtrefcount_broot_space_calc(mp, be16_to_cpu(bb->bb_level), + return xfs_rtrefcount_broot_space_calc(be16_to_cpu(bb->bb_level), be16_to_cpu(bb->bb_numrecs)); } diff --git a/fs/xfs/scrub/rtrefcount_repair.c b/fs/xfs/scrub/rtrefcount_repair.c index 2b939960c7ddcd..c78a6d2990c586 100644 --- a/fs/xfs/scrub/rtrefcount_repair.c +++ b/fs/xfs/scrub/rtrefcount_repair.c @@ -596,8 +596,7 @@ xrep_rtrefc_iroot_size( unsigned int nr_this_level, void *priv) { - return xfs_rtrefcount_broot_space_calc(cur->bc_mp, level, - nr_this_level); + return xfs_rtrefcount_broot_space_calc(level, nr_this_level); } /* From 0f08a3703f5f5091f7bada957500641c96528369 Mon Sep 17 00:00:00 2001 From: Eric Sandeen Date: Mon, 14 Sep 2026 17:28:55 -0500 Subject: [PATCH 0083/1352] xfs: remove unused mp argument from xfs_rtrefcount_broot_space() Signed-off-by: Eric Sandeen Reviewed-by: Carlos Maiolino Reviewed-by: Christoph Hellwig Reviewed-by: Darrick J. Wong Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_rtrefcount_btree.c | 2 +- fs/xfs/libxfs/xfs_rtrefcount_btree.h | 2 +- fs/xfs/scrub/inode_repair.c | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/fs/xfs/libxfs/xfs_rtrefcount_btree.c b/fs/xfs/libxfs/xfs_rtrefcount_btree.c index 7df33b5edfda56..e91ff14577f9ad 100644 --- a/fs/xfs/libxfs/xfs_rtrefcount_btree.c +++ b/fs/xfs/libxfs/xfs_rtrefcount_btree.c @@ -602,7 +602,7 @@ xfs_rtrefcountbt_from_disk( unsigned int maxrecs; unsigned int rblocklen; - rblocklen = xfs_rtrefcount_broot_space(mp, dblock); + rblocklen = xfs_rtrefcount_broot_space(dblock); xfs_btree_init_block(mp, rblock, &xfs_rtrefcountbt_ops, 0, 0, I_INO(ip)); diff --git a/fs/xfs/libxfs/xfs_rtrefcount_btree.h b/fs/xfs/libxfs/xfs_rtrefcount_btree.h index aeef004ffdc91e..9ab6ecf90ba515 100644 --- a/fs/xfs/libxfs/xfs_rtrefcount_btree.h +++ b/fs/xfs/libxfs/xfs_rtrefcount_btree.h @@ -145,7 +145,7 @@ xfs_rtrefcount_broot_space_calc( * btree root block. */ static inline size_t -xfs_rtrefcount_broot_space(struct xfs_mount *mp, struct xfs_rtrefcount_root *bb) +xfs_rtrefcount_broot_space(struct xfs_rtrefcount_root *bb) { return xfs_rtrefcount_broot_space_calc(be16_to_cpu(bb->bb_level), be16_to_cpu(bb->bb_numrecs)); diff --git a/fs/xfs/scrub/inode_repair.c b/fs/xfs/scrub/inode_repair.c index 1e0434f3d61518..5bc2bcb78051a5 100644 --- a/fs/xfs/scrub/inode_repair.c +++ b/fs/xfs/scrub/inode_repair.c @@ -1405,7 +1405,7 @@ xrep_dinode_ensure_forkoff( break; case XFS_METAFILE_RTREFCOUNT: rcdr = XFS_DFORK_PTR(dip, XFS_DATA_FORK); - dfork_min = xfs_rtrefcount_broot_space(sc->mp, rcdr); + dfork_min = xfs_rtrefcount_broot_space(rcdr); break; default: dfork_min = 0; From 1eaea71e8b998be7d8f745040b186df8d73759f4 Mon Sep 17 00:00:00 2001 From: Eric Sandeen Date: Mon, 14 Sep 2026 17:28:56 -0500 Subject: [PATCH 0084/1352] xfs: remove unused mp argument from xfs_rtrmap_broot_space_calc() Signed-off-by: Eric Sandeen Reviewed-by: Carlos Maiolino Reviewed-by: Christoph Hellwig Reviewed-by: Darrick J. Wong Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_rtrmap_btree.c | 6 +++--- fs/xfs/libxfs/xfs_rtrmap_btree.h | 3 +-- fs/xfs/scrub/rtrmap_repair.c | 2 +- 3 files changed, 5 insertions(+), 6 deletions(-) diff --git a/fs/xfs/libxfs/xfs_rtrmap_btree.c b/fs/xfs/libxfs/xfs_rtrmap_btree.c index a15e460a1ec7f4..5901d7efd3f676 100644 --- a/fs/xfs/libxfs/xfs_rtrmap_btree.c +++ b/fs/xfs/libxfs/xfs_rtrmap_btree.c @@ -412,7 +412,7 @@ xfs_rtrmapbt_broot_realloc( unsigned int old_size = ifp->if_broot_bytes; const unsigned int level = cur->bc_nlevels - 1; - new_size = xfs_rtrmap_broot_space_calc(mp, level, new_numrecs); + new_size = xfs_rtrmap_broot_space_calc(level, new_numrecs); /* Handle the nop case quietly. */ if (new_size == old_size) @@ -895,7 +895,7 @@ xfs_iformat_rtrmap( } broot = xfs_broot_alloc(xfs_ifork_ptr(ip, XFS_DATA_FORK), - xfs_rtrmap_broot_space_calc(mp, level, numrecs)); + xfs_rtrmap_broot_space_calc(level, numrecs)); if (broot) xfs_rtrmapbt_from_disk(ip, dfp, dsize, broot); return 0; @@ -980,7 +980,7 @@ xfs_rtrmapbt_create( ASSERT(ifp->if_bytes == 0); /* Initialize the empty incore btree root. */ - broot = xfs_broot_realloc(ifp, xfs_rtrmap_broot_space_calc(mp, 0, 0)); + broot = xfs_broot_realloc(ifp, xfs_rtrmap_broot_space_calc(0, 0)); if (broot) xfs_btree_init_block(mp, broot, &xfs_rtrmapbt_ops, 0, 0, I_INO(ip)); diff --git a/fs/xfs/libxfs/xfs_rtrmap_btree.h b/fs/xfs/libxfs/xfs_rtrmap_btree.h index e328fd62a149d1..c59a144b4bbfbe 100644 --- a/fs/xfs/libxfs/xfs_rtrmap_btree.h +++ b/fs/xfs/libxfs/xfs_rtrmap_btree.h @@ -140,7 +140,6 @@ xfs_rtrmap_broot_ptr_addr( */ static inline size_t xfs_rtrmap_broot_space_calc( - struct xfs_mount *mp, unsigned int level, unsigned int nrecs) { @@ -159,7 +158,7 @@ xfs_rtrmap_broot_space_calc( static inline size_t xfs_rtrmap_broot_space(struct xfs_mount *mp, struct xfs_rtrmap_root *bb) { - return xfs_rtrmap_broot_space_calc(mp, be16_to_cpu(bb->bb_level), + return xfs_rtrmap_broot_space_calc(be16_to_cpu(bb->bb_level), be16_to_cpu(bb->bb_numrecs)); } diff --git a/fs/xfs/scrub/rtrmap_repair.c b/fs/xfs/scrub/rtrmap_repair.c index a2b72e61edf5fe..5cfa4470c57c0f 100644 --- a/fs/xfs/scrub/rtrmap_repair.c +++ b/fs/xfs/scrub/rtrmap_repair.c @@ -693,7 +693,7 @@ xrep_rtrmap_iroot_size( unsigned int nr_this_level, void *priv) { - return xfs_rtrmap_broot_space_calc(cur->bc_mp, level, nr_this_level); + return xfs_rtrmap_broot_space_calc(level, nr_this_level); } /* From be0168dffd6911621bdf5e0d20b958c636685b87 Mon Sep 17 00:00:00 2001 From: Eric Sandeen Date: Mon, 14 Sep 2026 17:28:57 -0500 Subject: [PATCH 0085/1352] xfs: remove unused mp argument from xfs_calc_default_atomic_ioend_reservation() Signed-off-by: Eric Sandeen Reviewed-by: Carlos Maiolino Reviewed-by: Christoph Hellwig Reviewed-by: Darrick J. Wong Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_trans_resv.c | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/fs/xfs/libxfs/xfs_trans_resv.c b/fs/xfs/libxfs/xfs_trans_resv.c index 3151e97ca8ff6d..57f682f0b54a48 100644 --- a/fs/xfs/libxfs/xfs_trans_resv.c +++ b/fs/xfs/libxfs/xfs_trans_resv.c @@ -1292,7 +1292,6 @@ xfs_calc_namespace_reservations( STATIC void xfs_calc_default_atomic_ioend_reservation( - struct xfs_mount *mp, struct xfs_trans_resv *resp) { /* Pick a default that will scale reasonably for the log size. */ @@ -1398,7 +1397,7 @@ xfs_trans_resv_calc( * Now that we've finished computing the static reservations, we can * compute the dynamic reservation for atomic writes. */ - xfs_calc_default_atomic_ioend_reservation(mp, resp); + xfs_calc_default_atomic_ioend_reservation(resp); } /* @@ -1508,7 +1507,7 @@ xfs_calc_atomic_write_log_geometry( ASSERT(blockcount > 0); - xfs_calc_default_atomic_ioend_reservation(mp, M_RES(mp)); + xfs_calc_default_atomic_ioend_reservation(M_RES(mp)); per_intent = xfs_calc_atomic_write_ioend_geometry(mp, &step_size); @@ -1545,7 +1544,7 @@ xfs_calc_atomic_write_reservation( * use the defaults. */ if (blockcount == 0) { - xfs_calc_default_atomic_ioend_reservation(mp, M_RES(mp)); + xfs_calc_default_atomic_ioend_reservation(M_RES(mp)); return 0; } From 323b208c3c38673f31379db2eb4b07acf740d825 Mon Sep 17 00:00:00 2001 From: Eric Sandeen Date: Mon, 14 Sep 2026 17:28:58 -0500 Subject: [PATCH 0086/1352] xfs: remove unused mp argument from xfs_verify_dablk() Signed-off-by: Eric Sandeen Reviewed-by: Carlos Maiolino Reviewed-by: Christoph Hellwig Reviewed-by: Darrick J. Wong Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_types.c | 1 - fs/xfs/libxfs/xfs_types.h | 2 +- fs/xfs/scrub/bmap.c | 5 ++--- 3 files changed, 3 insertions(+), 5 deletions(-) diff --git a/fs/xfs/libxfs/xfs_types.c b/fs/xfs/libxfs/xfs_types.c index 67c947a47f1478..b5b0b6334d111a 100644 --- a/fs/xfs/libxfs/xfs_types.c +++ b/fs/xfs/libxfs/xfs_types.c @@ -222,7 +222,6 @@ xfs_verify_icount( /* Sanity-checking of dir/attr block offsets. */ bool xfs_verify_dablk( - struct xfs_mount *mp, xfs_fileoff_t dabno) { xfs_dablk_t max_dablk = -1U; diff --git a/fs/xfs/libxfs/xfs_types.h b/fs/xfs/libxfs/xfs_types.h index f6f4f2d4b5dbf5..d217220d2054d4 100644 --- a/fs/xfs/libxfs/xfs_types.h +++ b/fs/xfs/libxfs/xfs_types.h @@ -277,7 +277,7 @@ bool xfs_verify_rtbno(struct xfs_mount *mp, xfs_rtblock_t rtbno); bool xfs_verify_rtbext(struct xfs_mount *mp, xfs_rtblock_t rtbno, xfs_filblks_t len); bool xfs_verify_icount(struct xfs_mount *mp, unsigned long long icount); -bool xfs_verify_dablk(struct xfs_mount *mp, xfs_fileoff_t off); +bool xfs_verify_dablk(xfs_fileoff_t off); void xfs_icount_range(struct xfs_mount *mp, unsigned long long *min, unsigned long long *max); bool xfs_verify_fileoff(struct xfs_mount *mp, xfs_fileoff_t off); diff --git a/fs/xfs/scrub/bmap.c b/fs/xfs/scrub/bmap.c index 4f3c7f681bd921..868f502a44f2cf 100644 --- a/fs/xfs/scrub/bmap.c +++ b/fs/xfs/scrub/bmap.c @@ -453,18 +453,17 @@ xchk_bmap_dirattr_extent( struct xchk_bmap_info *info, struct xfs_bmbt_irec *irec) { - struct xfs_mount *mp = ip->i_mount; xfs_fileoff_t off; if (!S_ISDIR(VFS_I(ip)->i_mode) && info->whichfork != XFS_ATTR_FORK) return; - if (!xfs_verify_dablk(mp, irec->br_startoff)) + if (!xfs_verify_dablk(irec->br_startoff)) xchk_fblock_set_corrupt(info->sc, info->whichfork, irec->br_startoff); off = irec->br_startoff + irec->br_blockcount - 1; - if (!xfs_verify_dablk(mp, off)) + if (!xfs_verify_dablk(off)) xchk_fblock_set_corrupt(info->sc, info->whichfork, off); } From 06a83d006c8cf90c109a06b0c749194fd2301ccd Mon Sep 17 00:00:00 2001 From: Eric Sandeen Date: Mon, 14 Sep 2026 17:28:59 -0500 Subject: [PATCH 0087/1352] xfs: remove unused mp argument from xfs_verify_fileoff() Signed-off-by: Eric Sandeen Reviewed-by: Carlos Maiolino Reviewed-by: Christoph Hellwig Reviewed-by: Darrick J. Wong Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_types.c | 5 ++--- fs/xfs/libxfs/xfs_types.h | 2 +- fs/xfs/scrub/inode_repair.c | 2 +- fs/xfs/scrub/quota.c | 2 +- fs/xfs/scrub/quota_repair.c | 4 ++-- fs/xfs/xfs_inode.c | 2 +- fs/xfs/xfs_super.c | 2 +- 7 files changed, 9 insertions(+), 10 deletions(-) diff --git a/fs/xfs/libxfs/xfs_types.c b/fs/xfs/libxfs/xfs_types.c index b5b0b6334d111a..32e6c0aaf46703 100644 --- a/fs/xfs/libxfs/xfs_types.c +++ b/fs/xfs/libxfs/xfs_types.c @@ -232,7 +232,6 @@ xfs_verify_dablk( /* Check that a file block offset does not exceed the maximum. */ bool xfs_verify_fileoff( - struct xfs_mount *mp, xfs_fileoff_t off) { return off <= XFS_MAX_FILEOFF; @@ -248,8 +247,8 @@ xfs_verify_fileext( if (off + len <= off) return false; - if (!xfs_verify_fileoff(mp, off)) + if (!xfs_verify_fileoff(off)) return false; - return xfs_verify_fileoff(mp, off + len - 1); + return xfs_verify_fileoff(off + len - 1); } diff --git a/fs/xfs/libxfs/xfs_types.h b/fs/xfs/libxfs/xfs_types.h index d217220d2054d4..adae8114968014 100644 --- a/fs/xfs/libxfs/xfs_types.h +++ b/fs/xfs/libxfs/xfs_types.h @@ -280,7 +280,7 @@ bool xfs_verify_icount(struct xfs_mount *mp, unsigned long long icount); bool xfs_verify_dablk(xfs_fileoff_t off); void xfs_icount_range(struct xfs_mount *mp, unsigned long long *min, unsigned long long *max); -bool xfs_verify_fileoff(struct xfs_mount *mp, xfs_fileoff_t off); +bool xfs_verify_fileoff(xfs_fileoff_t off); bool xfs_verify_fileext(struct xfs_mount *mp, xfs_fileoff_t off, xfs_fileoff_t len); diff --git a/fs/xfs/scrub/inode_repair.c b/fs/xfs/scrub/inode_repair.c index 5bc2bcb78051a5..abfcab86928e62 100644 --- a/fs/xfs/scrub/inode_repair.c +++ b/fs/xfs/scrub/inode_repair.c @@ -933,7 +933,7 @@ xrep_dinode_bad_bmbt_fork( fkp = xfs_bmdr_key_addr(dfp, i); fileoff = be64_to_cpu(fkp->br_startoff); - if (!xfs_verify_fileoff(sc->mp, fileoff)) + if (!xfs_verify_fileoff(fileoff)) return true; fpp = xfs_bmdr_ptr_addr(dfp, i, dmxr); diff --git a/fs/xfs/scrub/quota.c b/fs/xfs/scrub/quota.c index 222812fe202c21..8c6ba1240fd53a 100644 --- a/fs/xfs/scrub/quota.c +++ b/fs/xfs/scrub/quota.c @@ -89,7 +89,7 @@ xchk_quota_item_bmap( int nmaps = 1; int error; - if (!xfs_verify_fileoff(mp, offset)) { + if (!xfs_verify_fileoff(offset)) { xchk_fblock_set_corrupt(sc, XFS_DATA_FORK, offset); return 0; } diff --git a/fs/xfs/scrub/quota_repair.c b/fs/xfs/scrub/quota_repair.c index 59302e8afc7ef0..9ab373c1994c1d 100644 --- a/fs/xfs/scrub/quota_repair.c +++ b/fs/xfs/scrub/quota_repair.c @@ -116,8 +116,8 @@ xrep_quota_item_bmap( int error; /* The computed file offset should always be valid. */ - if (!xfs_verify_fileoff(mp, offset)) { - ASSERT(xfs_verify_fileoff(mp, offset)); + if (!xfs_verify_fileoff(offset)) { + ASSERT(xfs_verify_fileoff(offset)); return -EFSCORRUPTED; } dq->q_fileoffset = offset; diff --git a/fs/xfs/xfs_inode.c b/fs/xfs/xfs_inode.c index 9165b161f0c662..15b62574b8d486 100644 --- a/fs/xfs/xfs_inode.c +++ b/fs/xfs/xfs_inode.c @@ -1064,7 +1064,7 @@ xfs_itruncate_extents_flags( * the page cache can't scale that far. */ first_unmap_block = XFS_B_TO_FSB(mp, (xfs_ufsize_t)new_size); - if (!xfs_verify_fileoff(mp, first_unmap_block)) { + if (!xfs_verify_fileoff(first_unmap_block)) { WARN_ON_ONCE(first_unmap_block > XFS_MAX_FILEOFF); return 0; } diff --git a/fs/xfs/xfs_super.c b/fs/xfs/xfs_super.c index b24db75eaedc5f..2edc2a4978835f 100644 --- a/fs/xfs/xfs_super.c +++ b/fs/xfs/xfs_super.c @@ -1889,7 +1889,7 @@ xfs_fs_fill_super( * Avoid integer overflow by comparing the maximum bmbt offset to the * maximum pagecache offset in units of fs blocks. */ - if (!xfs_verify_fileoff(mp, XFS_B_TO_FSBT(mp, MAX_LFS_FILESIZE))) { + if (!xfs_verify_fileoff(XFS_B_TO_FSBT(mp, MAX_LFS_FILESIZE))) { xfs_warn(mp, "MAX_LFS_FILESIZE block offset (%llu) exceeds extent map maximum (%llu)!", XFS_B_TO_FSBT(mp, MAX_LFS_FILESIZE), From 1c1058646d3bb5fc1b8ca4504c898e9dce520407 Mon Sep 17 00:00:00 2001 From: Eric Sandeen Date: Mon, 14 Sep 2026 17:29:00 -0500 Subject: [PATCH 0088/1352] xfs: remove unused mp argument from xfs_verify_fileext() (Follows from last patch) Signed-off-by: Eric Sandeen Reviewed-by: Carlos Maiolino Reviewed-by: Christoph Hellwig Reviewed-by: Darrick J. Wong Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_bmap.c | 2 +- fs/xfs/libxfs/xfs_rmap.c | 4 ++-- fs/xfs/libxfs/xfs_types.c | 1 - fs/xfs/libxfs/xfs_types.h | 3 +-- fs/xfs/scrub/bmap.c | 6 ++---- fs/xfs/scrub/bmap_repair.c | 4 ++-- fs/xfs/xfs_bmap_item.c | 2 +- fs/xfs/xfs_exchmaps_item.c | 4 ++-- fs/xfs/xfs_rmap_item.c | 2 +- 9 files changed, 12 insertions(+), 16 deletions(-) diff --git a/fs/xfs/libxfs/xfs_bmap.c b/fs/xfs/libxfs/xfs_bmap.c index ca147dfaed05b4..ae91f63455c5de 100644 --- a/fs/xfs/libxfs/xfs_bmap.c +++ b/fs/xfs/libxfs/xfs_bmap.c @@ -6087,7 +6087,7 @@ xfs_bmap_validate_extent_raw( int whichfork, struct xfs_bmbt_irec *irec) { - if (!xfs_verify_fileext(mp, irec->br_startoff, irec->br_blockcount)) + if (!xfs_verify_fileext(irec->br_startoff, irec->br_blockcount)) return __this_address; if (rtfile && whichfork == XFS_DATA_FORK) { diff --git a/fs/xfs/libxfs/xfs_rmap.c b/fs/xfs/libxfs/xfs_rmap.c index 37780a0526d8ea..14aef87837a9cc 100644 --- a/fs/xfs/libxfs/xfs_rmap.c +++ b/fs/xfs/libxfs/xfs_rmap.c @@ -260,7 +260,7 @@ xfs_rmap_check_irec( /* Check for a valid fork offset, if applicable. */ if (is_inode && !is_bmbt && - !xfs_verify_fileext(mp, irec->rm_offset, irec->rm_blockcount)) + !xfs_verify_fileext(irec->rm_offset, irec->rm_blockcount)) return __this_address; return NULL; @@ -310,7 +310,7 @@ xfs_rtrmap_check_inode_irec( return __this_address; if (!xfs_verify_rgbext(rtg, irec->rm_startblock, irec->rm_blockcount)) return __this_address; - if (!xfs_verify_fileext(mp, irec->rm_offset, irec->rm_blockcount)) + if (!xfs_verify_fileext(irec->rm_offset, irec->rm_blockcount)) return __this_address; return NULL; } diff --git a/fs/xfs/libxfs/xfs_types.c b/fs/xfs/libxfs/xfs_types.c index 32e6c0aaf46703..f195a04dbf666f 100644 --- a/fs/xfs/libxfs/xfs_types.c +++ b/fs/xfs/libxfs/xfs_types.c @@ -240,7 +240,6 @@ xfs_verify_fileoff( /* Check that a range of file block offsets do not exceed the maximum. */ bool xfs_verify_fileext( - struct xfs_mount *mp, xfs_fileoff_t off, xfs_fileoff_t len) { diff --git a/fs/xfs/libxfs/xfs_types.h b/fs/xfs/libxfs/xfs_types.h index adae8114968014..19dd5e7c8b1279 100644 --- a/fs/xfs/libxfs/xfs_types.h +++ b/fs/xfs/libxfs/xfs_types.h @@ -281,7 +281,6 @@ bool xfs_verify_dablk(xfs_fileoff_t off); void xfs_icount_range(struct xfs_mount *mp, unsigned long long *min, unsigned long long *max); bool xfs_verify_fileoff(xfs_fileoff_t off); -bool xfs_verify_fileext(struct xfs_mount *mp, xfs_fileoff_t off, - xfs_fileoff_t len); +bool xfs_verify_fileext(xfs_fileoff_t off, xfs_fileoff_t len); #endif /* __XFS_TYPES_H__ */ diff --git a/fs/xfs/scrub/bmap.c b/fs/xfs/scrub/bmap.c index 868f502a44f2cf..c190590bc56216 100644 --- a/fs/xfs/scrub/bmap.c +++ b/fs/xfs/scrub/bmap.c @@ -485,7 +485,7 @@ xchk_bmap_iextent( xchk_fblock_set_corrupt(info->sc, info->whichfork, irec->br_startoff); - if (!xfs_verify_fileext(mp, irec->br_startoff, irec->br_blockcount)) + if (!xfs_verify_fileext(irec->br_startoff, irec->br_blockcount)) xchk_fblock_set_corrupt(info->sc, info->whichfork, irec->br_startoff); @@ -876,8 +876,6 @@ xchk_bmap_iextent_delalloc( struct xchk_bmap_info *info, struct xfs_bmbt_irec *irec) { - struct xfs_mount *mp = info->sc->mp; - /* * Check for out-of-order extents. This record could have come * from the incore list, for which there is no ordering check. @@ -887,7 +885,7 @@ xchk_bmap_iextent_delalloc( xchk_fblock_set_corrupt(info->sc, info->whichfork, irec->br_startoff); - if (!xfs_verify_fileext(mp, irec->br_startoff, irec->br_blockcount)) + if (!xfs_verify_fileext(irec->br_startoff, irec->br_blockcount)) xchk_fblock_set_corrupt(info->sc, info->whichfork, irec->br_startoff); diff --git a/fs/xfs/scrub/bmap_repair.c b/fs/xfs/scrub/bmap_repair.c index eabffba47776e9..03af6cb92fcf5e 100644 --- a/fs/xfs/scrub/bmap_repair.c +++ b/fs/xfs/scrub/bmap_repair.c @@ -211,7 +211,7 @@ xrep_bmap_check_fork_rmap( /* Check the file offset range. */ if (!(rec->rm_flags & XFS_RMAP_BMBT_BLOCK) && - !xfs_verify_fileext(sc->mp, rec->rm_offset, rec->rm_blockcount)) + !xfs_verify_fileext(rec->rm_offset, rec->rm_blockcount)) return -EFSCORRUPTED; /* No contradictory flags. */ @@ -389,7 +389,7 @@ xrep_bmap_check_rtfork_rmap( return -EFSCORRUPTED; /* Check the file offsets and physical extents. */ - if (!xfs_verify_fileext(sc->mp, rec->rm_offset, rec->rm_blockcount)) + if (!xfs_verify_fileext(rec->rm_offset, rec->rm_blockcount)) return -EFSCORRUPTED; /* Check that this is within the rtgroup. */ diff --git a/fs/xfs/xfs_bmap_item.c b/fs/xfs/xfs_bmap_item.c index aa5b41629747d2..62be62342fea17 100644 --- a/fs/xfs/xfs_bmap_item.c +++ b/fs/xfs/xfs_bmap_item.c @@ -442,7 +442,7 @@ xfs_bui_validate( if (!xfs_verify_ino(mp, map->me_owner)) return false; - if (!xfs_verify_fileext(mp, map->me_startoff, map->me_len)) + if (!xfs_verify_fileext(map->me_startoff, map->me_len)) return false; if (map->me_flags & XFS_BMAP_EXTENT_REALTIME) diff --git a/fs/xfs/xfs_exchmaps_item.c b/fs/xfs/xfs_exchmaps_item.c index dd5d92ca1010fe..dd104bf778baf8 100644 --- a/fs/xfs/xfs_exchmaps_item.c +++ b/fs/xfs/xfs_exchmaps_item.c @@ -341,10 +341,10 @@ xfs_xmi_validate( !xfs_verify_ino(mp, xlf->xmi_inode2)) return false; - if (!xfs_verify_fileext(mp, xlf->xmi_startoff1, xlf->xmi_blockcount)) + if (!xfs_verify_fileext(xlf->xmi_startoff1, xlf->xmi_blockcount)) return false; - if (!xfs_verify_fileext(mp, xlf->xmi_startoff2, xlf->xmi_blockcount)) + if (!xfs_verify_fileext(xlf->xmi_startoff2, xlf->xmi_blockcount)) return false; if (xlf->xmi_flags & XFS_EXCHMAPS_SET_SIZES) { diff --git a/fs/xfs/xfs_rmap_item.c b/fs/xfs/xfs_rmap_item.c index 000cff1ce324f0..be2d5d6fe86337 100644 --- a/fs/xfs/xfs_rmap_item.c +++ b/fs/xfs/xfs_rmap_item.c @@ -494,7 +494,7 @@ xfs_rui_validate_map( !xfs_verify_ino(mp, map->me_owner)) return false; - if (!xfs_verify_fileext(mp, map->me_startoff, map->me_len)) + if (!xfs_verify_fileext(map->me_startoff, map->me_len)) return false; if (isrt) From 20c79db3d8b5aa61d27604f9d0a582c4644c5a69 Mon Sep 17 00:00:00 2001 From: Ard Biesheuvel Date: Sat, 5 Sep 2026 09:25:17 +0200 Subject: [PATCH 0089/1352] lib/ucs2_string: Drop arbitrary input size limit and associated WARN() It's not really the job of library code to WARN and potentially bring down the system (with panic_on_warn=1) on a condition that is fairly arbitrary to begin with. So drop the WARN_ON_ONCE() as well as the condition from ucs2_strscpy(). Reviewed-by: Vincent Mailhol Signed-off-by: Ard Biesheuvel --- lib/ucs2_string.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/lib/ucs2_string.c b/lib/ucs2_string.c index 1f7dd4eb640a87..d66fef9c9b8138 100644 --- a/lib/ucs2_string.c +++ b/lib/ucs2_string.c @@ -57,7 +57,7 @@ ssize_t ucs2_strscpy(ucs2_char_t *dst, const ucs2_char_t *src, size_t count) * Ensure that we have a valid amount of space. We need to store at * least one NUL-character. */ - if (count == 0 || WARN_ON_ONCE(count > INT_MAX / sizeof(*dst))) + if (count == 0) return -E2BIG; /* From f1c91a9a7ba98ff3eab4a14a4f443f40b1cbbc28 Mon Sep 17 00:00:00 2001 From: Ard Biesheuvel Date: Tue, 8 Sep 2026 09:08:08 +0200 Subject: [PATCH 0090/1352] lib/ucs2_string: Suppress modinfo when __DISABLE_EXPORTS is set Allow the UCS-2 string library to be reused in the EFI stub, by suppressing the modinfo data that is usually emitted so that the library can be built as a module. Reviewed-by: Vincent Mailhol Signed-off-by: Ard Biesheuvel --- lib/ucs2_string.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/lib/ucs2_string.c b/lib/ucs2_string.c index d66fef9c9b8138..f75fb4f7961a29 100644 --- a/lib/ucs2_string.c +++ b/lib/ucs2_string.c @@ -165,5 +165,7 @@ ucs2_as_utf8(u8 *dest, const ucs2_char_t *src, unsigned long maxlength) } EXPORT_SYMBOL(ucs2_as_utf8); +#ifndef __DISABLE_EXPORTS MODULE_DESCRIPTION("UCS2 string handling"); MODULE_LICENSE("GPL v2"); +#endif From d62858c982bcdfb971afe7d4834dd3186604a0ec Mon Sep 17 00:00:00 2001 From: Ard Biesheuvel Date: Wed, 9 Sep 2026 10:10:03 +0200 Subject: [PATCH 0091/1352] lib/ucs2_string: Split out ucs2_as_utf8_l() taking a separate limit ucs2_as_utf8() takes a maxlength argument, which specifies how many bytes the function is permitted to store into the destination buffer. The same value is used as an upper bound for the ucs2_strnlen() invocation, which is reasonable in the general case, as each UCS-2 character produces at least one byte of UTF-8 output, and so there is never a need to process more than 'maxlength' UCS-2 characters. However, if the UCS-2 string is not NUL terminated, ucs2_strnlen() may read past the end of the buffer if 'maxlength' is set to a high value. Current callers pass UCS-2 strings that are expected to be NUL terminated, but for processing the load options in the EFI stub, a version is needed that takes a separate limit argument. So split that off from the current implementation. Reviewed-by: Vincent Mailhol Signed-off-by: Ard Biesheuvel --- include/linux/ucs2_string.h | 11 ++++++++++- lib/ucs2_string.c | 15 ++++++++------- 2 files changed, 18 insertions(+), 8 deletions(-) diff --git a/include/linux/ucs2_string.h b/include/linux/ucs2_string.h index c499ae809c7d99..74f23ca5a96787 100644 --- a/include/linux/ucs2_string.h +++ b/include/linux/ucs2_string.h @@ -14,7 +14,16 @@ ssize_t ucs2_strscpy(ucs2_char_t *dst, const ucs2_char_t *src, size_t count); int ucs2_strncmp(const ucs2_char_t *a, const ucs2_char_t *b, size_t len); unsigned long ucs2_utf8size(const ucs2_char_t *src); +unsigned long +ucs2_as_utf8_l(u8 *dest, const ucs2_char_t *src, unsigned long limit, + unsigned long maxlength); + +static inline unsigned long ucs2_as_utf8(u8 *dest, const ucs2_char_t *src, - unsigned long maxlength); + unsigned long maxlength) +{ + return ucs2_as_utf8_l(dest, src, ucs2_strnlen(src, maxlength), + maxlength); +} #endif /* _LINUX_UCS2_STRING_H_ */ diff --git a/lib/ucs2_string.c b/lib/ucs2_string.c index f75fb4f7961a29..6c067b4280b9e4 100644 --- a/lib/ucs2_string.c +++ b/lib/ucs2_string.c @@ -125,18 +125,19 @@ ucs2_utf8size(const ucs2_char_t *src) EXPORT_SYMBOL(ucs2_utf8size); /* - * copy at most maxlength bytes of whole utf8 characters to dest from the - * ucs2 string src. + * Copy at most @limit whole utf8 characters to @dest from the ucs2 string + * @src, using no more than @maxlength bytes of buffer space. * - * The return value is the number of characters copied, not including the - * final NUL character. + * The return value is the number of bytes copied, not including the final NUL + * character. No NUL character will be appended if the output length equals + * @maxlength. */ unsigned long -ucs2_as_utf8(u8 *dest, const ucs2_char_t *src, unsigned long maxlength) +ucs2_as_utf8_l(u8 *dest, const ucs2_char_t *src, unsigned long limit, + unsigned long maxlength) { unsigned int i; unsigned long j = 0; - unsigned long limit = ucs2_strnlen(src, maxlength); for (i = 0; maxlength && i < limit; i++) { u16 c = src[i]; @@ -163,7 +164,7 @@ ucs2_as_utf8(u8 *dest, const ucs2_char_t *src, unsigned long maxlength) dest[j] = '\0'; return j; } -EXPORT_SYMBOL(ucs2_as_utf8); +EXPORT_SYMBOL(ucs2_as_utf8_l); #ifndef __DISABLE_EXPORTS MODULE_DESCRIPTION("UCS2 string handling"); From 3dc73845074fa32035e3e47fecf4c02b826296f0 Mon Sep 17 00:00:00 2001 From: Ard Biesheuvel Date: Mon, 7 Sep 2026 15:18:39 +0200 Subject: [PATCH 0092/1352] efi/libstub: Use ucs2_string library for UTF-16 to UTF-8 conversion Don't rely on sprintf() with a wide string conversion modifier to convert the command line from UTF-16 to UTF-8. Instead, use the existing ucs2 string library routine that does the same. Note that while UEFI claims support for UTF-16, in practice it ignores surrogate pairs entirely, and so the simplified UCS-2 character set (where each character takes up exactly 2 bytes) is sufficient here. This removes the only user of sprintf() in the EFI stub, so drop that function as well. Since boot memory is plentiful on UEFI systems, just establish a worst case upper bound for the size of the buffer (which can never exceed COMMAND_LINE_SIZE), and allocate that first. Then, perform the conversion, and only fall back to processing the command line character by character if that resulted in truncation. This makes the common execution path much simpler. Note that this no longer truncates the command line at the first newline, but there is no evidence that this has ever been needed. Reviewed-by: Vincent Mailhol Signed-off-by: Ard Biesheuvel --- drivers/firmware/efi/libstub/Makefile | 3 +- .../firmware/efi/libstub/efi-stub-helper.c | 98 +++++++------------ drivers/firmware/efi/libstub/vsprintf.c | 11 --- 3 files changed, 39 insertions(+), 73 deletions(-) diff --git a/drivers/firmware/efi/libstub/Makefile b/drivers/firmware/efi/libstub/Makefile index 77a2b2d74f3f62..12c0c7deb5cbe0 100644 --- a/drivers/firmware/efi/libstub/Makefile +++ b/drivers/firmware/efi/libstub/Makefile @@ -66,7 +66,8 @@ KBUILD_AFLAGS := $(KBUILD_CFLAGS) -D__ASSEMBLY__ lib-y := efi-stub-helper.o gop.o secureboot.o tpm.o \ file.o mem.o random.o randomalloc.o pci.o \ skip_spaces.o lib-cmdline.o lib-ctype.o \ - alignedmem.o printk.o vsprintf.o + alignedmem.o printk.o vsprintf.o \ + lib-ucs2_string.o # include the stub's libfdt dependencies from lib/ when needed libfdt-deps := fdt_rw.c fdt_ro.c fdt_wip.c fdt.c \ diff --git a/drivers/firmware/efi/libstub/efi-stub-helper.c b/drivers/firmware/efi/libstub/efi-stub-helper.c index f27f2e1f001997..3dc3654923017a 100644 --- a/drivers/firmware/efi/libstub/efi-stub-helper.c +++ b/drivers/firmware/efi/libstub/efi-stub-helper.c @@ -12,6 +12,7 @@ #include #include #include +#include #include #include @@ -334,81 +335,56 @@ char *efi_convert_cmdline(efi_loaded_image_t *image) { const efi_char16_t *options = efi_table_attr(image, load_options); u32 options_size = efi_table_attr(image, load_options_size); - int options_bytes = 0, safe_options_bytes = 0; /* UTF-8 bytes */ - unsigned long cmdline_addr = 0; - const efi_char16_t *s2; - bool in_quote = false; + unsigned long options_chars = 0; + unsigned long cmdline_bytes; efi_status_t status; - u32 options_chars; + char *cmdline_addr; if (options_size > 0) efi_measure_tagged_event((unsigned long)options, options_size, EFISTUB_EVT_LOAD_OPTIONS); efi_apply_loadoptions_quirk((const void **)&options, &options_size); - options_chars = options_size / sizeof(efi_char16_t); - - if (options) { - s2 = options; - while (options_bytes < COMMAND_LINE_SIZE && options_chars--) { - efi_char16_t c = *s2++; - - if (c < 0x80) { - if (c == L'\0' || c == L'\n') - break; - if (c == L'"') - in_quote = !in_quote; - else if (!in_quote && isspace((char)c)) - safe_options_bytes = options_bytes; - - options_bytes++; - continue; - } - - /* - * Get the number of UTF-8 bytes corresponding to a - * UTF-16 character. - * The first part handles everything in the BMP. - */ - options_bytes += 2 + (c >= 0x800); - /* - * Add one more byte for valid surrogate pairs. Invalid - * surrogates will be replaced with 0xfffd and take up - * only 3 bytes. - */ - if ((c & 0xfc00) == 0xd800) { - /* - * If the very last word is a high surrogate, - * we must ignore it since we can't access the - * low surrogate. - */ - if (!options_chars) { - options_bytes -= 3; - } else if ((*s2 & 0xfc00) == 0xdc00) { - options_bytes++; - options_chars--; - s2++; - } - } - } - if (options_bytes >= COMMAND_LINE_SIZE) { - options_bytes = safe_options_bytes; - efi_err("Command line is too long: truncated to %d bytes\n", - options_bytes); - } - } + if (options) + options_chars = ucs2_strnlen(options, + options_size / sizeof(efi_char16_t)); - options_bytes++; /* NUL termination */ + /* Each UCS-2 char takes up at most 3 UTF-8 bytes */ + cmdline_bytes = min(3 * options_chars, COMMAND_LINE_SIZE - 1) + 3; - status = efi_bs_call(allocate_pool, EFI_LOADER_DATA, options_bytes, + status = efi_bs_call(allocate_pool, EFI_LOADER_DATA, cmdline_bytes, (void **)&cmdline_addr); if (status != EFI_SUCCESS) return NULL; - snprintf((char *)cmdline_addr, options_bytes, "%.*ls", - options_bytes - 1, options); + if (ucs2_as_utf8_l(cmdline_addr, options, options_chars, + cmdline_bytes) >= COMMAND_LINE_SIZE) { + /* + * The output fills up the entire buffer, and may have been + * truncated. Work backwards through the buffer to find a safe + * truncation point (i.e., a blank character not inside a + * quoted string). + */ + int safe_pos[2] = {}; + int in_quote = 0; + + for (int i = COMMAND_LINE_SIZE - 1; i >= 0; i--) { + char c = cmdline_addr[i]; + + if (!c) + return cmdline_addr; + else if (c == '"') + in_quote ^= 1; + else if (!safe_pos[in_quote] && isspace(c)) + safe_pos[in_quote] = i; + } + + efi_err("Command line is too long: truncated to %d bytes\n", + safe_pos[in_quote]); + cmdline_addr[safe_pos[in_quote]] = '\0'; + } - return (char *)cmdline_addr; + return cmdline_addr; } /** diff --git a/drivers/firmware/efi/libstub/vsprintf.c b/drivers/firmware/efi/libstub/vsprintf.c index 71c71c222346a4..dba1366791729a 100644 --- a/drivers/firmware/efi/libstub/vsprintf.c +++ b/drivers/firmware/efi/libstub/vsprintf.c @@ -551,14 +551,3 @@ int vsnprintf(char *buf, size_t size, const char *fmt, va_list ap) return pos; } - -int snprintf(char *buf, size_t size, const char *fmt, ...) -{ - va_list args; - int i; - - va_start(args, fmt); - i = vsnprintf(buf, size, fmt, args); - va_end(args); - return i; -} From 9bcc2af24b8671fc9ba3f59502f5469288a74934 Mon Sep 17 00:00:00 2001 From: Ard Biesheuvel Date: Sun, 6 Sep 2026 13:15:12 +0200 Subject: [PATCH 0093/1352] efi/libstub: Avoid efi_puts() for compile time constant strings efi_puts() performs a UTF-8 to UTF-16 conversion on its input, as the EFI console's native character set is UTF-16. This is pointless for compile time constant strings, since we can simply define those as UTF-16 to begin with. This takes slightly more space, but removes any runtime handling of those strings, simplifying the code. Note that efi_puts() also performs LF to CR-LF conversion, so this needs to be taken into account as well. Reviewed-by: Vincent Mailhol Signed-off-by: Ard Biesheuvel --- drivers/firmware/efi/libstub/gop.c | 6 +++--- drivers/firmware/efi/libstub/printk.c | 4 ++-- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/drivers/firmware/efi/libstub/gop.c b/drivers/firmware/efi/libstub/gop.c index 80dc8cfeb33e96..6919e28ba92bde 100644 --- a/drivers/firmware/efi/libstub/gop.c +++ b/drivers/firmware/efi/libstub/gop.c @@ -309,12 +309,12 @@ static u32 choose_mode_list(efi_graphics_output_protocol_t *gop) efi_status_t status; efi_printk("Available graphics modes are 0-%u\n", max_mode-1); - efi_puts(" * = current mode\n" - " - = unusable mode\n"); + efi_char16_puts(L" * = current mode\r\n" + " - = unusable mode\r\n"); choose_mode(gop, match_list, (void *)cur_mode); - efi_puts("\nPress any key to continue (or wait 10 seconds)\n"); + efi_char16_puts(L"\r\nPress any key to continue (or wait 10 seconds)\r\n"); status = efi_wait_for_key(10 * EFI_USEC_PER_SEC, &key); if (status != EFI_SUCCESS && status != EFI_TIMEOUT) { efi_err("Unable to read key, continuing in 10 seconds\n"); diff --git a/drivers/firmware/efi/libstub/printk.c b/drivers/firmware/efi/libstub/printk.c index bc599212c05dd7..f36639886d004a 100644 --- a/drivers/firmware/efi/libstub/printk.c +++ b/drivers/firmware/efi/libstub/printk.c @@ -136,7 +136,7 @@ int efi_printk(const char *fmt, ...) return 0; if (loglevel >= 0) - efi_puts("EFI stub: "); + efi_char16_puts(L"EFI stub: "); fmt = printk_skip_level(fmt); @@ -146,7 +146,7 @@ int efi_printk(const char *fmt, ...) efi_puts(printf_buf); if (printed >= sizeof(printf_buf)) { - efi_puts("[Message truncated]\n"); + efi_char16_puts(L"[Message truncated]\r\n"); return -1; } From 10bfb3ed534efe4a93c8a87e2941a77623554d55 Mon Sep 17 00:00:00 2001 From: Ard Biesheuvel Date: Sun, 6 Sep 2026 13:23:41 +0200 Subject: [PATCH 0094/1352] efi/libstub: Output UTF-16 directly from vsnprintf() The only remaining users of vsnprintf() in the EFI stub are the diagnostic printk()'s, which are emitted to the console and not recorded for posterity. The EFI console uses UTF-16 (or actually, UCS-2) natively, and so all non-UTF16 strings that are emitted need to be converted. Given the stub's vsnprintf() support for wide strings (using the %ls conversion modifier), which uses UTF-16 to UTF-8 conversion internally, the final conversion to UTF-16 needs to support not just plain ASCII but UTF-8 as well. This is all pointless, of course, and it makes more sense to use UTF-16 internally. This removes the need for UTF-16 to UTF-8 conversion in vsnprintf(), and given that all non-wide string inputs to vsnprintf() that exist in the stub today are compile time constant ASCII strings, the need to convert UTF-8 to UTF-16 disappears as well. So implement efi_vsnprintf() taking a const char *fmt as before, but outputting a efi_char16_t[] that can be passed to the EFI console directly, rather than via efi_puts(), leaving the latter unused and therefore removed. Note that efi_puts() performs LF to CR-LF conversion internally, so add this capability to efi_vsnprintf() as well. Reviewed-by: Vincent Mailhol Signed-off-by: Ard Biesheuvel --- drivers/firmware/efi/libstub/efistub.h | 5 +- drivers/firmware/efi/libstub/printk.c | 91 ++--------------------- drivers/firmware/efi/libstub/vsprintf.c | 96 ++++--------------------- 3 files changed, 22 insertions(+), 170 deletions(-) diff --git a/drivers/firmware/efi/libstub/efistub.h b/drivers/firmware/efi/libstub/efistub.h index fd91fc15ec810b..36056c6247820b 100644 --- a/drivers/firmware/efi/libstub/efistub.h +++ b/drivers/firmware/efi/libstub/efistub.h @@ -1078,9 +1078,10 @@ efi_status_t check_platform_features(void); void *get_efi_config_table(efi_guid_t guid); -/* NOTE: These functions do not print a trailing newline after the string */ void efi_char16_puts(efi_char16_t *); -void efi_puts(const char *str); + +int efi_vsnprintf(efi_char16_t *buf, size_t size, const char *fmt, va_list ap, + bool crlf); __printf(1, 2) int efi_printk(char const *fmt, ...); diff --git a/drivers/firmware/efi/libstub/printk.c b/drivers/firmware/efi/libstub/printk.c index f36639886d004a..0a18cfe325283e 100644 --- a/drivers/firmware/efi/libstub/printk.c +++ b/drivers/firmware/efi/libstub/printk.c @@ -23,98 +23,20 @@ void efi_char16_puts(efi_char16_t *str) output_string, str); } -static -u32 utf8_to_utf32(const u8 **s8) -{ - u32 c32; - u8 c0, cx; - size_t clen, i; - - c0 = cx = *(*s8)++; - /* - * The position of the most-significant 0 bit gives us the length of - * a multi-octet encoding. - */ - for (clen = 0; cx & 0x80; ++clen) - cx <<= 1; - /* - * If the 0 bit is in position 8, this is a valid single-octet - * encoding. If the 0 bit is in position 7 or positions 1-3, the - * encoding is invalid. - * In either case, we just return the first octet. - */ - if (clen < 2 || clen > 4) - return c0; - /* Get the bits from the first octet. */ - c32 = cx >> clen--; - for (i = 0; i < clen; ++i) { - /* Trailing octets must have 10 in most significant bits. */ - cx = (*s8)[i] ^ 0x80; - if (cx & 0xc0) - return c0; - c32 = (c32 << 6) | cx; - } - /* - * Check for validity: - * - The character must be in the Unicode range. - * - It must not be a surrogate. - * - It must be encoded using the correct number of octets. - */ - if (c32 > 0x10ffff || - (c32 & 0xf800) == 0xd800 || - clen != (c32 >= 0x80) + (c32 >= 0x800) + (c32 >= 0x10000)) - return c0; - *s8 += clen; - return c32; -} - -/** - * efi_puts() - Write a UTF-8 encoded string to the console - * @str: UTF-8 encoded string - */ -void efi_puts(const char *str) -{ - efi_char16_t buf[128]; - size_t pos = 0, lim = ARRAY_SIZE(buf); - const u8 *s8 = (const u8 *)str; - u32 c32; - - while (*s8) { - if (*s8 == '\n') - buf[pos++] = L'\r'; - c32 = utf8_to_utf32(&s8); - if (c32 < 0x10000) { - /* Characters in plane 0 use a single word. */ - buf[pos++] = c32; - } else { - /* - * Characters in other planes encode into a surrogate - * pair. - */ - buf[pos++] = (0xd800 - (0x10000 >> 10)) + (c32 >> 10); - buf[pos++] = 0xdc00 + (c32 & 0x3ff); - } - if (*s8 == '\0' || pos >= lim - 2) { - buf[pos] = L'\0'; - efi_char16_puts(buf); - pos = 0; - } - } -} - /** * efi_printk() - Print a kernel message * @fmt: format string * * The first letter of the format string is used to determine the logging level * of the message. If the level is less then the current EFI logging level, the - * message is suppressed. The message will be truncated to 255 bytes. + * message is suppressed. The message will be truncated to 255 characters + * (ignoring surrogates). * * Return: number of printed characters */ int efi_printk(const char *fmt, ...) { - char printf_buf[256]; + efi_char16_t printf_buf[256]; va_list args; int printed; int loglevel = printk_get_level(fmt); @@ -141,11 +63,12 @@ int efi_printk(const char *fmt, ...) fmt = printk_skip_level(fmt); va_start(args, fmt); - printed = vsnprintf(printf_buf, sizeof(printf_buf), fmt, args); + printed = efi_vsnprintf(printf_buf, ARRAY_SIZE(printf_buf), fmt, args, + true); va_end(args); - efi_puts(printf_buf); - if (printed >= sizeof(printf_buf)) { + efi_char16_puts(printf_buf); + if (printed >= ARRAY_SIZE(printf_buf)) { efi_char16_puts(L"[Message truncated]\r\n"); return -1; } diff --git a/drivers/firmware/efi/libstub/vsprintf.c b/drivers/firmware/efi/libstub/vsprintf.c index dba1366791729a..bd32af6b4f4db4 100644 --- a/drivers/firmware/efi/libstub/vsprintf.c +++ b/drivers/firmware/efi/libstub/vsprintf.c @@ -14,10 +14,14 @@ #include #include +#include #include #include #include #include +#include + +#include "efistub.h" static int skip_atoi(const char **s) @@ -239,58 +243,6 @@ char get_sign(long long *num, int flags) return 0; } -static -size_t utf16s_utf8nlen(const u16 *s16, size_t maxlen) -{ - size_t len, clen; - - for (len = 0; len < maxlen && *s16; len += clen) { - u16 c0 = *s16++; - - /* First, get the length for a BMP character */ - clen = 1 + (c0 >= 0x80) + (c0 >= 0x800); - if (len + clen > maxlen) - break; - /* - * If this is a high surrogate, and we're already at maxlen, we - * can't include the character if it's a valid surrogate pair. - * Avoid accessing one extra word just to check if it's valid - * or not. - */ - if ((c0 & 0xfc00) == 0xd800) { - if (len + clen == maxlen) - break; - if ((*s16 & 0xfc00) == 0xdc00) { - ++s16; - ++clen; - } - } - } - - return len; -} - -static -u32 utf16_to_utf32(const u16 **s16) -{ - u16 c0, c1; - - c0 = *(*s16)++; - /* not a surrogate */ - if ((c0 & 0xf800) != 0xd800) - return c0; - /* invalid: low surrogate instead of high */ - if (c0 & 0x0400) - return 0xfffd; - c1 = **s16; - /* invalid: missing low surrogate */ - if ((c1 & 0xfc00) != 0xdc00) - return 0xfffd; - /* valid surrogate pair */ - ++(*s16); - return (0x10000 - (0xd800 << 10) - 0xdc00) + (c0 << 10) + c1; -} - #define PUTC(c) \ do { \ if (pos < size) \ @@ -298,7 +250,8 @@ do { \ ++pos; \ } while (0); -int vsnprintf(char *buf, size_t size, const char *fmt, va_list ap) +int efi_vsnprintf(efi_char16_t *buf, size_t size, const char *fmt, va_list ap, + bool crlf) { /* The maximum space required is to print a 64-bit number in octal */ char tmp[(sizeof(unsigned long long) * 8 + 2) / 3]; @@ -336,6 +289,8 @@ int vsnprintf(char *buf, size_t size, const char *fmt, va_list ap) for (pos = 0; *fmt; ++fmt) { if (*fmt != '%' || *++fmt == '%') { + if (crlf && *fmt == '\n') + PUTC('\r'); PUTC(*fmt); continue; } @@ -400,7 +355,7 @@ int vsnprintf(char *buf, size_t size, const char *fmt, va_list ap) else if (qualifier == 'l') { wstring: flags |= WIDE; - precision = len = utf16s_utf8nlen((const u16 *)s, precision); + precision = len = ucs2_strnlen((const u16 *)s, precision); goto output; } precision = len = strnlen(s, precision); @@ -505,36 +460,9 @@ int vsnprintf(char *buf, size_t size, const char *fmt, va_list ap) if (flags & WIDE) { const u16 *ws = (const u16 *)s; - while (len-- > 0) { - u32 c32 = utf16_to_utf32(&ws); - u8 *s8; - size_t clen; - - if (c32 < 0x80) { - PUTC(c32); - continue; - } - - /* Number of trailing octets */ - clen = 1 + (c32 >= 0x800) + (c32 >= 0x10000); - - len -= clen; - s8 = (u8 *)&buf[pos]; - - /* Avoid writing partial character */ - PUTC('\0'); - pos += clen; - if (pos >= size) - continue; - - /* Set high bits of leading octet */ - *s8 = (0xf00 >> 1) >> clen; - /* Write trailing octets in reverse order */ - for (s8 += clen; clen; --clen, c32 >>= 6) - *s8-- = 0x80 | (c32 & 0x3f); - /* Set low bits of leading octet */ - *s8 |= c32; - } + if (pos < size) + memcpy(&buf[pos], ws, min(len, size - pos) * sizeof(*ws)); + pos += len; } else { while (len-- > 0) PUTC(*s++); From f06dffe7be93051b66497da79e07e241f15fbf33 Mon Sep 17 00:00:00 2001 From: Ard Biesheuvel Date: Fri, 4 Sep 2026 17:46:58 +0200 Subject: [PATCH 0095/1352] efi/libstub: Add support for printing human readable GUIDs Add support for the %pUl printk conversion specifier, which takes a pointer to a GUID and prints it in the usual format: aaaaaaaa-bbbb-cccc-dddd-dddddddddddd Co-developed-by: Vincent Mailhol Signed-off-by: Vincent Mailhol Reviewed-by: Vincent Mailhol Signed-off-by: Ard Biesheuvel --- drivers/firmware/efi/libstub/vsprintf.c | 40 +++++++++++++++++++++---- 1 file changed, 35 insertions(+), 5 deletions(-) diff --git a/drivers/firmware/efi/libstub/vsprintf.c b/drivers/firmware/efi/libstub/vsprintf.c index bd32af6b4f4db4..7f6b891a338ac7 100644 --- a/drivers/firmware/efi/libstub/vsprintf.c +++ b/drivers/firmware/efi/libstub/vsprintf.c @@ -113,6 +113,9 @@ char *put_dec(char *end, unsigned long long n) return p; } +/* we are called with base 8, 10 or 16, only, thus don't need "G..." */ +static const char digits[16] = "0123456789ABCDEF"; /* "GHIJKLMNOPQRSTUVWXYZ"; */ + static char *number(char *end, unsigned long long num, int base, char locase) { @@ -121,9 +124,6 @@ char *number(char *end, unsigned long long num, int base, char locase) * produces same digits or (maybe lowercased) letters */ - /* we are called with base 8, 10 or 16, only, thus don't need "G..." */ - static const char digits[16] = "0123456789ABCDEF"; /* "GHIJKLMNOPQRSTUVWXYZ"; */ - switch (base) { case 10: if (num != 0) @@ -144,6 +144,29 @@ char *number(char *end, unsigned long long num, int base, char locase) return end; } +static char *guid_to_str(const efi_guid_t *guid, char *out, char locase) +{ + static const u8 guid_index[UUID_SIZE] = { + 3, 2, 1, 0, 5, 4, 7, 6, 8, 9, 10, 11, 12, 13, 14, 15, + }; + + for (int i = 0, p = 0; i < ARRAY_SIZE(guid_index); i++) { + u8 byte = guid->b[guid_index[i]]; + + out[p++] = locase | digits[byte >> 4]; + out[p++] = locase | digits[byte & 0xf]; + + switch (i) { + case 3: + case 5: + case 7: + case 9: + out[p++] = '-'; + } + } + return out; +} + #define ZEROPAD 1 /* pad with zero */ #define SIGN 2 /* unsigned/signed long */ #define PLUS 4 /* show plus */ @@ -253,8 +276,7 @@ do { \ int efi_vsnprintf(efi_char16_t *buf, size_t size, const char *fmt, va_list ap, bool crlf) { - /* The maximum space required is to print a 64-bit number in octal */ - char tmp[(sizeof(unsigned long long) * 8 + 2) / 3]; + char tmp[UUID_STRING_LEN]; char *tmp_end = &tmp[ARRAY_SIZE(tmp)]; long long num; int base; @@ -367,6 +389,14 @@ int efi_vsnprintf(efi_char16_t *buf, size_t size, const char *fmt, va_list ap, break; case 'p': + if (fmt[1] == 'U' && (fmt[2] | 0x20) == 'l') { + flags &= LEFT; + s = guid_to_str(va_arg(args, efi_guid_t *), tmp, fmt[2] & 0x20); + precision = len = UUID_STRING_LEN; + fmt += 2; + goto output; + } + if (precision < 0) precision = 2 * sizeof(void *); fallthrough; From 1b2a518428720d166ce955b67384ca3dcf274a97 Mon Sep 17 00:00:00 2001 From: Ard Biesheuvel Date: Sun, 6 Sep 2026 13:45:03 +0200 Subject: [PATCH 0096/1352] efi/libstub: Add efi_snprintf() to construct wide strings The native EFI character set is UTF-16 (or in practice, UCS-2). Implement efi_snprintf() to construct UTF-16 strings using printf style templates. This will be used in a subsequent patch to set the LoaderDevicePartUUID EFI variable. Reviewed-by: Vincent Mailhol Signed-off-by: Ard Biesheuvel --- drivers/firmware/efi/libstub/efistub.h | 1 + drivers/firmware/efi/libstub/vsprintf.c | 11 +++++++++++ 2 files changed, 12 insertions(+) diff --git a/drivers/firmware/efi/libstub/efistub.h b/drivers/firmware/efi/libstub/efistub.h index 36056c6247820b..880c1d0c464bc9 100644 --- a/drivers/firmware/efi/libstub/efistub.h +++ b/drivers/firmware/efi/libstub/efistub.h @@ -1084,6 +1084,7 @@ int efi_vsnprintf(efi_char16_t *buf, size_t size, const char *fmt, va_list ap, bool crlf); __printf(1, 2) int efi_printk(char const *fmt, ...); +__printf(3, 4) int efi_snprintf(efi_char16_t *buf, size_t size, const char *fmt, ...); void efi_free(unsigned long size, unsigned long addr); DEFINE_FREE(efi_pool, void *, if (_T) efi_bs_call(free_pool, _T)); diff --git a/drivers/firmware/efi/libstub/vsprintf.c b/drivers/firmware/efi/libstub/vsprintf.c index 7f6b891a338ac7..7fd6589f44ef91 100644 --- a/drivers/firmware/efi/libstub/vsprintf.c +++ b/drivers/firmware/efi/libstub/vsprintf.c @@ -509,3 +509,14 @@ int efi_vsnprintf(efi_char16_t *buf, size_t size, const char *fmt, va_list ap, return pos; } + +int efi_snprintf(efi_char16_t *buf, size_t size, const char *fmt, ...) +{ + va_list args; + int i; + + va_start(args, fmt); + i = efi_vsnprintf(buf, size, fmt, args, false); + va_end(args); + return i; +} From 9925e4ce78365f6378ae7facdc987b0b4ef67d37 Mon Sep 17 00:00:00 2001 From: Vincent Mailhol Date: Sun, 6 Sep 2026 23:52:11 +0200 Subject: [PATCH 0097/1352] efi/libstub: add initial Boot Loader Interface support MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The Boot Loader Interface (BLI) [1] defines EFI variables that expose boot loader state to the running OS. LoaderInfo identifies the boot loader, while LoaderDevicePartUUID records the GPT partition UUID of the partition containing it. LoaderDevicePartUUID is used, for example, by systemd-gpt-auto-generator [2] to identify the disk the boot loader was launched from and automatically detect and mount partitions on it. GRUB [3] and systemd-boot [4] populate these variables, but when the kernel is started directly by EFI firmware, there is no conventional external boot loader to provide them. In that case, because the EFI stub performs the boot loader role, it should provide the variables itself. Use LoaderInfo as a sentinel: if it is already set by an earlier boot stage or cannot be set, bail out. Otherwise, populate the other BLI variables. Parse the loaded image device path, extract the GUID signature from its GPT HD() node and publish it under the Linux loader entry vendor GUID as the volatile LoaderDevicePartUUID EFI variable. Install the efi_bli_set_variables() hook in both the generic efi-stub.c path and the x86-specific x86-stub.c path. [1] The Boot Loader Interface Link: https://systemd.io/BOOT_LOADER_INTERFACE/ [2] systemd-gpt-auto-generator Link: https://www.freedesktop.org/software/systemd/man/latest/systemd-gpt-auto-generator.html [3] GRUB -- §16.2 bli Link: https://www.gnu.org/software/grub/manual/grub/html_node/bli_005fmodule.html [4] systemd -- systemd-boot UEFI Boot Manager Link: https://github.com/systemd/systemd/blob/main/docs/BOOT.md?plain=1#L102 Signed-off-by: Vincent Mailhol [ardb: - constify 'image' pointer parameter - pass efi_guid_t* to efi_snprintf()] Signed-off-by: Ard Biesheuvel --- drivers/firmware/efi/libstub/Makefile | 2 +- drivers/firmware/efi/libstub/bli.c | 87 +++++++++++++++++++++++++ drivers/firmware/efi/libstub/efi-stub.c | 1 + drivers/firmware/efi/libstub/efistub.h | 2 + drivers/firmware/efi/libstub/x86-stub.c | 1 + include/linux/efi.h | 22 +++++++ 6 files changed, 114 insertions(+), 1 deletion(-) create mode 100644 drivers/firmware/efi/libstub/bli.c diff --git a/drivers/firmware/efi/libstub/Makefile b/drivers/firmware/efi/libstub/Makefile index 12c0c7deb5cbe0..564773c89d1464 100644 --- a/drivers/firmware/efi/libstub/Makefile +++ b/drivers/firmware/efi/libstub/Makefile @@ -66,7 +66,7 @@ KBUILD_AFLAGS := $(KBUILD_CFLAGS) -D__ASSEMBLY__ lib-y := efi-stub-helper.o gop.o secureboot.o tpm.o \ file.o mem.o random.o randomalloc.o pci.o \ skip_spaces.o lib-cmdline.o lib-ctype.o \ - alignedmem.o printk.o vsprintf.o \ + alignedmem.o printk.o vsprintf.o bli.o \ lib-ucs2_string.o # include the stub's libfdt dependencies from lib/ when needed diff --git a/drivers/firmware/efi/libstub/bli.c b/drivers/firmware/efi/libstub/bli.c new file mode 100644 index 00000000000000..b2407f63b743d2 --- /dev/null +++ b/drivers/firmware/efi/libstub/bli.c @@ -0,0 +1,87 @@ +// SPDX-License-Identifier: GPL-2.0 + +#include + +#include +#include +#include + +#include "efistub.h" + +static efi_guid_t loader_entry_guid = LINUX_EFI_LOADER_ENTRY_GUID; + +static const struct efi_hd_dev_path * +efi_bli_find_hd_node(const struct efi_dev_path *path) +{ + const struct efi_dev_path *node; + u16 node_len; + + for (node = path; + node->header.type != EFI_DEV_END_PATH && + node->header.type != EFI_DEV_END_PATH2; + node = (const void *)node + node_len) { + node_len = get_unaligned_le16(&node->header.length); + + if (node_len < sizeof(node->header)) + return NULL; + + if (node->header.type != EFI_DEV_MEDIA || + node->header.sub_type != EFI_DEV_MEDIA_HARD_DRIVE) + continue; + + if (node_len < sizeof(node->hd)) + return NULL; + + if (node->hd.partition_format != EFI_HD_PARTITION_FORMAT_GPT || + node->hd.signature_type != EFI_HD_SIGNATURE_TYPE_GUID) + continue; + + return &node->hd; + } + + return NULL; +} + +static void efi_bli_populate_loader_part_uuid(const efi_loaded_image_t *image) +{ + static efi_guid_t device_path_guid = EFI_DEVICE_PATH_PROTOCOL_GUID; + efi_char16_t partuuid[UUID_STRING_LEN + 1]; + const struct efi_hd_dev_path *hd_node; + const struct efi_dev_path *path; + + if (efi_bs_call(handle_protocol, efi_table_attr(image, device_handle), + &device_path_guid, (void **)&path) != EFI_SUCCESS) + return; + + hd_node = efi_bli_find_hd_node(path); + if (!hd_node) + return; + + if (efi_snprintf(partuuid, ARRAY_SIZE(partuuid), "%pUl", + &hd_node->signature) != UUID_STRING_LEN) + return; + + set_efi_var(L"LoaderDevicePartUUID", &loader_entry_guid, + EFI_VARIABLE_BOOTSERVICE_ACCESS | EFI_VARIABLE_RUNTIME_ACCESS, + sizeof(partuuid), partuuid); +} + +void efi_bli_set_variables(const efi_loaded_image_t *image) +{ + static efi_char16_t loader_info[] = L"Linux EFI stub " UTS_RELEASE; + unsigned long size = 0; + + if (!image) + return; + + if (get_efi_var(L"LoaderInfo", &loader_entry_guid, + NULL, &size, NULL) != EFI_NOT_FOUND) + return; + + if (set_efi_var(L"LoaderInfo", &loader_entry_guid, + EFI_VARIABLE_BOOTSERVICE_ACCESS | EFI_VARIABLE_RUNTIME_ACCESS, + sizeof(loader_info), loader_info) != EFI_SUCCESS) + return; + + efi_bli_populate_loader_part_uuid(image); +} diff --git a/drivers/firmware/efi/libstub/efi-stub.c b/drivers/firmware/efi/libstub/efi-stub.c index 42d6073bcd062a..2a95f4ea104a53 100644 --- a/drivers/firmware/efi/libstub/efi-stub.c +++ b/drivers/firmware/efi/libstub/efi-stub.c @@ -165,6 +165,7 @@ efi_status_t efi_stub_common(efi_handle_t handle, dpy = setup_primary_display(); efi_retrieve_eventlog(); + efi_bli_set_variables(image); /* Ask the firmware to clear memory on unclean shutdown */ efi_enable_reset_attack_mitigation(); diff --git a/drivers/firmware/efi/libstub/efistub.h b/drivers/firmware/efi/libstub/efistub.h index 880c1d0c464bc9..4f9e7ae28b6c27 100644 --- a/drivers/firmware/efi/libstub/efistub.h +++ b/drivers/firmware/efi/libstub/efistub.h @@ -1072,6 +1072,8 @@ efi_status_t efi_random_alloc(unsigned long size, unsigned long align, int memory_type, unsigned long alloc_min, unsigned long alloc_max); +void efi_bli_set_variables(const efi_loaded_image_t *image); + efi_status_t efi_random_get_seed(void); efi_status_t check_platform_features(void); diff --git a/drivers/firmware/efi/libstub/x86-stub.c b/drivers/firmware/efi/libstub/x86-stub.c index cef32e2c82d8fd..b762f7f37f2800 100644 --- a/drivers/firmware/efi/libstub/x86-stub.c +++ b/drivers/firmware/efi/libstub/x86-stub.c @@ -1014,6 +1014,7 @@ void __noreturn efi_stub_entry(efi_handle_t handle, efi_random_get_seed(); efi_retrieve_eventlog(); + efi_bli_set_variables(image); setup_graphics(boot_params); diff --git a/include/linux/efi.h b/include/linux/efi.h index aa15ff88539bdf..4bc47df2d8a722 100644 --- a/include/linux/efi.h +++ b/include/linux/efi.h @@ -957,6 +957,17 @@ extern int efi_status_to_err(efi_status_t status); #define EFI_DEV_END_INSTANCE 0x01 #define EFI_DEV_END_ENTIRE 0xFF +enum efi_hd_partition_format { + EFI_HD_PARTITION_FORMAT_MBR = 1, + EFI_HD_PARTITION_FORMAT_GPT, +}; + +enum efi_hd_signature_type { + EFI_HD_SIGNATURE_TYPE_NONE, + EFI_HD_SIGNATURE_TYPE_MBR, + EFI_HD_SIGNATURE_TYPE_GUID, +}; + struct efi_generic_dev_path { u8 type; u8 sub_type; @@ -988,6 +999,16 @@ struct efi_rel_offset_dev_path { u64 ending_offset; } __packed; +struct efi_hd_dev_path { + struct efi_generic_dev_path header; + u32 partition_number; + u64 partition_start; + u64 partition_size; + efi_guid_t signature; + u8 partition_format; + u8 signature_type; +} __packed; + struct efi_mem_mapped_dev_path { struct efi_generic_dev_path header; u32 memory_type; @@ -1007,6 +1028,7 @@ struct efi_dev_path { struct efi_pci_dev_path pci; struct efi_vendor_dev_path vendor; struct efi_rel_offset_dev_path rel_offset; + struct efi_hd_dev_path hd; }; } __packed; From 552e0bb24d5e6bf4bee6dcaade3e1c1ca89a6f58 Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Thu, 10 Sep 2026 22:42:33 +0300 Subject: [PATCH 0098/1352] drm/i915/display: stop using the configurable fence timeout MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit i915 has the Kconfig option DRM_I915_FENCE_TIMEOUT, defaulting to 10 seconds. xe doesn't use it, instead defaulting to MAX_SCHEDULE_TIMEOUT. Unify the behaviour by switching to dma_fence_wait() which defaults to MAX_SCHEDULE_TIMEOUT. As dma_fence_wait() returns 0 when the fence was signaled, update the return value check. This should be possible now that commit 3aef9c94c02b ("drm/i915: Perform full wedge on display reset") has been merged, and we no longer rely on the timeout. v3: Try this again with full wedge v2: Use dma_fence_wait(), fix return value check (Maarten) Reviewed-by: Maarten Lankhorst Cc: Ville Syrjälä Link: https://patch.msgid.link/1880e49c9270049d5dfd944a1cbf712d664d02aa.1789069102.git.jani.nikula@intel.com Signed-off-by: Jani Nikula --- drivers/gpu/drm/i915/display/intel_display.c | 6 ++---- .../gpu/drm/xe/compat-i915-headers/i915_config.h | 16 ---------------- 2 files changed, 2 insertions(+), 20 deletions(-) delete mode 100644 drivers/gpu/drm/xe/compat-i915-headers/i915_config.h diff --git a/drivers/gpu/drm/i915/display/intel_display.c b/drivers/gpu/drm/i915/display/intel_display.c index 9151ea6c15ab84..579b03034c46c2 100644 --- a/drivers/gpu/drm/i915/display/intel_display.c +++ b/drivers/gpu/drm/i915/display/intel_display.c @@ -50,7 +50,6 @@ #include "g4x_dp.h" #include "g4x_hdmi.h" #include "hsw_ips.h" -#include "i915_config.h" #include "i9xx_plane.h" #include "i9xx_plane_regs.h" #include "i9xx_wm.h" @@ -7298,9 +7297,8 @@ static void intel_atomic_commit_fence_wait(struct intel_atomic_state *intel_stat for_each_new_plane_in_state(&intel_state->base, plane, new_plane_state, i) { if (new_plane_state->fence) { - ret = dma_fence_wait_timeout(new_plane_state->fence, false, - i915_fence_timeout()); - if (ret <= 0) + ret = dma_fence_wait(new_plane_state->fence, false); + if (ret < 0) break; dma_fence_put(new_plane_state->fence); diff --git a/drivers/gpu/drm/xe/compat-i915-headers/i915_config.h b/drivers/gpu/drm/xe/compat-i915-headers/i915_config.h deleted file mode 100644 index d4522203e2dd8f..00000000000000 --- a/drivers/gpu/drm/xe/compat-i915-headers/i915_config.h +++ /dev/null @@ -1,16 +0,0 @@ -/* SPDX-License-Identifier: MIT */ -/* - * Copyright © 2023 Intel Corporation - */ - -#ifndef __I915_CONFIG_H__ -#define __I915_CONFIG_H__ - -#include - -static inline unsigned long i915_fence_timeout(void) -{ - return MAX_SCHEDULE_TIMEOUT; -} - -#endif /* __I915_CONFIG_H__ */ From c2e1495aff758181d420100f0a42abbc779109c0 Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Thu, 10 Sep 2026 22:42:34 +0300 Subject: [PATCH 0099/1352] drm/i915/display: use struct intel_atomic_state *state variable naming MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The struct intel_atomic_state * pointer variables and parameters are named "state" everywhere. Follow suit in intel_atomic_commit_fence_wait(). Cc: Ville Syrjälä Cc: Maarten Lankhorst Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/988f1253004e68d565f3915cbd24182ca95dee49.1789069102.git.jani.nikula@intel.com Signed-off-by: Jani Nikula --- drivers/gpu/drm/i915/display/intel_display.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_display.c b/drivers/gpu/drm/i915/display/intel_display.c index 579b03034c46c2..1c2e50c6d98a5c 100644 --- a/drivers/gpu/drm/i915/display/intel_display.c +++ b/drivers/gpu/drm/i915/display/intel_display.c @@ -7288,14 +7288,14 @@ static void skl_commit_modeset_enables(struct intel_atomic_state *state) drm_WARN_ON(display->drm, update_pipes); } -static void intel_atomic_commit_fence_wait(struct intel_atomic_state *intel_state) +static void intel_atomic_commit_fence_wait(struct intel_atomic_state *state) { struct drm_plane *plane; struct drm_plane_state *new_plane_state; long ret; int i; - for_each_new_plane_in_state(&intel_state->base, plane, new_plane_state, i) { + for_each_new_plane_in_state(&state->base, plane, new_plane_state, i) { if (new_plane_state->fence) { ret = dma_fence_wait(new_plane_state->fence, false); if (ret < 0) From 855c59d537e4505e2c50062a8f0c39dd49897f38 Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Thu, 10 Sep 2026 22:42:35 +0300 Subject: [PATCH 0100/1352] drm/i915/display: reduce indent in intel_atomic_commit_fence_wait() MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Continue early in the loop for no fence in order to reduce indent. v2: Rebase Cc: Ville Syrjälä Cc: Maarten Lankhorst Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/8ee588437fb214a20c57d8c181f28c57b81cacaa.1789069102.git.jani.nikula@intel.com Signed-off-by: Jani Nikula --- drivers/gpu/drm/i915/display/intel_display.c | 15 ++++++++------- 1 file changed, 8 insertions(+), 7 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_display.c b/drivers/gpu/drm/i915/display/intel_display.c index 1c2e50c6d98a5c..983e84bca1eeda 100644 --- a/drivers/gpu/drm/i915/display/intel_display.c +++ b/drivers/gpu/drm/i915/display/intel_display.c @@ -7296,14 +7296,15 @@ static void intel_atomic_commit_fence_wait(struct intel_atomic_state *state) int i; for_each_new_plane_in_state(&state->base, plane, new_plane_state, i) { - if (new_plane_state->fence) { - ret = dma_fence_wait(new_plane_state->fence, false); - if (ret < 0) - break; + if (!new_plane_state->fence) + continue; - dma_fence_put(new_plane_state->fence); - new_plane_state->fence = NULL; - } + ret = dma_fence_wait(new_plane_state->fence, false); + if (ret < 0) + break; + + dma_fence_put(new_plane_state->fence); + new_plane_state->fence = NULL; } } From e6e7a4b15efe0a3f43e52d27355d0383c0391fcd Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Thu, 10 Sep 2026 22:42:36 +0300 Subject: [PATCH 0101/1352] drm/i915/display: debug log about fence wait errors MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Currently there's no indication if fence wait failed. Debug log about it. v2: Rebase, no timeout anymore Cc: Ville Syrjälä Cc: Maarten Lankhorst Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/6b670a146bfcfd0cc2c7641cc8d72df96e356ef9.1789069102.git.jani.nikula@intel.com Signed-off-by: Jani Nikula --- drivers/gpu/drm/i915/display/intel_display.c | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/drivers/gpu/drm/i915/display/intel_display.c b/drivers/gpu/drm/i915/display/intel_display.c index 983e84bca1eeda..edaa29944d23cf 100644 --- a/drivers/gpu/drm/i915/display/intel_display.c +++ b/drivers/gpu/drm/i915/display/intel_display.c @@ -7290,6 +7290,7 @@ static void skl_commit_modeset_enables(struct intel_atomic_state *state) static void intel_atomic_commit_fence_wait(struct intel_atomic_state *state) { + struct intel_display *display = to_intel_display(state); struct drm_plane *plane; struct drm_plane_state *new_plane_state; long ret; @@ -7300,8 +7301,11 @@ static void intel_atomic_commit_fence_wait(struct intel_atomic_state *state) continue; ret = dma_fence_wait(new_plane_state->fence, false); - if (ret < 0) + if (ret < 0) { + drm_dbg_kms(display->drm, "[PLANE:%d:%s] fence wait failed (%pe)\n", + plane->base.id, plane->name, ERR_PTR(ret)); break; + } dma_fence_put(new_plane_state->fence); new_plane_state->fence = NULL; From 24b09db25f4ca2ec801d6c7b28473687b7ee18ad Mon Sep 17 00:00:00 2001 From: Ankit Nautiyal Date: Tue, 15 Sep 2026 22:16:42 +0530 Subject: [PATCH 0102/1352] drm/i915/dip: Add new file to handle Data Island Packet hardware Add new files intel_dip.c, intel_dip.h, and intel_dip_regs.h to handle low level hardware programming related to Data Island Packets. Currently only programming of the Transmission Line for HDMI 2.1 Extended Metadata Packet (EMP) and DP Adaptive-Sync Secondary Data Packet (SDP) is added (MMIO register EMP_AS_SDP_TL). This will serve as a common place for DIP related code, which is currently scattered across DP and HDMI files. A TODO has been added for extracting the remaining DIP helpers. Signed-off-by: Ankit Nautiyal Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/20260915164657.3429075-2-ankit.k.nautiyal@intel.com --- drivers/gpu/drm/i915/Makefile | 1 + drivers/gpu/drm/i915/display/intel_dip.c | 39 +++++++++++++++++++ drivers/gpu/drm/i915/display/intel_dip.h | 38 ++++++++++++++++++ drivers/gpu/drm/i915/display/intel_dip_regs.h | 18 +++++++++ drivers/gpu/drm/i915/display/intel_vrr.c | 1 + drivers/gpu/drm/i915/display/intel_vrr_regs.h | 6 --- drivers/gpu/drm/xe/Makefile | 1 + 7 files changed, 98 insertions(+), 6 deletions(-) create mode 100644 drivers/gpu/drm/i915/display/intel_dip.c create mode 100644 drivers/gpu/drm/i915/display/intel_dip.h create mode 100644 drivers/gpu/drm/i915/display/intel_dip_regs.h diff --git a/drivers/gpu/drm/i915/Makefile b/drivers/gpu/drm/i915/Makefile index a83fa8be0aba1b..d4e7764621e181 100644 --- a/drivers/gpu/drm/i915/Makefile +++ b/drivers/gpu/drm/i915/Makefile @@ -349,6 +349,7 @@ i915-y += \ display/intel_cx0_phy.o \ display/intel_ddi.o \ display/intel_ddi_buf_trans.o \ + display/intel_dip.o \ display/intel_display_device.o \ display/intel_display_trace.o \ display/intel_dkl_phy.o \ diff --git a/drivers/gpu/drm/i915/display/intel_dip.c b/drivers/gpu/drm/i915/display/intel_dip.c new file mode 100644 index 00000000000000..2e2bdb2b199cab --- /dev/null +++ b/drivers/gpu/drm/i915/display/intel_dip.c @@ -0,0 +1,39 @@ +// SPDX-License-Identifier: MIT +/* + * Copyright © 2026 Intel Corporation + * + */ + +#include "intel_de.h" +#include "intel_dip.h" +#include "intel_dip_regs.h" +#include "intel_display_types.h" + +u16 intel_dip_read_emp_as_sdp_tl(const struct intel_crtc_state *crtc_state) +{ + struct intel_display *display = to_intel_display(crtc_state); + enum transcoder cpu_transcoder = crtc_state->cpu_transcoder; + u32 val; + + if (!HAS_EMP_AS_SDP_TL(display)) + return 0; + + val = intel_de_read(display, EMP_AS_SDP_TL(display, cpu_transcoder)); + return REG_FIELD_GET(EMP_AS_SDP_DB_TL_MASK, val); +} + +void intel_dip_write_emp_as_sdp_tl(const struct intel_crtc_state *crtc_state) +{ + struct intel_display *display = to_intel_display(crtc_state); + enum transcoder cpu_transcoder = crtc_state->cpu_transcoder; + + if (!HAS_EMP_AS_SDP_TL(display)) + return; + /* + * Since currently we support VRR only for DP/eDP, so this is programmed + * only for Adaptive Sync SDP to Vsync start. + */ + intel_de_write(display, + EMP_AS_SDP_TL(display, cpu_transcoder), + EMP_AS_SDP_DB_TL(crtc_state->vrr.vsync_start)); +} diff --git a/drivers/gpu/drm/i915/display/intel_dip.h b/drivers/gpu/drm/i915/display/intel_dip.h new file mode 100644 index 00000000000000..25bae4a04d6ba7 --- /dev/null +++ b/drivers/gpu/drm/i915/display/intel_dip.h @@ -0,0 +1,38 @@ +/* SPDX-License-Identifier: MIT */ +/* + * Copyright © 2026 Intel Corporation + */ + +#ifndef __INTEL_DIP_H__ +#define __INTEL_DIP_H__ + +#include "intel_display_device.h" + +/* + * Video DIP (Data Island Packet) helpers. + * + * This file contains helpers for programming video DIP related hardware. + * + * TODO: Currently, this is only used for programming EMP_AS_SDP_TL i.e. to + * program Transmission Line for HDMI 2.1 Extended Metadata Packet (EMP) and + * DP Adaptive Sync (AS) Secondary Data Packet (SDP). However, all low level + * DIP buffer read/write and related helpers should be extracted here later. + */ + +struct intel_crtc_state; + +/* + * EMP AS SDP TL: Extended Metadata Packet (EMP) Adaptive Sync (AS) + * Secondary Data Packet (SDP) Transmission Line (TL). + * + * Starting with BMG (display ver 14.01) and LNL+ (display ver 20+), + * the AS SDP transmission line is programmable via the EMP AS SDP TL + * register. + */ +#define HAS_EMP_AS_SDP_TL(__display) (DISPLAY_VERx100(__display) == 1401 || \ + DISPLAY_VER(__display) >= 20) + +u16 intel_dip_read_emp_as_sdp_tl(const struct intel_crtc_state *crtc_state); +void intel_dip_write_emp_as_sdp_tl(const struct intel_crtc_state *crtc_state); + +#endif /* __INTEL_DIP_H__ */ diff --git a/drivers/gpu/drm/i915/display/intel_dip_regs.h b/drivers/gpu/drm/i915/display/intel_dip_regs.h new file mode 100644 index 00000000000000..c4e1dab1ea46c7 --- /dev/null +++ b/drivers/gpu/drm/i915/display/intel_dip_regs.h @@ -0,0 +1,18 @@ +/* SPDX-License-Identifier: MIT */ +/* + * Copyright © 2026 Intel Corporation + */ + +#ifndef __INTEL_DIP_REGS_H__ +#define __INTEL_DIP_REGS_H__ + +#include "intel_display_reg_defs.h" + +/* EMP (Extended Metadata Packet) AS (Adaptive Sync) SDP Transmission Line */ +#define _EMP_AS_SDP_TL_A 0x60204 +#define EMP_AS_SDP_TL(display, trans) _MMIO_TRANS2((display), (trans), _EMP_AS_SDP_TL_A) +#define EMP_AS_SDP_DB_TL_MASK REG_GENMASK(12, 0) +#define EMP_AS_SDP_DB_TL(db_transmit_line) REG_FIELD_PREP(EMP_AS_SDP_DB_TL_MASK, \ + (db_transmit_line)) + +#endif /* __INTEL_DIP_REGS_H__ */ diff --git a/drivers/gpu/drm/i915/display/intel_vrr.c b/drivers/gpu/drm/i915/display/intel_vrr.c index e36db11744405c..fd8b6f829cfe90 100644 --- a/drivers/gpu/drm/i915/display/intel_vrr.c +++ b/drivers/gpu/drm/i915/display/intel_vrr.c @@ -17,6 +17,7 @@ #include "intel_cmtg.h" #include "intel_crtc.h" #include "intel_de.h" +#include "intel_dip_regs.h" #include "intel_display_limits.h" #include "intel_display_regs.h" #include "intel_display_types.h" diff --git a/drivers/gpu/drm/i915/display/intel_vrr_regs.h b/drivers/gpu/drm/i915/display/intel_vrr_regs.h index 9d4d6573a149c9..ba8631cbc672a3 100644 --- a/drivers/gpu/drm/i915/display/intel_vrr_regs.h +++ b/drivers/gpu/drm/i915/display/intel_vrr_regs.h @@ -174,12 +174,6 @@ #define VRR_VSYNC_START_MASK REG_GENMASK(12, 0) #define VRR_VSYNC_START(vsync_start) REG_FIELD_PREP(VRR_VSYNC_START_MASK, (vsync_start)) -/* Common register for HDMI EMP and DP AS SDP */ -#define _EMP_AS_SDP_TL_A 0x60204 -#define EMP_AS_SDP_TL(display, trans) _MMIO_TRANS2((display), (trans), _EMP_AS_SDP_TL_A) -#define EMP_AS_SDP_DB_TL_MASK REG_GENMASK(12, 0) -#define EMP_AS_SDP_DB_TL(db_transmit_line) REG_FIELD_PREP(EMP_AS_SDP_DB_TL_MASK, (db_transmit_line)) - #define _TRANS_CMRR_M_LO_A 0x604F0 #define TRANS_CMRR_M_LO(display, trans) _MMIO_TRANS2((display), (trans), _TRANS_CMRR_M_LO_A) diff --git a/drivers/gpu/drm/xe/Makefile b/drivers/gpu/drm/xe/Makefile index 67b8b54776399a..ba4404896f2fe0 100644 --- a/drivers/gpu/drm/xe/Makefile +++ b/drivers/gpu/drm/xe/Makefile @@ -259,6 +259,7 @@ xe-$(CONFIG_DRM_XE_DISPLAY) += \ i915-display/intel_ddi.o \ i915-display/intel_ddi_buf_trans.o \ i915-display/intel_de.o \ + i915-display/intel_dip.o \ i915-display/intel_display.o \ i915-display/intel_display_conversion.o \ i915-display/intel_display_device.o \ From 7310befb0593cf8953b893a3c45c0bc3766aaf30 Mon Sep 17 00:00:00 2001 From: Ankit Nautiyal Date: Tue, 15 Sep 2026 22:16:43 +0530 Subject: [PATCH 0103/1352] drm/i915/vrr: Use the helper to write EMP_AS_SDP_TL register Now that intel_dip_write_emp_as_sdp_tl() exists, use it instead of the open-coded EMP_AS_SDP_TL programming sequence in intel_vrr_set_transcoder_timings(). Signed-off-by: Ankit Nautiyal Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/20260915164657.3429075-3-ankit.k.nautiyal@intel.com --- drivers/gpu/drm/i915/display/intel_vrr.c | 13 ++----------- 1 file changed, 2 insertions(+), 11 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_vrr.c b/drivers/gpu/drm/i915/display/intel_vrr.c index fd8b6f829cfe90..a516ba34e0d02a 100644 --- a/drivers/gpu/drm/i915/display/intel_vrr.c +++ b/drivers/gpu/drm/i915/display/intel_vrr.c @@ -17,6 +17,7 @@ #include "intel_cmtg.h" #include "intel_crtc.h" #include "intel_de.h" +#include "intel_dip.h" #include "intel_dip_regs.h" #include "intel_display_limits.h" #include "intel_display_regs.h" @@ -711,17 +712,7 @@ void intel_vrr_set_transcoder_timings(const struct intel_crtc_state *crtc_state) VRR_VSYNC_END(crtc_state->vrr.vsync_end) | VRR_VSYNC_START(crtc_state->vrr.vsync_start)); - /* - * For BMG and LNL+ onwards the EMP_AS_SDP_TL is used for programming - * double buffering point and transmission line for VRR packets for - * HDMI2.1/DP/eDP/DP->HDMI2.1 PCON. - * Since currently we support VRR only for DP/eDP, so this is programmed - * to for Adaptive Sync SDP to Vsync start. - */ - if (DISPLAY_VERx100(display) == 1401 || DISPLAY_VER(display) >= 20) - intel_de_write(display, - EMP_AS_SDP_TL(display, cpu_transcoder), - EMP_AS_SDP_DB_TL(crtc_state->vrr.vsync_start)); + intel_dip_write_emp_as_sdp_tl(crtc_state); } void From c23ee65d20ff2afcf6108106a58ff0155e48e10e Mon Sep 17 00:00:00 2001 From: Ankit Nautiyal Date: Tue, 15 Sep 2026 22:16:44 +0530 Subject: [PATCH 0104/1352] drm/i915/intel_dip: Add check for DP encoder At the moment, the common register for programming Transmission line for Extended Metadata Packet and Adaptive-Sync Secondary Data Packet (EMP_AS_SDP_TL) is only used to program Adaptive-Sync SDP (AS SDP). Since VRR and Video Timing Extended Metadat Packet (VTEMP) are not yet implemented for HDMI, add an explicit check to write the register only for DP encoders (that may use AS SDP) and reset the register for non-DP encoders. In subsequent changes, instead of directly writing the value, appropriate helpers will be called, that will supply the transmission line for these packets. Signed-off-by: Ankit Nautiyal Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/20260915164657.3429075-4-ankit.k.nautiyal@intel.com --- drivers/gpu/drm/i915/display/intel_dip.c | 11 ++++++++--- 1 file changed, 8 insertions(+), 3 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_dip.c b/drivers/gpu/drm/i915/display/intel_dip.c index 2e2bdb2b199cab..b28352a9e23f0a 100644 --- a/drivers/gpu/drm/i915/display/intel_dip.c +++ b/drivers/gpu/drm/i915/display/intel_dip.c @@ -26,14 +26,19 @@ void intel_dip_write_emp_as_sdp_tl(const struct intel_crtc_state *crtc_state) { struct intel_display *display = to_intel_display(crtc_state); enum transcoder cpu_transcoder = crtc_state->cpu_transcoder; + u32 transmission_line = 0; if (!HAS_EMP_AS_SDP_TL(display)) return; /* - * Since currently we support VRR only for DP/eDP, so this is programmed - * only for Adaptive Sync SDP to Vsync start. + * Since we currently support VRR only for DP/eDP, program the register + * for Adaptive Sync SDP using vsync start. For non-DP encoders, + * the register is reset to 0. */ + if (intel_crtc_has_dp_encoder(crtc_state)) + transmission_line = crtc_state->vrr.vsync_start; + intel_de_write(display, EMP_AS_SDP_TL(display, cpu_transcoder), - EMP_AS_SDP_DB_TL(crtc_state->vrr.vsync_start)); + EMP_AS_SDP_DB_TL(transmission_line)); } From 538f65410ee580e8ffaed4eb60a7b7de824442c7 Mon Sep 17 00:00:00 2001 From: Ankit Nautiyal Date: Tue, 15 Sep 2026 22:16:45 +0530 Subject: [PATCH 0105/1352] drm/i915/dip: Add helper to get AS SDP Transmission Line Introduce a DIP helper to compute the Adaptive Sync SDP transmission line and use it when programming the EMP_AS_SDP_TL register. Currently the AS SDP transmission line is programmed to the T1 position. This can be extended in the future to support programming the T2 position as well. While at it, improve the documentation: the AS SDP transmission line corresponds to the T1 position, which maps to the start of the VSYNC pulse. v2: - Move the helper into intel_dip.c and make it static, since intel_dip.c is its only caller. - Drop the now unused prototype from intel_dp.h. - Add the check HAS_EMP_AS_SDP_TL(). (Suraj) Signed-off-by: Ankit Nautiyal Reviewed-by: Suraj Kandpal (#v1) Link: https://patch.msgid.link/20260915164657.3429075-5-ankit.k.nautiyal@intel.com --- drivers/gpu/drm/i915/display/intel_dip.c | 19 ++++++++++++++++++- 1 file changed, 18 insertions(+), 1 deletion(-) diff --git a/drivers/gpu/drm/i915/display/intel_dip.c b/drivers/gpu/drm/i915/display/intel_dip.c index b28352a9e23f0a..0277b15e1c82d4 100644 --- a/drivers/gpu/drm/i915/display/intel_dip.c +++ b/drivers/gpu/drm/i915/display/intel_dip.c @@ -9,6 +9,23 @@ #include "intel_dip_regs.h" #include "intel_display_types.h" +static int intel_dip_get_as_sdp_transmission_line(const struct intel_crtc_state *crtc_state) +{ + struct intel_display *display = to_intel_display(crtc_state); + + if (!HAS_EMP_AS_SDP_TL(display)) + return 0; + + /* + * EMP_AS_SDP_TL defines the T1 position as the default AS SDP + * Transmission Line, which corresponds to the start of the + * VSYNC pulse. + * + * Use the T1 position for now. + */ + return crtc_state->vrr.vsync_start; +} + u16 intel_dip_read_emp_as_sdp_tl(const struct intel_crtc_state *crtc_state) { struct intel_display *display = to_intel_display(crtc_state); @@ -36,7 +53,7 @@ void intel_dip_write_emp_as_sdp_tl(const struct intel_crtc_state *crtc_state) * the register is reset to 0. */ if (intel_crtc_has_dp_encoder(crtc_state)) - transmission_line = crtc_state->vrr.vsync_start; + transmission_line = intel_dip_get_as_sdp_transmission_line(crtc_state); intel_de_write(display, EMP_AS_SDP_TL(display, cpu_transcoder), From 5aa79d2e99f6f9e2e94b77c74a9eabf2fad03030 Mon Sep 17 00:00:00 2001 From: Ankit Nautiyal Date: Tue, 15 Sep 2026 22:16:46 +0530 Subject: [PATCH 0106/1352] drm/i915/display: Add crtc state for DIP transmission lines The Adaptive Sync SDP is currently the only packet with a programmable transmission line. Make a structure struct intel_dip for Data Island Packets. Add a member to track Adaptive-Sync SDP transmission line. Include the new member in the pipe configuration comparison. This will pave the way for supporting more packets' programmable transmission lines, including the common base SDP transmission line introduced with Xe3p_lpd. v2: - Move struct intel_dip from intel_dip.h to intel_display_types.h (Jani) - Move PIPE_CONF_CHECK for emp_as_sdp_tl in !fastset block. (Sashiko) Signed-off-by: Ankit Nautiyal Reviewed-by: Suraj Kandpal (#v1) Link: https://patch.msgid.link/20260915164657.3429075-6-ankit.k.nautiyal@intel.com --- drivers/gpu/drm/i915/display/intel_display.c | 1 + drivers/gpu/drm/i915/display/intel_display_types.h | 10 ++++++++++ 2 files changed, 11 insertions(+) diff --git a/drivers/gpu/drm/i915/display/intel_display.c b/drivers/gpu/drm/i915/display/intel_display.c index edaa29944d23cf..3a495da360c654 100644 --- a/drivers/gpu/drm/i915/display/intel_display.c +++ b/drivers/gpu/drm/i915/display/intel_display.c @@ -5601,6 +5601,7 @@ intel_pipe_config_compare(const struct intel_crtc_state *current_config, PIPE_CONF_CHECK_I(vrr.dc_balance.max_increase); PIPE_CONF_CHECK_I(vrr.dc_balance.max_decrease); PIPE_CONF_CHECK_I(vrr.dc_balance.vblank_target); + PIPE_CONF_CHECK_I(dip.emp_as_sdp_tl); } if (!fastset || intel_vrr_always_use_vrr_tg(display)) { diff --git a/drivers/gpu/drm/i915/display/intel_display_types.h b/drivers/gpu/drm/i915/display/intel_display_types.h index a5f18ac8a7d06e..334179685ea2da 100644 --- a/drivers/gpu/drm/i915/display/intel_display_types.h +++ b/drivers/gpu/drm/i915/display/intel_display_types.h @@ -1007,6 +1007,14 @@ struct intel_casf { bool enable; }; +struct intel_dip { + /* + * DIP Transmission line, relative to the Vtotal. + * The programmed transmit line is (Vtotal - value) + */ + u16 emp_as_sdp_tl; +}; + struct intel_crtc_state { /* * uapi (drm) state. This is the software state shown to userspace. @@ -1315,6 +1323,8 @@ struct intel_crtc_state { struct drm_dp_as_sdp as_sdp; } infoframes; + struct intel_dip dip; + u8 eld[MAX_ELD_BYTES]; /* HDMI scrambling status */ From 50dae5bab56e25458a53b4869434789e29517034 Mon Sep 17 00:00:00 2001 From: Ankit Nautiyal Date: Tue, 15 Sep 2026 22:16:47 +0530 Subject: [PATCH 0107/1352] drm/i915/dip: Store and use AS SDP transmission line from crtc state The driver currently computes the Adaptive Sync SDP transmission line directly at programming time. Instead, compute and store the AS SDP transmission line in the crtc state and use it when programming the EMP_AS_SDP_TL register. We get the clear picture about the SDPs and guardband only in intel_dp_sdp_compute_config_late() therefore we must configure the AS SDP transmission line at this point when AS SDP is enabled in crtc_state. This prepares the ground for supporting programmable transmission lines for additional DP SDPs. While moving the helper into intel_dip.c, drop the intel_crtc_has_dp_encoder() check instead of relocating it. It was needed in the old VRR write path shared by other encoderes as well, but intel_dip_sdp_tl_compute_config_late() is only reached via DP, so HDMI never sets crtc_state->dip.emp_as_sdp_tl and it stays 0 by default. v2: - Move the helper into intel_dip.c and drop the intel_crtc_has_dp_encoder() check. - Drop the redundant checks. (Suraj) Signed-off-by: Ankit Nautiyal Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/20260915164657.3429075-7-ankit.k.nautiyal@intel.com --- drivers/gpu/drm/i915/display/intel_ddi.c | 2 ++ drivers/gpu/drm/i915/display/intel_dip.c | 20 +++++++++++--------- drivers/gpu/drm/i915/display/intel_dip.h | 3 +++ drivers/gpu/drm/i915/display/intel_dp.c | 3 +++ 4 files changed, 19 insertions(+), 9 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_ddi.c b/drivers/gpu/drm/i915/display/intel_ddi.c index 9b3b526e5e55b3..dacb4b7588a02b 100644 --- a/drivers/gpu/drm/i915/display/intel_ddi.c +++ b/drivers/gpu/drm/i915/display/intel_ddi.c @@ -49,6 +49,7 @@ #include "intel_ddi.h" #include "intel_ddi_buf_trans.h" #include "intel_de.h" +#include "intel_dip.h" #include "intel_display_power.h" #include "intel_display_regs.h" #include "intel_display_types.h" @@ -4235,6 +4236,7 @@ static void intel_ddi_get_config(struct intel_encoder *encoder, intel_read_dp_sdp(encoder, pipe_config, HDMI_PACKET_TYPE_GAMUT_METADATA); intel_read_dp_sdp(encoder, pipe_config, DP_SDP_VSC); intel_read_dp_sdp(encoder, pipe_config, DP_SDP_ADAPTIVE_SYNC); + intel_dip_sdp_transmission_line_get_config(pipe_config); intel_audio_codec_get_config(encoder, pipe_config); } diff --git a/drivers/gpu/drm/i915/display/intel_dip.c b/drivers/gpu/drm/i915/display/intel_dip.c index 0277b15e1c82d4..d1acc7eb5a3915 100644 --- a/drivers/gpu/drm/i915/display/intel_dip.c +++ b/drivers/gpu/drm/i915/display/intel_dip.c @@ -43,19 +43,21 @@ void intel_dip_write_emp_as_sdp_tl(const struct intel_crtc_state *crtc_state) { struct intel_display *display = to_intel_display(crtc_state); enum transcoder cpu_transcoder = crtc_state->cpu_transcoder; - u32 transmission_line = 0; if (!HAS_EMP_AS_SDP_TL(display)) return; - /* - * Since we currently support VRR only for DP/eDP, program the register - * for Adaptive Sync SDP using vsync start. For non-DP encoders, - * the register is reset to 0. - */ - if (intel_crtc_has_dp_encoder(crtc_state)) - transmission_line = intel_dip_get_as_sdp_transmission_line(crtc_state); intel_de_write(display, EMP_AS_SDP_TL(display, cpu_transcoder), - EMP_AS_SDP_DB_TL(transmission_line)); + EMP_AS_SDP_DB_TL(crtc_state->dip.emp_as_sdp_tl)); +} + +void intel_dip_sdp_tl_compute_config_late(struct intel_crtc_state *crtc_state) +{ + crtc_state->dip.emp_as_sdp_tl = intel_dip_get_as_sdp_transmission_line(crtc_state); +} + +void intel_dip_sdp_transmission_line_get_config(struct intel_crtc_state *crtc_state) +{ + crtc_state->dip.emp_as_sdp_tl = intel_dip_read_emp_as_sdp_tl(crtc_state); } diff --git a/drivers/gpu/drm/i915/display/intel_dip.h b/drivers/gpu/drm/i915/display/intel_dip.h index 25bae4a04d6ba7..20f9aeb85c3914 100644 --- a/drivers/gpu/drm/i915/display/intel_dip.h +++ b/drivers/gpu/drm/i915/display/intel_dip.h @@ -35,4 +35,7 @@ struct intel_crtc_state; u16 intel_dip_read_emp_as_sdp_tl(const struct intel_crtc_state *crtc_state); void intel_dip_write_emp_as_sdp_tl(const struct intel_crtc_state *crtc_state); +void intel_dip_sdp_tl_compute_config_late(struct intel_crtc_state *crtc_state); +void intel_dip_sdp_transmission_line_get_config(struct intel_crtc_state *crtc_state); + #endif /* __INTEL_DIP_H__ */ diff --git a/drivers/gpu/drm/i915/display/intel_dp.c b/drivers/gpu/drm/i915/display/intel_dp.c index c44d584cc07a75..f55ae30bebf8fc 100644 --- a/drivers/gpu/drm/i915/display/intel_dp.c +++ b/drivers/gpu/drm/i915/display/intel_dp.c @@ -61,6 +61,7 @@ #include "intel_cx0_phy.h" #include "intel_ddi.h" #include "intel_de.h" +#include "intel_dip.h" #include "intel_display_driver.h" #include "intel_display_jiffies.h" #include "intel_display_utils.h" @@ -7362,6 +7363,8 @@ int intel_dp_sdp_compute_config_late(struct intel_crtc_state *crtc_state) return -EINVAL; } + intel_dip_sdp_tl_compute_config_late(crtc_state); + return 0; } From ba4062e3d4d454bf422c8a748254e9a635fe716c Mon Sep 17 00:00:00 2001 From: Arun R Murthy Date: Tue, 15 Sep 2026 22:16:48 +0530 Subject: [PATCH 0108/1352] drm/i915/dip_regs: Add register definitions for common SDP Transmission Line Add registers definitions for common SDP transmission line CMN_SDP_TL and CMN_SDP_TL_STGR_CTL. v2: Move all registers to intel_dip_regs.h (Ankit) Bspec: 74384 Signed-off-by: Arun R Murthy Signed-off-by: Ankit Nautiyal Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/20260915164657.3429075-8-ankit.k.nautiyal@intel.com --- drivers/gpu/drm/i915/display/intel_dip_regs.h | 20 +++++++++++++++++++ 1 file changed, 20 insertions(+) diff --git a/drivers/gpu/drm/i915/display/intel_dip_regs.h b/drivers/gpu/drm/i915/display/intel_dip_regs.h index c4e1dab1ea46c7..d7e40a61391357 100644 --- a/drivers/gpu/drm/i915/display/intel_dip_regs.h +++ b/drivers/gpu/drm/i915/display/intel_dip_regs.h @@ -15,4 +15,24 @@ #define EMP_AS_SDP_DB_TL(db_transmit_line) REG_FIELD_PREP(EMP_AS_SDP_DB_TL_MASK, \ (db_transmit_line)) +/* COMMON SDP TRANSMISSION LINE */ +#define _CMN_SDP_TL_A 0x6020c +#define CMN_SDP_TL(display, trans) _MMIO_TRANS2(display, (trans), _CMN_SDP_TL_A) +#define TRANSMISSION_LINE_ENABLE REG_BIT(31) +#define BASE_TRANSMISSION_LINE_MASK REG_GENMASK(12, 0) +#define BASE_TRANSMISSION_LINE(x) REG_FIELD_PREP(BASE_TRANSMISSION_LINE_MASK, x) + +#define _CMN_SDP_TL_STGR_CTL_A 0x60214 +#define CMN_SDP_TL_STGR_CTL(display, trans) _MMIO_TRANS2(display, (trans), \ + _CMN_SDP_TL_STGR_CTL_A) +#define VSC_EXT_STAGGER_MASK REG_GENMASK(11, 8) +#define VSC_EXT_STAGGER(x) REG_FIELD_PREP(VSC_EXT_STAGGER_MASK, x) +#define VSC_EXT_STAGGER_DEFAULT 0x2 +#define PPS_STAGGER_MASK REG_GENMASK(7, 4) +#define PPS_STAGGER(x) REG_FIELD_PREP(PPS_STAGGER_MASK, x) +#define PPS_STAGGER_DEFAULT 0x1 +#define GMP_STAGGER_MASK REG_GENMASK(3, 0) +#define GMP_STAGGER(x) REG_FIELD_PREP(GMP_STAGGER_MASK, x) +#define GMP_STAGGER_DEFAULT 0x0 + #endif /* __INTEL_DIP_REGS_H__ */ From 45b6a7f1b82261400d53b8d9edba93f59c3a0b7a Mon Sep 17 00:00:00 2001 From: Ankit Nautiyal Date: Tue, 15 Sep 2026 22:16:49 +0530 Subject: [PATCH 0109/1352] drm/i915/dip: Add HAS_COMMON_SDP_TL macro Add a helper macro to detect CMN SDP TL support on platforms with display version 35 and above. v2: Use prefix drm/i915/dip. (Suraj) Signed-off-by: Ankit Nautiyal Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/20260915164657.3429075-9-ankit.k.nautiyal@intel.com --- drivers/gpu/drm/i915/display/intel_dip.h | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/drivers/gpu/drm/i915/display/intel_dip.h b/drivers/gpu/drm/i915/display/intel_dip.h index 20f9aeb85c3914..8ff22eb5663b47 100644 --- a/drivers/gpu/drm/i915/display/intel_dip.h +++ b/drivers/gpu/drm/i915/display/intel_dip.h @@ -32,6 +32,16 @@ struct intel_crtc_state; #define HAS_EMP_AS_SDP_TL(__display) (DISPLAY_VERx100(__display) == 1401 || \ DISPLAY_VER(__display) >= 20) +/* + * CMN SDP TL: Common Secondary Data Packet Transmission Line. + * + * Xe3p_lpd introduces new register CMN_SDP_TL to program a common SDP + * Transmission line that will be used by the Hardware to position the + * SDPs. Along with this, another new register CMN_SDP_TL_STGR_CTL is + * also added to stagger the different SDPs. + */ +#define HAS_COMMON_SDP_TL(__display) (DISPLAY_VER(__display) >= 35) + u16 intel_dip_read_emp_as_sdp_tl(const struct intel_crtc_state *crtc_state); void intel_dip_write_emp_as_sdp_tl(const struct intel_crtc_state *crtc_state); From 8aeed61a7d64c2e6bf1337848bc4960fadf9dfd1 Mon Sep 17 00:00:00 2001 From: Ankit Nautiyal Date: Tue, 15 Sep 2026 22:16:50 +0530 Subject: [PATCH 0110/1352] drm/i915/dip: Store SDP transmission lines in crtc_state Currently the driver only programs the transmission line for the Adaptive-Sync SDP, while the hardware controls the transmission lines for other SDPs. Starting with Xe3p_lpd, the hardware allows the driver to program transmission lines for additional DP SDPs. Prepare for this by adding fields to struct intel_crtc_state to store SDP transmission lines, and include them in pipe config comparison. The SDP transmission line fields track vrr.vsync_start/vtotal, which are allowed to change during a seamless LRR fastset. Guard their pipe config comparison under !fastset, same as vrr.vsync_start/vsync_end, so a fastset is not unnecessarily turned into a full modeset. Signed-off-by: Ankit Nautiyal Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/20260915164657.3429075-10-ankit.k.nautiyal@intel.com --- drivers/gpu/drm/i915/display/intel_display.c | 5 +++++ drivers/gpu/drm/i915/display/intel_display_types.h | 6 ++++++ 2 files changed, 11 insertions(+) diff --git a/drivers/gpu/drm/i915/display/intel_display.c b/drivers/gpu/drm/i915/display/intel_display.c index 3a495da360c654..e3e0d2f8faefdc 100644 --- a/drivers/gpu/drm/i915/display/intel_display.c +++ b/drivers/gpu/drm/i915/display/intel_display.c @@ -5602,6 +5602,11 @@ intel_pipe_config_compare(const struct intel_crtc_state *current_config, PIPE_CONF_CHECK_I(vrr.dc_balance.max_decrease); PIPE_CONF_CHECK_I(vrr.dc_balance.vblank_target); PIPE_CONF_CHECK_I(dip.emp_as_sdp_tl); + PIPE_CONF_CHECK_I(dip.gmp_sdp_tl); + PIPE_CONF_CHECK_I(dip.pps_sdp_tl); + PIPE_CONF_CHECK_I(dip.vsc_sdp_tl); + PIPE_CONF_CHECK_I(dip.vsc_ext_sdp_tl); + PIPE_CONF_CHECK_I(dip.cmn_sdp_tl); } if (!fastset || intel_vrr_always_use_vrr_tg(display)) { diff --git a/drivers/gpu/drm/i915/display/intel_display_types.h b/drivers/gpu/drm/i915/display/intel_display_types.h index 334179685ea2da..79f30660c2b69b 100644 --- a/drivers/gpu/drm/i915/display/intel_display_types.h +++ b/drivers/gpu/drm/i915/display/intel_display_types.h @@ -1013,6 +1013,12 @@ struct intel_dip { * The programmed transmit line is (Vtotal - value) */ u16 emp_as_sdp_tl; + u16 gmp_sdp_tl; + u16 pps_sdp_tl; + u16 vsc_sdp_tl; + u16 vsc_ext_sdp_tl; + /* Common SDP Base transmission line (Xe3p_lpd+) */ + u16 cmn_sdp_tl; }; struct intel_crtc_state { From 589a07df22d6e150571eea49b606fde120545846 Mon Sep 17 00:00:00 2001 From: Ankit Nautiyal Date: Tue, 15 Sep 2026 22:16:51 +0530 Subject: [PATCH 0111/1352] drm/i915/dp: Introduce helpers to enable/disable CMN SDP Transmission line Introduce helpers to program or disable CMN_SDP_TL and stagger registers using the state stored in crtc_state. v2: - Use HAS_COMMON_SDP_TL(display) instead of checking crtc_state->dip.cmn_sdp_tl, since 0 is a valid transmission line value. (Sashiko) Signed-off-by: Ankit Nautiyal Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/20260915164657.3429075-11-ankit.k.nautiyal@intel.com --- drivers/gpu/drm/i915/display/intel_dip.c | 56 ++++++++++++++++++++++++ drivers/gpu/drm/i915/display/intel_dip.h | 2 + 2 files changed, 58 insertions(+) diff --git a/drivers/gpu/drm/i915/display/intel_dip.c b/drivers/gpu/drm/i915/display/intel_dip.c index d1acc7eb5a3915..f8e8577b275588 100644 --- a/drivers/gpu/drm/i915/display/intel_dip.c +++ b/drivers/gpu/drm/i915/display/intel_dip.c @@ -4,6 +4,8 @@ * */ +#include + #include "intel_de.h" #include "intel_dip.h" #include "intel_dip_regs.h" @@ -61,3 +63,57 @@ void intel_dip_sdp_transmission_line_get_config(struct intel_crtc_state *crtc_st { crtc_state->dip.emp_as_sdp_tl = intel_dip_read_emp_as_sdp_tl(crtc_state); } + +static int intel_dip_sdp_tl_to_stagger(const struct intel_crtc_state *crtc_state, + u16 sdp_transmission_line) +{ + return sdp_transmission_line - crtc_state->dip.cmn_sdp_tl; +} + +void intel_dip_cmn_sdp_transmission_line_enable(const struct intel_crtc_state *crtc_state) +{ + struct intel_display *display = to_intel_display(crtc_state); + enum transcoder cpu_transcoder = crtc_state->cpu_transcoder; + int gmp_stagger; + int pps_stagger; + int vsc_ext_stagger; + + if (!HAS_COMMON_SDP_TL(display)) + return; + + gmp_stagger = intel_dip_sdp_tl_to_stagger(crtc_state, + crtc_state->dip.gmp_sdp_tl); + + pps_stagger = intel_dip_sdp_tl_to_stagger(crtc_state, + crtc_state->dip.pps_sdp_tl); + + vsc_ext_stagger = intel_dip_sdp_tl_to_stagger(crtc_state, + crtc_state->dip.vsc_ext_sdp_tl); + + if (drm_WARN_ON(display->drm, gmp_stagger < 0)) + return; + if (drm_WARN_ON(display->drm, pps_stagger < 0)) + return; + if (drm_WARN_ON(display->drm, vsc_ext_stagger < 0)) + return; + + intel_de_write(display, CMN_SDP_TL_STGR_CTL(display, cpu_transcoder), + GMP_STAGGER(gmp_stagger) | + PPS_STAGGER(pps_stagger) | + VSC_EXT_STAGGER(vsc_ext_stagger)); + + intel_de_write(display, CMN_SDP_TL(display, cpu_transcoder), + TRANSMISSION_LINE_ENABLE | + BASE_TRANSMISSION_LINE(crtc_state->dip.cmn_sdp_tl)); +} + +void intel_dip_cmn_sdp_transmission_line_disable(const struct intel_crtc_state *old_crtc_state) +{ + struct intel_display *display = to_intel_display(old_crtc_state); + enum transcoder cpu_transcoder = old_crtc_state->cpu_transcoder; + + if (!HAS_COMMON_SDP_TL(display)) + return; + + intel_de_write(display, CMN_SDP_TL(display, cpu_transcoder), 0); +} diff --git a/drivers/gpu/drm/i915/display/intel_dip.h b/drivers/gpu/drm/i915/display/intel_dip.h index 8ff22eb5663b47..7c64c84cb1beb2 100644 --- a/drivers/gpu/drm/i915/display/intel_dip.h +++ b/drivers/gpu/drm/i915/display/intel_dip.h @@ -47,5 +47,7 @@ void intel_dip_write_emp_as_sdp_tl(const struct intel_crtc_state *crtc_state); void intel_dip_sdp_tl_compute_config_late(struct intel_crtc_state *crtc_state); void intel_dip_sdp_transmission_line_get_config(struct intel_crtc_state *crtc_state); +void intel_dip_cmn_sdp_transmission_line_enable(const struct intel_crtc_state *crtc_state); +void intel_dip_cmn_sdp_transmission_line_disable(const struct intel_crtc_state *old_crtc_state); #endif /* __INTEL_DIP_H__ */ From bfa597238073d1859bef12bdd530d6e8258eec3b Mon Sep 17 00:00:00 2001 From: Ankit Nautiyal Date: Tue, 15 Sep 2026 22:16:52 +0530 Subject: [PATCH 0112/1352] drm/i915/dip: Enable Common SDP Transmission line Enable programming of the common SDP transmission line on platforms that support it. Compute and program the common base transmission line and per-SDP stagger values from the crtc state during modeset, and disable the feature on pipe disable. Currently, the stagger values are set as per the default policy of the Hardware. This can be optimized later if we come up with a specific driver policy to sequence the SDPs better. v2: Add WARN if the Common Transmission Line is more than Guardband + SCL. (Suraj) Signed-off-by: Ankit Nautiyal Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/20260915164657.3429075-12-ankit.k.nautiyal@intel.com --- drivers/gpu/drm/i915/display/intel_ddi.c | 3 + drivers/gpu/drm/i915/display/intel_dip.c | 96 ++++++++++++++++++++++++ 2 files changed, 99 insertions(+) diff --git a/drivers/gpu/drm/i915/display/intel_ddi.c b/drivers/gpu/drm/i915/display/intel_ddi.c index dacb4b7588a02b..3cdb06e81130b7 100644 --- a/drivers/gpu/drm/i915/display/intel_ddi.c +++ b/drivers/gpu/drm/i915/display/intel_ddi.c @@ -2737,6 +2737,8 @@ static void mtl_ddi_pre_enable_dp(struct intel_atomic_state *state, /* 6.o Configure and enable FEC if needed */ intel_ddi_enable_fec(encoder, crtc_state); + intel_dip_cmn_sdp_transmission_line_enable(crtc_state); + /* 7.a 128b/132b SST. */ if (!is_mst && intel_dp_is_uhbr(crtc_state)) { /* VCPID 1, start slot 0 for 128b/132b, tu slots */ @@ -3124,6 +3126,7 @@ static void intel_ddi_buf_disable(struct intel_encoder *encoder, DP_TP_CTL_ENABLE, 0); } + intel_dip_cmn_sdp_transmission_line_disable(crtc_state); intel_ddi_disable_fec(encoder, crtc_state); if (DISPLAY_VER(display) < 14) diff --git a/drivers/gpu/drm/i915/display/intel_dip.c b/drivers/gpu/drm/i915/display/intel_dip.c index f8e8577b275588..f447150ff7bc03 100644 --- a/drivers/gpu/drm/i915/display/intel_dip.c +++ b/drivers/gpu/drm/i915/display/intel_dip.c @@ -10,6 +10,7 @@ #include "intel_dip.h" #include "intel_dip_regs.h" #include "intel_display_types.h" +#include "intel_hdmi.h" static int intel_dip_get_as_sdp_transmission_line(const struct intel_crtc_state *crtc_state) { @@ -54,14 +55,109 @@ void intel_dip_write_emp_as_sdp_tl(const struct intel_crtc_state *crtc_state) EMP_AS_SDP_DB_TL(crtc_state->dip.emp_as_sdp_tl)); } +static int intel_dip_sdp_stagger_to_tl(struct intel_crtc_state *crtc_state, + int stagger) +{ + return crtc_state->dip.cmn_sdp_tl + stagger; +} + +static +void intel_dip_cmn_sdp_tl_compute_config_late(struct intel_crtc_state *crtc_state) +{ + struct intel_display *display = to_intel_display(crtc_state); + bool as_sdp; + + if (!HAS_COMMON_SDP_TL(display)) + return; + + as_sdp = crtc_state->infoframes.enable & + intel_hdmi_infoframe_enable(DP_SDP_ADAPTIVE_SYNC); + /* + * When AS SDP is enabled : + * - The common SDP Transmission Line matches the EMP SDP Transmission Line. + * + * When AS SDP is disabled: + * - Bspec mentions the positions as lines of delayed vblank. + * - Guardband = 1st line of delayed vblank + * - Common SDP Transmission line is set to 2nd line of delayed vblank. + */ + + if (as_sdp) + crtc_state->dip.cmn_sdp_tl = crtc_state->dip.emp_as_sdp_tl; + else + crtc_state->dip.cmn_sdp_tl = crtc_state->vrr.guardband - 1; + + if (drm_WARN_ON(display->drm, + crtc_state->dip.cmn_sdp_tl >= + crtc_state->vrr.guardband + crtc_state->set_context_latency)) + return; + + /* + * Currently we are programming the default stagger values, but these + * can be optimized if required, based on number of SDPs enabled. + * + * Default values of the Transmission lines for SDPs other than AS SDP: + * VSC : CMN SDP Transmission line + * GMP : CMN SDP Transmission line + * PPS : CMN SDP Transmission line + 1 + * VSC_EXT: CMN SDP Transmission line + 2 + */ + crtc_state->dip.vsc_sdp_tl = crtc_state->dip.cmn_sdp_tl; + crtc_state->dip.gmp_sdp_tl = + intel_dip_sdp_stagger_to_tl(crtc_state, GMP_STAGGER_DEFAULT); + crtc_state->dip.pps_sdp_tl = + intel_dip_sdp_stagger_to_tl(crtc_state, PPS_STAGGER_DEFAULT); + crtc_state->dip.vsc_ext_sdp_tl = + intel_dip_sdp_stagger_to_tl(crtc_state, VSC_EXT_STAGGER_DEFAULT); +} + void intel_dip_sdp_tl_compute_config_late(struct intel_crtc_state *crtc_state) { crtc_state->dip.emp_as_sdp_tl = intel_dip_get_as_sdp_transmission_line(crtc_state); + + intel_dip_cmn_sdp_tl_compute_config_late(crtc_state); +} + +static +void intel_dip_cmn_sdp_transmission_line_get_config(struct intel_crtc_state *crtc_state) +{ + struct intel_display *display = to_intel_display(crtc_state); + enum transcoder cpu_transcoder = crtc_state->cpu_transcoder; + u16 vsc_ext_stagger, pps_stagger, gmp_stagger; + u32 val; + + if (!HAS_COMMON_SDP_TL(display)) + return; + + val = intel_de_read(display, CMN_SDP_TL(display, cpu_transcoder)); + + if (!(val & TRANSMISSION_LINE_ENABLE)) + return; + + crtc_state->dip.cmn_sdp_tl = REG_FIELD_GET(BASE_TRANSMISSION_LINE_MASK, val); + + /* SDP VSC uses same transmission line as CMN base transmission line */ + crtc_state->dip.vsc_sdp_tl = crtc_state->dip.cmn_sdp_tl; + + val = intel_de_read(display, CMN_SDP_TL_STGR_CTL(display, cpu_transcoder)); + + vsc_ext_stagger = REG_FIELD_GET(VSC_EXT_STAGGER_MASK, val); + pps_stagger = REG_FIELD_GET(PPS_STAGGER_MASK, val); + gmp_stagger = REG_FIELD_GET(GMP_STAGGER_MASK, val); + + crtc_state->dip.vsc_ext_sdp_tl = + intel_dip_sdp_stagger_to_tl(crtc_state, vsc_ext_stagger); + crtc_state->dip.pps_sdp_tl = + intel_dip_sdp_stagger_to_tl(crtc_state, pps_stagger); + crtc_state->dip.gmp_sdp_tl = + intel_dip_sdp_stagger_to_tl(crtc_state, gmp_stagger); } void intel_dip_sdp_transmission_line_get_config(struct intel_crtc_state *crtc_state) { crtc_state->dip.emp_as_sdp_tl = intel_dip_read_emp_as_sdp_tl(crtc_state); + + intel_dip_cmn_sdp_transmission_line_get_config(crtc_state); } static int intel_dip_sdp_tl_to_stagger(const struct intel_crtc_state *crtc_state, From d6795b64b172b24622d33a3e0c7c1c7d15870fd2 Mon Sep 17 00:00:00 2001 From: Ankit Nautiyal Date: Tue, 15 Sep 2026 22:16:53 +0530 Subject: [PATCH 0113/1352] drm/i915/dp: Account VSC SDP in min guardband The transmission line for VSC SDP is the same as AS SDP (EMP_AS_SDP_TL) when AS SDP is enabled. Otherwise, it uses the second line of delayed vblank. For VSC without AS SDP, this requires 2 lines plus 1 setup line, so the VRR guardband must be at least 3 lines. When both AS SDP and VSC SDP are enabled, the guardband requirement is already accounted for during optimized guardband calculation and the final guardband validation in compute_config_late(). However, when VSC SDP is enabled without AS SDP, the VSC SDP requirement is not checked explicitly. Even in the unlikely case where the optimized guardband is clamped to vblank length, it cannot fall below 5 lines since such modes are already pruned. Still, for completeness, account for VSC SDP and ensure a minimum guardband of 3 lines. Signed-off-by: Ankit Nautiyal Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/20260915164657.3429075-13-ankit.k.nautiyal@intel.com --- drivers/gpu/drm/i915/display/intel_dp.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/drivers/gpu/drm/i915/display/intel_dp.c b/drivers/gpu/drm/i915/display/intel_dp.c index f55ae30bebf8fc..27f03f36eabbcd 100644 --- a/drivers/gpu/drm/i915/display/intel_dp.c +++ b/drivers/gpu/drm/i915/display/intel_dp.c @@ -7397,6 +7397,8 @@ int intel_dp_get_lines_for_sdp(const struct intel_crtc_state *crtc_state, u32 ty return 8; case DP_SDP_PPS: return 7; + case DP_SDP_VSC: + return 3; case DP_SDP_ADAPTIVE_SYNC: return crtc_state->vrr.vsync_start + 1; default: @@ -7428,6 +7430,11 @@ int intel_dp_sdp_min_guardband(const struct intel_crtc_state *crtc_state, sdp_guardband = max(sdp_guardband, intel_dp_get_lines_for_sdp(crtc_state, DP_SDP_ADAPTIVE_SYNC)); + if (crtc_state->infoframes.enable & + intel_hdmi_infoframe_enable(DP_SDP_VSC)) + sdp_guardband = max(sdp_guardband, + intel_dp_get_lines_for_sdp(crtc_state, DP_SDP_VSC)); + return sdp_guardband; } From 571631ac5b30c2156087505be915de67182b3753 Mon Sep 17 00:00:00 2001 From: Ankit Nautiyal Date: Tue, 15 Sep 2026 22:16:54 +0530 Subject: [PATCH 0114/1352] drm/i915/dp: Adjust SDP guardband requirement for CMN_SDP_TL Once CMN_SDP_TL is enabled, GMP/PPS/VSC/VSC_EXT/AS SDPs are no longer positioned relative to the guardband: they are anchored via CMN_SDP_TL/CMN_SDP_TL_STGR_CTL instead. As per Bspec 68921, SDP Setup is 0 in this mode, so the old per-packet guardband sizing (based on GMP/PPS/AS-SDP being enabled) no longer applies for GMP/PPS/VSC/VSC_EXT. Since we are using the default stagger values for now, size the guardband such that the max default transmission line can be supported, similar to when CMN SDP TL is not set: base : 2nd line of delayed vblank GMP : 2 + GMP_STAGGER VSC_EXT: 2 + VSC_EXT_STAGGER VSC : 2 PPS : 2 + PPS_STAGGER SDP Setup = 1 + MAX(GMP, VSC_EXT, VSC, PPS setup lines) Add intel_dp_get_lines_for_cmn_sdp_tl() and route it via the existing intel_dp_get_lines_for_sdp(). The AS SDP check in intel_dp_sdp_min_guardband() still adds vrr.vsync_start + 1 to the guardband, since AS SDP positioning is unaffected by CMN_SDP_TL. v2: Add VSC min SDP guardband. (Sashiko) Bspec: 68921 Assisted-by: Copilot:claude-sonnet-4.5 Signed-off-by: Ankit Nautiyal Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/20260915164657.3429075-14-ankit.k.nautiyal@intel.com --- drivers/gpu/drm/i915/display/intel_dp.c | 52 ++++++++++++++++++++++++- 1 file changed, 50 insertions(+), 2 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_dp.c b/drivers/gpu/drm/i915/display/intel_dp.c index 27f03f36eabbcd..50ed615cf0f606 100644 --- a/drivers/gpu/drm/i915/display/intel_dp.c +++ b/drivers/gpu/drm/i915/display/intel_dp.c @@ -62,6 +62,7 @@ #include "intel_ddi.h" #include "intel_de.h" #include "intel_dip.h" +#include "intel_dip_regs.h" #include "intel_display_driver.h" #include "intel_display_jiffies.h" #include "intel_display_utils.h" @@ -7386,9 +7387,58 @@ int intel_dp_compute_config_late(struct intel_encoder *encoder, return 0; } +static +int intel_dp_get_lines_for_cmn_sdp_tl(u32 type) +{ + u32 stagger_val; + + /* + * Since we are using default stagger values similar to the case + * where CMN SDP TL is not set, the different SDP transmission + * lines are: + * base : 2nd line of delayed vblank: + * GMP : 2 + GMP_STAGGER + * VSC_EXT: 2 + VSC_EXT_STAGGER + * VSC : 2 + * PPS : 2 + PPS_STAGGER + * + * SDP Setup = 1 + MAX(GMP, VSC_EXT, VSC, PPS setup lines) + * + * For EMP_AS_SDP_TL guardband should be more than vrr.vsync_start. + */ + + switch (type) { + case DP_SDP_VSC_EXT_VESA: + case DP_SDP_VSC_EXT_CEA: + stagger_val = VSC_EXT_STAGGER_DEFAULT; + break; + case HDMI_PACKET_TYPE_GAMUT_METADATA: + stagger_val = GMP_STAGGER_DEFAULT; + break; + case DP_SDP_PPS: + stagger_val = PPS_STAGGER_DEFAULT; + break; + case DP_SDP_VSC: + stagger_val = 0; + break; + default: + return 0; + } + + return 1 + 2 + stagger_val; +} + static int intel_dp_get_lines_for_sdp(const struct intel_crtc_state *crtc_state, u32 type) { + struct intel_display *display = to_intel_display(crtc_state); + + if (type == DP_SDP_ADAPTIVE_SYNC) + return crtc_state->vrr.vsync_start + 1; + + if (HAS_COMMON_SDP_TL(display)) + return intel_dp_get_lines_for_cmn_sdp_tl(type); + switch (type) { case DP_SDP_VSC_EXT_VESA: case DP_SDP_VSC_EXT_CEA: @@ -7399,8 +7449,6 @@ int intel_dp_get_lines_for_sdp(const struct intel_crtc_state *crtc_state, u32 ty return 7; case DP_SDP_VSC: return 3; - case DP_SDP_ADAPTIVE_SYNC: - return crtc_state->vrr.vsync_start + 1; default: break; } From b452f1fe719c47e06fef219757c50be3e71a20c6 Mon Sep 17 00:00:00 2001 From: Ankit Nautiyal Date: Tue, 15 Sep 2026 22:16:55 +0530 Subject: [PATCH 0115/1352] drm/i915/display: Dump DIP Transmission lines Add DIP transmission lines to the CRTC state dump. Signed-off-by: Ankit Nautiyal Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/20260915164657.3429075-15-ankit.k.nautiyal@intel.com --- drivers/gpu/drm/i915/display/intel_crtc_state_dump.c | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/drivers/gpu/drm/i915/display/intel_crtc_state_dump.c b/drivers/gpu/drm/i915/display/intel_crtc_state_dump.c index ad4f362e0c09d1..92675a88b969be 100644 --- a/drivers/gpu/drm/i915/display/intel_crtc_state_dump.c +++ b/drivers/gpu/drm/i915/display/intel_crtc_state_dump.c @@ -251,6 +251,15 @@ void intel_crtc_state_dump(const struct intel_crtc_state *pipe_config, str_enabled_disabled(pipe_config->has_panel_replay), str_enabled_disabled(pipe_config->enable_psr2_sel_fetch)); drm_printf(&p, "minimum hblank: %d\n", pipe_config->min_hblank); + + drm_printf(&p, "DIP Transmission Lines: EMP/AS SDP: %u\n", + pipe_config->dip.emp_as_sdp_tl); + drm_printf(&p, "DIP Transmission Lines: Common Base SDP: %u, GMP SDP: %u, PPS SDP: %u, VSC SDP: %u, VSC_EXT SDP: %u\n", + pipe_config->dip.cmn_sdp_tl, + pipe_config->dip.gmp_sdp_tl, + pipe_config->dip.pps_sdp_tl, + pipe_config->dip.vsc_sdp_tl, + pipe_config->dip.vsc_ext_sdp_tl); } drm_printf(&p, "audio: %i, infoframes: %i, infoframes enabled: 0x%x\n", From 0a18e1a795b0e3735619dc0035317bf2e009f06f Mon Sep 17 00:00:00 2001 From: Leon Romanovsky Date: Sun, 30 Aug 2026 15:31:49 +0300 Subject: [PATCH 0116/1352] PCI/P2PDMA: Update DMABUF lifecycle docs after move_notify() rename MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Commit 95308225e5ba ("dma-buf: Rename dma_buf_move_notify() to dma_buf_invalidate_mappings()") left the DMABUF section of the P2PDMA documentation pointing at move_notify(), a symbol that no longer exists. Readers grepping for it find nothing, and this is the only place in Documentation/ describing the revocation requirement. Name the current function and record that importers which cannot unmap within bounded time have to be rejected at attach time, which is what makes the synchronous unmap on remove() achievable. Fixes: 95308225e5ba ("dma-buf: Rename dma_buf_move_notify() to dma_buf_invalidate_mappings()") Signed-off-by: Leon Romanovsky Signed-off-by: Bjorn Helgaas Reviewed-by: Christian König Reviewed-by: Logan Gunthorpe Link: https://patch.msgid.link/20260830-doc-p2p-move-v1-1-61a388620588@nvidia.com --- Documentation/driver-api/pci/p2pdma.rst | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/Documentation/driver-api/pci/p2pdma.rst b/Documentation/driver-api/pci/p2pdma.rst index d3f406cca69409..63cff9e4d2c9ac 100644 --- a/Documentation/driver-api/pci/p2pdma.rst +++ b/Documentation/driver-api/pci/p2pdma.rst @@ -167,9 +167,11 @@ In this case the initiator and target pci_devices are known and the P2P subsyste is used to determine the mapping type. The phys_addr_t-based DMA API is used to establish the dma_addr_t. -Lifecycle is controlled by DMABUF move_notify(). When the exporting driver wants +Lifecycle is controlled by DMABUF revocation. When the exporting driver wants to remove() it must deliver an invalidation shutdown to all DMABUF importing -drivers through move_notify() and synchronously DMA unmap all the MMIO. +drivers through dma_buf_invalidate_mappings() and synchronously DMA unmap all +the MMIO. Importers unable to complete that unmap within bounded time have to +be rejected when they attach, which dma_buf_attach_revocable() checks for. No importing driver can continue to have a DMA map to the MMIO after the exporting driver has destroyed its p2p_provider. From bd43b8e94024171bedbffb24787f0e093871f61b Mon Sep 17 00:00:00 2001 From: Leon Romanovsky Date: Sun, 30 Aug 2026 14:16:19 +0300 Subject: [PATCH 0117/1352] PCI/P2PDMA: Do not tear down the allocate attribute on registration failure pci_p2pdma_add_resource() installs pci_p2pdma_unmap_mappings() as a devres action with the devres allocated p2p_pgmap as its data, and only then adds the range to the pool: error = devm_add_action_or_reset(&pdev->dev, pci_p2pdma_unmap_mappings, p2p_pgmap); if (error) goto pages_free; p2pdma = rcu_dereference_protected(pdev->p2pdma, 1); error = gen_pool_add_owner(p2pdma->pool, ...); if (error) goto pages_free; The action removes the allocate attribute for the whole device, which tears down existing userspace mappings of every BAR already registered on it. Both failures here get that wrong, in opposite ways. devm_add_action_or_reset() runs the action when it cannot allocate its devres node, so an -ENOMEM while registering a second BAR unmaps the first one. Use devm_add_action() and let the error path unwind only what this call created. gen_pool_add_owner() allocates a chunk and can also fail with -ENOMEM. There the action is registered, and the error path frees p2p_pgmap with devm_kfree() while leaving the action pointing at it. On unbind devres runs the action and pci_p2pdma_unmap_mappings() dereferences p2p_pgmap->mem->owner->kobj, which is freed memory. Give that failure its own label and drop the action with devm_remove_action(), which removes it without running it. Fixes: 7e9c7ef83d78 ("PCI/P2PDMA: Allow userspace VMA allocations through sysfs") Fixes: f58ef9d1d135 ("PCI/P2PDMA: Separate the mmap() support from the core logic") Signed-off-by: Leon Romanovsky Signed-off-by: Bjorn Helgaas Tested-by: Tushar Dave Reviewed-by: Logan Gunthorpe Reviewed-by: Jason Gunthorpe Link: https://patch.msgid.link/20260830-batch-p2p-fixes-v1-1-5044e8dfbe2e@nvidia.com --- drivers/pci/p2pdma.c | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/drivers/pci/p2pdma.c b/drivers/pci/p2pdma.c index 772640a08bf948..8e4129c7e9fb53 100644 --- a/drivers/pci/p2pdma.c +++ b/drivers/pci/p2pdma.c @@ -440,8 +440,8 @@ int pci_p2pdma_add_resource(struct pci_dev *pdev, int bar, size_t size, goto pgmap_free; } - error = devm_add_action_or_reset(&pdev->dev, pci_p2pdma_unmap_mappings, - p2p_pgmap); + error = devm_add_action(&pdev->dev, pci_p2pdma_unmap_mappings, + p2p_pgmap); if (error) goto pages_free; @@ -451,13 +451,15 @@ int pci_p2pdma_add_resource(struct pci_dev *pdev, int bar, size_t size, range_len(&pgmap->range), dev_to_node(&pdev->dev), &pgmap->ref); if (error) - goto pages_free; + goto mappings_remove; pci_info(pdev, "added peer-to-peer DMA memory %#llx-%#llx\n", pgmap->range.start, pgmap->range.end); return 0; +mappings_remove: + devm_remove_action(&pdev->dev, pci_p2pdma_unmap_mappings, p2p_pgmap); pages_free: devm_memunmap_pages(&pdev->dev, pgmap); pgmap_free: From 3d02f70940f9143d86ca5d95b3f1eca14f03db3b Mon Sep 17 00:00:00 2001 From: Leon Romanovsky Date: Sun, 30 Aug 2026 14:16:20 +0300 Subject: [PATCH 0118/1352] PCI/P2PDMA: Wait for RCU readers before freeing state pci_p2pmem_find_many() scans all PCI devices without locking or protection against driver unbind, including devices with poolless P2PDMA state. pci_has_p2pmem() may observe pdev->p2pdma just before driver unbind clears it, while pci_p2pdma_release() skips the grace period when no pool is present. This allows devres to free the object while it is still in use. Clear the pointer with RCU_INIT_POINTER() and always wait for pre-existing RCU readers before returning. The same grace period continues to protect gen_pool users for pool-backed providers. Fixes: 372d6d1b8ae3 ("PCI/P2PDMA: Refactor to separate core P2P functionality from memory allocation") Signed-off-by: Leon Romanovsky Signed-off-by: Bjorn Helgaas Tested-by: Tushar Dave Reviewed-by: Logan Gunthorpe Reviewed-by: Jason Gunthorpe Cc: Alex Williamson Cc: Matt Evans Link: https://patch.msgid.link/20260830-batch-p2p-fixes-v1-2-5044e8dfbe2e@nvidia.com --- drivers/pci/p2pdma.c | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/drivers/pci/p2pdma.c b/drivers/pci/p2pdma.c index 8e4129c7e9fb53..d23f23965cbcbf 100644 --- a/drivers/pci/p2pdma.c +++ b/drivers/pci/p2pdma.c @@ -236,9 +236,8 @@ static void pci_p2pdma_release(void *data) return; /* Flush and disable pci_alloc_p2p_mem() */ - pdev->p2pdma = NULL; - if (p2pdma->pool) - synchronize_rcu(); + RCU_INIT_POINTER(pdev->p2pdma, NULL); + synchronize_rcu(); xa_destroy(&p2pdma->map_types); if (!p2pdma->pool) From 2c6079599aa0f95fdd6e1942529b09478bafcb56 Mon Sep 17 00:00:00 2001 From: Leon Romanovsky Date: Sun, 30 Aug 2026 14:16:21 +0300 Subject: [PATCH 0119/1352] PCI/P2PDMA: Restrict the p2pmem search to pool-backed providers pci_p2pmem_find_many() exists to pick a provider that the caller will then allocate from with pci_alloc_p2pmem(), which goes straight to the gen_pool: ret = (void *)gen_pool_alloc_owner(p2pdma->pool, size, (void **) &ref); pci_has_p2pmem() does not ask for that pool, only for the published flag. The two used to be equivalent, because a provider could only exist by way of pci_p2pdma_add_resource(), which always creates the pool. pcim_p2pdma_init() broke that. It registers a provider for the DMABUF path and never creates a pool, so pdev->p2pdma is set while p2pdma->pool stays NULL. Nothing publishes such a provider today, so the search cannot return one yet, but the flag alone no longer says what the caller needs. Ask for the pool as well, so the search covers the providers its result is used for. A later patch documents the pdev->p2pdma lifetime and RCU rules. Signed-off-by: Leon Romanovsky Signed-off-by: Bjorn Helgaas Tested-by: Tushar Dave Reviewed-by: Logan Gunthorpe Reviewed-by: Jason Gunthorpe Link: https://patch.msgid.link/20260830-batch-p2p-fixes-v1-3-5044e8dfbe2e@nvidia.com --- drivers/pci/p2pdma.c | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/drivers/pci/p2pdma.c b/drivers/pci/p2pdma.c index d23f23965cbcbf..ad896e28d667a0 100644 --- a/drivers/pci/p2pdma.c +++ b/drivers/pci/p2pdma.c @@ -876,7 +876,13 @@ static bool pci_has_p2pmem(struct pci_dev *pdev) rcu_read_lock(); p2pdma = rcu_dereference(pdev->p2pdma); - res = p2pdma && p2pdma->p2pmem_published; + + /* + * The callers hand the result to pci_alloc_p2pmem(), so only a + * provider backed by a pool is of any use here. pcim_p2pdma_init() + * creates providers without one. + */ + res = p2pdma && p2pdma->pool && p2pdma->p2pmem_published; rcu_read_unlock(); return res; From 2600739d5a9b6b4dc5139ca618c76cdf695e8589 Mon Sep 17 00:00:00 2001 From: Leon Romanovsky Date: Sun, 30 Aug 2026 14:16:22 +0300 Subject: [PATCH 0120/1352] PCI/P2PDMA: Safely terminate ACS redirect lists seq_buf marks an overflow by setting len to size + 1. The ACS diagnostic path unconditionally writes a terminator to buffer[len - 1], so a path with enough ACS ports to fill the 128-byte buffer writes one byte beyond the buffer when verbose diagnostics are requested. Use seq_buf_str() to terminate truncated output safely and remove the final semicolon only when the buffer did not overflow. Fixes: 52916982af48 ("PCI/P2PDMA: Support peer-to-peer memory") Signed-off-by: Leon Romanovsky Signed-off-by: Bjorn Helgaas Tested-by: Tushar Dave Reviewed-by: Logan Gunthorpe Reviewed-by: Jason Gunthorpe Link: https://patch.msgid.link/20260830-batch-p2p-fixes-v1-4-5044e8dfbe2e@nvidia.com --- drivers/pci/p2pdma.c | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/drivers/pci/p2pdma.c b/drivers/pci/p2pdma.c index ad896e28d667a0..f76510e43da2b7 100644 --- a/drivers/pci/p2pdma.c +++ b/drivers/pci/p2pdma.c @@ -780,11 +780,13 @@ calc_map_type_and_dist(struct pci_dev *provider, struct pci_dev *client, } if (verbose) { - acs_list.buffer[acs_list.len-1] = 0; /* drop final semicolon */ + /* Drop the final semicolon; the list is not empty here */ + if (!seq_buf_has_overflowed(&acs_list)) + acs_list.buffer[acs_list.len - 1] = '\0'; pci_warn(client, "ACS redirect is set between the client and provider (%s)\n", pci_name(provider)); pci_warn(client, "to disable ACS redirect for this path, add the kernel parameter: pci=disable_acs_redir=%s\n", - acs_list.buffer); + seq_buf_str(&acs_list)); } acs_redirects = true; From b9964341607f7954184f7ce5e7d8b937e44952b7 Mon Sep 17 00:00:00 2001 From: Leon Romanovsky Date: Sun, 30 Aug 2026 14:16:23 +0300 Subject: [PATCH 0121/1352] PCI/P2PDMA: Gate the host bridge whitelist warning on verbose calc_map_type_and_dist() prints every other diagnostic under its verbose argument, but reaches the "Host bridge not in P2PDMA whitelist" warning through host_bridge_whitelist(), which it hands acs_redirects instead. A caller that asked for a silent answer still gets the warning whenever any port on the path has an ACS redirect bit set, the CPU is not whitelisted by cpu_supports_p2pdma(), and the host bridge is not in pci_p2pdma_whitelist[]. pci_p2pmem_find_many() is such a caller. It sweeps every device with published p2pmem and asks for the distance to each client with verbose=false, and pci_p2pdma_distance_many() recomputes rather than consulting the map_types cache, so the warning repeats on every sweep. The argument was never meant to say "ACS redirects were found". When commit cf201bfe8cdc ("PCI/P2PDMA: Warn if host bridge not in whitelist") added it, acs_redirects was a bool pointer that the quiet entry point passed as NULL: if (verbose) map = calc_map_type_and_dist_warn(provider, pci_client, &distance); else map = calc_map_type_and_dist(provider, pci_client, &distance, NULL, NULL); so the argument was true on exactly the path that commit describes. Folding the two entry points into one verbose flag turned the pointer into a value and left the call site alone, silently narrowing the warning to paths that carry an ACS redirect. Pass verbose. This also restores the warning for a verbose caller that takes the host bridge route with no ACS redirect on the path, which until now was told it could not use peer-to-peer DMA without being told which vendor and device would have to be added to the whitelist. Fixes: d1b8dc09dd71 ("PCI/P2PDMA: Simplify distance calculation") Signed-off-by: Leon Romanovsky Signed-off-by: Bjorn Helgaas Tested-by: Tushar Dave Reviewed-by: Jason Gunthorpe Reviewed-by: Logan Gunthorpe Link: https://patch.msgid.link/20260830-batch-p2p-fixes-v1-5-5044e8dfbe2e@nvidia.com --- drivers/pci/p2pdma.c | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/drivers/pci/p2pdma.c b/drivers/pci/p2pdma.c index f76510e43da2b7..beb1fe20c9de54 100644 --- a/drivers/pci/p2pdma.c +++ b/drivers/pci/p2pdma.c @@ -717,7 +717,6 @@ calc_map_type_and_dist(struct pci_dev *provider, struct pci_dev *client, { enum pci_p2pdma_map_type map_type = PCI_P2PDMA_MAP_THRU_HOST_BRIDGE; struct pci_dev *a = provider, *b = client, *bb; - bool acs_redirects = false; struct pci_p2pdma *p2pdma; struct seq_buf acs_list; int acs_cnt = 0; @@ -788,11 +787,10 @@ calc_map_type_and_dist(struct pci_dev *provider, struct pci_dev *client, pci_warn(client, "to disable ACS redirect for this path, add the kernel parameter: pci=disable_acs_redir=%s\n", seq_buf_str(&acs_list)); } - acs_redirects = true; map_through_host_bridge: if (!cpu_supports_p2pdma() && - !host_bridge_whitelist(provider, client, acs_redirects)) { + !host_bridge_whitelist(provider, client, verbose)) { if (verbose) pci_warn(client, "cannot be used for peer-to-peer DMA as the client and provider (%s) do not share an upstream bridge or whitelisted host bridge\n", pci_name(provider)); From 76048abb841c649ebc2cc9df9da7a039201ec1c8 Mon Sep 17 00:00:00 2001 From: Zongmin Zhou Date: Fri, 18 Sep 2026 15:36:50 +0800 Subject: [PATCH 0122/1352] riscv: hwprobe: use _BITULL() rather than BIT() in MIPS vendor uapi header BIT() is a kernel-internal macro that is not available to userspace, but the MIPS vendor extension uapi header uses it without defining or including it. Any userspace program that includes this header and uses RISCV_HWPROBE_VENDOR_EXT_XMIPSEXECTL fails to build. Use _BITULL(0) from linux/const.h instead, which keeps the value at 1, so there is no ABI change. Fixes: bb4b0f8a1bcb ("riscv: hwprobe: Add MIPS vendor extension probing") Cc: stable@vger.kernel.org Signed-off-by: Zongmin Zhou Reviewed-by: Jesse Taube Link: https://patch.msgid.link/20260918073650.42586-1-min_halo@163.com Signed-off-by: Paul Walmsley --- arch/riscv/include/uapi/asm/vendor/mips.h | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/arch/riscv/include/uapi/asm/vendor/mips.h b/arch/riscv/include/uapi/asm/vendor/mips.h index e65ab268b26551..a31c23afe74fb2 100644 --- a/arch/riscv/include/uapi/asm/vendor/mips.h +++ b/arch/riscv/include/uapi/asm/vendor/mips.h @@ -1,3 +1,5 @@ /* SPDX-License-Identifier: GPL-2.0 WITH Linux-syscall-note */ -#define RISCV_HWPROBE_VENDOR_EXT_XMIPSEXECTL BIT(0) +#include + +#define RISCV_HWPROBE_VENDOR_EXT_XMIPSEXECTL _BITULL(0) From 79138f9e3dd1a688fd024a8304a4ca642aed3c8e Mon Sep 17 00:00:00 2001 From: Arnd Bergmann Date: Wed, 16 Sep 2026 10:37:21 +0200 Subject: [PATCH 0123/1352] riscv: limit sifive errata to CONFIG_64BIT There are two errata for this vendor, and they both individually depend on CONFIG_64BIT already. However, an 32-bit allmodconfig produces this warning from clang for a condition that can never be true. arch/riscv/errata/sifive/errata.c:29:14: error: result of comparison of constant 9223372036854775815 with expression of type 'unsigned long' is always true [-Werror,-Wtautological-constant-out-of-range-compare] 29 | if (arch_id != 0x8000000000000007 || | ~~~~~~~ ^ ~~~~~~~~~~~~~~~~~~ arch/riscv/errata/sifive/errata.c:42:14: error: result of comparison of constant 9223372036854775815 with expression of type 'unsigned long' is always true [-Werror,-Wtautological-constant-out-of-range-compare] 42 | if (arch_id != 0x8000000000000007 && arch_id != 0x1) | ~~~~~~~ ^ ~~~~~~~~~~~~~~~~~~ Signed-off-by: Arnd Bergmann Link: https://patch.msgid.link/20260916083752.47310-1-arnd@kernel.org Signed-off-by: Paul Walmsley --- arch/riscv/Kconfig.errata | 2 +- arch/riscv/Kconfig.socs | 1 + 2 files changed, 2 insertions(+), 1 deletion(-) diff --git a/arch/riscv/Kconfig.errata b/arch/riscv/Kconfig.errata index 3c945d086c7d02..38ade0ea4e9cb7 100644 --- a/arch/riscv/Kconfig.errata +++ b/arch/riscv/Kconfig.errata @@ -46,7 +46,7 @@ config ERRATA_MIPS_P8700_PAUSE_OPCODE config ERRATA_SIFIVE bool "SiFive errata" - depends on RISCV_ALTERNATIVE + depends on RISCV_ALTERNATIVE && 64BIT help All SiFive errata Kconfig depend on this Kconfig. Disabling this Kconfig will disable all SiFive errata. Please say "Y" diff --git a/arch/riscv/Kconfig.socs b/arch/riscv/Kconfig.socs index 429e0758930687..cecbce0c498384 100644 --- a/arch/riscv/Kconfig.socs +++ b/arch/riscv/Kconfig.socs @@ -34,6 +34,7 @@ config ARCH_RENESAS config ARCH_SIFIVE bool "SiFive SoCs" select ERRATA_SIFIVE + depends on 64BIT help This enables support for SiFive SoC platform hardware. From 90f6065b5c2a40e86149e8e59ad0501469a83cc6 Mon Sep 17 00:00:00 2001 From: Troy Mitchell Date: Tue, 8 Sep 2026 21:03:37 +0800 Subject: [PATCH 0124/1352] riscv: vector: Fix data pointer constraints in context save/restore The standard vector save/restore asm advances datap but declares it as input-only. An inlined caller reusing the original pointer may therefore use the advanced address instead. Declare datap as read-write so the compiler can preserve the original pointer when needed. Fixes: 03c3fcd9941a ("riscv: Introduce struct/helpers to save/restore per-task Vector state") Signed-off-by: Troy Mitchell Reviewed-by: GUO Ren (XuanTie) Reviewed-by: Aurelien Jarno Reviewed-by: Andy Chiu Link: https://patch.msgid.link/20260908-riscv-vector-asm-fix-v1-1-147f314efb2b@linux.dev Signed-off-by: Paul Walmsley --- arch/riscv/include/asm/vector.h | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/arch/riscv/include/asm/vector.h b/arch/riscv/include/asm/vector.h index fffe72a7720804..c7fd6d50a7a47a 100644 --- a/arch/riscv/include/asm/vector.h +++ b/arch/riscv/include/asm/vector.h @@ -230,7 +230,7 @@ static inline void __riscv_v_vstate_save(struct __riscv_v_ext_state *save_to, "add %1, %1, %0\n\t" "vse8.v v24, (%1)\n\t" ".option pop\n\t" - : "=&r" (vl) : "r" (datap) : "memory"); + : "=&r" (vl), "+r" (datap) : : "memory"); } riscv_v_disable(); } @@ -266,7 +266,7 @@ static inline void __riscv_v_vstate_restore(struct __riscv_v_ext_state *restore_ "add %1, %1, %0\n\t" "vle8.v v24, (%1)\n\t" ".option pop\n\t" - : "=&r" (vl) : "r" (datap) : "memory"); + : "=&r" (vl), "+r" (datap) : : "memory"); } __vstate_csr_restore(restore_from); riscv_v_disable(); From e1116094833337fe9480fb9527a59c7637b96118 Mon Sep 17 00:00:00 2001 From: Aurelien Jarno Date: Fri, 11 Sep 2026 21:42:17 +0200 Subject: [PATCH 0125/1352] riscv: vector: Fix output operands in context save The __vstate_csr_save() inline assembly declares vcsr as an output operand while the assembly only writes vstart, vtype, and vl. Remove it as it is later saved by the C code. Fixes: d863910eabaf ("riscv: vector: Support xtheadvector save/restore") Signed-off-by: Aurelien Jarno Link: https://patch.msgid.link/20260911194218.3379486-1-aurelien@aurel32.net Signed-off-by: Paul Walmsley --- arch/riscv/include/asm/vector.h | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/arch/riscv/include/asm/vector.h b/arch/riscv/include/asm/vector.h index c7fd6d50a7a47a..1f8dcae3af517e 100644 --- a/arch/riscv/include/asm/vector.h +++ b/arch/riscv/include/asm/vector.h @@ -137,8 +137,8 @@ static __always_inline void __vstate_csr_save(struct __riscv_v_ext_state *dest) "csrr %0, " __stringify(CSR_VSTART) "\n\t" "csrr %1, " __stringify(CSR_VTYPE) "\n\t" "csrr %2, " __stringify(CSR_VL) "\n\t" - : "=r" (dest->vstart), "=r" (dest->vtype), "=r" (dest->vl), - "=r" (dest->vcsr) : :); + : "=r" (dest->vstart), "=r" (dest->vtype), "=r" (dest->vl) + : :); if (has_xtheadvector()) { unsigned long status; From f72bffbaa04dd1174526bd87b595a9c802b1b58a Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:27:49 +1000 Subject: [PATCH 0126/1352] nfsd: honour client-provided attributes for NFS4_CREATE_EXCLUSIVE4_1 When a file is created with a v4.1 OPEN which requests NFS4_CREATE_EXCLUSIVE4_1, the request can include attributes to be set. However when the mtime/atime are set to hold the verifier, the other ia_valid flags are cleared, so no attributes requested by the client are used. This code was originally written for NFSv3 where NFS3_CREATE_EXCLUSIVE never includes attributes. When it was updated for v4.1, the fact that an exclusive create CAN include attributes was not handled properly. Fixes: ac6721a13e5b ("nfsd41: make sure nfs server process OPEN with EXCLUSIVE4_1 correctly") Cc: stable@vger.kernel.org Reviewed-by: Jeff Layton Signed-off-by: NeilBrown Link: https://patch.msgid.link/20260717093001.1972119-2-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 50c07561e31f3c..f3f7f14a93683f 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -394,8 +394,8 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, if ((iap->ia_valid & ATTR_SIZE) && (iap->ia_size == 0)) iap->ia_valid &= ~ATTR_SIZE; if (nfsd4_create_is_exclusive(open->op_createmode)) { - iap->ia_valid = ATTR_MTIME | ATTR_ATIME | - ATTR_MTIME_SET|ATTR_ATIME_SET; + iap->ia_valid |= ATTR_MTIME | ATTR_ATIME | + ATTR_MTIME_SET|ATTR_ATIME_SET; iap->ia_mtime.tv_sec = v_mtime; iap->ia_atime.tv_sec = v_atime; iap->ia_mtime.tv_nsec = 0; From ae94c530e5e2074c3c200b0a4412adb10fd7ee7e Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:27:50 +1000 Subject: [PATCH 0127/1352] nfsd: move check_nfsd_access() call into nfsd_cross_mnt() Whenever we cross a mount point, we need to check_nfsd_access() for v4. So move the call into nfsd_cross_mnt() in the place where we actually do cross. This avoids the possibility of calling nfsd_cross_mnt() without the required check_nfsd_access(). Also remove the last arg from check_nfsd_access(), which is always false. nfsd_cross_mnt() now returns an nfserr rather than an errno. Signed-off-by: NeilBrown Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717093001.1972119-3-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/export.c | 6 ++--- fs/nfsd/export.h | 3 +-- fs/nfsd/nfs4proc.c | 2 +- fs/nfsd/nfs4xdr.c | 9 +------- fs/nfsd/vfs.c | 55 +++++++++++++++++++++++++--------------------- fs/nfsd/vfs.h | 4 ++-- 6 files changed, 37 insertions(+), 42 deletions(-) diff --git a/fs/nfsd/export.c b/fs/nfsd/export.c index b6e0c543e02891..76cde579372141 100644 --- a/fs/nfsd/export.c +++ b/fs/nfsd/export.c @@ -1890,21 +1890,19 @@ __be32 check_security_flavor(struct svc_export *exp, struct svc_rqst *rqstp, * check_nfsd_access - check if access to export is allowed. * @exp: svc_export that is being accessed. * @rqstp: svc_rqst attempting to access @exp. - * @may_bypass_gss: reduce strictness of authorization check * * Return values: * %nfs_ok if access is granted, or * %nfserr_wrongsec if access is denied */ -__be32 check_nfsd_access(struct svc_export *exp, struct svc_rqst *rqstp, - bool may_bypass_gss) +__be32 check_nfsd_access(struct svc_export *exp, struct svc_rqst *rqstp) { __be32 status; status = check_xprtsec_policy(exp, rqstp); if (status != nfs_ok) return status; - return check_security_flavor(exp, rqstp, may_bypass_gss); + return check_security_flavor(exp, rqstp, false); } /* diff --git a/fs/nfsd/export.h b/fs/nfsd/export.h index d2b09cd761453d..117fb28db1e024 100644 --- a/fs/nfsd/export.h +++ b/fs/nfsd/export.h @@ -104,8 +104,7 @@ int nfsexp_flags(struct svc_cred *cred, struct svc_export *exp); __be32 check_xprtsec_policy(struct svc_export *exp, struct svc_rqst *rqstp); __be32 check_security_flavor(struct svc_export *exp, struct svc_rqst *rqstp, bool may_bypass_gss); -__be32 check_nfsd_access(struct svc_export *exp, struct svc_rqst *rqstp, - bool may_bypass_gss); +__be32 check_nfsd_access(struct svc_export *exp, struct svc_rqst *rqstp); /* * Function declarations diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index f3f7f14a93683f..93d8e722aae80a 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -3328,7 +3328,7 @@ nfsd4_proc_compound(struct svc_rqst *rqstp) if (current_fh->fh_export && need_wrongsec_check(rqstp)) - op->status = check_nfsd_access(current_fh->fh_export, rqstp, false); + op->status = check_nfsd_access(current_fh->fh_export, rqstp); } encode_op: if (op->status == nfserr_replay_me) { diff --git a/fs/nfsd/nfs4xdr.c b/fs/nfsd/nfs4xdr.c index 606ddcb085c027..04755c41d87115 100644 --- a/fs/nfsd/nfs4xdr.c +++ b/fs/nfsd/nfs4xdr.c @@ -4574,8 +4574,6 @@ nfsd4_encode_entry4_fattr(struct nfsd4_readdir *cd, const char *name, * directly from the mountpoint dentry. */ if (nfsd_mountpoint(dentry, exp)) { - int err; - if (!(exp->ex_flags & NFSEXP_V4ROOT) && !attributes_need_mount(cd->rd_bmval)) { ignore_crossmnt = 1; @@ -4586,12 +4584,7 @@ nfsd4_encode_entry4_fattr(struct nfsd4_readdir *cd, const char *name, * Different "."/".." handling? Something else? * At least, add a comment here to explain.... */ - err = nfsd_cross_mnt(cd->rd_rqstp, &dentry, &exp); - if (err) { - nfserr = nfserrno(err); - goto out_put; - } - nfserr = check_nfsd_access(exp, cd->rd_rqstp, false); + nfserr = nfsd_cross_mnt(cd->rd_rqstp, &dentry, &exp); if (nfserr) goto out_put; crossed = true; diff --git a/fs/nfsd/vfs.c b/fs/nfsd/vfs.c index 8923a9910a08f2..7386062ae449aa 100644 --- a/fs/nfsd/vfs.c +++ b/fs/nfsd/vfs.c @@ -118,15 +118,15 @@ nfserrno (int errno) return nfserr_io; } -/* - * Called from nfsd_lookup and encode_dirent. Check if we have crossed +/* + * Called from nfsd_lookup and encode_dirent. Check if we have crossed * a mount point. - * Returns -EAGAIN or -ETIMEDOUT leaving *dpp and *expp unchanged, + * Returns an nfs error leaving *dpp and *expp unchanged, * or nfs_ok having possibly changed *dpp and *expp */ -int -nfsd_cross_mnt(struct svc_rqst *rqstp, struct dentry **dpp, - struct svc_export **expp) +__be32 +nfsd_cross_mnt(struct svc_rqst *rqstp, struct dentry **dpp, + struct svc_export **expp) { struct svc_export *exp = *expp, *exp2 = NULL; struct dentry *dentry = *dpp; @@ -134,6 +134,7 @@ nfsd_cross_mnt(struct svc_rqst *rqstp, struct dentry **dpp, .dentry = dget(dentry)}; unsigned int follow_flags = 0; int err = 0; + __be32 nfserr = nfs_ok; if (exp->ex_flags & NFSEXP_CROSSMOUNT) follow_flags = LOOKUP_AUTOMOUNT; @@ -163,23 +164,28 @@ nfsd_cross_mnt(struct svc_rqst *rqstp, struct dentry **dpp, err = 0; } else if (nfsd_v4client(rqstp) || (exp->ex_flags & NFSEXP_CROSSMOUNT) || EX_NOHIDE(exp2)) { - /* successfully crossed mount point */ - /* - * This is subtle: path.dentry is *not* on path.mnt - * at this point. The only reason we are safe is that - * original mnt is pinned down by exp, so we should - * put path *before* putting exp - */ - *dpp = path.dentry; - path.dentry = dentry; - *expp = exp2; - exp2 = exp; + nfserr = check_nfsd_access(exp, rqstp); + if (nfserr == nfs_ok) { + /* successfully crossed mount point */ + /* + * This is subtle: path.dentry is *not* on path.mnt + * at this point. The only reason we are safe is that + * original mnt is pinned down by exp, so we should + * put path *before* putting exp + */ + *dpp = path.dentry; + path.dentry = dentry; + *expp = exp2; + exp2 = exp; + } } out: path_put(&path); if (exp2) exp_put(exp2); - return err; + if (nfserr) + return nfserr; + return nfserrno(err); } static void follow_to_parent(struct path *path) @@ -277,10 +283,12 @@ nfsd_lookup_dentry(struct svc_rqst *rqstp, struct svc_fh *fhp, if (IS_ERR(dentry)) goto out_nfserr; if (nfsd_mountpoint(dentry, exp)) { - host_err = nfsd_cross_mnt(rqstp, &dentry, &exp); - if (host_err) { + __be32 nfserr = nfsd_cross_mnt(rqstp, &dentry, &exp); + + if (nfserr) { dput(dentry); - goto out_nfserr; + exp_put(exp); + return nfserr; } } } @@ -327,9 +335,6 @@ nfsd_lookup(struct svc_rqst *rqstp, struct svc_fh *fhp, const char *name, err = nfsd_lookup_dentry(rqstp, fhp, name, len, &exp, &dentry); if (err) return err; - err = check_nfsd_access(exp, rqstp, false); - if (err) - goto out; /* * Note: we compose the file handle now, but as the * dentry may be negative, it may need to be updated. @@ -337,7 +342,7 @@ nfsd_lookup(struct svc_rqst *rqstp, struct svc_fh *fhp, const char *name, err = fh_compose(resfh, exp, dentry, fhp); if (!err && d_really_is_negative(dentry)) err = nfserr_noent; -out: + dput(dentry); exp_put(exp); return err; diff --git a/fs/nfsd/vfs.h b/fs/nfsd/vfs.h index 4af2ff9e9dfeee..5554878781f464 100644 --- a/fs/nfsd/vfs.h +++ b/fs/nfsd/vfs.h @@ -76,8 +76,8 @@ static inline bool nfsd_attrs_valid(struct nfsd_attrs *attrs) } __be32 nfserrno (int errno); -int nfsd_cross_mnt(struct svc_rqst *rqstp, struct dentry **dpp, - struct svc_export **expp); +__be32 nfsd_cross_mnt(struct svc_rqst *rqstp, struct dentry **dpp, + struct svc_export **expp); __be32 nfsd_lookup(struct svc_rqst *, struct svc_fh *, const char *, unsigned int, struct svc_fh *); __be32 nfsd_lookup_dentry(struct svc_rqst *, struct svc_fh *, From cd7ef082dabaebace84916be35dde277f4907a3e Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:27:51 +1000 Subject: [PATCH 0128/1352] nfsd: correctly handle CREATE of mounted-on files Linux allows a file (non-directory) to be mounted on a file. nfsd mostly supports this if the crossmnt option is in effect. However if CREATE is used on an existing mounted-on file, the filehandle for the underlying file is returns. The client will then continue to use that filehandle. So cat /mnt/file will show the contents of the mounted file as expected, but if the dcache is flushed with "drop_caches" or similar, then >> /mnt/file cat /mnt/file will show the mounted-on file. For exclusive or checked creates this is not a problem as the creation will fail no matter which file is seen. For unchecked creates we need to see if the name is in the dcache, and if it is mounted. If so, we simply provide that filehandle, possibly truncating. This probably has always existed since before the git history. Signed-off-by: NeilBrown Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717093001.1972119-4-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs3proc.c | 19 ++++++++++++++++++- fs/nfsd/nfs4proc.c | 28 ++++++++++++++++++++++++++++ fs/nfsd/nfsproc.c | 17 ++++++++++++++++- 3 files changed, 62 insertions(+), 2 deletions(-) diff --git a/fs/nfsd/nfs3proc.c b/fs/nfsd/nfs3proc.c index 0904d953d10e07..1df3c719e0da6c 100644 --- a/fs/nfsd/nfs3proc.c +++ b/fs/nfsd/nfs3proc.c @@ -282,6 +282,7 @@ nfsd3_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, struct nfsd_attrs attrs = { .na_iattr = iap, }; + struct svc_export *exp; __u32 v_mtime, v_atime; struct inode *inode; __be32 status; @@ -320,7 +321,23 @@ nfsd3_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, goto out; } - status = fh_compose(resfhp, fhp->fh_export, child, fhp); + exp = exp_get(fhp->fh_export); + if (argp->createmode == NFS3_CREATE_UNCHECKED) { + /* + * If name is already in dcache we need to check for mountpoints + */ + if (d_is_reg(child) && + unlikely(nfsd_mountpoint(child, exp))) { + status = nfsd_cross_mnt(rqstp, &child, &exp); + if (status != nfs_ok) { + exp_put(exp); + goto out; + } + } + } + + status = fh_compose(resfhp, exp, child, fhp); + exp_put(exp); if (status != nfs_ok) goto out; diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 93d8e722aae80a..424221677fa0ee 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -271,6 +271,34 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, parent = fhp->fh_dentry; inode = d_inode(parent); + if (open->op_createmode == NFS4_CREATE_UNCHECKED) { + /* + * If name is already in dcache we need to check for mountpoints + */ + child = try_lookup_noperm(&QSTR_LEN(open->op_fname, + open->op_fnamelen), + parent); + if (child && !IS_ERR(child) && d_is_reg(child) && + unlikely(nfsd_mountpoint(child, fhp->fh_export))) { + struct svc_export *exp = exp_get(fhp->fh_export); + + status = nfsd_cross_mnt(rqstp, &child, &exp); + if (status == nfs_ok) + status = fh_compose(resfhp, exp, + child, fhp); + if (status == nfs_ok) + status = fh_fill_both_attrs(fhp); + open->op_truncate = + (iap->ia_valid & ATTR_SIZE) && + !iap->ia_size; + dput(child); + exp_put(exp); + return status; + } + if (!IS_ERR(child)) + dput(child); + } + host_err = fh_want_write(fhp); if (host_err) return nfserrno(host_err); diff --git a/fs/nfsd/nfsproc.c b/fs/nfsd/nfsproc.c index e2b5f8a241bea7..2a82fa64e47855 100644 --- a/fs/nfsd/nfsproc.c +++ b/fs/nfsd/nfsproc.c @@ -291,6 +291,7 @@ nfsd_proc_create(struct svc_rqst *rqstp) struct nfsd_attrs attrs = { .na_iattr = attr, }; + struct svc_export *exp; struct inode *inode; struct dentry *dchild; int type, mode; @@ -319,8 +320,22 @@ nfsd_proc_create(struct svc_rqst *rqstp) resp->status = nfserrno(PTR_ERR(dchild)); goto out_write; } + /* + * If name exists we need to check for mountpoints + */ + exp = exp_get(dirfhp->fh_export); + if (d_is_reg(dchild) && + unlikely(nfsd_mountpoint(dchild, exp))) { + resp->status = nfsd_cross_mnt(rqstp, &dchild, &exp); + if (resp->status != nfs_ok) { + exp_put(exp); + goto out_unlock; + } + } + fh_init(newfhp, NFS_FHSIZE); - resp->status = fh_compose(newfhp, dirfhp->fh_export, dchild, dirfhp); + resp->status = fh_compose(newfhp, exp, dchild, dirfhp); + exp_put(exp); if (!resp->status && d_really_is_negative(dchild)) resp->status = nfserr_noent; if (resp->status) { From 9245feafe20f5526c0b576eefa530962daff8d9d Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:27:52 +1000 Subject: [PATCH 0129/1352] nfsd: replace fh_fill_both_attrs() with fh_fill_post_noop() fh_fill_both_attrs() is only needed for open/create and is used in the case when the target already existed so no creating happens. As part of refactoring this code it is changed to call fh_fill_pre_attrs() once early on (so errors only need to be caught in one place) and then to use a new fh_fill_post_noop() when it is determined that no creation happened. fh_fill_pre_attrs() now stores the attrs (which it had to get all of anyway)_ in ->fh_post_attr. fh_fill_post_noop() simply marks them as valid. fh_fill_post_attrs() replaces them. This change involves moving fh_fill_pre_attrs() out of the inode_lock on the directory. This means that we cannot provide "atomic" wcc data so a new fh_fill_pre_attrs_unlocked() is provided which marks the attrs as non-atomic. This is unfortunate but inevitable if we are ever to allow concurrent updates in a directory (which can significantly improve performance in some cases). To get atomic pre/post attributes we will need to be able to ask the fs to provide them, or to request a lease on the directory for the duration of an operation. Note that we haven't provided pre/post attrs on WRITE requests for a long time for exactly this reason - we cannot lock the file to get them. Reviewed-by: Jeff Layton Signed-off-by: NeilBrown Link: https://patch.msgid.link/20260717093001.1972119-5-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 23 +++++++--------- fs/nfsd/nfsfh.c | 69 +++++++++++++++++++++++----------------------- fs/nfsd/nfsfh.h | 14 +++++++++- 3 files changed, 57 insertions(+), 49 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 424221677fa0ee..8e17e95d2cd058 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -286,8 +286,7 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, if (status == nfs_ok) status = fh_compose(resfhp, exp, child, fhp); - if (status == nfs_ok) - status = fh_fill_both_attrs(fhp); + fh_fill_post_noop(fhp); open->op_truncate = (iap->ia_valid & ATTR_SIZE) && !iap->ia_size; @@ -356,9 +355,7 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, /* NFSv4 protocol requires change attributes even though * no change happened. */ - status = fh_fill_both_attrs(fhp); - if (status != nfs_ok) - goto out; + fh_fill_post_noop(fhp); status = fh_compose(resfhp, fhp->fh_export, child, fhp); if (status != nfs_ok) @@ -405,9 +402,6 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, if (!IS_POSIXACL(inode)) iap->ia_mode &= ~current_umask(); - status = fh_fill_pre_attrs(fhp); - if (status != nfs_ok) - goto out; status = nfsd4_vfs_create(fhp, &child, open); if (status != nfs_ok) goto out; @@ -493,6 +487,9 @@ do_open_lookup(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate, stru fh_init(*resfh, NFS4_FHSIZE); open->op_truncate = false; + status = fh_fill_pre_attrs_unlocked(current_fh); + if (status) + goto out; if (open->op_create) { /* FIXME: check session persistence and pnfs flags. * The nfsv4.1 spec requires the following semantics: @@ -524,11 +521,11 @@ do_open_lookup(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate, stru } else { status = nfsd_lookup(rqstp, current_fh, open->op_fname, open->op_fnamelen, *resfh); - if (status == nfs_ok) - /* NFSv4 protocol requires change attributes even though - * no change happened. - */ - status = fh_fill_both_attrs(current_fh); + /* + * NFSv4 protocol requires change attributes even though + * no change happened. + */ + fh_fill_post_noop(current_fh); } if (status) goto out; diff --git a/fs/nfsd/nfsfh.c b/fs/nfsd/nfsfh.c index c7c60c35bdfc0e..c0a46784d525ae 100644 --- a/fs/nfsd/nfsfh.c +++ b/fs/nfsd/nfsfh.c @@ -782,34 +782,53 @@ __be32 fh_getattr(const struct svc_fh *fhp, struct kstat *stat) AT_STATX_SYNC_AS_STAT)); } -/** - * fh_fill_pre_attrs - Fill in pre-op attributes - * @fhp: file handle to be updated - * - */ -__be32 __must_check fh_fill_pre_attrs(struct svc_fh *fhp) +static __be32 __must_check __fh_fill_pre_attrs(struct svc_fh *fhp) { bool v4 = (fhp->fh_maxsize == NFS4_FHSIZE); - struct kstat stat; __be32 err; if (fhp->fh_no_wcc || fhp->fh_pre_saved) return nfs_ok; - err = fh_getattr(fhp, &stat); + err = fh_getattr(fhp, &fhp->fh_post_attr); if (err) return err; if (v4) - fhp->fh_pre_change = nfsd4_change_attribute(&stat); + fhp->fh_pre_change = fhp->fh_post_change = + nfsd4_change_attribute(&fhp->fh_post_attr); - fhp->fh_pre_mtime = stat.mtime; - fhp->fh_pre_ctime = stat.ctime; - fhp->fh_pre_size = stat.size; + fhp->fh_pre_mtime = fhp->fh_post_attr.mtime; + fhp->fh_pre_ctime = fhp->fh_post_attr.ctime; + fhp->fh_pre_size = fhp->fh_post_attr.size; fhp->fh_pre_saved = true; return nfs_ok; } +/** + * fh_fill_pre_attrs - Fill in pre-op attributes + * @fhp: file handle to be updated + * + * Post-op attrs are filled and pre-op attrs are copied + * from there. The post-op attrs can later be replaced by + * fh_fill_post_attrs() or activated by fh_fill_post_noop(). + * + * The inode must be locked. + * + * Returns: error from vfs_getattr() which must be checked. + */ +__be32 __must_check fh_fill_pre_attrs(struct svc_fh *fhp) +{ + lockdep_assert_held_write(&fhp->fh_dentry->d_inode->i_rwsem); + return __fh_fill_pre_attrs(fhp); +} + +__be32 __must_check fh_fill_pre_attrs_unlocked(struct svc_fh *fhp) +{ + fhp->fh_no_atomic_attr = true; + return __fh_fill_pre_attrs(fhp); +} + /** * fh_fill_post_attrs - Fill in post-op attributes * @fhp: file handle to be updated @@ -826,6 +845,9 @@ __be32 fh_fill_post_attrs(struct svc_fh *fhp) if (fhp->fh_post_saved) printk("nfsd: inode locked twice during operation.\n"); + if (!fhp->fh_no_atomic_attr) + lockdep_assert_held_write(&fhp->fh_dentry->d_inode->i_rwsem); + err = fh_getattr(fhp, &fhp->fh_post_attr); if (err) return err; @@ -837,29 +859,6 @@ __be32 fh_fill_post_attrs(struct svc_fh *fhp) return nfs_ok; } -/** - * fh_fill_both_attrs - Fill pre-op and post-op attributes - * @fhp: file handle to be updated - * - * This is used when the directory wasn't changed, but wcc attributes - * are needed anyway. - */ -__be32 __must_check fh_fill_both_attrs(struct svc_fh *fhp) -{ - __be32 err; - - err = fh_fill_post_attrs(fhp); - if (err) - return err; - - fhp->fh_pre_change = fhp->fh_post_change; - fhp->fh_pre_mtime = fhp->fh_post_attr.mtime; - fhp->fh_pre_ctime = fhp->fh_post_attr.ctime; - fhp->fh_pre_size = fhp->fh_post_attr.size; - fhp->fh_pre_saved = true; - return nfs_ok; -} - /* * Release a file handle. */ diff --git a/fs/nfsd/nfsfh.h b/fs/nfsd/nfsfh.h index cdeb5eea65a896..ab15b59ac7b3ba 100644 --- a/fs/nfsd/nfsfh.h +++ b/fs/nfsd/nfsfh.h @@ -337,6 +337,18 @@ static inline void fh_clear_pre_post_attrs(struct svc_fh *fhp) u64 nfsd4_change_attribute(const struct kstat *stat); __be32 __must_check fh_fill_pre_attrs(struct svc_fh *fhp); +__be32 __must_check fh_fill_pre_attrs_unlocked(struct svc_fh *fhp); __be32 fh_fill_post_attrs(struct svc_fh *fhp); -__be32 __must_check fh_fill_both_attrs(struct svc_fh *fhp); + +/** + * fh_fill_post_noop - Copy pre attrs to post attrs + * @fhp: file handle to be updated + * + * This is used when the directory wasn't changed, but wcc attributes + * are needed anyway. + */ +static inline void fh_fill_post_noop(struct svc_fh *fhp) +{ + fhp->fh_post_saved = true; +} #endif /* _LINUX_NFSD_NFSFH_H */ From 05369c7245bdb961d14c186ccf2474d35231e4de Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:27:53 +1000 Subject: [PATCH 0130/1352] nfsd: move fh_want_write() after preamble in nfsd4_create_file() As part of separating the nfsd-specific code from the VFS interaction code in nfsd4_create_file(), move fh_want_write() to just before we need it. Consequently errors in the "if" statement that this code is moved over can now be returned immediately rather than needing to "goto out". Also restructure that "if" statement to only test is_create_with_attrs() once. Reviewed-by: Jeff Layton Signed-off-by: NeilBrown Link: https://patch.msgid.link/20260717093001.1972119-6-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 31 +++++++++++++++++-------------- 1 file changed, 17 insertions(+), 14 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 8e17e95d2cd058..d3c629491ef10c 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -298,22 +298,18 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, dput(child); } - host_err = fh_want_write(fhp); - if (host_err) - return nfserrno(host_err); - - if (open->op_acl) { + if (!is_create_with_attrs(open)) { + /* No attrs to check */ + } else if (open->op_acl) { if (open->op_dpacl || open->op_pacl) { - status = nfserr_inval; - goto out; + /* Cannot specify both NFSv4 and Posix ACLs */ + return nfserr_inval; } - if (is_create_with_attrs(open)) { - status = nfsd4_acl_to_attr(NF4REG, open->op_acl, + status = nfsd4_acl_to_attr(NF4REG, open->op_acl, &attrs); - if (status) - goto out; - } - } else if (is_create_with_attrs(open)) { + if (status) + return status; + } else { /* The dpacl and pacl will get released by nfsd_attrs_free(). */ attrs.na_dpacl = open->op_dpacl; attrs.na_pacl = open->op_pacl; @@ -321,6 +317,12 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, open->op_pacl = NULL; } + host_err = fh_want_write(fhp); + if (host_err) { + status = nfserrno(host_err); + goto out_free; + } + child = start_creating(&nop_mnt_idmap, parent, &QSTR_LEN(open->op_fname, open->op_fnamelen)); if (IS_ERR(child)) { @@ -437,8 +439,9 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, open->op_bmval[2] &= ~FATTR4_WORD2_POSIX_ACCESS_ACL; out: end_creating(child); - nfsd_attrs_free(&attrs); fh_drop_write(fhp); +out_free: + nfsd_attrs_free(&attrs); return status; } From 51478247c019eac1f1810d9cea1e7f5557893b16 Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:27:54 +1000 Subject: [PATCH 0131/1352] nfsd: move more nfs-specific code into preamble of nfsd4_create_file() Do NFS-specific prep before interacting with the VFS. We now add the verifier to iap early so it applies even when an EXCLUSIVE4_1 replay is detected based on that verifier, so we will set those attributes again. This should be harmless even though it will update ctime and i_version, and so will update the changeid seen by the client. It shouldn't matter because the resend implies that the client hasn't seen the file or its changeid. If some other client happens to have noticed the file, it might see an unnecessary changeid up, but that is of no consequence. Note that ctime would have been updated anyway if the client has included other attributes like an ACL. Reviewed-by: Jeff Layton Signed-off-by: NeilBrown Link: https://patch.msgid.link/20260717093001.1972119-7-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 55 +++++++++++++++++++++++----------------------- 1 file changed, 27 insertions(+), 28 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index d3c629491ef10c..9d17cc41c6af13 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -298,6 +298,9 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, dput(child); } + if (!IS_POSIXACL(inode)) + iap->ia_mode &= ~current_umask(); + if (!is_create_with_attrs(open)) { /* No attrs to check */ } else if (open->op_acl) { @@ -317,6 +320,30 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, open->op_pacl = NULL; } + v_mtime = 0; + v_atime = 0; + if (nfsd4_create_is_exclusive(open->op_createmode)) { + u32 *verifier = (u32 *)open->op_verf.data; + + /* + * Solaris 7 gets confused (bugid 4218508) if these have + * the high bit set, as do xfs filesystems without the + * "bigtime" feature. So just clear the high bits. If this + * is ever changed to use different attrs for storing the + * verifier, then do_open_lookup() will also need to be + * fixed accordingly. + */ + v_mtime = verifier[0] & 0x7fffffff; + v_atime = verifier[1] & 0x7fffffff; + + iap->ia_valid |= ATTR_MTIME | ATTR_ATIME | + ATTR_MTIME_SET|ATTR_ATIME_SET; + iap->ia_mtime.tv_sec = v_mtime; + iap->ia_atime.tv_sec = v_atime; + iap->ia_mtime.tv_nsec = 0; + iap->ia_atime.tv_nsec = 0; + } + host_err = fh_want_write(fhp); if (host_err) { status = nfserrno(host_err); @@ -336,23 +363,6 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, goto out; } - v_mtime = 0; - v_atime = 0; - if (nfsd4_create_is_exclusive(open->op_createmode)) { - u32 *verifier = (u32 *)open->op_verf.data; - - /* - * Solaris 7 gets confused (bugid 4218508) if these have - * the high bit set, as do xfs filesystems without the - * "bigtime" feature. So just clear the high bits. If this - * is ever changed to use different attrs for storing the - * verifier, then do_open_lookup() will also need to be - * fixed accordingly. - */ - v_mtime = verifier[0] & 0x7fffffff; - v_atime = verifier[1] & 0x7fffffff; - } - if (d_really_is_positive(child)) { /* NFSv4 protocol requires change attributes even though * no change happened. @@ -401,9 +411,6 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, goto out; } - if (!IS_POSIXACL(inode)) - iap->ia_mode &= ~current_umask(); - status = nfsd4_vfs_create(fhp, &child, open); if (status != nfs_ok) goto out; @@ -417,14 +424,6 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, /* A newly created file already has a file size of zero. */ if ((iap->ia_valid & ATTR_SIZE) && (iap->ia_size == 0)) iap->ia_valid &= ~ATTR_SIZE; - if (nfsd4_create_is_exclusive(open->op_createmode)) { - iap->ia_valid |= ATTR_MTIME | ATTR_ATIME | - ATTR_MTIME_SET|ATTR_ATIME_SET; - iap->ia_mtime.tv_sec = v_mtime; - iap->ia_atime.tv_sec = v_atime; - iap->ia_mtime.tv_nsec = 0; - iap->ia_atime.tv_nsec = 0; - } set_attr: status = nfsd_create_setattr(rqstp, fhp, resfhp, &attrs); From 9dea12369d0d39e1c920d3e44d87cf65a6203aca Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:27:55 +1000 Subject: [PATCH 0132/1352] nfsd: remove subtlety from nfsd4_create_file() nfsd4_create_file() has a switch with cases for NFS4_CREATE_EXCLUSIVE and NFS4_CREATE_EXCLUSIVE4_1 which are identical except for one line which is marked "subtle" in both cases. The difference boils down to a "goto". For the EXCLUSIVE case the target is "out:" which is after a setattr call. For EXCLUSIVE4_1 the target is "set_attr:" which is the start of that setattr call. In the EXCLUSIVE case 'attrs' will only contain the verifier. Setting these again is not harmful as discussed in the previous patch. It will also call commit_metadata(). In performance terms the cost of an extra 'commit' in the rare case of a replaying exclusive create is negligible. So we can safely "goto setattr" in both cases and thus simplify the code. Reviewed-by: Jeff Layton Signed-off-by: NeilBrown Link: https://patch.msgid.link/20260717093001.1972119-8-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 11 ++--------- 1 file changed, 2 insertions(+), 9 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 9d17cc41c6af13..07f5baec14f154 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -391,22 +391,15 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, status = nfserr_exist; break; case NFS4_CREATE_EXCLUSIVE: - if (inode_get_mtime_sec(d_inode(child)) == v_mtime && - inode_get_atime_sec(d_inode(child)) == v_atime && - d_inode(child)->i_size == 0) { - open->op_created = true; - break; /* subtle */ - } - status = nfserr_exist; - break; case NFS4_CREATE_EXCLUSIVE4_1: if (inode_get_mtime_sec(d_inode(child)) == v_mtime && inode_get_atime_sec(d_inode(child)) == v_atime && d_inode(child)->i_size == 0) { open->op_created = true; - goto set_attr; /* subtle */ + goto set_attr; } status = nfserr_exist; + break; } goto out; } From f233c7a0efa6fc6397884a072c8a2593fe73d632 Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:27:56 +1000 Subject: [PATCH 0133/1352] nfsd: in nfsd4_create_file() let VFS report if file was created. nfsd4_create_file() currently assumes that if a lookup failed but then a create succeeds, then the "create" operation actually created the file. With atomic_open this may not be the case - some other actor might have created the file between the lookup and the create. So we move the call to nfsd4_vfs_create() earlier and set ->op_created based on the FMODE_CREATED flag that it set. Then use "! ->op_created" to trigger nfserr_exist handling. The switch statement is split up into two if() statements. First we check for the possibility of a successful exclusive create and set ->op_create to true if appropriate. Then we check for NFS4_CREATE_UNCHECKED to decide if a pre-existing file means an error or success. This allows us to combine the two fh_compose() calls to one place. A subtle difference here is that we now must only pass O_EXCL to dentry_create() for NFS4_CREATE_GUARDED. For the EXCLUSIVE create modes we want a successful open even if the file already exists. We then check the verifier after the open succeeded to see if it was exclusive. The above requires changing dentry_create() to reliably set FMODE_CREATED when the file was actually created. Previously it only sets this flag when atomic_open is used. Signed-off-by: NeilBrown Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717093001.1972119-9-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/namei.c | 2 ++ fs/nfsd/nfs4proc.c | 69 ++++++++++++++++++++-------------------------- 2 files changed, 32 insertions(+), 39 deletions(-) diff --git a/fs/namei.c b/fs/namei.c index 20a6534ea3efff..d95249dd527c1b 100644 --- a/fs/namei.c +++ b/fs/namei.c @@ -5211,6 +5211,8 @@ struct file *dentry_create(struct path *path, int flags, umode_t mode, error = vfs_create(mnt_idmap(path->mnt), path->dentry, mode, NULL); if (!error) error = vfs_open(path, file); + if (!error) + file->f_mode |= FMODE_CREATED; } if (unlikely(error)) return ERR_PTR(error); diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 07f5baec14f154..4e62809d1890a6 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -211,7 +211,11 @@ nfsd4_vfs_create(struct svc_fh *fhp, struct dentry **child, int oflags; oflags = O_CREAT | O_LARGEFILE; - if (nfsd4_create_is_exclusive(open->op_createmode)) + /* + * For the EXCLUSIVE modes we do our own uniqueness tests + * so don't want O_EXCL. + */ + if (open->op_createmode == NFS4_CREATE_GUARDED) oflags |= O_EXCL; switch (open->op_share_access & NFS4_SHARE_ACCESS_BOTH) { @@ -361,22 +365,30 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, status = fh_verify(rqstp, fhp, S_IFDIR, NFSD_MAY_CREATE); if (status != nfs_ok) goto out; - } - if (d_really_is_positive(child)) { - /* NFSv4 protocol requires change attributes even though - * no change happened. - */ - fh_fill_post_noop(fhp); - - status = fh_compose(resfhp, fhp->fh_export, child, fhp); + status = nfsd4_vfs_create(fhp, &child, open); if (status != nfs_ok) goto out; + open->op_created = open->op_filp->f_mode & FMODE_CREATED; + } - switch (open->op_createmode) { - case NFS4_CREATE_UNCHECKED: - if (!d_is_reg(child)) - break; + status = fh_compose(resfhp, fhp->fh_export, child, fhp); + if (status != nfs_ok) + goto out; + + if (!open->op_created && + nfsd4_create_is_exclusive(open->op_createmode) && + inode_get_mtime_sec(d_inode(child)) == v_mtime && + inode_get_atime_sec(d_inode(child)) == v_atime && + d_inode(child)->i_size == 0) + open->op_created = true; + + if (!open->op_created) { + if (open->op_createmode == NFS4_CREATE_UNCHECKED) { + /* NFSv4 protocol requires change attributes + * even though no change happened. + */ + fh_fill_post_noop(fhp); /* * In NFSv4, we don't want to truncate the file @@ -384,41 +396,20 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, * some other reason. Furthermore, if the size is * nonzero, we should ignore it according to spec! */ - open->op_truncate = (iap->ia_valid & ATTR_SIZE) && - !iap->ia_size; - break; - case NFS4_CREATE_GUARDED: - status = nfserr_exist; - break; - case NFS4_CREATE_EXCLUSIVE: - case NFS4_CREATE_EXCLUSIVE4_1: - if (inode_get_mtime_sec(d_inode(child)) == v_mtime && - inode_get_atime_sec(d_inode(child)) == v_atime && - d_inode(child)->i_size == 0) { - open->op_created = true; - goto set_attr; - } + open->op_truncate = (d_is_reg(child) && + (iap->ia_valid & ATTR_SIZE) && + !iap->ia_size); + } else status = nfserr_exist; - break; - } goto out; } - - status = nfsd4_vfs_create(fhp, &child, open); - if (status != nfs_ok) - goto out; - open->op_created = true; + /* file was created */ fh_fill_post_attrs(fhp); - status = fh_compose(resfhp, fhp->fh_export, child, fhp); - if (status != nfs_ok) - goto out; - /* A newly created file already has a file size of zero. */ if ((iap->ia_valid & ATTR_SIZE) && (iap->ia_size == 0)) iap->ia_valid &= ~ATTR_SIZE; -set_attr: status = nfsd_create_setattr(rqstp, fhp, resfhp, &attrs); if (attrs.na_labelerr) From 87c6b26dc0ef2b229af5db5e961eec89c94f7234 Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:27:57 +1000 Subject: [PATCH 0134/1352] nfsd: nfsd4_create_file(): Move NFSD_MAY_CREATE check earlier We only need NFS_MAY_CREATE check if the file doesn't exist, but it is nfsd-specific code as it needs to check NFSEXP_READONLY and I want that to be separate from vfs-specific code, which eventually all be provided by the VFS. So move that check earlier, but hold the error status until needed. The if/else chain here looks a bit clumsy, but it will make a later patch cleaner. Reviewed-by: Jeff Layton Signed-off-by: NeilBrown Link: https://patch.msgid.link/20260717093001.1972119-10-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 21 ++++++++++++--------- 1 file changed, 12 insertions(+), 9 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 4e62809d1890a6..5f43a4a26f3dc2 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -261,7 +261,7 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, struct dentry *parent, *child = ERR_PTR(-EINVAL); __u32 v_mtime, v_atime; struct inode *inode; - __be32 status; + __be32 status, create_status; int host_err; if (name_is_dot_dotdot(open->op_fname, open->op_fnamelen)) @@ -348,6 +348,8 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, iap->ia_atime.tv_nsec = 0; } + create_status = fh_verify(rqstp, fhp, S_IFDIR, NFSD_MAY_CREATE); + host_err = fh_want_write(fhp); if (host_err) { status = nfserrno(host_err); @@ -361,16 +363,17 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, goto out; } - if (d_really_is_negative(child)) { - status = fh_verify(rqstp, fhp, S_IFDIR, NFSD_MAY_CREATE); - if (status != nfs_ok) - goto out; - + if (d_really_is_positive(child)) { + /* No creation needed */ + } else if (create_status) { + status = create_status; + } else { status = nfsd4_vfs_create(fhp, &child, open); - if (status != nfs_ok) - goto out; - open->op_created = open->op_filp->f_mode & FMODE_CREATED; + if (status == nfs_ok) + open->op_created = open->op_filp->f_mode & FMODE_CREATED; } + if (status != nfs_ok) + goto out; status = fh_compose(resfhp, fhp->fh_export, child, fhp); if (status != nfs_ok) From addb6151d6536a5bdf797518b3684182c5e89ea8 Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:27:58 +1000 Subject: [PATCH 0135/1352] nfsd: fh_want_write) failure need not be immediately fatal for nfsd4_create_file() If nfsd4_create_file() is asked to create a file, then failure to get write access to the mount need not be fatal if the file already exists. So we can delay handling the error until it is known if creation was needed, just like with the error from testing for write permission in parent. This is similar to want_write error handling in lookup_open() in fs/namei.c. Note that getting mnt write access to support O_RDWR is handled separately in do_dentry_open(), and op_truncate is handled in do_open_permission(), so nfsd doesn't need to be concerned with these. It only needs to be concerned with creation, and setattr. Signed-off-by: NeilBrown Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717093001.1972119-11-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 15 +++++++-------- 1 file changed, 7 insertions(+), 8 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 5f43a4a26f3dc2..527602698d380c 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -262,7 +262,7 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, __u32 v_mtime, v_atime; struct inode *inode; __be32 status, create_status; - int host_err; + int want_write_err; if (name_is_dot_dotdot(open->op_fname, open->op_fnamelen)) return nfserr_exist; @@ -350,11 +350,10 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, create_status = fh_verify(rqstp, fhp, S_IFDIR, NFSD_MAY_CREATE); - host_err = fh_want_write(fhp); - if (host_err) { - status = nfserrno(host_err); - goto out_free; - } + want_write_err = fh_want_write(fhp); + if (want_write_err) + /* Might still succeed if no create is needed */ + create_status = nfserrno(want_write_err); child = start_creating(&nop_mnt_idmap, parent, &QSTR_LEN(open->op_fname, open->op_fnamelen)); @@ -425,8 +424,8 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, open->op_bmval[2] &= ~FATTR4_WORD2_POSIX_ACCESS_ACL; out: end_creating(child); - fh_drop_write(fhp); -out_free: + if (!want_write_err) + fh_drop_write(fhp); nfsd_attrs_free(&attrs); return status; } From 0afc1941e7349a3bfed0ce510faf1ff73b76f57e Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:27:59 +1000 Subject: [PATCH 0136/1352] nfsd: (almost) always open file in nfsd4_create_file() If the file is found to already exist, open it anyway. This will normally be needed eventually anyway, and providing a consistently valid op_filp will simplify future changes. To simplify this, change nfsd_check_obj_isreg() to take a dentry. This doesn't apply in the case where the file was found in the dcache to be mounted-on. That takes a different path and doesn't require an early open. Signed-off-by: NeilBrown Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717093001.1972119-12-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 39 +++++++++++++++++++++++++++++++++++---- 1 file changed, 35 insertions(+), 4 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 527602698d380c..226993ca761aa5 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -169,9 +169,9 @@ do_open_permission(struct svc_rqst *rqstp, struct svc_fh *current_fh, struct nfs return fh_verify(rqstp, current_fh, S_IFREG, accmode); } -static __be32 nfsd_check_obj_isreg(struct svc_fh *fh, u32 minor_version) +static __be32 nfsd_check_obj_isreg(struct dentry *child, u32 minor_version) { - umode_t mode = d_inode(fh->fh_dentry)->i_mode; + umode_t mode = d_inode(child)->i_mode; if (S_ISREG(mode)) return nfs_ok; @@ -253,6 +253,8 @@ static __be32 nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, struct svc_fh *resfhp, struct nfsd4_open *open) { + struct nfsd4_compoundres *resp = rqstp->rq_resp; + struct nfsd4_compound_state *cstate = &resp->cstate; struct iattr *iap = &open->op_iattr; struct nfsd_attrs attrs = { .na_iattr = iap, @@ -363,7 +365,35 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, } if (d_really_is_positive(child)) { - /* No creation needed */ + /* + * open the file so that we consistently have a valid + * op_filp. + */ + struct path path = {.mnt = fhp->fh_export->ex_path.mnt, + .dentry = child, + }; + unsigned int oflags = O_LARGEFILE; + + switch (open->op_share_access & NFS4_SHARE_ACCESS_BOTH) { + case NFS4_SHARE_ACCESS_WRITE: + oflags |= O_WRONLY; + break; + case NFS4_SHARE_ACCESS_BOTH: + oflags |= O_RDWR; + break; + default: + oflags |= O_RDONLY; + } + + status = nfsd_check_obj_isreg(child, cstate->minorversion); + if (status == nfs_ok) { + open->op_filp = dentry_open(&path, oflags, + current_cred()); + if (IS_ERR(open->op_filp)) { + status = nfserrno(PTR_ERR(open->op_filp)); + open->op_filp = NULL; + } + } } else if (create_status) { status = create_status; } else { @@ -517,7 +547,8 @@ do_open_lookup(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate, stru } if (status) goto out; - status = nfsd_check_obj_isreg(*resfh, cstate->minorversion); + status = nfsd_check_obj_isreg((*resfh)->fh_dentry, + cstate->minorversion); if (status) goto out; From 76fbaacff550e5ee499b04043373a95d9f7d7948 Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:28:00 +1000 Subject: [PATCH 0137/1352] nfsd: reduce range of directory lock in nfsd4_create_file() We only need to hold the lock taken by start_creating() until the create has been attempted. Holding for longer can serve no purpose. The lock is currently held across the setattr call. This might be the intent but it serves no purpose. Holding the lock prevents the name from being removed or renamed, but it doesn't prevent a GETATTR or a racing SETATTR or an OPEN. Calling end_creating() puts the reference to 'child', but we can still use the reference that was stored in open->op_filp. Signed-off-by: NeilBrown Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717093001.1972119-13-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 226993ca761aa5..2ef67dd951be30 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -367,7 +367,7 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, if (d_really_is_positive(child)) { /* * open the file so that we consistently have a valid - * op_filp. + * op_filp and consequently a valid ->f_path.dentry. */ struct path path = {.mnt = fhp->fh_export->ex_path.mnt, .dentry = child, @@ -401,9 +401,12 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, if (status == nfs_ok) open->op_created = open->op_filp->f_mode & FMODE_CREATED; } + end_creating(child); if (status != nfs_ok) goto out; + child = open->op_filp->f_path.dentry; + status = fh_compose(resfhp, fhp->fh_export, child, fhp); if (status != nfs_ok) goto out; @@ -453,7 +456,6 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, if (attrs.na_paclerr) open->op_bmval[2] &= ~FATTR4_WORD2_POSIX_ACCESS_ACL; out: - end_creating(child); if (!want_write_err) fh_drop_write(fhp); nfsd_attrs_free(&attrs); From aa1f61016321fb9ab26d91eda2f1b766d99ee6b1 Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:28:01 +1000 Subject: [PATCH 0138/1352] nfsd: open-code nfsd4_vfs_create() into nfsd4_create_file() Having this sub function separate doesn't really add clarity, and merging allows for some refactoring and ultimately using a different VFS interface. Reviewed-by: Jeff Layton Signed-off-by: NeilBrown Link: https://patch.msgid.link/20260717093001.1972119-14-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 76 +++++++++++++++++++++------------------------- 1 file changed, 34 insertions(+), 42 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 2ef67dd951be30..ee8616c7918575 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -202,46 +202,6 @@ static inline bool nfsd4_create_is_exclusive(int createmode) createmode == NFS4_CREATE_EXCLUSIVE4_1; } -static __be32 -nfsd4_vfs_create(struct svc_fh *fhp, struct dentry **child, - struct nfsd4_open *open) -{ - struct file *filp; - struct path path; - int oflags; - - oflags = O_CREAT | O_LARGEFILE; - /* - * For the EXCLUSIVE modes we do our own uniqueness tests - * so don't want O_EXCL. - */ - if (open->op_createmode == NFS4_CREATE_GUARDED) - oflags |= O_EXCL; - - switch (open->op_share_access & NFS4_SHARE_ACCESS_BOTH) { - case NFS4_SHARE_ACCESS_WRITE: - oflags |= O_WRONLY; - break; - case NFS4_SHARE_ACCESS_BOTH: - oflags |= O_RDWR; - break; - default: - oflags |= O_RDONLY; - } - - path.mnt = fhp->fh_export->ex_path.mnt; - path.dentry = *child; - filp = dentry_create(&path, oflags, open->op_iattr.ia_mode, - current_cred()); - *child = path.dentry; - - if (IS_ERR(filp)) - return nfserrno(PTR_ERR(filp)); - - open->op_filp = filp; - return nfs_ok; -} - /* * Implement NFSv4's unchecked, guarded, and exclusive create * semantics for regular files. Open state for this new file is @@ -397,9 +357,41 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, } else if (create_status) { status = create_status; } else { - status = nfsd4_vfs_create(fhp, &child, open); - if (status == nfs_ok) + struct file *filp; + struct path path; + int oflags; + + oflags = O_CREAT | O_LARGEFILE; + /* + * For the EXCLUSIVE modes we do our own uniqueness tests + * so don't want O_EXCL. + */ + if (open->op_createmode == NFS4_CREATE_GUARDED) + oflags |= O_EXCL; + + switch (open->op_share_access & NFS4_SHARE_ACCESS_BOTH) { + case NFS4_SHARE_ACCESS_WRITE: + oflags |= O_WRONLY; + break; + case NFS4_SHARE_ACCESS_BOTH: + oflags |= O_RDWR; + break; + default: + oflags |= O_RDONLY; + } + + path.mnt = fhp->fh_export->ex_path.mnt; + path.dentry = child; + filp = dentry_create(&path, oflags, open->op_iattr.ia_mode, + current_cred()); + child = path.dentry; + + if (IS_ERR(filp)) { + status = nfserrno(PTR_ERR(filp)); + } else { + open->op_filp = filp; open->op_created = open->op_filp->f_mode & FMODE_CREATED; + } } end_creating(child); if (status != nfs_ok) From 0836b14023c6042014accb650a0de845c9769cd5 Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:28:02 +1000 Subject: [PATCH 0139/1352] nfsd: move some code out of the d_really_is_negative() branch in nfsd4_create_file() The benefit of this code movement isn't immediately obvious, but it will make it easier to switch to using vfs_lookup_open(). One immediate benefit is that common code in the d_is_positive() branch can be discarded. Reviewed-by: Jeff Layton Signed-off-by: NeilBrown Link: https://patch.msgid.link/20260717093001.1972119-15-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 73 ++++++++++++++++++---------------------------- 1 file changed, 28 insertions(+), 45 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index ee8616c7918575..6dff6013a068b4 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -220,7 +220,11 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, .na_iattr = iap, .na_seclabel = &open->op_label, }; + int oflags = O_CREAT | O_LARGEFILE; struct dentry *parent, *child = ERR_PTR(-EINVAL); + struct path path = { + .mnt = fhp->fh_export->ex_path.mnt, + }; __u32 v_mtime, v_atime; struct inode *inode; __be32 status, create_status; @@ -267,6 +271,24 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, if (!IS_POSIXACL(inode)) iap->ia_mode &= ~current_umask(); + /* + * For the EXCLUSIVE modes we do our own uniqueness tests + * so don't want O_EXCL. + */ + if (open->op_createmode == NFS4_CREATE_GUARDED) + oflags |= O_EXCL; + + switch (open->op_share_access & NFS4_SHARE_ACCESS_BOTH) { + case NFS4_SHARE_ACCESS_WRITE: + oflags |= O_WRONLY; + break; + case NFS4_SHARE_ACCESS_BOTH: + oflags |= O_RDWR; + break; + default: + oflags |= O_RDONLY; + } + if (!is_create_with_attrs(open)) { /* No attrs to check */ } else if (open->op_acl) { @@ -323,27 +345,13 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, status = nfserrno(PTR_ERR(child)); goto out; } + path.dentry = child; if (d_really_is_positive(child)) { /* * open the file so that we consistently have a valid * op_filp and consequently a valid ->f_path.dentry. */ - struct path path = {.mnt = fhp->fh_export->ex_path.mnt, - .dentry = child, - }; - unsigned int oflags = O_LARGEFILE; - - switch (open->op_share_access & NFS4_SHARE_ACCESS_BOTH) { - case NFS4_SHARE_ACCESS_WRITE: - oflags |= O_WRONLY; - break; - case NFS4_SHARE_ACCESS_BOTH: - oflags |= O_RDWR; - break; - default: - oflags |= O_RDONLY; - } status = nfsd_check_obj_isreg(child, cstate->minorversion); if (status == nfs_ok) { @@ -357,39 +365,14 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, } else if (create_status) { status = create_status; } else { - struct file *filp; - struct path path; - int oflags; - - oflags = O_CREAT | O_LARGEFILE; - /* - * For the EXCLUSIVE modes we do our own uniqueness tests - * so don't want O_EXCL. - */ - if (open->op_createmode == NFS4_CREATE_GUARDED) - oflags |= O_EXCL; - - switch (open->op_share_access & NFS4_SHARE_ACCESS_BOTH) { - case NFS4_SHARE_ACCESS_WRITE: - oflags |= O_WRONLY; - break; - case NFS4_SHARE_ACCESS_BOTH: - oflags |= O_RDWR; - break; - default: - oflags |= O_RDONLY; - } - - path.mnt = fhp->fh_export->ex_path.mnt; - path.dentry = child; - filp = dentry_create(&path, oflags, open->op_iattr.ia_mode, - current_cred()); + open->op_filp = dentry_create(&path, oflags, open->op_iattr.ia_mode, + current_cred()); child = path.dentry; - if (IS_ERR(filp)) { - status = nfserrno(PTR_ERR(filp)); + if (IS_ERR(open->op_filp)) { + status = nfserrno(PTR_ERR(open->op_filp)); + open->op_filp = NULL; } else { - open->op_filp = filp; open->op_created = open->op_filp->f_mode & FMODE_CREATED; } } From 2480e2592255d3a34a3e785830e390f24852bbd9 Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:28:03 +1000 Subject: [PATCH 0140/1352] nfsd: reduce want-write range in nfsd4_create_file() nfsd4_create_file() needs write access to the mount for two purposes: 1/ to create/open the file. 2/ to set attributes on the newly created (or pre-existing) file. Currently this is all handled by holding the write access across the open and the setattr. A subsequent patch will necessarily change how write access is gained for the open. So we reduce the range for the first want_write, and add another one to cover setattr. If we failed to get write access, it is only fatal if there were attrs to set. We call nfsd_create_setattr() if at all possible, even when no attrs, as it also calls commit_metadata and we need to be certain that the file creation has been synced. If the mount became read-only since the creation happened, we can safely assume that the sync happened as part of that. Signed-off-by: NeilBrown Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717093001.1972119-16-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 17 ++++++++++++++--- 1 file changed, 14 insertions(+), 3 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 6dff6013a068b4..5e047469ba78e9 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -343,6 +343,8 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, &QSTR_LEN(open->op_fname, open->op_fnamelen)); if (IS_ERR(child)) { status = nfserrno(PTR_ERR(child)); + if (!want_write_err) + fh_drop_write(fhp); goto out; } path.dentry = child; @@ -377,6 +379,8 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, } } end_creating(child); + if (!want_write_err) + fh_drop_write(fhp); if (status != nfs_ok) goto out; @@ -420,7 +424,16 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, if ((iap->ia_valid & ATTR_SIZE) && (iap->ia_size == 0)) iap->ia_valid &= ~ATTR_SIZE; - status = nfsd_create_setattr(rqstp, fhp, resfhp, &attrs); + /* We will need write access to set the attrs */ + want_write_err = fh_want_write(fhp); + if (!want_write_err) { + status = nfsd_create_setattr(rqstp, fhp, + resfhp, &attrs); + fh_drop_write(fhp); + } else if (nfsd_attrs_valid(&attrs)) { + /* Needed write access */ + status = nfserrno(want_write_err); + } if (attrs.na_labelerr) open->op_bmval[2] &= ~FATTR4_WORD2_SECURITY_LABEL; @@ -431,8 +444,6 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, if (attrs.na_paclerr) open->op_bmval[2] &= ~FATTR4_WORD2_POSIX_ACCESS_ACL; out: - if (!want_write_err) - fh_drop_write(fhp); nfsd_attrs_free(&attrs); return status; } From 4d2d0de2788679dc3d53fae27300054eda13d9b3 Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:28:04 +1000 Subject: [PATCH 0141/1352] nfsd: move v0 checking out of nfsd_check_obj_isreg() A future patch will use nfsd_check_obj_isreg() in a context where the protocol version is not easily available. So move the version check out and put it at the end of do_open_lookup(). Also change to return errno error code and use nfserrno() to convert to nfs error codes. Use -ELOOP for nfserr_symlink, which is an error indication a problem with symlinks. -EFTYPE is a good match for nfserr_wrong_type. Signed-off-by: NeilBrown Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717093001.1972119-17-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 29 ++++++++++++----------------- fs/nfsd/vfs.c | 4 +++- 2 files changed, 15 insertions(+), 18 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 5e047469ba78e9..7853bc379b9f11 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -169,23 +169,17 @@ do_open_permission(struct svc_rqst *rqstp, struct svc_fh *current_fh, struct nfs return fh_verify(rqstp, current_fh, S_IFREG, accmode); } -static __be32 nfsd_check_obj_isreg(struct dentry *child, u32 minor_version) +static __be32 nfsd_check_obj_isreg(struct dentry *child) { umode_t mode = d_inode(child)->i_mode; if (S_ISREG(mode)) - return nfs_ok; + return 0; if (S_ISDIR(mode)) - return nfserr_isdir; + return -EISDIR; if (S_ISLNK(mode)) - return nfserr_symlink; - - /* RFC 7530 - 16.16.6 */ - if (minor_version == 0) - return nfserr_symlink; - else - return nfserr_wrong_type; - + return -ELOOP; + return -EFTYPE; } static void nfsd4_set_open_owner_reply_cache(struct nfsd4_compound_state *cstate, struct nfsd4_open *open, struct svc_fh *resfh) @@ -213,8 +207,6 @@ static __be32 nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, struct svc_fh *resfhp, struct nfsd4_open *open) { - struct nfsd4_compoundres *resp = rqstp->rq_resp; - struct nfsd4_compound_state *cstate = &resp->cstate; struct iattr *iap = &open->op_iattr; struct nfsd_attrs attrs = { .na_iattr = iap, @@ -355,8 +347,8 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, * op_filp and consequently a valid ->f_path.dentry. */ - status = nfsd_check_obj_isreg(child, cstate->minorversion); - if (status == nfs_ok) { + status = nfserrno(nfsd_check_obj_isreg(child)); + if (!status) { open->op_filp = dentry_open(&path, oflags, current_cred()); if (IS_ERR(open->op_filp)) { @@ -535,8 +527,7 @@ do_open_lookup(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate, stru } if (status) goto out; - status = nfsd_check_obj_isreg((*resfh)->fh_dentry, - cstate->minorversion); + status = nfserrno(nfsd_check_obj_isreg((*resfh)->fh_dentry)); if (status) goto out; @@ -548,6 +539,10 @@ do_open_lookup(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate, stru status = do_open_permission(rqstp, *resfh, open, accmode); set_change_info(&open->op_cinfo, current_fh); out: + if (status == nfserr_wrong_type && cstate->minorversion == 0) + /* RFC 7530 - 16.16.6 */ + return nfserr_symlink; + return status; } diff --git a/fs/nfsd/vfs.c b/fs/nfsd/vfs.c index 7386062ae449aa..f65dad403ee08f 100644 --- a/fs/nfsd/vfs.c +++ b/fs/nfsd/vfs.c @@ -63,7 +63,7 @@ u64 nfsd_io_cache_write __read_mostly = NFSD_IO_BUFFERED; * it's an error we don't expect, log it once and return nfserr_io. */ __be32 -nfserrno (int errno) +nfserrno(int errno) { static struct { __be32 nfserr; @@ -107,6 +107,8 @@ nfserrno (int errno) { nfserr_perm, -ENOKEY }, { nfserr_no_grace, -ENOGRACE}, { nfserr_io, -EBADMSG }, + { nfserr_symlink, -ELOOP }, + { nfserr_wrong_type, -EFTYPE }, }; int i; From 720c9dc0d839553c6a432075f702f69daa683cd0 Mon Sep 17 00:00:00 2001 From: NeilBrown Date: Fri, 17 Jul 2026 19:28:05 +1000 Subject: [PATCH 0142/1352] nfsd: separate out VFS-specific code from nfsd4_create_file() All the code in nfsd4_create_file() that is VFS manipulation, with now NFS-specific knowledge, has been localised. Now we split that out into a separate function: do_lookup_open(). It is planned to provide a vfs_lookup_open() in vfs code which provides this functionality. This will share more code with the syscall open path, and make it easier to modify locking at the VFS level. Signed-off-by: NeilBrown Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717093001.1972119-18-neilb@ownmail.net Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 121 ++++++++++++++++++++++++--------------------- 1 file changed, 66 insertions(+), 55 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 7853bc379b9f11..2d43ff327b8708 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -169,7 +169,7 @@ do_open_permission(struct svc_rqst *rqstp, struct svc_fh *current_fh, struct nfs return fh_verify(rqstp, current_fh, S_IFREG, accmode); } -static __be32 nfsd_check_obj_isreg(struct dentry *child) +static int nfsd_check_obj_isreg(struct dentry *child) { umode_t mode = d_inode(child)->i_mode; @@ -196,6 +196,52 @@ static inline bool nfsd4_create_is_exclusive(int createmode) createmode == NFS4_CREATE_EXCLUSIVE4_1; } +static struct file *do_lookup_open(struct path *parent, + struct qstr *name, + unsigned int oflags, + umode_t mode) +{ + struct file *filp = NULL; + struct path path; + struct dentry *child; + int want_write_err = 0; + + want_write_err = mnt_want_write(parent->mnt); + + child = start_creating(&nop_mnt_idmap, parent->dentry, name); + if (IS_ERR(child)) { + filp = ERR_CAST(child); + goto out; + } + path.mnt = parent->mnt; + path.dentry = child; + + if (d_really_is_positive(child)) { + /* + * open the file so that we consistently have a valid + * op_filp and consequently a valid ->f_path.dentry. + */ + int err = nfsd_check_obj_isreg(child); + + if (err) + filp = ERR_PTR(err); + else + filp = dentry_open(&path, oflags, current_cred()); + } else if (!(oflags & O_CREAT)) { + filp = ERR_PTR(-ENOENT); + } else if (want_write_err) { + filp = ERR_PTR(want_write_err); + } else { + filp = dentry_create(&path, oflags, mode, current_cred()); + child = path.dentry; + } + end_creating(child); +out: + if (!want_write_err) + mnt_drop_write(parent->mnt); + return filp; +} + /* * Implement NFSv4's unchecked, guarded, and exclusive create * semantics for regular files. Open state for this new file is @@ -213,12 +259,12 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, .na_seclabel = &open->op_label, }; int oflags = O_CREAT | O_LARGEFILE; - struct dentry *parent, *child = ERR_PTR(-EINVAL); - struct path path = { + struct dentry *child = ERR_PTR(-EINVAL); + struct path parent = { .mnt = fhp->fh_export->ex_path.mnt, + .dentry = fhp->fh_dentry, }; __u32 v_mtime, v_atime; - struct inode *inode; __be32 status, create_status; int want_write_err; @@ -230,8 +276,6 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, status = fh_verify(rqstp, fhp, S_IFDIR, NFSD_MAY_EXEC); if (status != nfs_ok) return status; - parent = fhp->fh_dentry; - inode = d_inode(parent); if (open->op_createmode == NFS4_CREATE_UNCHECKED) { /* @@ -239,7 +283,7 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, */ child = try_lookup_noperm(&QSTR_LEN(open->op_fname, open->op_fnamelen), - parent); + parent.dentry); if (child && !IS_ERR(child) && d_is_reg(child) && unlikely(nfsd_mountpoint(child, fhp->fh_export))) { struct svc_export *exp = exp_get(fhp->fh_export); @@ -260,7 +304,7 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, dput(child); } - if (!IS_POSIXACL(inode)) + if (!IS_POSIXACL(d_inode(parent.dentry))) iap->ia_mode &= ~current_umask(); /* @@ -325,58 +369,25 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, } create_status = fh_verify(rqstp, fhp, S_IFDIR, NFSD_MAY_CREATE); - - want_write_err = fh_want_write(fhp); - if (want_write_err) + if (create_status) /* Might still succeed if no create is needed */ - create_status = nfserrno(want_write_err); - - child = start_creating(&nop_mnt_idmap, parent, - &QSTR_LEN(open->op_fname, open->op_fnamelen)); - if (IS_ERR(child)) { - status = nfserrno(PTR_ERR(child)); - if (!want_write_err) - fh_drop_write(fhp); + oflags &= ~O_CREAT; + + open->op_filp = do_lookup_open(&parent, + &QSTR_LEN(open->op_fname, + open->op_fnamelen), + oflags, + open->op_iattr.ia_mode); + if (IS_ERR(open->op_filp)) { + status = nfserrno(PTR_ERR(open->op_filp)); + open->op_filp = NULL; + if (status == nfserr_noent && create_status) + status = create_status; goto out; } - path.dentry = child; - - if (d_really_is_positive(child)) { - /* - * open the file so that we consistently have a valid - * op_filp and consequently a valid ->f_path.dentry. - */ - - status = nfserrno(nfsd_check_obj_isreg(child)); - if (!status) { - open->op_filp = dentry_open(&path, oflags, - current_cred()); - if (IS_ERR(open->op_filp)) { - status = nfserrno(PTR_ERR(open->op_filp)); - open->op_filp = NULL; - } - } - } else if (create_status) { - status = create_status; - } else { - open->op_filp = dentry_create(&path, oflags, open->op_iattr.ia_mode, - current_cred()); - child = path.dentry; - - if (IS_ERR(open->op_filp)) { - status = nfserrno(PTR_ERR(open->op_filp)); - open->op_filp = NULL; - } else { - open->op_created = open->op_filp->f_mode & FMODE_CREATED; - } - } - end_creating(child); - if (!want_write_err) - fh_drop_write(fhp); - if (status != nfs_ok) - goto out; child = open->op_filp->f_path.dentry; + open->op_created = open->op_filp->f_mode & FMODE_CREATED; status = fh_compose(resfhp, fhp->fh_export, child, fhp); if (status != nfs_ok) From d1c7c417d2894cc0fcfbfe0540efc48bb4eeed40 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Fri, 17 Jul 2026 14:41:07 -0400 Subject: [PATCH 0143/1352] NFSD: Move XDR encoding helpers out of xdr4.h These static inline helpers use the nfserr_resource macro, which pulls in the whole rack of NFS status codes. Move those helpers into the only file that uses them, to get rid of the nfserr macro dependency globally. These helper were originally placed in xdr4.h because I thought they would be utilized in the rest of the NFSv4 XDR code, but XDR translation is eventually to be subsumed by xdrgen instead. I'm not converting them now because that is much more churn than this patch is. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717184112.507548-2-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/blocklayoutxdr.c | 10 +++ fs/nfsd/nfs4xdr.c | 92 ++++++++++++++++++++++++++ fs/nfsd/xdr4.h | 139 --------------------------------------- 3 files changed, 102 insertions(+), 139 deletions(-) diff --git a/fs/nfsd/blocklayoutxdr.c b/fs/nfsd/blocklayoutxdr.c index f80dbc41fd5f01..51cc2a07d7b384 100644 --- a/fs/nfsd/blocklayoutxdr.c +++ b/fs/nfsd/blocklayoutxdr.c @@ -13,6 +13,16 @@ #define NFSDDBG_FACILITY NFSDDBG_PNFS +static __be32 +nfsd4_decode_deviceid4(struct xdr_stream *xdr, struct nfsd4_deviceid *devid) +{ + __be32 *p = xdr_inline_decode(xdr, NFS4_DEVICEID4_SIZE); + + if (unlikely(!p)) + return nfserr_bad_xdr; + svcxdr_decode_deviceid4(p, devid); + return nfs_ok; +} /** * nfsd4_block_encode_layoutget - encode block/scsi layout extent array diff --git a/fs/nfsd/nfs4xdr.c b/fs/nfsd/nfs4xdr.c index 04755c41d87115..05d64733094b2a 100644 --- a/fs/nfsd/nfs4xdr.c +++ b/fs/nfsd/nfs4xdr.c @@ -1919,6 +1919,17 @@ nfsd4_decode_get_dir_delegation(struct nfsd4_compoundargs *argp, } #ifdef CONFIG_NFSD_PNFS +static __be32 +nfsd4_decode_deviceid4(struct xdr_stream *xdr, struct nfsd4_deviceid *devid) +{ + __be32 *p = xdr_inline_decode(xdr, NFS4_DEVICEID4_SIZE); + + if (unlikely(!p)) + return nfserr_bad_xdr; + svcxdr_decode_deviceid4(p, devid); + return nfs_ok; +} + static __be32 nfsd4_decode_getdeviceinfo(struct nfsd4_compoundargs *argp, union nfsd4_op_u *u) @@ -2733,6 +2744,87 @@ nfsd4_decode_compound(struct nfsd4_compoundargs *argp) return true; } +static __always_inline __be32 +nfsd4_encode_bool(struct xdr_stream *xdr, bool val) +{ + __be32 *p = xdr_reserve_space(xdr, XDR_UNIT); + + if (unlikely(p == NULL)) + return nfserr_resource; + *p = val ? xdr_one : xdr_zero; + return nfs_ok; +} + +static __always_inline __be32 +nfsd4_encode_uint32_t(struct xdr_stream *xdr, u32 val) +{ + __be32 *p = xdr_reserve_space(xdr, XDR_UNIT); + + if (unlikely(p == NULL)) + return nfserr_resource; + *p = cpu_to_be32(val); + return nfs_ok; +} + +#define nfsd4_encode_aceflag4(x, v) nfsd4_encode_uint32_t(x, v) +#define nfsd4_encode_acemask4(x, v) nfsd4_encode_uint32_t(x, v) +#define nfsd4_encode_acetype4(x, v) nfsd4_encode_uint32_t(x, v) +#define nfsd4_encode_count4(x, v) nfsd4_encode_uint32_t(x, v) +#define nfsd4_encode_mode4(x, v) nfsd4_encode_uint32_t(x, v) +#define nfsd4_encode_nfs_lease4(x, v) nfsd4_encode_uint32_t(x, v) +#define nfsd4_encode_qop4(x, v) nfsd4_encode_uint32_t(x, v) +#define nfsd4_encode_sequenceid4(x, v) nfsd4_encode_uint32_t(x, v) +#define nfsd4_encode_slotid4(x, v) nfsd4_encode_uint32_t(x, v) + +static __always_inline __be32 +nfsd4_encode_uint64_t(struct xdr_stream *xdr, u64 val) +{ + __be32 *p = xdr_reserve_space(xdr, XDR_UNIT * 2); + + if (unlikely(p == NULL)) + return nfserr_resource; + put_unaligned_be64(val, p); + return nfs_ok; +} + +#define nfsd4_encode_changeid4(x, v) nfsd4_encode_uint64_t(x, v) +#define nfsd4_encode_nfs_cookie4(x, v) nfsd4_encode_uint64_t(x, v) +#define nfsd4_encode_length4(x, v) nfsd4_encode_uint64_t(x, v) +#define nfsd4_encode_offset4(x, v) nfsd4_encode_uint64_t(x, v) + +static __always_inline __be32 +nfsd4_encode_opaque_fixed(struct xdr_stream *xdr, const void *data, + size_t size) +{ + __be32 *p = xdr_reserve_space(xdr, xdr_align_size(size)); + size_t pad = xdr_pad_size(size); + + if (unlikely(p == NULL)) + return nfserr_resource; + memcpy(p, data, size); + if (pad) + memset((char *)p + size, 0, pad); + return nfs_ok; +} + +static __always_inline __be32 +nfsd4_encode_opaque(struct xdr_stream *xdr, const void *data, size_t size) +{ + size_t pad = xdr_pad_size(size); + __be32 *p; + + p = xdr_reserve_space(xdr, XDR_UNIT + xdr_align_size(size)); + if (unlikely(p == NULL)) + return nfserr_resource; + *p++ = cpu_to_be32(size); + memcpy(p, data, size); + if (pad) + memset((char *)p + size, 0, pad); + return nfs_ok; +} + +#define nfsd4_encode_component4(x, d, s) nfsd4_encode_opaque(x, d, s) + static __be32 nfsd4_encode_nfs_fh4(struct xdr_stream *xdr, const struct knfsd_fh *fh_handle) { diff --git a/fs/nfsd/xdr4.h b/fs/nfsd/xdr4.h index c7eda5bc833b1a..e833407859c8d8 100644 --- a/fs/nfsd/xdr4.h +++ b/fs/nfsd/xdr4.h @@ -50,134 +50,6 @@ #define HAS_CSTATE_FLAG(c, f) ((c)->sid_flags & (f)) #define CLEAR_CSTATE_FLAG(c, f) ((c)->sid_flags &= ~(f)) -/** - * nfsd4_encode_bool - Encode an XDR bool type result - * @xdr: target XDR stream - * @val: boolean value to encode - * - * Return values: - * %nfs_ok: @val encoded; @xdr advanced to next position - * %nfserr_resource: stream buffer space exhausted - */ -static __always_inline __be32 -nfsd4_encode_bool(struct xdr_stream *xdr, bool val) -{ - __be32 *p = xdr_reserve_space(xdr, XDR_UNIT); - - if (unlikely(p == NULL)) - return nfserr_resource; - *p = val ? xdr_one : xdr_zero; - return nfs_ok; -} - -/** - * nfsd4_encode_uint32_t - Encode an XDR uint32_t type result - * @xdr: target XDR stream - * @val: integer value to encode - * - * Return values: - * %nfs_ok: @val encoded; @xdr advanced to next position - * %nfserr_resource: stream buffer space exhausted - */ -static __always_inline __be32 -nfsd4_encode_uint32_t(struct xdr_stream *xdr, u32 val) -{ - __be32 *p = xdr_reserve_space(xdr, XDR_UNIT); - - if (unlikely(p == NULL)) - return nfserr_resource; - *p = cpu_to_be32(val); - return nfs_ok; -} - -#define nfsd4_encode_aceflag4(x, v) nfsd4_encode_uint32_t(x, v) -#define nfsd4_encode_acemask4(x, v) nfsd4_encode_uint32_t(x, v) -#define nfsd4_encode_acetype4(x, v) nfsd4_encode_uint32_t(x, v) -#define nfsd4_encode_count4(x, v) nfsd4_encode_uint32_t(x, v) -#define nfsd4_encode_mode4(x, v) nfsd4_encode_uint32_t(x, v) -#define nfsd4_encode_nfs_lease4(x, v) nfsd4_encode_uint32_t(x, v) -#define nfsd4_encode_qop4(x, v) nfsd4_encode_uint32_t(x, v) -#define nfsd4_encode_sequenceid4(x, v) nfsd4_encode_uint32_t(x, v) -#define nfsd4_encode_slotid4(x, v) nfsd4_encode_uint32_t(x, v) - -/** - * nfsd4_encode_uint64_t - Encode an XDR uint64_t type result - * @xdr: target XDR stream - * @val: integer value to encode - * - * Return values: - * %nfs_ok: @val encoded; @xdr advanced to next position - * %nfserr_resource: stream buffer space exhausted - */ -static __always_inline __be32 -nfsd4_encode_uint64_t(struct xdr_stream *xdr, u64 val) -{ - __be32 *p = xdr_reserve_space(xdr, XDR_UNIT * 2); - - if (unlikely(p == NULL)) - return nfserr_resource; - put_unaligned_be64(val, p); - return nfs_ok; -} - -#define nfsd4_encode_changeid4(x, v) nfsd4_encode_uint64_t(x, v) -#define nfsd4_encode_nfs_cookie4(x, v) nfsd4_encode_uint64_t(x, v) -#define nfsd4_encode_length4(x, v) nfsd4_encode_uint64_t(x, v) -#define nfsd4_encode_offset4(x, v) nfsd4_encode_uint64_t(x, v) - -/** - * nfsd4_encode_opaque_fixed - Encode a fixed-length XDR opaque type result - * @xdr: target XDR stream - * @data: pointer to data - * @size: length of data in bytes - * - * Return values: - * %nfs_ok: @data encoded; @xdr advanced to next position - * %nfserr_resource: stream buffer space exhausted - */ -static __always_inline __be32 -nfsd4_encode_opaque_fixed(struct xdr_stream *xdr, const void *data, - size_t size) -{ - __be32 *p = xdr_reserve_space(xdr, xdr_align_size(size)); - size_t pad = xdr_pad_size(size); - - if (unlikely(p == NULL)) - return nfserr_resource; - memcpy(p, data, size); - if (pad) - memset((char *)p + size, 0, pad); - return nfs_ok; -} - -/** - * nfsd4_encode_opaque - Encode a variable-length XDR opaque type result - * @xdr: target XDR stream - * @data: pointer to data - * @size: length of data in bytes - * - * Return values: - * %nfs_ok: @data encoded; @xdr advanced to next position - * %nfserr_resource: stream buffer space exhausted - */ -static __always_inline __be32 -nfsd4_encode_opaque(struct xdr_stream *xdr, const void *data, size_t size) -{ - size_t pad = xdr_pad_size(size); - __be32 *p; - - p = xdr_reserve_space(xdr, XDR_UNIT + xdr_align_size(size)); - if (unlikely(p == NULL)) - return nfserr_resource; - *p++ = cpu_to_be32(size); - memcpy(p, data, size); - if (pad) - memset((char *)p + size, 0, pad); - return nfs_ok; -} - -#define nfsd4_encode_component4(x, d, s) nfsd4_encode_opaque(x, d, s) - struct nfsd4_compound_state { struct svc_fh current_fh; struct svc_fh save_fh; @@ -642,17 +514,6 @@ svcxdr_decode_deviceid4(__be32 *p, struct nfsd4_deviceid *devid) return p; } -static inline __be32 -nfsd4_decode_deviceid4(struct xdr_stream *xdr, struct nfsd4_deviceid *devid) -{ - __be32 *p = xdr_inline_decode(xdr, NFS4_DEVICEID4_SIZE); - - if (unlikely(!p)) - return nfserr_bad_xdr; - svcxdr_decode_deviceid4(p, devid); - return nfs_ok; -} - struct nfsd4_layout_seg { u32 iomode; u64 offset; From 3b09fb8e2315e1c90d6edfecdde79df7cada5a41 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Fri, 17 Jul 2026 14:41:08 -0400 Subject: [PATCH 0144/1352] NFSD: Move pre-xdr'ed status codes out of nfsd.h nfsd.h is included by nearly every NFSD translation unit, so its include of reaches all of them, whether or not they touch NFSv4. That include existed solely for the block of pre-xdr'ed nfserr_* values at the end of the file: several of those values, such as nfserr_delay and nfserr_admin_revoked, are built from NFS4ERR_* constants defined in nfs4.h. The NFSD-internal error enum that follows the block (NFSERR_EOF and friends) is anchored at an impossible nfsstat4 value, thus it also needs nothing from nfs4.h. But these codes are used by all NFS versions, so their new home must be version-neutral. Move the pre-xdr'ed value block and the internal error enum into a new fs/nfsd/nfserr.h, which includes nfs4.h itself, and drop the nfs4.h include from nfsd.h. Include nfserr.h directly from each translation unit that references the pre-xdr'ed values or the internal error codes, rather than carrying it in a widely-included header. A translation unit that includes nfsd.h without using the error block no longer pulls in nfs4.h. The ones that reference the block can still reach it through nfserr.h. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717184112.507548-3-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/blocklayout.c | 1 + fs/nfsd/blocklayoutxdr.c | 1 + fs/nfsd/export.c | 1 + fs/nfsd/filecache.c | 1 + fs/nfsd/flexfilelayout.c | 1 + fs/nfsd/flexfilelayoutxdr.c | 1 + fs/nfsd/lockd.c | 1 + fs/nfsd/nfs2acl.c | 1 + fs/nfsd/nfs3acl.c | 1 + fs/nfsd/nfs3proc.c | 1 + fs/nfsd/nfs3xdr.c | 1 + fs/nfsd/nfs4acl.c | 1 + fs/nfsd/nfs4callback.c | 1 + fs/nfsd/nfs4idmap.c | 1 + fs/nfsd/nfs4layouts.c | 1 + fs/nfsd/nfs4proc.c | 1 + fs/nfsd/nfs4state.c | 1 + fs/nfsd/nfs4xdr.c | 1 + fs/nfsd/nfscache.c | 1 + fs/nfsd/nfsctl.c | 1 + fs/nfsd/nfsd.h | 143 -------------------------------- fs/nfsd/nfserr.h | 158 ++++++++++++++++++++++++++++++++++++ fs/nfsd/nfsfh.c | 1 + fs/nfsd/nfsproc.c | 1 + fs/nfsd/nfssvc.c | 2 + fs/nfsd/nfsxdr.c | 1 + fs/nfsd/vfs.c | 1 + 27 files changed, 184 insertions(+), 143 deletions(-) create mode 100644 fs/nfsd/nfserr.h diff --git a/fs/nfsd/blocklayout.c b/fs/nfsd/blocklayout.c index 5be7721c22c235..df02cf7464799c 100644 --- a/fs/nfsd/blocklayout.c +++ b/fs/nfsd/blocklayout.c @@ -9,6 +9,7 @@ #include +#include "nfserr.h" #include "blocklayoutxdr.h" #include "pnfs.h" #include "filecache.h" diff --git a/fs/nfsd/blocklayoutxdr.c b/fs/nfsd/blocklayoutxdr.c index 51cc2a07d7b384..a6589f5c878aca 100644 --- a/fs/nfsd/blocklayoutxdr.c +++ b/fs/nfsd/blocklayoutxdr.c @@ -8,6 +8,7 @@ #include #include "nfsd.h" +#include "nfserr.h" #include "blocklayoutxdr.h" #include "vfs.h" diff --git a/fs/nfsd/export.c b/fs/nfsd/export.c index 76cde579372141..23192eb7094fa7 100644 --- a/fs/nfsd/export.c +++ b/fs/nfsd/export.c @@ -21,6 +21,7 @@ #include #include "nfsd.h" +#include "nfserr.h" #include "nfsfh.h" #include "netns.h" #include "pnfs.h" diff --git a/fs/nfsd/filecache.c b/fs/nfsd/filecache.c index b9548eb17c77de..3539149cc75f77 100644 --- a/fs/nfsd/filecache.c +++ b/fs/nfsd/filecache.c @@ -43,6 +43,7 @@ #include "vfs.h" #include "nfsd.h" +#include "nfserr.h" #include "nfsfh.h" #include "netns.h" #include "filecache.h" diff --git a/fs/nfsd/flexfilelayout.c b/fs/nfsd/flexfilelayout.c index 6d531285ab439e..9f532418cac8ae 100644 --- a/fs/nfsd/flexfilelayout.c +++ b/fs/nfsd/flexfilelayout.c @@ -13,6 +13,7 @@ #include +#include "nfserr.h" #include "flexfilelayoutxdr.h" #include "pnfs.h" #include "vfs.h" diff --git a/fs/nfsd/flexfilelayoutxdr.c b/fs/nfsd/flexfilelayoutxdr.c index 374e52d3064a65..97d8a28dd3a04f 100644 --- a/fs/nfsd/flexfilelayoutxdr.c +++ b/fs/nfsd/flexfilelayoutxdr.c @@ -6,6 +6,7 @@ #include #include "nfsd.h" +#include "nfserr.h" #include "flexfilelayoutxdr.h" #define NFSDDBG_FACILITY NFSDDBG_PNFS diff --git a/fs/nfsd/lockd.c b/fs/nfsd/lockd.c index 72a5b499839d81..f5a4f352f8abf9 100644 --- a/fs/nfsd/lockd.c +++ b/fs/nfsd/lockd.c @@ -10,6 +10,7 @@ #include #include #include "nfsd.h" +#include "nfserr.h" #include "vfs.h" #define NFSDDBG_FACILITY NFSDDBG_LOCKD diff --git a/fs/nfsd/nfs2acl.c b/fs/nfsd/nfs2acl.c index 190f5a00190091..aba69dd278a136 100644 --- a/fs/nfsd/nfs2acl.c +++ b/fs/nfsd/nfs2acl.c @@ -6,6 +6,7 @@ */ #include "nfsd.h" +#include "nfserr.h" /* FIXME: nfsacl.h is a broken header */ #include #include diff --git a/fs/nfsd/nfs3acl.c b/fs/nfsd/nfs3acl.c index 6b6b289db63613..7183995182ab48 100644 --- a/fs/nfsd/nfs3acl.c +++ b/fs/nfsd/nfs3acl.c @@ -6,6 +6,7 @@ */ #include "nfsd.h" +#include "nfserr.h" /* FIXME: nfsacl.h is a broken header */ #include #include diff --git a/fs/nfsd/nfs3proc.c b/fs/nfsd/nfs3proc.c index 1df3c719e0da6c..4b3075c05b9793 100644 --- a/fs/nfsd/nfs3proc.c +++ b/fs/nfsd/nfs3proc.c @@ -13,6 +13,7 @@ #include "cache.h" #include "xdr3.h" #include "vfs.h" +#include "nfserr.h" #include "filecache.h" #include "trace.h" diff --git a/fs/nfsd/nfs3xdr.c b/fs/nfsd/nfs3xdr.c index e481804bb120c0..196bcc6edebb98 100644 --- a/fs/nfsd/nfs3xdr.c +++ b/fs/nfsd/nfs3xdr.c @@ -13,6 +13,7 @@ #include "auth.h" #include "netns.h" #include "vfs.h" +#include "nfserr.h" /* * Force construction of an empty post-op attr diff --git a/fs/nfsd/nfs4acl.c b/fs/nfsd/nfs4acl.c index 2c2f2fd89e8795..94f6ad381ebe5a 100644 --- a/fs/nfsd/nfs4acl.c +++ b/fs/nfsd/nfs4acl.c @@ -40,6 +40,7 @@ #include "nfsfh.h" #include "nfsd.h" +#include "nfserr.h" #include "acl.h" #include "vfs.h" diff --git a/fs/nfsd/nfs4callback.c b/fs/nfsd/nfs4callback.c index 19dc337502ca86..2939b5c6a5feac 100644 --- a/fs/nfsd/nfs4callback.c +++ b/fs/nfsd/nfs4callback.c @@ -37,6 +37,7 @@ #include #include #include "nfsd.h" +#include "nfserr.h" #include "state.h" #include "netns.h" #include "stats.h" diff --git a/fs/nfsd/nfs4idmap.c b/fs/nfsd/nfs4idmap.c index e9faf8b78f74a4..4e529759396342 100644 --- a/fs/nfsd/nfs4idmap.c +++ b/fs/nfsd/nfs4idmap.c @@ -41,6 +41,7 @@ #include "auth.h" #include "idmap.h" #include "nfsd.h" +#include "nfserr.h" #include "netns.h" #include "vfs.h" diff --git a/fs/nfsd/nfs4layouts.c b/fs/nfsd/nfs4layouts.c index 22bcb6d09f7037..4187202f9acca5 100644 --- a/fs/nfsd/nfs4layouts.c +++ b/fs/nfsd/nfs4layouts.c @@ -9,6 +9,7 @@ #include #include +#include "nfserr.h" #include "pnfs.h" #include "netns.h" #include "trace.h" diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 2d43ff327b8708..0bbf781d4ac591 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -51,6 +51,7 @@ #include "netns.h" #include "acl.h" #include "pnfs.h" +#include "nfserr.h" #include "trace.h" static bool inter_copy_offload_enable; diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c index 9c4adf3110aea4..47909a34d52b68 100644 --- a/fs/nfsd/nfs4state.c +++ b/fs/nfsd/nfs4state.c @@ -52,6 +52,7 @@ #include "vfs.h" #include "current_stateid.h" #include "stats.h" +#include "nfserr.h" #include "netns.h" #include "pnfs.h" diff --git a/fs/nfsd/nfs4xdr.c b/fs/nfsd/nfs4xdr.c index 05d64733094b2a..7ccc7c897b00cb 100644 --- a/fs/nfsd/nfs4xdr.c +++ b/fs/nfsd/nfs4xdr.c @@ -54,6 +54,7 @@ #include "xdr4.h" #include "vfs.h" #include "state.h" +#include "nfserr.h" #include "cache.h" #include "netns.h" #include "pnfs.h" diff --git a/fs/nfsd/nfscache.c b/fs/nfsd/nfscache.c index c7db532c852376..80364b91331a2a 100644 --- a/fs/nfsd/nfscache.c +++ b/fs/nfsd/nfscache.c @@ -19,6 +19,7 @@ #include #include "nfsd.h" +#include "nfserr.h" #include "netns.h" #include "stats.h" #include "cache.h" diff --git a/fs/nfsd/nfsctl.c b/fs/nfsd/nfsctl.c index 5abb2d4274c98e..4492856c76b4fe 100644 --- a/fs/nfsd/nfsctl.c +++ b/fs/nfsd/nfsctl.c @@ -23,6 +23,7 @@ #include "idmap.h" #include "nfsd.h" +#include "nfserr.h" #include "netns.h" #include "stats.h" #include "cache.h" diff --git a/fs/nfsd/nfsd.h b/fs/nfsd/nfsd.h index 76a69d9a4e7332..384a2498b6a295 100644 --- a/fs/nfsd/nfsd.h +++ b/fs/nfsd/nfsd.h @@ -15,7 +15,6 @@ #include #include #include -#include #include #include @@ -198,148 +197,6 @@ void nfsd_lockd_init(void); void nfsd_lockd_shutdown(void); -/* - * These macros provide pre-xdr'ed values for faster operation. - */ -#define nfs_ok cpu_to_be32(NFS_OK) -#define nfserr_perm cpu_to_be32(NFSERR_PERM) -#define nfserr_noent cpu_to_be32(NFSERR_NOENT) -#define nfserr_io cpu_to_be32(NFSERR_IO) -#define nfserr_nxio cpu_to_be32(NFSERR_NXIO) -#define nfserr_acces cpu_to_be32(NFSERR_ACCES) -#define nfserr_exist cpu_to_be32(NFSERR_EXIST) -#define nfserr_xdev cpu_to_be32(NFSERR_XDEV) -#define nfserr_nodev cpu_to_be32(NFSERR_NODEV) -#define nfserr_notdir cpu_to_be32(NFSERR_NOTDIR) -#define nfserr_isdir cpu_to_be32(NFSERR_ISDIR) -#define nfserr_inval cpu_to_be32(NFSERR_INVAL) -#define nfserr_fbig cpu_to_be32(NFSERR_FBIG) -#define nfserr_nospc cpu_to_be32(NFSERR_NOSPC) -#define nfserr_rofs cpu_to_be32(NFSERR_ROFS) -#define nfserr_mlink cpu_to_be32(NFSERR_MLINK) -#define nfserr_nametoolong cpu_to_be32(NFSERR_NAMETOOLONG) -#define nfserr_notempty cpu_to_be32(NFSERR_NOTEMPTY) -#define nfserr_dquot cpu_to_be32(NFSERR_DQUOT) -#define nfserr_stale cpu_to_be32(NFSERR_STALE) -#define nfserr_remote cpu_to_be32(NFSERR_REMOTE) -#define nfserr_wflush cpu_to_be32(NFSERR_WFLUSH) -#define nfserr_badhandle cpu_to_be32(NFSERR_BADHANDLE) -#define nfserr_notsync cpu_to_be32(NFSERR_NOT_SYNC) -#define nfserr_badcookie cpu_to_be32(NFSERR_BAD_COOKIE) -#define nfserr_notsupp cpu_to_be32(NFSERR_NOTSUPP) -#define nfserr_toosmall cpu_to_be32(NFSERR_TOOSMALL) -#define nfserr_serverfault cpu_to_be32(NFSERR_SERVERFAULT) -#define nfserr_badtype cpu_to_be32(NFSERR_BADTYPE) -#define nfserr_jukebox cpu_to_be32(NFSERR_JUKEBOX) -#define nfserr_denied cpu_to_be32(NFSERR_DENIED) -#define nfserr_deadlock cpu_to_be32(NFSERR_DEADLOCK) -#define nfserr_expired cpu_to_be32(NFSERR_EXPIRED) -#define nfserr_bad_cookie cpu_to_be32(NFSERR_BAD_COOKIE) -#define nfserr_same cpu_to_be32(NFSERR_SAME) -#define nfserr_clid_inuse cpu_to_be32(NFSERR_CLID_INUSE) -#define nfserr_stale_clientid cpu_to_be32(NFSERR_STALE_CLIENTID) -#define nfserr_resource cpu_to_be32(NFSERR_RESOURCE) -#define nfserr_moved cpu_to_be32(NFSERR_MOVED) -#define nfserr_nofilehandle cpu_to_be32(NFSERR_NOFILEHANDLE) -#define nfserr_minor_vers_mismatch cpu_to_be32(NFSERR_MINOR_VERS_MISMATCH) -#define nfserr_share_denied cpu_to_be32(NFSERR_SHARE_DENIED) -#define nfserr_stale_stateid cpu_to_be32(NFSERR_STALE_STATEID) -#define nfserr_old_stateid cpu_to_be32(NFSERR_OLD_STATEID) -#define nfserr_bad_stateid cpu_to_be32(NFSERR_BAD_STATEID) -#define nfserr_bad_seqid cpu_to_be32(NFSERR_BAD_SEQID) -#define nfserr_symlink cpu_to_be32(NFSERR_SYMLINK) -#define nfserr_not_same cpu_to_be32(NFSERR_NOT_SAME) -#define nfserr_lock_range cpu_to_be32(NFSERR_LOCK_RANGE) -#define nfserr_restorefh cpu_to_be32(NFSERR_RESTOREFH) -#define nfserr_attrnotsupp cpu_to_be32(NFSERR_ATTRNOTSUPP) -#define nfserr_bad_xdr cpu_to_be32(NFSERR_BAD_XDR) -#define nfserr_openmode cpu_to_be32(NFSERR_OPENMODE) -#define nfserr_badowner cpu_to_be32(NFSERR_BADOWNER) -#define nfserr_locks_held cpu_to_be32(NFSERR_LOCKS_HELD) -#define nfserr_op_illegal cpu_to_be32(NFSERR_OP_ILLEGAL) -#define nfserr_grace cpu_to_be32(NFSERR_GRACE) -#define nfserr_no_grace cpu_to_be32(NFSERR_NO_GRACE) -#define nfserr_reclaim_bad cpu_to_be32(NFSERR_RECLAIM_BAD) -#define nfserr_badname cpu_to_be32(NFSERR_BADNAME) -#define nfserr_admin_revoked cpu_to_be32(NFS4ERR_ADMIN_REVOKED) -#define nfserr_cb_path_down cpu_to_be32(NFSERR_CB_PATH_DOWN) -#define nfserr_locked cpu_to_be32(NFSERR_LOCKED) -#define nfserr_wrongsec cpu_to_be32(NFSERR_WRONGSEC) -#define nfserr_delay cpu_to_be32(NFS4ERR_DELAY) -#define nfserr_badiomode cpu_to_be32(NFS4ERR_BADIOMODE) -#define nfserr_badlayout cpu_to_be32(NFS4ERR_BADLAYOUT) -#define nfserr_bad_session_digest cpu_to_be32(NFS4ERR_BAD_SESSION_DIGEST) -#define nfserr_badsession cpu_to_be32(NFS4ERR_BADSESSION) -#define nfserr_badslot cpu_to_be32(NFS4ERR_BADSLOT) -#define nfserr_complete_already cpu_to_be32(NFS4ERR_COMPLETE_ALREADY) -#define nfserr_conn_not_bound_to_session cpu_to_be32(NFS4ERR_CONN_NOT_BOUND_TO_SESSION) -#define nfserr_deleg_already_wanted cpu_to_be32(NFS4ERR_DELEG_ALREADY_WANTED) -#define nfserr_back_chan_busy cpu_to_be32(NFS4ERR_BACK_CHAN_BUSY) -#define nfserr_layouttrylater cpu_to_be32(NFS4ERR_LAYOUTTRYLATER) -#define nfserr_layoutunavailable cpu_to_be32(NFS4ERR_LAYOUTUNAVAILABLE) -#define nfserr_nomatching_layout cpu_to_be32(NFS4ERR_NOMATCHING_LAYOUT) -#define nfserr_recallconflict cpu_to_be32(NFS4ERR_RECALLCONFLICT) -#define nfserr_unknown_layouttype cpu_to_be32(NFS4ERR_UNKNOWN_LAYOUTTYPE) -#define nfserr_seq_misordered cpu_to_be32(NFS4ERR_SEQ_MISORDERED) -#define nfserr_sequence_pos cpu_to_be32(NFS4ERR_SEQUENCE_POS) -#define nfserr_req_too_big cpu_to_be32(NFS4ERR_REQ_TOO_BIG) -#define nfserr_rep_too_big cpu_to_be32(NFS4ERR_REP_TOO_BIG) -#define nfserr_rep_too_big_to_cache cpu_to_be32(NFS4ERR_REP_TOO_BIG_TO_CACHE) -#define nfserr_retry_uncached_rep cpu_to_be32(NFS4ERR_RETRY_UNCACHED_REP) -#define nfserr_unsafe_compound cpu_to_be32(NFS4ERR_UNSAFE_COMPOUND) -#define nfserr_too_many_ops cpu_to_be32(NFS4ERR_TOO_MANY_OPS) -#define nfserr_op_not_in_session cpu_to_be32(NFS4ERR_OP_NOT_IN_SESSION) -#define nfserr_hash_alg_unsupp cpu_to_be32(NFS4ERR_HASH_ALG_UNSUPP) -#define nfserr_clientid_busy cpu_to_be32(NFS4ERR_CLIENTID_BUSY) -#define nfserr_pnfs_io_hole cpu_to_be32(NFS4ERR_PNFS_IO_HOLE) -#define nfserr_seq_false_retry cpu_to_be32(NFS4ERR_SEQ_FALSE_RETRY) -#define nfserr_bad_high_slot cpu_to_be32(NFS4ERR_BAD_HIGH_SLOT) -#define nfserr_deadsession cpu_to_be32(NFS4ERR_DEADSESSION) -#define nfserr_encr_alg_unsupp cpu_to_be32(NFS4ERR_ENCR_ALG_UNSUPP) -#define nfserr_pnfs_no_layout cpu_to_be32(NFS4ERR_PNFS_NO_LAYOUT) -#define nfserr_not_only_op cpu_to_be32(NFS4ERR_NOT_ONLY_OP) -#define nfserr_wrong_cred cpu_to_be32(NFS4ERR_WRONG_CRED) -#define nfserr_wrong_type cpu_to_be32(NFS4ERR_WRONG_TYPE) -#define nfserr_dirdeleg_unavail cpu_to_be32(NFS4ERR_DIRDELEG_UNAVAIL) -#define nfserr_reject_deleg cpu_to_be32(NFS4ERR_REJECT_DELEG) -#define nfserr_returnconflict cpu_to_be32(NFS4ERR_RETURNCONFLICT) -#define nfserr_deleg_revoked cpu_to_be32(NFS4ERR_DELEG_REVOKED) -#define nfserr_partner_notsupp cpu_to_be32(NFS4ERR_PARTNER_NOTSUPP) -#define nfserr_partner_no_auth cpu_to_be32(NFS4ERR_PARTNER_NO_AUTH) -#define nfserr_union_notsupp cpu_to_be32(NFS4ERR_UNION_NOTSUPP) -#define nfserr_offload_denied cpu_to_be32(NFS4ERR_OFFLOAD_DENIED) -#define nfserr_wrong_lfs cpu_to_be32(NFS4ERR_WRONG_LFS) -#define nfserr_badlabel cpu_to_be32(NFS4ERR_BADLABEL) -#define nfserr_file_open cpu_to_be32(NFS4ERR_FILE_OPEN) -#define nfserr_xattr2big cpu_to_be32(NFS4ERR_XATTR2BIG) -#define nfserr_noxattr cpu_to_be32(NFS4ERR_NOXATTR) - -/* - * Error codes for internal use. These are based at an impossible - * nfsstat4 value so that, once converted to be32, they cannot conflict - * with any value defined by the protocol (compare the nlm__int__* codes - * in fs/lockd/lockd.h). - */ -enum { -/* end-of-file indicator in readdir */ - NFSERR_EOF = 30000, -#define nfserr_eof cpu_to_be32(NFSERR_EOF) - -/* replay detected */ - NFSERR_REPLAY_ME, -#define nfserr_replay_me cpu_to_be32(NFSERR_REPLAY_ME) - -/* nfs41 replay detected */ - NFSERR_REPLAY_CACHE, -#define nfserr_replay_cache cpu_to_be32(NFSERR_REPLAY_CACHE) - -/* symlink found where dir expected - handled differently to - * other symlink found errors by NFSv3. - */ - NFSERR_SYMLINK_NOT_DIR, -#define nfserr_symlink_not_dir cpu_to_be32(NFSERR_SYMLINK_NOT_DIR) -}; - #ifdef CONFIG_NFSD_V4 /* before processing a COMPOUND operation, we have to check that there diff --git a/fs/nfsd/nfserr.h b/fs/nfsd/nfserr.h new file mode 100644 index 00000000000000..9b9df7aab220d7 --- /dev/null +++ b/fs/nfsd/nfserr.h @@ -0,0 +1,158 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +/* + * Pre-xdr'ed nfsd error values and nfsd-internal error codes. + * + * Separated from nfsd.h so that nfsd.h itself does not have to pull + * in : the NFS4ERR_* values used below are the only + * reason that include was needed. + */ + +#ifndef LINUX_NFSD_NFSERR_H +#define LINUX_NFSD_NFSERR_H + +#include +#include + +/* + * These macros provide pre-xdr'ed values for faster operation. + */ +#define nfs_ok cpu_to_be32(NFS_OK) +#define nfserr_perm cpu_to_be32(NFSERR_PERM) +#define nfserr_noent cpu_to_be32(NFSERR_NOENT) +#define nfserr_io cpu_to_be32(NFSERR_IO) +#define nfserr_nxio cpu_to_be32(NFSERR_NXIO) +#define nfserr_acces cpu_to_be32(NFSERR_ACCES) +#define nfserr_exist cpu_to_be32(NFSERR_EXIST) +#define nfserr_xdev cpu_to_be32(NFSERR_XDEV) +#define nfserr_nodev cpu_to_be32(NFSERR_NODEV) +#define nfserr_notdir cpu_to_be32(NFSERR_NOTDIR) +#define nfserr_isdir cpu_to_be32(NFSERR_ISDIR) +#define nfserr_inval cpu_to_be32(NFSERR_INVAL) +#define nfserr_fbig cpu_to_be32(NFSERR_FBIG) +#define nfserr_nospc cpu_to_be32(NFSERR_NOSPC) +#define nfserr_rofs cpu_to_be32(NFSERR_ROFS) +#define nfserr_mlink cpu_to_be32(NFSERR_MLINK) +#define nfserr_nametoolong cpu_to_be32(NFSERR_NAMETOOLONG) +#define nfserr_notempty cpu_to_be32(NFSERR_NOTEMPTY) +#define nfserr_dquot cpu_to_be32(NFSERR_DQUOT) +#define nfserr_stale cpu_to_be32(NFSERR_STALE) +#define nfserr_remote cpu_to_be32(NFSERR_REMOTE) +#define nfserr_wflush cpu_to_be32(NFSERR_WFLUSH) +#define nfserr_badhandle cpu_to_be32(NFSERR_BADHANDLE) +#define nfserr_notsync cpu_to_be32(NFSERR_NOT_SYNC) +#define nfserr_badcookie cpu_to_be32(NFSERR_BAD_COOKIE) +#define nfserr_notsupp cpu_to_be32(NFSERR_NOTSUPP) +#define nfserr_toosmall cpu_to_be32(NFSERR_TOOSMALL) +#define nfserr_serverfault cpu_to_be32(NFSERR_SERVERFAULT) +#define nfserr_badtype cpu_to_be32(NFSERR_BADTYPE) +#define nfserr_jukebox cpu_to_be32(NFSERR_JUKEBOX) +#define nfserr_denied cpu_to_be32(NFSERR_DENIED) +#define nfserr_deadlock cpu_to_be32(NFSERR_DEADLOCK) +#define nfserr_expired cpu_to_be32(NFSERR_EXPIRED) +#define nfserr_bad_cookie cpu_to_be32(NFSERR_BAD_COOKIE) +#define nfserr_same cpu_to_be32(NFSERR_SAME) +#define nfserr_clid_inuse cpu_to_be32(NFSERR_CLID_INUSE) +#define nfserr_stale_clientid cpu_to_be32(NFSERR_STALE_CLIENTID) +#define nfserr_resource cpu_to_be32(NFSERR_RESOURCE) +#define nfserr_moved cpu_to_be32(NFSERR_MOVED) +#define nfserr_nofilehandle cpu_to_be32(NFSERR_NOFILEHANDLE) +#define nfserr_minor_vers_mismatch cpu_to_be32(NFSERR_MINOR_VERS_MISMATCH) +#define nfserr_share_denied cpu_to_be32(NFSERR_SHARE_DENIED) +#define nfserr_stale_stateid cpu_to_be32(NFSERR_STALE_STATEID) +#define nfserr_old_stateid cpu_to_be32(NFSERR_OLD_STATEID) +#define nfserr_bad_stateid cpu_to_be32(NFSERR_BAD_STATEID) +#define nfserr_bad_seqid cpu_to_be32(NFSERR_BAD_SEQID) +#define nfserr_symlink cpu_to_be32(NFSERR_SYMLINK) +#define nfserr_not_same cpu_to_be32(NFSERR_NOT_SAME) +#define nfserr_lock_range cpu_to_be32(NFSERR_LOCK_RANGE) +#define nfserr_restorefh cpu_to_be32(NFSERR_RESTOREFH) +#define nfserr_attrnotsupp cpu_to_be32(NFSERR_ATTRNOTSUPP) +#define nfserr_bad_xdr cpu_to_be32(NFSERR_BAD_XDR) +#define nfserr_openmode cpu_to_be32(NFSERR_OPENMODE) +#define nfserr_badowner cpu_to_be32(NFSERR_BADOWNER) +#define nfserr_locks_held cpu_to_be32(NFSERR_LOCKS_HELD) +#define nfserr_op_illegal cpu_to_be32(NFSERR_OP_ILLEGAL) +#define nfserr_grace cpu_to_be32(NFSERR_GRACE) +#define nfserr_no_grace cpu_to_be32(NFSERR_NO_GRACE) +#define nfserr_reclaim_bad cpu_to_be32(NFSERR_RECLAIM_BAD) +#define nfserr_badname cpu_to_be32(NFSERR_BADNAME) +#define nfserr_admin_revoked cpu_to_be32(NFS4ERR_ADMIN_REVOKED) +#define nfserr_cb_path_down cpu_to_be32(NFSERR_CB_PATH_DOWN) +#define nfserr_locked cpu_to_be32(NFSERR_LOCKED) +#define nfserr_wrongsec cpu_to_be32(NFSERR_WRONGSEC) +#define nfserr_delay cpu_to_be32(NFS4ERR_DELAY) +#define nfserr_badiomode cpu_to_be32(NFS4ERR_BADIOMODE) +#define nfserr_badlayout cpu_to_be32(NFS4ERR_BADLAYOUT) +#define nfserr_bad_session_digest cpu_to_be32(NFS4ERR_BAD_SESSION_DIGEST) +#define nfserr_badsession cpu_to_be32(NFS4ERR_BADSESSION) +#define nfserr_badslot cpu_to_be32(NFS4ERR_BADSLOT) +#define nfserr_complete_already cpu_to_be32(NFS4ERR_COMPLETE_ALREADY) +#define nfserr_conn_not_bound_to_session cpu_to_be32(NFS4ERR_CONN_NOT_BOUND_TO_SESSION) +#define nfserr_deleg_already_wanted cpu_to_be32(NFS4ERR_DELEG_ALREADY_WANTED) +#define nfserr_back_chan_busy cpu_to_be32(NFS4ERR_BACK_CHAN_BUSY) +#define nfserr_layouttrylater cpu_to_be32(NFS4ERR_LAYOUTTRYLATER) +#define nfserr_layoutunavailable cpu_to_be32(NFS4ERR_LAYOUTUNAVAILABLE) +#define nfserr_nomatching_layout cpu_to_be32(NFS4ERR_NOMATCHING_LAYOUT) +#define nfserr_recallconflict cpu_to_be32(NFS4ERR_RECALLCONFLICT) +#define nfserr_unknown_layouttype cpu_to_be32(NFS4ERR_UNKNOWN_LAYOUTTYPE) +#define nfserr_seq_misordered cpu_to_be32(NFS4ERR_SEQ_MISORDERED) +#define nfserr_sequence_pos cpu_to_be32(NFS4ERR_SEQUENCE_POS) +#define nfserr_req_too_big cpu_to_be32(NFS4ERR_REQ_TOO_BIG) +#define nfserr_rep_too_big cpu_to_be32(NFS4ERR_REP_TOO_BIG) +#define nfserr_rep_too_big_to_cache cpu_to_be32(NFS4ERR_REP_TOO_BIG_TO_CACHE) +#define nfserr_retry_uncached_rep cpu_to_be32(NFS4ERR_RETRY_UNCACHED_REP) +#define nfserr_unsafe_compound cpu_to_be32(NFS4ERR_UNSAFE_COMPOUND) +#define nfserr_too_many_ops cpu_to_be32(NFS4ERR_TOO_MANY_OPS) +#define nfserr_op_not_in_session cpu_to_be32(NFS4ERR_OP_NOT_IN_SESSION) +#define nfserr_hash_alg_unsupp cpu_to_be32(NFS4ERR_HASH_ALG_UNSUPP) +#define nfserr_clientid_busy cpu_to_be32(NFS4ERR_CLIENTID_BUSY) +#define nfserr_pnfs_io_hole cpu_to_be32(NFS4ERR_PNFS_IO_HOLE) +#define nfserr_seq_false_retry cpu_to_be32(NFS4ERR_SEQ_FALSE_RETRY) +#define nfserr_bad_high_slot cpu_to_be32(NFS4ERR_BAD_HIGH_SLOT) +#define nfserr_deadsession cpu_to_be32(NFS4ERR_DEADSESSION) +#define nfserr_encr_alg_unsupp cpu_to_be32(NFS4ERR_ENCR_ALG_UNSUPP) +#define nfserr_pnfs_no_layout cpu_to_be32(NFS4ERR_PNFS_NO_LAYOUT) +#define nfserr_not_only_op cpu_to_be32(NFS4ERR_NOT_ONLY_OP) +#define nfserr_wrong_cred cpu_to_be32(NFS4ERR_WRONG_CRED) +#define nfserr_wrong_type cpu_to_be32(NFS4ERR_WRONG_TYPE) +#define nfserr_dirdeleg_unavail cpu_to_be32(NFS4ERR_DIRDELEG_UNAVAIL) +#define nfserr_reject_deleg cpu_to_be32(NFS4ERR_REJECT_DELEG) +#define nfserr_returnconflict cpu_to_be32(NFS4ERR_RETURNCONFLICT) +#define nfserr_deleg_revoked cpu_to_be32(NFS4ERR_DELEG_REVOKED) +#define nfserr_partner_notsupp cpu_to_be32(NFS4ERR_PARTNER_NOTSUPP) +#define nfserr_partner_no_auth cpu_to_be32(NFS4ERR_PARTNER_NO_AUTH) +#define nfserr_union_notsupp cpu_to_be32(NFS4ERR_UNION_NOTSUPP) +#define nfserr_offload_denied cpu_to_be32(NFS4ERR_OFFLOAD_DENIED) +#define nfserr_wrong_lfs cpu_to_be32(NFS4ERR_WRONG_LFS) +#define nfserr_badlabel cpu_to_be32(NFS4ERR_BADLABEL) +#define nfserr_file_open cpu_to_be32(NFS4ERR_FILE_OPEN) +#define nfserr_xattr2big cpu_to_be32(NFS4ERR_XATTR2BIG) +#define nfserr_noxattr cpu_to_be32(NFS4ERR_NOXATTR) + +/* + * Error codes for internal use. These are based at an impossible + * nfsstat4 value so that, once converted to be32, they cannot conflict + * with any value defined by the protocol (compare the nlm__int__* codes + * in fs/lockd/lockd.h). + */ +enum { +/* end-of-file indicator in readdir */ + NFSERR_EOF = 30000, +#define nfserr_eof cpu_to_be32(NFSERR_EOF) + +/* replay detected */ + NFSERR_REPLAY_ME, +#define nfserr_replay_me cpu_to_be32(NFSERR_REPLAY_ME) + +/* nfs41 replay detected */ + NFSERR_REPLAY_CACHE, +#define nfserr_replay_cache cpu_to_be32(NFSERR_REPLAY_CACHE) + +/* symlink found where dir expected - handled differently to + * other symlink found errors by NFSv3. + */ + NFSERR_SYMLINK_NOT_DIR, +#define nfserr_symlink_not_dir cpu_to_be32(NFSERR_SYMLINK_NOT_DIR) +}; + +#endif /* LINUX_NFSD_NFSERR_H */ diff --git a/fs/nfsd/nfsfh.c b/fs/nfsd/nfsfh.c index c0a46784d525ae..fd721a5a6b37bd 100644 --- a/fs/nfsd/nfsfh.c +++ b/fs/nfsd/nfsfh.c @@ -13,6 +13,7 @@ #include #include #include "nfsd.h" +#include "nfserr.h" #include "netns.h" #include "stats.h" #include "vfs.h" diff --git a/fs/nfsd/nfsproc.c b/fs/nfsd/nfsproc.c index 2a82fa64e47855..919acfba356a4f 100644 --- a/fs/nfsd/nfsproc.c +++ b/fs/nfsd/nfsproc.c @@ -10,6 +10,7 @@ #include "cache.h" #include "xdr.h" #include "vfs.h" +#include "nfserr.h" #include "trace.h" #define NFSDDBG_FACILITY NFSDDBG_PROC diff --git a/fs/nfsd/nfssvc.c b/fs/nfsd/nfssvc.c index 2edf716ea022ff..7f6ffbe7be289b 100644 --- a/fs/nfsd/nfssvc.c +++ b/fs/nfsd/nfssvc.c @@ -25,7 +25,9 @@ #include #include #include + #include "nfsd.h" +#include "nfserr.h" #include "cache.h" #include "vfs.h" #include "netns.h" diff --git a/fs/nfsd/nfsxdr.c b/fs/nfsd/nfsxdr.c index 019f0cc971a7aa..0961c13d6ab1d7 100644 --- a/fs/nfsd/nfsxdr.c +++ b/fs/nfsd/nfsxdr.c @@ -8,6 +8,7 @@ #include #include "vfs.h" +#include "nfserr.h" #include "xdr.h" #include "auth.h" diff --git a/fs/nfsd/vfs.c b/fs/nfsd/vfs.c index f65dad403ee08f..68fc2a45b64cc2 100644 --- a/fs/nfsd/vfs.c +++ b/fs/nfsd/vfs.c @@ -43,6 +43,7 @@ #endif /* CONFIG_NFSD_V4 */ #include "nfsd.h" +#include "nfserr.h" #include "netns.h" #include "stats.h" #include "vfs.h" From 0bb7728dcfc25b3ab320b4a8a93d5529a04b269c Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Fri, 17 Jul 2026 14:41:09 -0400 Subject: [PATCH 0145/1352] NFSD: Remove two unused NFSv4 constants Neither COMPOUND_SLACK_SPACE nor NFSD_COURTESY_CLIENT_TIMEOUT has a remaining user. COMPOUND_SLACK_SPACE lost its last reference in commit ea8d7720b274 ("nfsd4: remove redundant encode buffer size checking"), which deleted the encode buffer-space check the macro fed; the comment above it still describes that departed check. NFSD_COURTESY_CLIENT_TIMEOUT is likewise unreferenced: the courteous-server code expires clients through the laundromat's reaper and conflict paths, never a fixed 24-hour timer. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717184112.507548-4-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfsd.h | 7 ------- 1 file changed, 7 deletions(-) diff --git a/fs/nfsd/nfsd.h b/fs/nfsd/nfsd.h index 384a2498b6a295..27e5384cd84902 100644 --- a/fs/nfsd/nfsd.h +++ b/fs/nfsd/nfsd.h @@ -204,21 +204,14 @@ void nfsd_lockd_shutdown(void); * we might process an operation with side effects, and be unable to * tell the client that the operation succeeded. * - * COMPOUND_SLACK_SPACE - this is the minimum bytes of buffer space - * needed to encode an "ordinary" _successful_ operation. (GETATTR, - * READ, READDIR, and READLINK have their own buffer checks.) if we - * fall below this level, we fail the next operation with NFS4ERR_RESOURCE. - * * COMPOUND_ERR_SLACK_SPACE - this is the minimum bytes of buffer space * needed to encode an operation which has failed with NFS4ERR_RESOURCE. * care is taken to ensure that we never fall below this level for any * reason. */ -#define COMPOUND_SLACK_SPACE 140 /* OP_GETFH */ #define COMPOUND_ERR_SLACK_SPACE 16 /* OP_SETATTR */ #define NFSD_LAUNDROMAT_MINTIMEOUT 1 /* seconds */ -#define NFSD_COURTESY_CLIENT_TIMEOUT (24 * 60 * 60) /* seconds */ #define NFSD_CLIENT_MAX_TRIM_PER_RUN 128 #define NFS4_CLIENTS_PER_GB 1024 #define NFSD_DELEGRETURN_TIMEOUT (HZ / 34) /* 30ms */ From 80b634cff1a0e6aa1dc2af5a13a76b92055ea57e Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Fri, 17 Jul 2026 14:41:10 -0400 Subject: [PATCH 0146/1352] NFSD: Relocate NFSv4-internal constants to state.h The COMPOUND encode-slack sizes and the state-management timeouts at the tail of nfsd.h are evaluated only by NFSv4 code (nfs4state.c, nfs4proc.c, and nfs4xdr.c). They nonetheless sit in nfsd.h, where every NFSD translation unit, including the NFSv2 and NFSv3 paths that have no use for them, has to parse them. All three consumers already reach state.h through xdr4.h, so move the block there. nfsd.h keeps the NFSv4 prototypes for now. Only the pure constants move. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717184112.507548-5-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfsd.h | 18 ------------------ fs/nfsd/state.h | 19 +++++++++++++++++++ 2 files changed, 19 insertions(+), 18 deletions(-) diff --git a/fs/nfsd/nfsd.h b/fs/nfsd/nfsd.h index 27e5384cd84902..1886f6d4292910 100644 --- a/fs/nfsd/nfsd.h +++ b/fs/nfsd/nfsd.h @@ -199,24 +199,6 @@ void nfsd_lockd_shutdown(void); #ifdef CONFIG_NFSD_V4 -/* before processing a COMPOUND operation, we have to check that there - * is enough space in the buffer for XDR encode to succeed. otherwise, - * we might process an operation with side effects, and be unable to - * tell the client that the operation succeeded. - * - * COMPOUND_ERR_SLACK_SPACE - this is the minimum bytes of buffer space - * needed to encode an operation which has failed with NFS4ERR_RESOURCE. - * care is taken to ensure that we never fall below this level for any - * reason. - */ -#define COMPOUND_ERR_SLACK_SPACE 16 /* OP_SETATTR */ - -#define NFSD_LAUNDROMAT_MINTIMEOUT 1 /* seconds */ -#define NFSD_CLIENT_MAX_TRIM_PER_RUN 128 -#define NFS4_CLIENTS_PER_GB 1024 -#define NFSD_DELEGRETURN_TIMEOUT (HZ / 34) /* 30ms */ -#define NFSD_CB_GETATTR_TIMEOUT NFSD_DELEGRETURN_TIMEOUT - extern int nfsd4_is_junction(struct dentry *dentry); extern int register_cld_notifier(void); extern void unregister_cld_notifier(void); diff --git a/fs/nfsd/state.h b/fs/nfsd/state.h index 2d00a411c6634e..c4627dc91e2067 100644 --- a/fs/nfsd/state.h +++ b/fs/nfsd/state.h @@ -45,6 +45,25 @@ #include "nfsfh.h" #include "nfsd.h" +/* + * Before processing a COMPOUND operation, we have to check that there + * is enough space in the buffer for XDR encode to succeed. otherwise, + * we might process an operation with side effects, and be unable to + * tell the client that the operation succeeded. + * + * COMPOUND_ERR_SLACK_SPACE - this is the minimum bytes of buffer space + * needed to encode an operation which has failed with NFS4ERR_RESOURCE. + * care is taken to ensure that we never fall below this level for any + * reason. + */ +#define COMPOUND_ERR_SLACK_SPACE 16 /* OP_SETATTR */ + +#define NFSD_LAUNDROMAT_MINTIMEOUT 1 /* seconds */ +#define NFSD_CLIENT_MAX_TRIM_PER_RUN 128 +#define NFS4_CLIENTS_PER_GB 1024 +#define NFSD_DELEGRETURN_TIMEOUT (HZ / 34) /* 30ms */ +#define NFSD_CB_GETATTR_TIMEOUT NFSD_DELEGRETURN_TIMEOUT + typedef struct { u32 cl_boot; u32 cl_id; From 5858bae6d02be8248a3e1ab0d4fe132f2940e1a7 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Fri, 17 Jul 2026 14:41:11 -0400 Subject: [PATCH 0147/1352] NFSD: Evacuate NFSv4 entry-point prototypes from nfsd.h nfsd.h is included by nearly every NFSD translation unit, yet the two blocks of NFSv4 lifecycle and control prototypes it carries are referenced by only seven of them (out of over two dozen). The remaining consumers, including the NFSv2 and NFSv3 ACL and XDR paths, parse these declarations for no benefit. The declarations cannot simply move into an NFSv4-only header such as state.h: nfssvc.c, nfsctl.c, vfs.c, and export.c call the routines unconditionally and rely on the CONFIG_NFSD_V4=n stubs, and pulling the heavy state.h types into those lean translation units to obtain a handful of prototypes would trade one form of coupling for a worse one. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717184112.507548-6-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/export.c | 1 + fs/nfsd/nfs4ctl.h | 83 +++++++++++++++++++++++++++++++++++++++++++ fs/nfsd/nfs4proc.c | 1 + fs/nfsd/nfs4recover.c | 1 + fs/nfsd/nfs4state.c | 1 + fs/nfsd/nfsctl.c | 1 + fs/nfsd/nfsd.h | 63 -------------------------------- fs/nfsd/nfssvc.c | 1 + fs/nfsd/vfs.c | 1 + 9 files changed, 90 insertions(+), 63 deletions(-) create mode 100644 fs/nfsd/nfs4ctl.h diff --git a/fs/nfsd/export.c b/fs/nfsd/export.c index 23192eb7094fa7..e5a0f1ababe627 100644 --- a/fs/nfsd/export.c +++ b/fs/nfsd/export.c @@ -22,6 +22,7 @@ #include "nfsd.h" #include "nfserr.h" +#include "nfs4ctl.h" #include "nfsfh.h" #include "netns.h" #include "pnfs.h" diff --git a/fs/nfsd/nfs4ctl.h b/fs/nfsd/nfs4ctl.h new file mode 100644 index 00000000000000..bcec4c4ef1d536 --- /dev/null +++ b/fs/nfsd/nfs4ctl.h @@ -0,0 +1,83 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +/* + * Entry points by which the knfsd core drives the optional NFSv4 + * subsystem: state lifecycle, the laundromat workqueue, the recovery + * directory, junctions, the CLD notifier, and leases-net setup. + * + * Separated from nfsd.h so that the many translation units that + * include nfsd.h but call none of these -- among them the NFSv2 and + * NFSv3 paths -- do not have to parse them. The CONFIG_NFSD_V4=n + * stubs let the version-agnostic callers invoke the routines + * unconditionally. + */ + +#ifndef LINUX_NFSD_NFS4CTL_H +#define LINUX_NFSD_NFS4CTL_H + +#include +#include + +struct net; +struct inode; +struct dentry; +struct svc_rqst; +struct nfsd_net; + +#ifdef CONFIG_NFSD_V4 +extern unsigned long max_delegations; +int nfsd4_init_slabs(void); +void nfsd4_free_slabs(void); +int nfs4_state_start(void); +int nfs4_state_start_net(struct net *net); +void nfs4_state_shutdown(void); +void nfs4_state_shutdown_net(struct net *net); +int nfs4_reset_recoverydir(char *recdir); +char * nfs4_recoverydir(void); +bool nfsd4_spo_must_allow(struct svc_rqst *rqstp); +int nfsd4_create_laundry_wq(void); +void nfsd4_destroy_laundry_wq(void); +bool nfsd_wait_for_delegreturn(struct svc_rqst *rqstp, struct inode *inode); + +extern int nfsd4_is_junction(struct dentry *dentry); +extern int register_cld_notifier(void); +extern void unregister_cld_notifier(void); +#ifdef CONFIG_NFSD_V4_2_INTER_SSC +extern void nfsd4_ssc_init_umount_work(struct nfsd_net *nn); +#endif + +extern void nfsd4_init_leases_net(struct nfsd_net *nn); + +#else /* CONFIG_NFSD_V4 */ +static inline int nfsd4_init_slabs(void) { return 0; } +static inline void nfsd4_free_slabs(void) { } +static inline int nfs4_state_start(void) { return 0; } +static inline int nfs4_state_start_net(struct net *net) { return 0; } +static inline void nfs4_state_shutdown(void) { } +static inline void nfs4_state_shutdown_net(struct net *net) { } +static inline int nfs4_reset_recoverydir(char *recdir) { return 0; } +static inline char * nfs4_recoverydir(void) {return NULL; } +static inline bool nfsd4_spo_must_allow(struct svc_rqst *rqstp) +{ + return false; +} +static inline int nfsd4_create_laundry_wq(void) { return 0; }; +static inline void nfsd4_destroy_laundry_wq(void) {}; +static inline bool nfsd_wait_for_delegreturn(struct svc_rqst *rqstp, + struct inode *inode) +{ + return false; +} + +static inline int nfsd4_is_junction(struct dentry *dentry) +{ + return 0; +} + +static inline void nfsd4_init_leases_net(struct nfsd_net *nn) { }; + +#define register_cld_notifier() 0 +#define unregister_cld_notifier() do { } while(0) + +#endif /* CONFIG_NFSD_V4 */ + +#endif /* LINUX_NFSD_NFS4CTL_H */ diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 0bbf781d4ac591..d4265ee3c73dee 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -46,6 +46,7 @@ #include "idmap.h" #include "cache.h" #include "xdr4.h" +#include "nfs4ctl.h" #include "vfs.h" #include "current_stateid.h" #include "netns.h" diff --git a/fs/nfsd/nfs4recover.c b/fs/nfsd/nfs4recover.c index d513971fb119d5..aee3a0b22d1cb0 100644 --- a/fs/nfsd/nfs4recover.c +++ b/fs/nfsd/nfs4recover.c @@ -47,6 +47,7 @@ #include #include "nfsd.h" +#include "nfs4ctl.h" #include "state.h" #include "vfs.h" #include "netns.h" diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c index 47909a34d52b68..9c766a063b19ad 100644 --- a/fs/nfsd/nfs4state.c +++ b/fs/nfsd/nfs4state.c @@ -48,6 +48,7 @@ #include #include "xdr4.h" +#include "nfs4ctl.h" #include "xdr4cb.h" #include "vfs.h" #include "current_stateid.h" diff --git a/fs/nfsd/nfsctl.c b/fs/nfsd/nfsctl.c index 4492856c76b4fe..c0f10517470be7 100644 --- a/fs/nfsd/nfsctl.c +++ b/fs/nfsd/nfsctl.c @@ -24,6 +24,7 @@ #include "idmap.h" #include "nfsd.h" #include "nfserr.h" +#include "nfs4ctl.h" #include "netns.h" #include "stats.h" #include "cache.h" diff --git a/fs/nfsd/nfsd.h b/fs/nfsd/nfsd.h index 1886f6d4292910..1cccf5a3c034bd 100644 --- a/fs/nfsd/nfsd.h +++ b/fs/nfsd/nfsd.h @@ -151,45 +151,6 @@ static inline int nfsd_v4client(struct svc_rqst *rq) return rq && rq->rq_prog == NFS_PROGRAM && rq->rq_vers == 4; } -/* - * NFSv4 State - */ -#ifdef CONFIG_NFSD_V4 -extern unsigned long max_delegations; -int nfsd4_init_slabs(void); -void nfsd4_free_slabs(void); -int nfs4_state_start(void); -int nfs4_state_start_net(struct net *net); -void nfs4_state_shutdown(void); -void nfs4_state_shutdown_net(struct net *net); -int nfs4_reset_recoverydir(char *recdir); -char * nfs4_recoverydir(void); -bool nfsd4_spo_must_allow(struct svc_rqst *rqstp); -int nfsd4_create_laundry_wq(void); -void nfsd4_destroy_laundry_wq(void); -bool nfsd_wait_for_delegreturn(struct svc_rqst *rqstp, struct inode *inode); -#else -static inline int nfsd4_init_slabs(void) { return 0; } -static inline void nfsd4_free_slabs(void) { } -static inline int nfs4_state_start(void) { return 0; } -static inline int nfs4_state_start_net(struct net *net) { return 0; } -static inline void nfs4_state_shutdown(void) { } -static inline void nfs4_state_shutdown_net(struct net *net) { } -static inline int nfs4_reset_recoverydir(char *recdir) { return 0; } -static inline char * nfs4_recoverydir(void) {return NULL; } -static inline bool nfsd4_spo_must_allow(struct svc_rqst *rqstp) -{ - return false; -} -static inline int nfsd4_create_laundry_wq(void) { return 0; }; -static inline void nfsd4_destroy_laundry_wq(void) {}; -static inline bool nfsd_wait_for_delegreturn(struct svc_rqst *rqstp, - struct inode *inode) -{ - return false; -} -#endif - /* * lockd binding */ @@ -197,28 +158,4 @@ void nfsd_lockd_init(void); void nfsd_lockd_shutdown(void); -#ifdef CONFIG_NFSD_V4 - -extern int nfsd4_is_junction(struct dentry *dentry); -extern int register_cld_notifier(void); -extern void unregister_cld_notifier(void); -#ifdef CONFIG_NFSD_V4_2_INTER_SSC -extern void nfsd4_ssc_init_umount_work(struct nfsd_net *nn); -#endif - -extern void nfsd4_init_leases_net(struct nfsd_net *nn); - -#else /* CONFIG_NFSD_V4 */ -static inline int nfsd4_is_junction(struct dentry *dentry) -{ - return 0; -} - -static inline void nfsd4_init_leases_net(struct nfsd_net *nn) { }; - -#define register_cld_notifier() 0 -#define unregister_cld_notifier() do { } while(0) - -#endif /* CONFIG_NFSD_V4 */ - #endif /* LINUX_NFSD_NFSD_H */ diff --git a/fs/nfsd/nfssvc.c b/fs/nfsd/nfssvc.c index 7f6ffbe7be289b..9cc8489978a367 100644 --- a/fs/nfsd/nfssvc.c +++ b/fs/nfsd/nfssvc.c @@ -28,6 +28,7 @@ #include "nfsd.h" #include "nfserr.h" +#include "nfs4ctl.h" #include "cache.h" #include "vfs.h" #include "netns.h" diff --git a/fs/nfsd/vfs.c b/fs/nfsd/vfs.c index 68fc2a45b64cc2..cbd1e34a548a83 100644 --- a/fs/nfsd/vfs.c +++ b/fs/nfsd/vfs.c @@ -44,6 +44,7 @@ #include "nfsd.h" #include "nfserr.h" +#include "nfs4ctl.h" #include "netns.h" #include "stats.h" #include "vfs.h" From 5af52f4bd8eed4cff553b979db252bc944b35f6b Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Fri, 17 Jul 2026 14:41:12 -0400 Subject: [PATCH 0148/1352] NFSD: Move nfsd_v4client() out of nfsd.h nfsd_v4client() is the last user in nfsd.h of XDR-defined item references. Once this helper moves out, and the other XDR-specific headers nfsd.h includes have nothing left to provide, and can be dropped. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260717184112.507548-7-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfsd.h | 5 +---- fs/nfsd/nfssvc.c | 5 +++++ 2 files changed, 6 insertions(+), 4 deletions(-) diff --git a/fs/nfsd/nfsd.h b/fs/nfsd/nfsd.h index 1cccf5a3c034bd..64315890eef589 100644 --- a/fs/nfsd/nfsd.h +++ b/fs/nfsd/nfsd.h @@ -146,10 +146,7 @@ extern u64 nfsd_io_cache_write __read_mostly; extern int nfsd_max_blksize; -static inline int nfsd_v4client(struct svc_rqst *rq) -{ - return rq && rq->rq_prog == NFS_PROGRAM && rq->rq_vers == 4; -} +bool nfsd_v4client(struct svc_rqst *rqstp); /* * lockd binding diff --git a/fs/nfsd/nfssvc.c b/fs/nfsd/nfssvc.c index 9cc8489978a367..c04ef9d180ceab 100644 --- a/fs/nfsd/nfssvc.c +++ b/fs/nfsd/nfssvc.c @@ -207,6 +207,11 @@ int nfsd_minorversion(struct nfsd_net *nn, u32 minorversion, enum vers_op change return 0; } +bool nfsd_v4client(struct svc_rqst *rqstp) +{ + return rqstp && rqstp->rq_prog == NFS_PROGRAM && rqstp->rq_vers == 4; +} + bool nfsd_net_try_get(struct net *net) __must_hold(rcu) { struct nfsd_net *nn = net_generic(net, nfsd_net_id); From 355521b9374ae13da91bd4ba57eee9bf473ae727 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Mon, 20 Jul 2026 12:37:24 -0400 Subject: [PATCH 0149/1352] NFSD: Map flex file layout IDs through the request's user namespace nfsd4_ff_proc_layoutget() and nfsd4_ff_encode_layoutget() translate the file's owner and group with init_user_ns, but every other identity nfsd places on the wire goes through nfsd_user_namespace(). When the transport carries a credential from a non-initial user namespace, the flex file layout reports host-global IDs. The client copies those IDs into the AUTH_SYS credential it presents to the data server, and svcauth_unix_accept() resolves that credential in the transport's namespace, so data server I/O runs under an identity unrelated to the file's owner. Switching to the request's namespace introduces a second hazard. from_kuid() returns (uid_t)-1 when the target namespace has no mapping for the owner, and the IOMODE_READ arm adds one to that result to derive an identity for which the data server denies writes. The addition would wrap to zero, handing the client uid 0 instead of an identity distinct from the owner. Translate both IDs in nfsd4_ff_proc_layoutget(), which has the svc_rqst, and carry the wire values in struct pnfs_ff_layout. from_kuid_munged() substitutes overflowuid for an unmapped owner and thus never returns (uid_t)-1, so the increment cannot wrap to zero. Fixes: 9b9960a0ca47 ("nfsd: Add a super simple flex file server") Cc: stable@vger.kernel.org Reported-by: sashiko-bot Closes: https://sashiko.dev/#/patchset/20260720141442.783935-1-cel@kernel.org?part=3 Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260720163724.810227-1-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/flexfilelayout.c | 20 ++++++++++++-------- fs/nfsd/flexfilelayoutxdr.c | 4 ++-- fs/nfsd/flexfilelayoutxdr.h | 5 +++-- 3 files changed, 17 insertions(+), 12 deletions(-) diff --git a/fs/nfsd/flexfilelayout.c b/fs/nfsd/flexfilelayout.c index 9f532418cac8ae..c6e6b9106a8393 100644 --- a/fs/nfsd/flexfilelayout.c +++ b/fs/nfsd/flexfilelayout.c @@ -15,6 +15,7 @@ #include "nfserr.h" #include "flexfilelayoutxdr.h" +#include "auth.h" #include "pnfs.h" #include "vfs.h" @@ -24,10 +25,10 @@ static __be32 nfsd4_ff_proc_layoutget(struct svc_rqst *rqstp, struct inode *inode, const struct svc_fh *fhp, struct nfsd4_layoutget *args) { + struct user_namespace *userns = nfsd_user_namespace(rqstp); struct nfsd4_layout_seg *seg = &args->lg_seg; u32 device_generation = 0; int error; - uid_t u; struct pnfs_ff_layout *fl; @@ -50,13 +51,16 @@ nfsd4_ff_proc_layoutget(struct svc_rqst *rqstp, struct inode *inode, fl->flags = FF_FLAGS_NO_LAYOUTCOMMIT | FF_FLAGS_NO_IO_THRU_MDS | FF_FLAGS_NO_READ_IO; - /* Do not allow a IOMODE_READ segment to have write pemissions */ - if (seg->iomode == IOMODE_READ) { - u = from_kuid(&init_user_ns, inode->i_uid) + 1; - fl->uid = make_kuid(&init_user_ns, u); - } else - fl->uid = inode->i_uid; - fl->gid = inode->i_gid; + fl->uid = from_kuid_munged(userns, inode->i_uid); + fl->gid = from_kgid_munged(userns, inode->i_gid); + + /* + * Do not allow an IOMODE_READ segment to have write permissions. + * The group is left intact so group-readable files stay readable; + * nfsd_setuser() squashes an unmapped uid to the export's anon ID. + */ + if (seg->iomode == IOMODE_READ) + fl->uid++; error = nfsd4_set_deviceid(&fl->deviceid, fhp, device_generation); if (error) diff --git a/fs/nfsd/flexfilelayoutxdr.c b/fs/nfsd/flexfilelayoutxdr.c index 97d8a28dd3a04f..c12bb7c371b0e5 100644 --- a/fs/nfsd/flexfilelayoutxdr.c +++ b/fs/nfsd/flexfilelayoutxdr.c @@ -33,8 +33,8 @@ nfsd4_ff_encode_layoutget(struct xdr_stream *xdr, fh_len = 4 + xdr_align_size(fl->fh.size); - uid.len = sprintf(uid.buf, "%u", from_kuid(&init_user_ns, fl->uid)); - gid.len = sprintf(gid.buf, "%u", from_kgid(&init_user_ns, fl->gid)); + uid.len = sprintf(uid.buf, "%u", fl->uid); + gid.len = sprintf(gid.buf, "%u", fl->gid); /* data server entry: deviceid + efficiency + stateid + fh list + * user + group + flags + stats_collect_hint diff --git a/fs/nfsd/flexfilelayoutxdr.h b/fs/nfsd/flexfilelayoutxdr.h index 6d5a1066a903c1..3e1876d49db234 100644 --- a/fs/nfsd/flexfilelayoutxdr.h +++ b/fs/nfsd/flexfilelayoutxdr.h @@ -35,8 +35,9 @@ struct pnfs_ff_device_addr { struct pnfs_ff_layout { u32 flags; u32 stats_collect_hint; - kuid_t uid; - kgid_t gid; + /* Values to encode; nfsd4_ff_proc_layoutget() has mapped these */ + u32 uid; + u32 gid; struct nfsd4_deviceid deviceid; stateid_t stateid; struct nfs_fh fh; From 68b8730da8a32670b519cd1f753fe9cd55ba14ae Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Mon, 20 Jul 2026 10:14:40 -0400 Subject: [PATCH 0150/1352] NFS: Add linux/nfs_fh.h Plenty of spots around the kernel need the full definition of struct nfs_fh but not the cred, sunrpc, and uapi dependencies that linux/nfs.h pulls in along with it. Relocate struct nfs_fh to its own header, and include that header in linux/nfs.h so existing consumers keep building. Over time, consumers can then replace #include with #include While relocating the code, add kernel-doc comments for the FH operations and convert nfs_compare_fh() to return bool. Link: https://patch.msgid.link/20260720141442.783935-2-cel@kernel.org Signed-off-by: Chuck Lever --- include/linux/nfs.h | 39 ++------------------------ include/linux/nfs_fh.h | 63 ++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 65 insertions(+), 37 deletions(-) create mode 100644 include/linux/nfs_fh.h diff --git a/include/linux/nfs.h b/include/linux/nfs.h index 0906a0b40c6aa5..0e2a0b1e306156 100644 --- a/include/linux/nfs.h +++ b/include/linux/nfs.h @@ -11,8 +11,8 @@ #include #include #include -#include -#include +#include + #include /* The LOCALIO program is entirely private to Linux and is @@ -22,30 +22,6 @@ #define LOCALIOPROC_NULL 0 #define LOCALIOPROC_UUID_IS_LOCAL 1 -/* - * This is the kernel NFS client file handle representation - */ -#define NFS_MAXFHSIZE 128 -struct nfs_fh { - unsigned short size; - unsigned char data[NFS_MAXFHSIZE]; -}; - -/* - * Returns a zero iff the size and data fields match. - * Checks only "size" bytes in the data field. - */ -static inline int nfs_compare_fh(const struct nfs_fh *a, const struct nfs_fh *b) -{ - return a->size != b->size || memcmp(a->data, b->data, a->size) != 0; -} - -static inline void nfs_copy_fh(struct nfs_fh *target, const struct nfs_fh *source) -{ - target->size = source->size; - memcpy(target->data, source->data, source->size); -} - enum nfs3_stable_how { NFS_UNSTABLE = 0, NFS_DATA_SYNC = 1, @@ -55,15 +31,4 @@ enum nfs3_stable_how { NFS_INVALID_STABLE_HOW = -1 }; -/** - * nfs_fhandle_hash - calculate the crc32 hash for the filehandle - * @fh - pointer to filehandle - * - * returns a crc32 hash for the filehandle that is compatible with - * the one displayed by "wireshark". - */ -static inline u32 nfs_fhandle_hash(const struct nfs_fh *fh) -{ - return ~crc32_le(0xFFFFFFFF, &fh->data[0], fh->size); -} #endif /* _LINUX_NFS_H */ diff --git a/include/linux/nfs_fh.h b/include/linux/nfs_fh.h new file mode 100644 index 00000000000000..49dfc5ec60fec8 --- /dev/null +++ b/include/linux/nfs_fh.h @@ -0,0 +1,63 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +/* + * struct nfs_fh is an NFS version-agnostic data structure that + * stores an NFS file handle. It is also commonly used in NFS + * related APIs. + */ +#ifndef _LINUX_NFS_FH_H +#define _LINUX_NFS_FH_H + +#include +#include +#include + +/* + * The largest file handle size today is an NFSv4 file handle, + * which can be up to 128 octets long. + */ +#define NFS_MAXFHSIZE 128 +struct nfs_fh { + unsigned short size; + unsigned char data[NFS_MAXFHSIZE]; +}; + +/** + * nfs_compare_fh - Compare two NFS file handles + * @a: An NFS file handle to be compared + * @b: An NFS file handle to be compared + * + * Checks only "size" bytes in each data field. + * + * Return: %false if the two file handles are equal, otherwise %true + */ +static inline bool nfs_compare_fh(const struct nfs_fh *a, const struct nfs_fh *b) +{ + return a->size != b->size || memcmp(a->data, b->data, a->size) != 0; +} + +/** + * nfs_copy_fh - Copy an NFS file handle + * @target: Destination file handle + * @source: Source file handle + * + * Copies source->size bytes of file handle data into target. + */ +static inline void nfs_copy_fh(struct nfs_fh *target, const struct nfs_fh *source) +{ + target->size = source->size; + memcpy(target->data, source->data, source->size); +} + +/** + * nfs_fhandle_hash - Calculate the crc32 hash for the filehandle + * @fh: An NFS file handle to hash + * + * Return: a crc32 hash for the filehandle that is compatible with + * the one displayed by "wireshark" + */ +static inline u32 nfs_fhandle_hash(const struct nfs_fh *fh) +{ + return ~crc32_le(0xFFFFFFFF, &fh->data[0], fh->size); +} + +#endif /* _LINUX_NFS_FH_H */ From ae186707f5cb11e8c8808d1a399afa6e3bf95d24 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Mon, 20 Jul 2026 10:14:41 -0400 Subject: [PATCH 0151/1352] lockd: Switch linux/nfs.h to linux/nfs_fh.h lockd references "struct nfs_fh" but none of the other definitions in linux/nfs.h. That header also pulls in cred.h, several sunrpc headers, and uapi/linux/nfs.h, none of which lockd needs. A new linux/nfs_fh.h provides "struct nfs_fh" and its helpers without the rest of that surface. Switch lockd's xdr.h to linux/nfs_fh.h, and drop the now-redundant linux/nfs.h includes from svc.c and trace.h. Link: https://patch.msgid.link/20260720141442.783935-3-cel@kernel.org Signed-off-by: Chuck Lever --- fs/lockd/svc.c | 1 - fs/lockd/trace.h | 1 - fs/lockd/xdr.h | 2 +- 3 files changed, 1 insertion(+), 3 deletions(-) diff --git a/fs/lockd/svc.c b/fs/lockd/svc.c index ee90e743064afb..f0e1a58c910666 100644 --- a/fs/lockd/svc.c +++ b/fs/lockd/svc.c @@ -36,7 +36,6 @@ #include #include #include -#include #include "lockd.h" #include "netns.h" diff --git a/fs/lockd/trace.h b/fs/lockd/trace.h index a11d04e8c835eb..1f79955ea0f5f6 100644 --- a/fs/lockd/trace.h +++ b/fs/lockd/trace.h @@ -7,7 +7,6 @@ #include #include -#include #include "lockd.h" diff --git a/fs/lockd/xdr.h b/fs/lockd/xdr.h index a1126cca98c6ce..56b9796aa39d96 100644 --- a/fs/lockd/xdr.h +++ b/fs/lockd/xdr.h @@ -10,7 +10,7 @@ #include #include -#include +#include #include #define SM_MAXSTRLEN 1024 From 63bf8250daf46ac69d5e5b84bac48ca3168af620 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Mon, 20 Jul 2026 10:14:42 -0400 Subject: [PATCH 0152/1352] NFSD: Use struct knfsd_fh in struct pnfs_ff_layout The file handle held in struct pnfs_ff_layout is copied directly out of a struct svc_fh, whose fh_handle member is a struct knfsd_fh. Storing the layout's copy as struct nfs_fh instead forced an open-coded field-by-field copy between two unrelated structures. Hold the layout's file handle in struct knfsd_fh so the copy uses the canonical fh_copy_shallow() helper and server code no longer reaches into a separate file handle representation. Cc: Thomas Haynes Link: https://patch.msgid.link/20260720141442.783935-4-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/flexfilelayout.c | 3 +-- fs/nfsd/flexfilelayoutxdr.c | 4 ++-- fs/nfsd/flexfilelayoutxdr.h | 3 ++- 3 files changed, 5 insertions(+), 5 deletions(-) diff --git a/fs/nfsd/flexfilelayout.c b/fs/nfsd/flexfilelayout.c index c6e6b9106a8393..0deb913493a398 100644 --- a/fs/nfsd/flexfilelayout.c +++ b/fs/nfsd/flexfilelayout.c @@ -66,8 +66,7 @@ nfsd4_ff_proc_layoutget(struct svc_rqst *rqstp, struct inode *inode, if (error) goto out_error; - fl->fh.size = fhp->fh_handle.fh_size; - memcpy(fl->fh.data, &fhp->fh_handle.fh_raw, fl->fh.size); + fh_copy_shallow(&fl->fh, &fhp->fh_handle); /* Give whole file layout segments */ seg->offset = 0; diff --git a/fs/nfsd/flexfilelayoutxdr.c b/fs/nfsd/flexfilelayoutxdr.c index c12bb7c371b0e5..e297100a2ac3ce 100644 --- a/fs/nfsd/flexfilelayoutxdr.c +++ b/fs/nfsd/flexfilelayoutxdr.c @@ -31,7 +31,7 @@ nfsd4_ff_encode_layoutget(struct xdr_stream *xdr, struct ff_idmap uid; struct ff_idmap gid; - fh_len = 4 + xdr_align_size(fl->fh.size); + fh_len = 4 + xdr_align_size(fl->fh.fh_size); uid.len = sprintf(uid.buf, "%u", fl->uid); gid.len = sprintf(gid.buf, "%u", fl->gid); @@ -69,7 +69,7 @@ nfsd4_ff_encode_layoutget(struct xdr_stream *xdr, sizeof(stateid_opaque_t)); *p++ = cpu_to_be32(1); /* single file handle */ - p = xdr_encode_opaque(p, fl->fh.data, fl->fh.size); + p = xdr_encode_opaque(p, fl->fh.fh_raw, fl->fh.fh_size); p = xdr_encode_opaque(p, uid.buf, uid.len); p = xdr_encode_opaque(p, gid.buf, gid.len); diff --git a/fs/nfsd/flexfilelayoutxdr.h b/fs/nfsd/flexfilelayoutxdr.h index 3e1876d49db234..f7d1dd0708ec46 100644 --- a/fs/nfsd/flexfilelayoutxdr.h +++ b/fs/nfsd/flexfilelayoutxdr.h @@ -6,6 +6,7 @@ #define _NFSD_FLEXFILELAYOUTXDR_H 1 #include +#include "nfsfh.h" #include "xdr4.h" #define FF_FLAGS_NO_LAYOUTCOMMIT 1 @@ -40,7 +41,7 @@ struct pnfs_ff_layout { u32 gid; struct nfsd4_deviceid deviceid; stateid_t stateid; - struct nfs_fh fh; + struct knfsd_fh fh; }; __be32 nfsd4_ff_encode_getdeviceinfo(struct xdr_stream *xdr, From 9d606fbb1cee7a29e7b88f28dd32f154f96cb39b Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Tue, 21 Jul 2026 12:23:03 -0400 Subject: [PATCH 0153/1352] nfs_common: Remove unused nfs_ssc_client_ops infrastructure Clean up: Commit 75333d48f922 ("NFSD: fix use-after-free in __nfs42_ssc_open()") addressed a use-after-free bug by removing the nfsd4_interssc_disconnect() function. Post-copy clean-up was then delegated to NFSD's laundromat. Since that commit, the nfs_do_sb_deactive() wrapper function and the entire nfs_ssc_client_ops infrastructure no longer have any consumers. This includes nfs_do_sb_deactive(), struct nfs_ssc_client_ops, nfs_ssc_register(), nfs_ssc_unregister(), and related registrations in the NFS client. Cc: Olga Kornievskaia Cc: Dai Ngo Link: https://patch.msgid.link/20260721162306.894558-2-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfs/super.c | 25 ------------------------ fs/nfs_common/nfs_ssc.c | 42 ----------------------------------------- fs/nfsd/nfs4proc.c | 2 -- include/linux/nfs_ssc.h | 20 -------------------- 4 files changed, 89 deletions(-) diff --git a/fs/nfs/super.c b/fs/nfs/super.c index cb19f1540d9841..23292680adbd5d 100644 --- a/fs/nfs/super.c +++ b/fs/nfs/super.c @@ -58,7 +58,6 @@ #include #include -#include #include @@ -92,12 +91,6 @@ const struct super_operations nfs_sops = { }; EXPORT_SYMBOL_GPL(nfs_sops); -#ifdef CONFIG_NFS_V4_2 -static const struct nfs_ssc_client_ops nfs_ssc_clnt_ops_tbl = { - .sco_sb_deactive = nfs_sb_deactive, -}; -#endif - #if IS_ENABLED(CONFIG_NFS_V4) static int __init register_nfs4_fs(void) { @@ -119,18 +112,6 @@ static void unregister_nfs4_fs(void) } #endif -#ifdef CONFIG_NFS_V4_2 -static void nfs_ssc_register_ops(void) -{ - nfs_ssc_register(&nfs_ssc_clnt_ops_tbl); -} - -static void nfs_ssc_unregister_ops(void) -{ - nfs_ssc_unregister(&nfs_ssc_clnt_ops_tbl); -} -#endif /* CONFIG_NFS_V4_2 */ - static struct shrinker *acl_shrinker; /* @@ -163,9 +144,6 @@ int __init register_nfs_fs(void) shrinker_register(acl_shrinker); -#ifdef CONFIG_NFS_V4_2 - nfs_ssc_register_ops(); -#endif return 0; error_3: nfs_unregister_sysctl(); @@ -185,9 +163,6 @@ void __exit unregister_nfs_fs(void) shrinker_free(acl_shrinker); nfs_unregister_sysctl(); unregister_nfs4_fs(); -#ifdef CONFIG_NFS_V4_2 - nfs_ssc_unregister_ops(); -#endif unregister_filesystem(&nfs_fs_type); } diff --git a/fs/nfs_common/nfs_ssc.c b/fs/nfs_common/nfs_ssc.c index 832246b22c5175..8d8b7344ab4fe5 100644 --- a/fs/nfs_common/nfs_ssc.c +++ b/fs/nfs_common/nfs_ssc.c @@ -47,45 +47,3 @@ void nfs42_ssc_unregister(const struct nfs4_ssc_client_ops *ops) } EXPORT_SYMBOL_GPL(nfs42_ssc_unregister); #endif /* CONFIG_NFS_V4_2 */ - -#ifdef CONFIG_NFS_V4_2 -/** - * nfs_ssc_register - install the NFS_FS client ops in the nfs_ssc_client_tbl - * @ops: NFS_FS ops to be installed - * - * Return values: - * None - */ -void nfs_ssc_register(const struct nfs_ssc_client_ops *ops) -{ - nfs_ssc_client_tbl.ssc_nfs_ops = ops; -} -EXPORT_SYMBOL_GPL(nfs_ssc_register); - -/** - * nfs_ssc_unregister - uninstall the NFS_FS client ops from - * the nfs_ssc_client_tbl - * @ops: ops to be uninstalled - * - * Return values: - * None - */ -void nfs_ssc_unregister(const struct nfs_ssc_client_ops *ops) -{ - if (nfs_ssc_client_tbl.ssc_nfs_ops != ops) - return; - nfs_ssc_client_tbl.ssc_nfs_ops = NULL; -} -EXPORT_SYMBOL_GPL(nfs_ssc_unregister); - -#else -void nfs_ssc_register(const struct nfs_ssc_client_ops *ops) -{ -} -EXPORT_SYMBOL_GPL(nfs_ssc_register); - -void nfs_ssc_unregister(const struct nfs_ssc_client_ops *ops) -{ -} -EXPORT_SYMBOL_GPL(nfs_ssc_unregister); -#endif /* CONFIG_NFS_V4_2 */ diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index d4265ee3c73dee..8a2742d248c00f 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -1698,8 +1698,6 @@ extern struct file *nfs42_ssc_open(struct vfsmount *ss_mnt, nfs4_stateid *stateid); extern void nfs42_ssc_close(struct file *filep); -extern void nfs_sb_deactive(struct super_block *sb); - #define NFSD42_INTERSSC_MOUNTOPS "vers=4.2,addr=%s,sec=sys" /* diff --git a/include/linux/nfs_ssc.h b/include/linux/nfs_ssc.h index 22265b1ff08005..ba236dba8975cb 100644 --- a/include/linux/nfs_ssc.h +++ b/include/linux/nfs_ssc.h @@ -21,16 +21,8 @@ struct nfs4_ssc_client_ops { void (*sco_close)(struct file *filep); }; -/* - * NFS_FS - */ -struct nfs_ssc_client_ops { - void (*sco_sb_deactive)(struct super_block *sb); -}; - struct nfs_ssc_client_ops_tbl { const struct nfs4_ssc_client_ops *ssc_nfs4_ops; - const struct nfs_ssc_client_ops *ssc_nfs_ops; }; extern void nfs42_ssc_register_ops(void); @@ -67,15 +59,3 @@ struct nfsd4_ssc_umount_item { struct vfsmount *nsui_vfsmount; char nsui_ipaddr[RPC_MAX_ADDRBUFLEN + 1]; }; - -/* - * NFS_FS - */ -extern void nfs_ssc_register(const struct nfs_ssc_client_ops *ops); -extern void nfs_ssc_unregister(const struct nfs_ssc_client_ops *ops); - -static inline void nfs_do_sb_deactive(struct super_block *sb) -{ - if (nfs_ssc_client_tbl.ssc_nfs_ops) - (*nfs_ssc_client_tbl.ssc_nfs_ops->sco_sb_deactive)(sb); -} From cbc1c08f9add73608bffe7478e7214f2e078b602 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Tue, 21 Jul 2026 12:23:04 -0400 Subject: [PATCH 0154/1352] NFSD: Hoist nfs42_ssc_open() into fs/nfs_common/nfs_ssc.c Refactor: The infrastructure and details for calling the client's ssc_open method can be hidden in nfs_ssc.c. This reduces the SSC footprint in fs/nfsd/nfs4proc.c, a step toward removing that file's dependency on , which indirectly includes . The open and close functions are named "nfsd42_" since they are meant to be invoked only by NFSD. Cc: Olga Kornievskaia Cc: Dai Ngo Link: https://patch.msgid.link/20260721162306.894558-3-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfs_common/nfs_ssc.c | 56 +++++++++++++++++++++++++++++++++++++++-- fs/nfsd/nfs4proc.c | 17 +++---------- include/linux/nfs_ssc.h | 23 +++++++---------- 3 files changed, 66 insertions(+), 30 deletions(-) diff --git a/fs/nfs_common/nfs_ssc.c b/fs/nfs_common/nfs_ssc.c index 8d8b7344ab4fe5..a8e79ec687018a 100644 --- a/fs/nfs_common/nfs_ssc.c +++ b/fs/nfs_common/nfs_ssc.c @@ -12,9 +12,61 @@ #include #include "../nfs/nfs4_fs.h" +struct nfs_ssc_client_ops_tbl { + const struct nfs4_ssc_client_ops *ssc_nfs4_ops; +}; -struct nfs_ssc_client_ops_tbl nfs_ssc_client_tbl; -EXPORT_SYMBOL_GPL(nfs_ssc_client_tbl); +static struct nfs_ssc_client_ops_tbl nfs_ssc_client_tbl __read_mostly; + +/** + * nfsd42_ssc_open - Open a file to be used for server-to-server copy + * @ss_mnt: active mount point on which the source file resides + * @src_fh: file handle of the source file to be copied + * @stateid: stateid to use for COPY operation + * + * Caller must close the returned file using nfsd42_ssc_close(). + * + * Return: an open file, or an ERR_PTR on error + */ +struct file *nfsd42_ssc_open(struct vfsmount *ss_mnt, struct nfs_fh *src_fh, + nfs4_stateid *stateid) +{ + /* + * Built under CONFIG_NFS_V4_2_SSC_HELPER, which the NFS client + * enables on its own. The dispatch below is live only when the + * server also sets CONFIG_NFSD_V4_2_INTER_SSC; without it the + * source file cannot be opened, so callers get -EIO. + */ +#if IS_ENABLED(CONFIG_NFSD_V4_2_INTER_SSC) + const struct nfs4_ssc_client_ops *ops = nfs_ssc_client_tbl.ssc_nfs4_ops; + + if (ops) + return ops->sco_open(ss_mnt, src_fh, stateid); +#endif + + return ERR_PTR(-EIO); +} +EXPORT_SYMBOL_GPL(nfsd42_ssc_open); + +/** + * nfsd42_ssc_close - Close a file opened with nfsd42_ssc_open() + * @filp: struct file to be closed + * + * The real cleanup happens unconditionally in nfsd4_cleanup_inter_ssc(). + * The vfsmount is pinned until this function is called, preventing + * the client from unregistering its SSC ops. + */ +void nfsd42_ssc_close(struct file *filp) +{ + /* Live only under CONFIG_NFSD_V4_2_INTER_SSC; see nfsd42_ssc_open(). */ +#if IS_ENABLED(CONFIG_NFSD_V4_2_INTER_SSC) + const struct nfs4_ssc_client_ops *ops = nfs_ssc_client_tbl.ssc_nfs4_ops; + + if (ops) + ops->sco_close(filp); +#endif +} +EXPORT_SYMBOL_GPL(nfsd42_ssc_close); #ifdef CONFIG_NFS_V4_2 /** diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 8a2742d248c00f..aacfa80bddd864 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -1693,11 +1693,6 @@ void nfsd4_cancel_copy_by_sb(struct net *net, struct super_block *sb) #ifdef CONFIG_NFSD_V4_2_INTER_SSC -extern struct file *nfs42_ssc_open(struct vfsmount *ss_mnt, - struct nfs_fh *src_fh, - nfs4_stateid *stateid); -extern void nfs42_ssc_close(struct file *filep); - #define NFSD42_INTERSSC_MOUNTOPS "vers=4.2,addr=%s,sec=sys" /* @@ -1920,7 +1915,7 @@ nfsd4_cleanup_inter_ssc(struct nfsd4_ssc_umount_item *nsui, struct file *filp, struct nfsd_net *nn = net_generic(dst->nf_net, nfsd_net_id); long timeout = msecs_to_jiffies(nfsd4_ssc_umount_timeout); - nfs42_ssc_close(filp); + nfsd42_ssc_close(filp); fput(filp); spin_lock(&nn->nfsd_ssc_lock); @@ -1952,12 +1947,6 @@ nfsd4_cleanup_inter_ssc(struct nfsd4_ssc_umount_item *nsui, struct file *filp, { } -static struct file *nfs42_ssc_open(struct vfsmount *ss_mnt, - struct nfs_fh *src_fh, - nfs4_stateid *stateid) -{ - return NULL; -} #endif /* CONFIG_NFSD_V4_2_INTER_SSC */ static __be32 @@ -2180,8 +2169,8 @@ static int nfsd4_do_async_copy(void *data) if (nfsd4_ssc_is_inter(copy)) { struct file *filp; - filp = nfs42_ssc_open(copy->ss_nsui->nsui_vfsmount, - ©->c_fh, ©->stateid); + filp = nfsd42_ssc_open(copy->ss_nsui->nsui_vfsmount, + ©->c_fh, ©->stateid); if (IS_ERR(filp)) { switch (PTR_ERR(filp)) { case -EBADF: diff --git a/include/linux/nfs_ssc.h b/include/linux/nfs_ssc.h index ba236dba8975cb..fc0d5d48dec2a7 100644 --- a/include/linux/nfs_ssc.h +++ b/include/linux/nfs_ssc.h @@ -10,8 +10,6 @@ #include #include -extern struct nfs_ssc_client_ops_tbl nfs_ssc_client_tbl; - /* * NFS_V4 */ @@ -21,29 +19,26 @@ struct nfs4_ssc_client_ops { void (*sco_close)(struct file *filep); }; -struct nfs_ssc_client_ops_tbl { - const struct nfs4_ssc_client_ops *ssc_nfs4_ops; -}; - extern void nfs42_ssc_register_ops(void); extern void nfs42_ssc_unregister_ops(void); extern void nfs42_ssc_register(const struct nfs4_ssc_client_ops *ops); extern void nfs42_ssc_unregister(const struct nfs4_ssc_client_ops *ops); -#ifdef CONFIG_NFSD_V4_2_INTER_SSC -static inline struct file *nfs42_ssc_open(struct vfsmount *ss_mnt, - struct nfs_fh *src_fh, nfs4_stateid *stateid) +#if IS_ENABLED(CONFIG_NFS_V4_2_SSC_HELPER) +struct file *nfsd42_ssc_open(struct vfsmount *ss_mnt, struct nfs_fh *src_fh, + nfs4_stateid *stateid); +void nfsd42_ssc_close(struct file *filp); +#else +static inline struct file *nfsd42_ssc_open(struct vfsmount *ss_mnt, + struct nfs_fh *src_fh, + nfs4_stateid *stateid) { - if (nfs_ssc_client_tbl.ssc_nfs4_ops) - return (*nfs_ssc_client_tbl.ssc_nfs4_ops->sco_open)(ss_mnt, src_fh, stateid); return ERR_PTR(-EIO); } -static inline void nfs42_ssc_close(struct file *filep) +static inline void nfsd42_ssc_close(struct file *filp) { - if (nfs_ssc_client_tbl.ssc_nfs4_ops) - (*nfs_ssc_client_tbl.ssc_nfs4_ops->sco_close)(filep); } #endif From f18c812458bc887e1b5efab565896247f01f5efe Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Tue, 21 Jul 2026 12:23:05 -0400 Subject: [PATCH 0155/1352] nfs_common: Synchronize access to the SSC client ops table nfsd42_ssc_open() and nfsd42_ssc_close() load ssc_nfs4_ops without synchronization while nfs42_ssc_register() and nfs42_ssc_unregister() store to it. Those reads are safe today only through a non-obvious invariant: an inter-server copy holds an active vers=4.2 mount of the source across both calls, the mount pins the nfsv4 module through the nfs_client's cl_nfs_mod reference, and unregister runs only at nfsv4 module exit, so it cannot run while a call is in flight. Replace that implicit contract with synchronization local to the broker, so its safety no longer rests on a caller in another subsystem. Read the pointer under RCU so a reader observes it atomically as a valid table or NULL. nfs42_ssc_unregister() stores NULL and then calls synchronize_rcu(), so it cannot return while a reader still holds the pointer. The two readers need different handling because one sleeps and the other does not. sco_close() does not sleep, so nfsd42_ssc_close() runs it to completion inside the RCU read-side section and the synchronize_rcu() in unregister waits for it. __nfs42_ssc_open() does sleep -- it issues a GETATTR RPC to the source server and allocates with GFP_KERNEL -- so it must not run inside an RCU read-side section. Pin the provider module with try_module_get() while still under rcu_read_lock(), drop the lock, invoke the open, then release the module. The reference keeps the provider mapped across the sleep without relying on the caller's mount. If the table has already been torn down the copy gets -EIO. Cc: Olga Kornievskaia Cc: Dai Ngo Link: https://patch.msgid.link/20260721162306.894558-4-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfs/nfs4file.c | 1 + fs/nfs_common/nfs_ssc.c | 40 ++++++++++++++++++++++++++++++---------- include/linux/nfs_ssc.h | 1 + 3 files changed, 32 insertions(+), 10 deletions(-) diff --git a/fs/nfs/nfs4file.c b/fs/nfs/nfs4file.c index 6401f6363f7534..9a434f5dda8d36 100644 --- a/fs/nfs/nfs4file.c +++ b/fs/nfs/nfs4file.c @@ -402,6 +402,7 @@ static void __nfs42_ssc_close(struct file *filep) } static const struct nfs4_ssc_client_ops nfs4_ssc_clnt_ops_tbl = { + .owner = THIS_MODULE, .sco_open = __nfs42_ssc_open, .sco_close = __nfs42_ssc_close, }; diff --git a/fs/nfs_common/nfs_ssc.c b/fs/nfs_common/nfs_ssc.c index a8e79ec687018a..ef158008b80330 100644 --- a/fs/nfs_common/nfs_ssc.c +++ b/fs/nfs_common/nfs_ssc.c @@ -13,7 +13,7 @@ #include "../nfs/nfs4_fs.h" struct nfs_ssc_client_ops_tbl { - const struct nfs4_ssc_client_ops *ssc_nfs4_ops; + const struct nfs4_ssc_client_ops __rcu *ssc_nfs4_ops; }; static struct nfs_ssc_client_ops_tbl nfs_ssc_client_tbl __read_mostly; @@ -38,10 +38,24 @@ struct file *nfsd42_ssc_open(struct vfsmount *ss_mnt, struct nfs_fh *src_fh, * source file cannot be opened, so callers get -EIO. */ #if IS_ENABLED(CONFIG_NFSD_V4_2_INTER_SSC) - const struct nfs4_ssc_client_ops *ops = nfs_ssc_client_tbl.ssc_nfs4_ops; + const struct nfs4_ssc_client_ops *ops; + struct file *res; - if (ops) - return ops->sco_open(ss_mnt, src_fh, stateid); + /* + * sco_open() sleeps and must not run inside an RCU read-side + * section. Pin the provider module so the open runs with the + * module held; try_module_get() fails once unregister begins, + * and the copy then gets -EIO. + */ + rcu_read_lock(); + ops = rcu_dereference(nfs_ssc_client_tbl.ssc_nfs4_ops); + if (ops && try_module_get(ops->owner)) { + rcu_read_unlock(); + res = ops->sco_open(ss_mnt, src_fh, stateid); + module_put(ops->owner); + return res; + } + rcu_read_unlock(); #endif return ERR_PTR(-EIO); @@ -53,17 +67,21 @@ EXPORT_SYMBOL_GPL(nfsd42_ssc_open); * @filp: struct file to be closed * * The real cleanup happens unconditionally in nfsd4_cleanup_inter_ssc(). - * The vfsmount is pinned until this function is called, preventing - * the client from unregistering its SSC ops. + * The client ops table is read under RCU; nfs42_ssc_unregister() calls + * synchronize_rcu() so unregistration cannot complete while a close is + * in flight. */ void nfsd42_ssc_close(struct file *filp) { /* Live only under CONFIG_NFSD_V4_2_INTER_SSC; see nfsd42_ssc_open(). */ #if IS_ENABLED(CONFIG_NFSD_V4_2_INTER_SSC) - const struct nfs4_ssc_client_ops *ops = nfs_ssc_client_tbl.ssc_nfs4_ops; + const struct nfs4_ssc_client_ops *ops; + rcu_read_lock(); + ops = rcu_dereference(nfs_ssc_client_tbl.ssc_nfs4_ops); if (ops) ops->sco_close(filp); + rcu_read_unlock(); #endif } EXPORT_SYMBOL_GPL(nfsd42_ssc_close); @@ -78,7 +96,7 @@ EXPORT_SYMBOL_GPL(nfsd42_ssc_close); */ void nfs42_ssc_register(const struct nfs4_ssc_client_ops *ops) { - nfs_ssc_client_tbl.ssc_nfs4_ops = ops; + rcu_assign_pointer(nfs_ssc_client_tbl.ssc_nfs4_ops, ops); } EXPORT_SYMBOL_GPL(nfs42_ssc_register); @@ -92,10 +110,12 @@ EXPORT_SYMBOL_GPL(nfs42_ssc_register); */ void nfs42_ssc_unregister(const struct nfs4_ssc_client_ops *ops) { - if (nfs_ssc_client_tbl.ssc_nfs4_ops != ops) + if (rcu_dereference_protected(nfs_ssc_client_tbl.ssc_nfs4_ops, + true) != ops) return; - nfs_ssc_client_tbl.ssc_nfs4_ops = NULL; + rcu_assign_pointer(nfs_ssc_client_tbl.ssc_nfs4_ops, NULL); + synchronize_rcu(); } EXPORT_SYMBOL_GPL(nfs42_ssc_unregister); #endif /* CONFIG_NFS_V4_2 */ diff --git a/include/linux/nfs_ssc.h b/include/linux/nfs_ssc.h index fc0d5d48dec2a7..b392d56a4dc252 100644 --- a/include/linux/nfs_ssc.h +++ b/include/linux/nfs_ssc.h @@ -14,6 +14,7 @@ * NFS_V4 */ struct nfs4_ssc_client_ops { + struct module *owner; struct file *(*sco_open)(struct vfsmount *ss_mnt, struct nfs_fh *src_fh, nfs4_stateid *stateid); void (*sco_close)(struct file *filep); From 8a0d15a960436ae46583c05f8dae1f99982a7605 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Tue, 21 Jul 2026 12:23:06 -0400 Subject: [PATCH 0156/1352] NFSD: Split linux/nfs_ssc.h The nfs_ssc.h header contains both client- and server-side data structures, which means each of those implementations has to pull in headers from the other. Create a linux/nfsd_ssc.h for the server side APIs which no longer includes uapi/linux/nfs.h either directly or indirectly. Because nfsd_ssc.h drops the transitive include of the NFS client headers, fs/nfsd/nfs4proc.c now includes directly for filemap_check_wb_err(). struct nfsd4_ssc_umount_item is private to nfsd. Move it into fs/nfsd/xdr4.h alongside its only consumers rather than into the exported nfsd_ssc.h. As an added clean-up, add missing header guard macros and the struct file and struct vfsmount forward declarations the server prototypes need. Cc: Olga Kornievskaia Cc: Dai Ngo Link: https://patch.msgid.link/20260721162306.894558-5-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfs_common/nfs_ssc.c | 2 +- fs/nfsd/nfs4proc.c | 3 ++- fs/nfsd/nfs4state.c | 2 +- fs/nfsd/xdr4.h | 13 ++++++++++++ include/linux/nfs_ssc.h | 45 ++++++++++------------------------------ include/linux/nfsd_ssc.h | 38 +++++++++++++++++++++++++++++++++ 6 files changed, 66 insertions(+), 37 deletions(-) create mode 100644 include/linux/nfsd_ssc.h diff --git a/fs/nfs_common/nfs_ssc.c b/fs/nfs_common/nfs_ssc.c index ef158008b80330..e521e3c836fe7b 100644 --- a/fs/nfs_common/nfs_ssc.c +++ b/fs/nfs_common/nfs_ssc.c @@ -10,7 +10,7 @@ #include #include #include -#include "../nfs/nfs4_fs.h" +#include struct nfs_ssc_client_ops_tbl { const struct nfs4_ssc_client_ops __rcu *ssc_nfs4_ops; diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index aacfa80bddd864..8ffabe7a480dd7 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -38,9 +38,10 @@ #include #include #include +#include #include -#include +#include #include "attr4.h" #include "idmap.h" diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c index 9c766a063b19ad..a130d4bcf85031 100644 --- a/fs/nfsd/nfs4state.c +++ b/fs/nfsd/nfs4state.c @@ -45,7 +45,7 @@ #include #include #include -#include +#include #include "xdr4.h" #include "nfs4ctl.h" diff --git a/fs/nfsd/xdr4.h b/fs/nfsd/xdr4.h index e833407859c8d8..7bbb375874ef80 100644 --- a/fs/nfsd/xdr4.h +++ b/fs/nfsd/xdr4.h @@ -597,6 +597,19 @@ struct nfsd4_cb_offload { u32 co_referring_seqno; }; +struct nfsd4_ssc_umount_item { + struct list_head nsui_list; + bool nsui_busy; + /* + * nsui_refcnt inited to 2, 1 on list and 1 for consumer. Entry + * is removed when refcnt drops to 1 and nsui_expire expires. + */ + refcount_t nsui_refcnt; + unsigned long nsui_expire; + struct vfsmount *nsui_vfsmount; + char nsui_ipaddr[RPC_MAX_ADDRBUFLEN + 1]; +}; + struct nfsd4_copy { /* request */ stateid_t cp_src_stateid; diff --git a/include/linux/nfs_ssc.h b/include/linux/nfs_ssc.h index b392d56a4dc252..c199ea23e7eb77 100644 --- a/include/linux/nfs_ssc.h +++ b/include/linux/nfs_ssc.h @@ -2,17 +2,22 @@ /* * include/linux/nfs_ssc.h * + * NFSv4.2 server-to-server copy, NFS client side APIs + * * Author: Dai Ngo * * Copyright (c) 2020, Oracle and/or its affiliates. */ -#include -#include +#ifndef _LINUX_NFS_SSC_H +#define _LINUX_NFS_SSC_H + +#include +#include + +struct file; +struct vfsmount; -/* - * NFS_V4 - */ struct nfs4_ssc_client_ops { struct module *owner; struct file *(*sco_open)(struct vfsmount *ss_mnt, @@ -26,32 +31,4 @@ extern void nfs42_ssc_unregister_ops(void); extern void nfs42_ssc_register(const struct nfs4_ssc_client_ops *ops); extern void nfs42_ssc_unregister(const struct nfs4_ssc_client_ops *ops); -#if IS_ENABLED(CONFIG_NFS_V4_2_SSC_HELPER) -struct file *nfsd42_ssc_open(struct vfsmount *ss_mnt, struct nfs_fh *src_fh, - nfs4_stateid *stateid); -void nfsd42_ssc_close(struct file *filp); -#else -static inline struct file *nfsd42_ssc_open(struct vfsmount *ss_mnt, - struct nfs_fh *src_fh, - nfs4_stateid *stateid) -{ - return ERR_PTR(-EIO); -} - -static inline void nfsd42_ssc_close(struct file *filp) -{ -} -#endif - -struct nfsd4_ssc_umount_item { - struct list_head nsui_list; - bool nsui_busy; - /* - * nsui_refcnt inited to 2, 1 on list and 1 for consumer. Entry - * is removed when refcnt drops to 1 and nsui_expire expires. - */ - refcount_t nsui_refcnt; - unsigned long nsui_expire; - struct vfsmount *nsui_vfsmount; - char nsui_ipaddr[RPC_MAX_ADDRBUFLEN + 1]; -}; +#endif /* _LINUX_NFS_SSC_H */ diff --git a/include/linux/nfsd_ssc.h b/include/linux/nfsd_ssc.h new file mode 100644 index 00000000000000..7001410f01c299 --- /dev/null +++ b/include/linux/nfsd_ssc.h @@ -0,0 +1,38 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +/* + * include/linux/nfsd_ssc.h + * + * NFSv4.2 server-to-server copy, NFS server side APIs + * + * Author: Dai Ngo + * + * Copyright (c) 2020, Oracle and/or its affiliates. + */ + +#ifndef _LINUX_NFSD_SSC_H +#define _LINUX_NFSD_SSC_H + +#include +#include + +struct file; +struct vfsmount; + +#if IS_ENABLED(CONFIG_NFS_V4_2_SSC_HELPER) +struct file *nfsd42_ssc_open(struct vfsmount *ss_mnt, struct nfs_fh *src_fh, + nfs4_stateid *stateid); +void nfsd42_ssc_close(struct file *filp); +#else +static inline struct file *nfsd42_ssc_open(struct vfsmount *ss_mnt, + struct nfs_fh *src_fh, + nfs4_stateid *stateid) +{ + return ERR_PTR(-EIO); +} + +static inline void nfsd42_ssc_close(struct file *filp) +{ +} +#endif + +#endif /* _LINUX_NFSD_SSC_H */ From 0c5eb13e2e52614c65713f622f95b4f9e31613eb Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Thu, 23 Jul 2026 14:20:42 -0400 Subject: [PATCH 0157/1352] NFS: Move definition of enum nfs3_stable_how Clean up: enum nfs3_stable_how was introduced in NFSv3. NFSv2 has no stable_how on the wire; its write path passes NFS_FILE_SYNC only as a placeholder that the protocol ignores. The stable_how constants describe an NFSv3 wire value, so they belong in linux/nfs3.h. Link: https://patch.msgid.link/20260723182043.990391-2-cel@kernel.org Signed-off-by: Chuck Lever --- include/linux/nfs.h | 9 --------- include/linux/nfs3.h | 8 ++++++++ include/trace/misc/nfs.h | 1 + 3 files changed, 9 insertions(+), 9 deletions(-) diff --git a/include/linux/nfs.h b/include/linux/nfs.h index 0e2a0b1e306156..0e2b210c103b69 100644 --- a/include/linux/nfs.h +++ b/include/linux/nfs.h @@ -22,13 +22,4 @@ #define LOCALIOPROC_NULL 0 #define LOCALIOPROC_UUID_IS_LOCAL 1 -enum nfs3_stable_how { - NFS_UNSTABLE = 0, - NFS_DATA_SYNC = 1, - NFS_FILE_SYNC = 2, - - /* used by direct.c to mark verf as invalid */ - NFS_INVALID_STABLE_HOW = -1 -}; - #endif /* _LINUX_NFS_H */ diff --git a/include/linux/nfs3.h b/include/linux/nfs3.h index 404b8f724fc956..1d18da0860d52b 100644 --- a/include/linux/nfs3.h +++ b/include/linux/nfs3.h @@ -7,6 +7,14 @@ #include +enum nfs3_stable_how { + NFS_UNSTABLE = 0, + NFS_DATA_SYNC = 1, + NFS_FILE_SYNC = 2, + + /* used to mark verf as invalid */ + NFS_INVALID_STABLE_HOW = -1 +}; /* Number of 32bit words in post_op_attr */ #define NFS3_POST_OP_ATTR_WORDS 22 diff --git a/include/trace/misc/nfs.h b/include/trace/misc/nfs.h index a394b4d38e18fa..b5fb77d7954b35 100644 --- a/include/trace/misc/nfs.h +++ b/include/trace/misc/nfs.h @@ -8,6 +8,7 @@ */ #include +#include #include #include From 6dcddbb70b0882b46439491264fea0fcd8c52428 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Thu, 23 Jul 2026 14:20:43 -0400 Subject: [PATCH 0158/1352] NFSD: Replace nfsd_write()'s "stable" argument with "iocb_flags" The current nfsd_write() API is not NFS version-agnostic, as it relies on callers to pass an NFSv3 stable_how value to determine the persistence of the requested WRITE. NFSv2 does not use a stable-how value on the wire, and NFSv4 has its own stable_how4 (though stable_how and stable_how4 happen to share the same numeric values). To remove the dependence on NFSv3-specific XDR values from NFSD's generic VFS APIs, replace nfsd_write()'s stable argument with an argument that passes a set of IOCB flags instead of an XDR-defined value. The NFSv4 WRITE and COPY paths had been borrowing the NFSv3 stable_how constants for their own on-the-wire stable values, relying on the numeric coincidence noted above. Convert those sites to the stable_how4 enumerators so the v4 code expresses its own protocol's values directly, with no change in behavior. While here, bound-check the decoded NFSv3 WRITE stable value, as the NFSv4 WRITE decoder already does, and make the nfsd3_writeargs stable field unsigned to suit. The larger benefit is one less NFSv4 dependency on nfs3.h. Link: https://patch.msgid.link/20260723182043.990391-3-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfs3proc.c | 17 ++++++++++++++++- fs/nfsd/nfs3xdr.c | 2 ++ fs/nfsd/nfs4proc.c | 18 ++++++++++++++++-- fs/nfsd/nfs4xdr.c | 2 +- fs/nfsd/nfsproc.c | 2 +- fs/nfsd/vfs.c | 30 ++++++++++-------------------- fs/nfsd/vfs.h | 6 ++++-- fs/nfsd/xdr3.h | 2 +- include/linux/nfs4.h | 6 ++++++ 9 files changed, 57 insertions(+), 28 deletions(-) diff --git a/fs/nfsd/nfs3proc.c b/fs/nfsd/nfs3proc.c index 4b3075c05b9793..19ab0a713d8212 100644 --- a/fs/nfsd/nfs3proc.c +++ b/fs/nfsd/nfs3proc.c @@ -49,6 +49,20 @@ static bool nfsd3_time_in_range(const struct iattr *iap) return true; } +static int nfsd3_iocb_flags(enum nfs3_stable_how how) +{ + switch (how) { + case NFS_FILE_SYNC: + /* persist data and timestamps */ + return IOCB_DSYNC | IOCB_SYNC; + case NFS_DATA_SYNC: + /* persist data only */ + return IOCB_DSYNC; + default: + return 0; + } +} + static __be32 nfsd3_map_status(__be32 status) { switch (status) { @@ -261,7 +275,8 @@ nfsd3_proc_write(struct svc_rqst *rqstp) resp->committed = argp->stable; resp->status = nfsd_write(rqstp, &resp->fh, argp->offset, &argp->payload, &cnt, - resp->committed, resp->verf); + nfsd3_iocb_flags(resp->committed), + resp->verf); resp->count = cnt; resp->status = nfsd3_map_status(resp->status); return rpc_success; diff --git a/fs/nfsd/nfs3xdr.c b/fs/nfsd/nfs3xdr.c index 196bcc6edebb98..090cea8e545dc6 100644 --- a/fs/nfsd/nfs3xdr.c +++ b/fs/nfsd/nfs3xdr.c @@ -557,6 +557,8 @@ nfs3svc_decode_writeargs(struct svc_rqst *rqstp, struct xdr_stream *xdr) return false; if (xdr_stream_decode_u32(xdr, &args->stable) < 0) return false; + if (args->stable > NFS_FILE_SYNC) + return false; /* opaque data */ if (xdr_stream_decode_u32(xdr, &args->len) < 0) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 8ffabe7a480dd7..33dc92d48ce0f7 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -72,6 +72,20 @@ MODULE_PARM_DESC(nfsd4_ssc_umount_timeout, #define NFSDDBG_FACILITY NFSDDBG_PROC +static int nfsd4_iocb_flags(enum stable_how4 how) +{ + switch (how) { + case FILE_SYNC4: + /* persist data and timestamps */ + return IOCB_DSYNC | IOCB_SYNC; + case DATA_SYNC4: + /* persist data only */ + return IOCB_DSYNC; + default: + return 0; + } +} + static u32 nfsd_attrmask[] = { NFSD_WRITEABLE_ATTRS_WORD0, NFSD_WRITEABLE_ATTRS_WORD1, @@ -1418,7 +1432,7 @@ nfsd4_write(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate, write->wr_how_written = write->wr_stable_how; status = nfsd_vfs_write(rqstp, &cstate->current_fh, nf, write->wr_offset, &write->wr_payload, - &cnt, write->wr_how_written, + &cnt, nfsd4_iocb_flags(write->wr_how_written), (__be32 *)write->wr_verifier.data); nfsd_file_put(nf); @@ -2002,7 +2016,7 @@ static void nfsd4_init_copy_res(struct nfsd4_copy *copy, bool sync) { copy->cp_res.wr_stable_how = test_bit(NFSD4_COPY_F_COMMITTED, ©->cp_flags) ? - NFS_FILE_SYNC : NFS_UNSTABLE; + FILE_SYNC4 : UNSTABLE4; nfsd4_copy_set_sync(copy, sync); } diff --git a/fs/nfsd/nfs4xdr.c b/fs/nfsd/nfs4xdr.c index 7ccc7c897b00cb..a47eb544b99f6f 100644 --- a/fs/nfsd/nfs4xdr.c +++ b/fs/nfsd/nfs4xdr.c @@ -1607,7 +1607,7 @@ nfsd4_decode_write(struct nfsd4_compoundargs *argp, union nfsd4_op_u *u) return nfserr_bad_xdr; if (xdr_stream_decode_u32(argp->xdr, &write->wr_stable_how) < 0) return nfserr_bad_xdr; - if (write->wr_stable_how > NFS_FILE_SYNC) + if (write->wr_stable_how > FILE_SYNC4) return nfserr_bad_xdr; if (xdr_stream_decode_u32(argp->xdr, &write->wr_buflen) < 0) return nfserr_bad_xdr; diff --git a/fs/nfsd/nfsproc.c b/fs/nfsd/nfsproc.c index 919acfba356a4f..48541ef7644f4d 100644 --- a/fs/nfsd/nfsproc.c +++ b/fs/nfsd/nfsproc.c @@ -266,7 +266,7 @@ nfsd_proc_write(struct svc_rqst *rqstp) fh_copy(&resp->fh, &argp->fh); resp->status = nfsd_write(rqstp, &resp->fh, argp->offset, - &argp->payload, &cnt, NFS_DATA_SYNC, NULL); + &argp->payload, &cnt, IOCB_DSYNC, NULL); if (resp->status == nfs_ok) resp->status = fh_getattr(&resp->fh, &resp->stat); else if (resp->status == nfserr_jukebox) diff --git a/fs/nfsd/vfs.c b/fs/nfsd/vfs.c index cbd1e34a548a83..807e09521e0c3e 100644 --- a/fs/nfsd/vfs.c +++ b/fs/nfsd/vfs.c @@ -1433,7 +1433,7 @@ nfsd_direct_write(struct svc_rqst *rqstp, struct svc_fh *fhp, * @offset: Byte offset of start * @payload: xdr_buf containing the write payload * @cnt: IN: number of bytes to write, OUT: number of bytes actually written - * @stable: An NFS stable_how value + * @iocb_flags: VFS IOCB_* flags expressing the requested write stability * @verf: NFS WRITE verifier * * Upon return, caller must invoke fh_put on @fhp. @@ -1445,7 +1445,7 @@ __be32 nfsd_vfs_write(struct svc_rqst *rqstp, struct svc_fh *fhp, struct nfsd_file *nf, loff_t offset, const struct xdr_buf *payload, unsigned long *cnt, - int stable, __be32 *verf) + int iocb_flags, __be32 *verf) { struct nfsd_net *nn = net_generic(SVC_NET(rqstp), nfsd_net_id); struct file *file = nf->nf_file; @@ -1482,21 +1482,11 @@ nfsd_vfs_write(struct svc_rqst *rqstp, struct svc_fh *fhp, exp = fhp->fh_export; if (!EX_ISSYNC(exp)) - stable = NFS_UNSTABLE; + iocb_flags = 0; init_sync_kiocb(&kiocb, file); kiocb.ki_pos = offset; - if (likely(!fhp->fh_use_wgather)) { - switch (stable) { - case NFS_FILE_SYNC: - /* persist data and timestamps */ - kiocb.ki_flags |= IOCB_DSYNC | IOCB_SYNC; - break; - case NFS_DATA_SYNC: - /* persist data only */ - kiocb.ki_flags |= IOCB_DSYNC; - break; - } - } + if (likely(!fhp->fh_use_wgather)) + kiocb.ki_flags |= iocb_flags; nvecs = xdr_buf_to_bvec(rqstp->rq_bvec, rqstp->rq_maxpages, payload); if (nvecs < 0) { @@ -1537,7 +1527,7 @@ nfsd_vfs_write(struct svc_rqst *rqstp, struct svc_fh *fhp, goto out_nfserr; } - if (stable && fhp->fh_use_wgather) { + if (iocb_flags && fhp->fh_use_wgather) { host_err = wait_for_concurrent_writes(file); if (host_err < 0) commit_reset_write_verifier(nn, rqstp, host_err); @@ -1628,7 +1618,7 @@ __be32 nfsd_read(struct svc_rqst *rqstp, struct svc_fh *fhp, * @offset: Byte offset of start * @payload: xdr_buf containing the write payload * @cnt: IN: number of bytes to write, OUT: number of bytes actually written - * @stable: An NFS stable_how value + * @iocb_flags: VFS IOCB_* flags expressing the requested write stability * @verf: NFS WRITE verifier * * Upon return, caller must invoke fh_put on @fhp. @@ -1638,8 +1628,8 @@ __be32 nfsd_read(struct svc_rqst *rqstp, struct svc_fh *fhp, */ __be32 nfsd_write(struct svc_rqst *rqstp, struct svc_fh *fhp, loff_t offset, - const struct xdr_buf *payload, unsigned long *cnt, int stable, - __be32 *verf) + const struct xdr_buf *payload, unsigned long *cnt, + int iocb_flags, __be32 *verf) { struct nfsd_file *nf; __be32 err; @@ -1651,7 +1641,7 @@ nfsd_write(struct svc_rqst *rqstp, struct svc_fh *fhp, loff_t offset, goto out; err = nfsd_vfs_write(rqstp, fhp, nf, offset, payload, cnt, - stable, verf); + iocb_flags, verf); nfsd_file_put(nf); out: trace_nfsd_write_done(rqstp, fhp, offset, *cnt); diff --git a/fs/nfsd/vfs.h b/fs/nfsd/vfs.h index 5554878781f464..aa7679d4c54adf 100644 --- a/fs/nfsd/vfs.h +++ b/fs/nfsd/vfs.h @@ -135,11 +135,13 @@ __be32 nfsd_read(struct svc_rqst *rqstp, struct svc_fh *fhp, u32 *eof); __be32 nfsd_write(struct svc_rqst *rqstp, struct svc_fh *fhp, loff_t offset, const struct xdr_buf *payload, - unsigned long *cnt, int stable, __be32 *verf); + unsigned long *cnt, int iocb_flags, + __be32 *verf); __be32 nfsd_vfs_write(struct svc_rqst *rqstp, struct svc_fh *fhp, struct nfsd_file *nf, loff_t offset, const struct xdr_buf *payload, - unsigned long *cnt, int stable, __be32 *verf); + unsigned long *cnt, int iocb_flags, + __be32 *verf); __be32 nfsd_readlink(struct svc_rqst *, struct svc_fh *, char *, int *); __be32 nfsd_symlink(struct svc_rqst *, struct svc_fh *, diff --git a/fs/nfsd/xdr3.h b/fs/nfsd/xdr3.h index 344203874b4c47..cad875d1423136 100644 --- a/fs/nfsd/xdr3.h +++ b/fs/nfsd/xdr3.h @@ -39,7 +39,7 @@ struct nfsd3_writeargs { svc_fh fh; __u64 offset; __u32 count; - int stable; + __u32 stable; __u32 len; struct xdr_buf payload; }; diff --git a/include/linux/nfs4.h b/include/linux/nfs4.h index 1a3981c26b23bf..41b7cdcc674f12 100644 --- a/include/linux/nfs4.h +++ b/include/linux/nfs4.h @@ -263,6 +263,12 @@ enum why_no_delegation4 { /* new to v4.1 */ WND4_IS_DIR = 8, }; +enum stable_how4 { + UNSTABLE4 = 0, + DATA_SYNC4 = 1, + FILE_SYNC4 = 2, +}; + enum lock_type4 { NFS4_UNLOCK_LT = 0, NFS4_READ_LT = 1, From 5c41f4cbe2d5f3d6a78d5ac43cf0de5687c9ce7c Mon Sep 17 00:00:00 2001 From: Ameer Hamza Date: Sun, 26 Jul 2026 17:46:58 +0500 Subject: [PATCH 0159/1352] nfsd: fix race between client_info_show() and free_client() client_info_show() renders /proc/fs/nfsd/clients//info and walks clp->cl_sessions under clp->cl_lock to print each session's slot counts. free_client() tears down the same list without taking cl_lock, and is the only unlocked mutator of cl_sessions. A reader can observe a client mid-teardown because get_nfsdfs_clp() pins the nfs4_client but not its sessions: free_client() frees every session before calling nfsd_client_rmdir(), so an in-flight seq_file reader can follow a list_del()'d node whose ->next now holds LIST_POISON1 and take a general protection fault: Oops: general protection fault, probably for non-canonical address 0xdead00000000014c CPU: 1 UID: 0 PID: 132488 Comm: cat RIP: 0010:client_info_show+0x2bf/0x3d0 RAX: dead000000000100 Call Trace: seq_read_iter+0x12a/0x4b0 seq_read+0xf1/0x130 vfs_read+0xbf/0x350 ksys_read+0x6f/0xf0 do_syscall_64+0x8b/0xcb0 entry_SYSCALL_64_after_hwframe+0x76/0x7e Kernel panic - not syncing: Fatal exception Detach cl_sessions onto a local reaplist under cl_lock, then free the sessions after dropping the lock. Removing entries from cl_sessions under cl_lock matches unhash_session(), and the detach-then-reap shape matches how __destroy_client() reaps cl_delegations. The sessions cannot be freed while cl_lock is held, since free_session() calls nfsd4_del_conns(), which re-acquires it. Reported-by: Nicholas Wolff Fixes: 601c8cb349c2 ("nfsd: add session slot count to /proc/fs/nfsd/clients/*/info") Cc: stable@vger.kernel.org Signed-off-by: Ameer Hamza Link: https://patch.msgid.link/20260726124658.1715711-1-ameer.hamza@truenas.com Signed-off-by: Chuck Lever --- fs/nfsd/nfs4state.c | 12 +++++++++--- 1 file changed, 9 insertions(+), 3 deletions(-) diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c index a130d4bcf85031..58e300c2a5ef72 100644 --- a/fs/nfsd/nfs4state.c +++ b/fs/nfsd/nfs4state.c @@ -2791,10 +2791,16 @@ void nfsd4_put_client(struct nfs4_client *clp) static void free_client(struct nfs4_client *clp) { - while (!list_empty(&clp->cl_sessions)) { + LIST_HEAD(reaplist); + + /* client_info_show() walks cl_sessions under cl_lock */ + spin_lock(&clp->cl_lock); + list_splice_init(&clp->cl_sessions, &reaplist); + spin_unlock(&clp->cl_lock); + while (!list_empty(&reaplist)) { struct nfsd4_session *ses; - ses = list_entry(clp->cl_sessions.next, struct nfsd4_session, - se_perclnt); + ses = list_entry(reaplist.next, struct nfsd4_session, + se_perclnt); list_del(&ses->se_perclnt); WARN_ON_ONCE(atomic_read(&ses->se_ref)); free_session(ses); From 51751892329a6094d6b2fb053e0b7a518399a5d7 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Tue, 28 Jul 2026 12:59:07 -0400 Subject: [PATCH 0160/1352] NFSD: Move the RPC program definition for LOCALIO Clean up: The definitions for the LOCALIO program are not needed by most files that include linux/nfs.h. Following the convention used by most other in-kernel RPC program implementations, relocate the LOCALIO program definitions to a localio-specific header. Reviewed-by: NeilBrown Reviewed-by: Mike Snitzer Link: https://patch.msgid.link/20260728165911.462534-2-cel@kernel.org Signed-off-by: Chuck Lever --- include/linux/nfs.h | 7 ------- include/linux/nfslocalio.h | 8 ++++++++ 2 files changed, 8 insertions(+), 7 deletions(-) diff --git a/include/linux/nfs.h b/include/linux/nfs.h index 0e2b210c103b69..8c2818db43c51f 100644 --- a/include/linux/nfs.h +++ b/include/linux/nfs.h @@ -15,11 +15,4 @@ #include -/* The LOCALIO program is entirely private to Linux and is - * NOT part of the uapi. - */ -#define NFS_LOCALIO_PROGRAM 400122 -#define LOCALIOPROC_NULL 0 -#define LOCALIOPROC_UUID_IS_LOCAL 1 - #endif /* _LINUX_NFS_H */ diff --git a/include/linux/nfslocalio.h b/include/linux/nfslocalio.h index 3d91043254e64a..d2b39e6e6c6acf 100644 --- a/include/linux/nfslocalio.h +++ b/include/linux/nfslocalio.h @@ -16,6 +16,14 @@ #include #include +/* + * The LOCALIO program is entirely private to Linux and is NOT part of + * the uapi. + */ +#define NFS_LOCALIO_PROGRAM 400122 +#define LOCALIOPROC_NULL 0 +#define LOCALIOPROC_UUID_IS_LOCAL 1 + struct nfs_client; struct nfs_file_localio; From 7bb4c6963815374dd230df7734cca2b0061b433b Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Tue, 28 Jul 2026 12:59:08 -0400 Subject: [PATCH 0161/1352] nfs_common: Remove "#include " from linux/nfslocalio.h Clean up: linux/nfslocalio.h pulls in linux/nfs.h only for the definition of struct nfs_fh, which now lives in linux/nfs_fh.h. Replace linux/nfs.h with linux/nfs_fh.h so that nfslocalio.h no longer carries uapi/linux/nfs.h into its consumers. Reviewed-by: NeilBrown Reviewed-by: Mike Snitzer Link: https://patch.msgid.link/20260728165911.462534-3-cel@kernel.org Signed-off-by: Chuck Lever --- include/linux/nfslocalio.h | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/include/linux/nfslocalio.h b/include/linux/nfslocalio.h index d2b39e6e6c6acf..8ce4d978a6367e 100644 --- a/include/linux/nfslocalio.h +++ b/include/linux/nfslocalio.h @@ -13,7 +13,8 @@ #include #include #include -#include +#include + #include /* From c7017a5e090ba276b629c5ac1a12a6009923b6a8 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Tue, 28 Jul 2026 12:59:09 -0400 Subject: [PATCH 0162/1352] NFSD: Tighten header includes in localio.c As a prerequisite to converting NFSD to use xdrgen more broadly, NFSD source files should not depend on NFS client headers. fs/nfsd/localio.c is server-side LOCALIO code, yet it pulled in three of them: , the client inode header (struct nfs_inode, NFS_I(), writeback helpers), which server code never uses; , whose only referenced symbol is decode_opaque_fixed(), a static inline that exists to remap the error return to -EIO for client call sites; and the catch-all . Convert the UUID decoder to call the canonical SUNRPC primitive xdr_stream_decode_opaque_fixed() directly. It is shared by client and server, performs the identical bounds check, and is already reachable through . With the wrapper gone, localio.c references no symbol from , and with that header gone, none of the NFSv3 definitions its structs embed are needed here. Drop all three client includes and add what the file actually uses: struct nfs_fh comes from , included directly rather than through nfslocalio.h's conditional re-export, and NFS4_FHSIZE from . enum nfs_stat and nfs_stat_to_errno continue to come from the already-included . Reviewed-by: NeilBrown Reviewed-by: Mike Snitzer Link: https://patch.msgid.link/20260728165911.462534-4-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/localio.c | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/fs/nfsd/localio.c b/fs/nfsd/localio.c index c458c01e94783a..4110be02b75092 100644 --- a/fs/nfsd/localio.c +++ b/fs/nfsd/localio.c @@ -11,11 +11,10 @@ #include #include #include -#include +#include #include +#include #include -#include -#include #include #include "nfsd.h" @@ -179,7 +178,7 @@ static bool localio_decode_uuidarg(struct svc_rqst *rqstp, struct localio_uuidarg *argp = rqstp->rq_argp; u8 uuid[UUID_SIZE]; - if (decode_opaque_fixed(xdr, uuid, UUID_SIZE)) + if (xdr_stream_decode_opaque_fixed(xdr, uuid, UUID_SIZE) < 0) return false; import_uuid(&argp->uuid, uuid); From 35cff9c29ba33afd2011b120f5a2b266076aae90 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Tue, 28 Jul 2026 12:59:10 -0400 Subject: [PATCH 0163/1352] NFSD: Name the fh_maxsize value that carries no NFS version nfsd_set_fh_dentry() selects behavior specific to an NFS protocol version by matching fh_maxsize against NFS_FHSIZE, NFS3_FHSIZE, or NFS4_FHSIZE. A filehandle that reaches NFSD outside an NFS request has no such version. nlm_fopen() opts out of the switch by passing a bare 0, which matches no arm, and the literal says nothing about why, so an adjacent comment has to carry it. Reviewed-by: NeilBrown Reviewed-by: Mike Snitzer Link: https://patch.msgid.link/20260728165911.462534-5-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/lockd.c | 3 +-- fs/nfsd/nfsfh.c | 2 ++ fs/nfsd/nfsfh.h | 14 ++++++++++++++ 3 files changed, 17 insertions(+), 2 deletions(-) diff --git a/fs/nfsd/lockd.c b/fs/nfsd/lockd.c index f5a4f352f8abf9..5ec0f545606323 100644 --- a/fs/nfsd/lockd.c +++ b/fs/nfsd/lockd.c @@ -34,8 +34,7 @@ static int nlm_fopen(struct svc_rqst *rqstp, struct nfs_fh *f, int access; struct svc_fh fh; - /* must initialize before using! but maxsize doesn't matter */ - fh_init(&fh,0); + fh_init(&fh, NFSD_FHSIZE_UNSPEC); fh.fh_handle.fh_size = f->size; memcpy(&fh.fh_handle.fh_raw, f->data, f->size); fh.fh_export = NULL; diff --git a/fs/nfsd/nfsfh.c b/fs/nfsd/nfsfh.c index fd721a5a6b37bd..b1f3c22af52586 100644 --- a/fs/nfsd/nfsfh.c +++ b/fs/nfsd/nfsfh.c @@ -335,6 +335,8 @@ static __be32 nfsd_set_fh_dentry(struct svc_rqst *rqstp, struct net *net, } switch (fhp->fh_maxsize) { + case NFSD_FHSIZE_UNSPEC: + break; case NFS4_FHSIZE: if (dentry->d_sb->s_export_op->flags & EXPORT_OP_NOATOMIC_ATTR) fhp->fh_no_atomic_attr = true; diff --git a/fs/nfsd/nfsfh.h b/fs/nfsd/nfsfh.h index ab15b59ac7b3ba..7d8e3f0153073f 100644 --- a/fs/nfsd/nfsfh.h +++ b/fs/nfsd/nfsfh.h @@ -246,6 +246,20 @@ fh_copy_shallow(struct knfsd_fh *dst, const struct knfsd_fh *src) memcpy(&dst->fh_raw, &src->fh_raw, src->fh_size); } +#define NFSD_FHSIZE_UNSPEC 0 + +/** + * fh_init - Prepare a file handle for fh_compose() or fh_verify() + * @fhp: File handle to initialize + * @maxsize: Largest file handle, in bytes, to build in @fhp + * + * @maxsize bounds the handle fh_compose() may build: NFS_FHSIZE, + * NFS3_FHSIZE, and NFS4_FHSIZE additionally select version-specific + * handling in fh_verify(). Callers that only verify an incoming + * handle pass NFSD_FHSIZE_UNSPEC, which cannot be composed. + * + * Return: @fhp + */ static __inline__ struct svc_fh * fh_init(struct svc_fh *fhp, int maxsize) { From f47b7fb268bcd6efa2b0c93a470f89040ad51103 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Tue, 28 Jul 2026 12:59:11 -0400 Subject: [PATCH 0164/1352] NFSD: Don't apply NFS version-specific behavior to LOCALIO requests LOCALIO serves NFS clients of every version through one entry point, so no protocol version is associated with such a request. nfsd_set_fh_dentry() selects version-specific behavior anyway: its switch keys off fh_maxsize, and nfsd_open_local_fh() passes NFS4_FHSIZE because that is the size of the buffer it copies into, so LOCALIO lands in the NFSv4 arm. fh_getattr() keys off fh_maxsize too and does run on a LOCALIO open, adding STATX_BTIME and STATX_CHANGE_COOKIE to the mask it requests: work on filesystems that compute them for a caller that never reads them. nfsd_open_local_fh() only verifies a handle it received, so it has no maximum size to state. Pass NFSD_FHSIZE_UNSPEC as nlm_fopen() already does, which selects the switch arm that applies no version-specific behavior, and state the bound on the copy out of struct nfs_fh as NFS_MAXFHSIZE. Suggested-by: NeilBrown Reviewed-by: NeilBrown Reviewed-by: Mike Snitzer Link: https://patch.msgid.link/20260728165911.462534-6-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/localio.c | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/fs/nfsd/localio.c b/fs/nfsd/localio.c index 4110be02b75092..33b56d1b3f440b 100644 --- a/fs/nfsd/localio.c +++ b/fs/nfsd/localio.c @@ -11,7 +11,6 @@ #include #include #include -#include #include #include #include @@ -54,7 +53,7 @@ nfsd_open_local_fh(struct net *net, struct auth_domain *dom, struct nfsd_file *localio; __be32 beres; - if (nfs_fh->size > NFS4_FHSIZE) + if (nfs_fh->size > NFS_MAXFHSIZE) return ERR_PTR(-EINVAL); if (!nfsd_net_try_get(net)) @@ -67,7 +66,7 @@ nfsd_open_local_fh(struct net *net, struct auth_domain *dom, return localio; /* nfs_fh -> svc_fh */ - fh_init(&fh, NFS4_FHSIZE); + fh_init(&fh, NFSD_FHSIZE_UNSPEC); fh.fh_handle.fh_size = nfs_fh->size; memcpy(fh.fh_handle.fh_raw, nfs_fh->data, nfs_fh->size); From 675cc207187b0d0ce59365924aa3c0a50ff9dd04 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sun, 2 Aug 2026 13:04:28 -0400 Subject: [PATCH 0165/1352] NFSD: Budget the CB_SEQUENCE opcode and referring call array count cb_sequence_enc_sz counts the session ID, the four scalar fields, and one referring call list. encode_cb_sequence4args() also emits the CB_SEQUENCE opcode and the csa_referring_call_lists array count, so the macro falls two XDR words short. Every NFS4_enc_cb_*_sz built on it is short by the same two words. NFSD_CB_MAX_REQ_SZ derives from NFS4_enc_cb_recall_sz, so the two missing CB_SEQUENCE words shrink the ca_maxrequestsize that check_backchannel_attrs() accepts by eight bytes. Count both words. The minimum a client must advertise rises by those eight bytes. The short count cannot overrun the send buffer. The macro sizes p_arglen, and rq_callsize adds two credential slacks on top of that. The only client affected is one whose ca_maxrequestsize falls inside those eight bytes. No backport is needed. Link: https://patch.msgid.link/20260802-nfsd-deleg-destroy-badhandle-v1-1-323aa7196055@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/xdr4cb.h | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/fs/nfsd/xdr4cb.h b/fs/nfsd/xdr4cb.h index b06d0170d7c43b..04d3e321a972db 100644 --- a/fs/nfsd/xdr4cb.h +++ b/fs/nfsd/xdr4cb.h @@ -6,14 +6,14 @@ #define cb_compound_enc_hdr_sz 4 #define cb_compound_dec_hdr_sz (3 + (NFS4_MAXTAGLEN >> 2)) #define sessionid_sz (NFS4_MAX_SESSIONID_LEN >> 2) +#define op_enc_sz 1 #define enc_referring_call4_sz (1 + 1) #define enc_referring_call_list4_sz (sessionid_sz + 1 + \ enc_referring_call4_sz) -#define cb_sequence_enc_sz (sessionid_sz + 4 + \ - enc_referring_call_list4_sz) +#define cb_sequence_enc_sz (op_enc_sz + sessionid_sz + 4 + \ + 1 + enc_referring_call_list4_sz) #define cb_sequence_dec_sz (op_dec_sz + sessionid_sz + 4) -#define op_enc_sz 1 #define op_dec_sz 2 #define enc_nfs4_fh_sz (1 + (NFS4_FHSIZE >> 2)) #define enc_stateid_sz (NFS4_STATEID_SIZE >> 2) From e0399b80d1aca9847bc281faa5f3b0803404f062 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sun, 2 Aug 2026 13:04:29 -0400 Subject: [PATCH 0166/1352] NFSD: Budget the CB_RECALL truncate field NFS4_enc_cb_recall_sz counts the CB_RECALL opcode, the stateid, and the file handle. encode_cb_recall4args() also emits the truncate field, so the macro falls one XDR word short. NFSD_CB_MAX_REQ_SZ derives from this macro, so the minimum ca_maxrequestsize a client must advertise rises by four bytes. The field has been unbudgeted since the macro was written. Neither consumer of the macro justifies a backport. rq_callsize covers p_arglen with two credential slacks. The only client affected is one whose ca_maxrequestsize falls inside those four bytes. Link: https://patch.msgid.link/20260802-nfsd-deleg-destroy-badhandle-v1-2-323aa7196055@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/xdr4cb.h | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/nfsd/xdr4cb.h b/fs/nfsd/xdr4cb.h index 04d3e321a972db..a7c8dc355d1aa8 100644 --- a/fs/nfsd/xdr4cb.h +++ b/fs/nfsd/xdr4cb.h @@ -19,8 +19,8 @@ #define enc_stateid_sz (NFS4_STATEID_SIZE >> 2) #define NFS4_enc_cb_recall_sz (cb_compound_enc_hdr_sz + \ cb_sequence_enc_sz + \ - 1 + enc_stateid_sz + \ - enc_nfs4_fh_sz) + op_enc_sz + enc_stateid_sz + \ + 1 + enc_nfs4_fh_sz) #define NFS4_dec_cb_recall_sz (cb_compound_dec_hdr_sz + \ cb_sequence_dec_sz + \ From d557a0da5d121c4b6d91376353bc9a3df9cb2c79 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sun, 2 Aug 2026 13:04:30 -0400 Subject: [PATCH 0167/1352] NFSD: Budget the CB_LAYOUTRECALL recall stateid NFS4_enc_cb_layout_sz counts the opcode, the three scalar fields, the file handle, and the offset and length hypers. encode_cb_layout4args() also emits the layoutrecall4 discriminator and the recall stateid, so the macro falls five XDR words short. This macro sizes p_arglen and nothing else. rq_callsize pads that with two credential slacks, so the shortfall has never reached the send buffer. No backport is needed. Link: https://patch.msgid.link/20260802-nfsd-deleg-destroy-badhandle-v1-3-323aa7196055@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/xdr4cb.h | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/fs/nfsd/xdr4cb.h b/fs/nfsd/xdr4cb.h index a7c8dc355d1aa8..5285934b3a54b9 100644 --- a/fs/nfsd/xdr4cb.h +++ b/fs/nfsd/xdr4cb.h @@ -27,8 +27,9 @@ op_dec_sz) #define NFS4_enc_cb_layout_sz (cb_compound_enc_hdr_sz + \ cb_sequence_enc_sz + \ - 1 + 3 + \ - enc_nfs4_fh_sz + 4) + op_enc_sz + 3 + 1 + \ + enc_nfs4_fh_sz + 4 + \ + enc_stateid_sz) #define NFS4_dec_cb_layout_sz (cb_compound_dec_hdr_sz + \ cb_sequence_dec_sz + \ op_dec_sz) From 409f8aaa74b80064f1638bdda4e04f1aa7a7a2d2 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sun, 2 Aug 2026 13:04:31 -0400 Subject: [PATCH 0168/1352] NFSD: Budget the CB_OFFLOAD opcode NFS4_enc_cb_offload_sz counts the file handle, the stateid, and the offload information. encode_cb_offload4args() emits an opcode ahead of all three, so the macro falls one XDR word short. This macro sizes p_arglen and nothing else. rq_callsize pads that with two credential slacks, so the shortfall has never reached the send buffer. No backport is needed. Link: https://patch.msgid.link/20260802-nfsd-deleg-destroy-badhandle-v1-4-323aa7196055@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/xdr4cb.h | 1 + 1 file changed, 1 insertion(+) diff --git a/fs/nfsd/xdr4cb.h b/fs/nfsd/xdr4cb.h index 5285934b3a54b9..21d8280edead82 100644 --- a/fs/nfsd/xdr4cb.h +++ b/fs/nfsd/xdr4cb.h @@ -58,6 +58,7 @@ XDR_QUADLEN(NFS4_VERIFIER_SIZE)) #define NFS4_enc_cb_offload_sz (cb_compound_enc_hdr_sz + \ cb_sequence_enc_sz + \ + op_enc_sz + \ enc_nfs4_fh_sz + \ enc_stateid_sz + \ enc_cb_offload_info_sz) From 7cec21e7488c4de99e90561395ee23c141148e40 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sun, 2 Aug 2026 13:04:32 -0400 Subject: [PATCH 0169/1352] NFSD: Budget the CB_NOTIFY_LOCK opcode NFS4_enc_cb_notify_lock_sz counts the lock owner and the file handle. nfs4_xdr_enc_cb_notify_lock() emits an opcode ahead of both, so the macro falls one XDR word short. This macro sizes p_arglen and nothing else. rq_callsize pads that with two credential slacks, so the shortfall has never reached the send buffer. No backport is needed. Link: https://patch.msgid.link/20260802-nfsd-deleg-destroy-badhandle-v1-5-323aa7196055@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/xdr4cb.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/nfsd/xdr4cb.h b/fs/nfsd/xdr4cb.h index 21d8280edead82..01bb54f03d2577 100644 --- a/fs/nfsd/xdr4cb.h +++ b/fs/nfsd/xdr4cb.h @@ -48,7 +48,7 @@ #define NFS4_enc_cb_notify_lock_sz (cb_compound_enc_hdr_sz + \ cb_sequence_enc_sz + \ - 2 + 1 + \ + op_enc_sz + 2 + 1 + \ XDR_QUADLEN(NFS4_OPAQUE_LIMIT) + \ enc_nfs4_fh_sz) #define NFS4_dec_cb_notify_lock_sz (cb_compound_dec_hdr_sz + \ From 2c063b0a14cda4ee899a34e8e40816caea7cb1cd Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sun, 2 Aug 2026 13:04:33 -0400 Subject: [PATCH 0170/1352] NFSD: Budget the CB_RECALL_ANY opcode NFS4_enc_cb_recall_any_sz counts the objects-to-keep field, the bitmap array length, and the bitmap word. encode_cb_recallany4args() emits an opcode ahead of all three, so the macro falls one XDR word short. This macro sizes p_arglen and nothing else. rq_callsize pads that with two credential slacks, so the shortfall has never reached the send buffer. No backport is needed. Link: https://patch.msgid.link/20260802-nfsd-deleg-destroy-badhandle-v1-6-323aa7196055@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/xdr4cb.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/nfsd/xdr4cb.h b/fs/nfsd/xdr4cb.h index 01bb54f03d2577..838f8629821f48 100644 --- a/fs/nfsd/xdr4cb.h +++ b/fs/nfsd/xdr4cb.h @@ -67,7 +67,7 @@ op_dec_sz) #define NFS4_enc_cb_recall_any_sz (cb_compound_enc_hdr_sz + \ cb_sequence_enc_sz + \ - 1 + 1 + 1) + op_enc_sz + 1 + 1 + 1) #define NFS4_dec_cb_recall_any_sz (cb_compound_dec_hdr_sz + \ cb_sequence_dec_sz + \ op_dec_sz) From 10b5466de3de1cdbda2b8471d7a6db1df0aa3145 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sun, 2 Aug 2026 13:04:34 -0400 Subject: [PATCH 0171/1352] NFSD: Correct locking documentation for delegation sc_status The comment above the SC_STATUS_ flags states that nn->deleg_lock protects sc_status for delegation stateids, but only the transitions made while a delegation is hashed are taken under that lock. This comment was accurate until commit c88c150a467f ("nfsd: fix possible badness in FREE_STATEID") set SC_STATUS_CLOSED under ->cl_lock. Commit 8dd91e8d31fe ("nfsd: fix race between laundromat and free_stateid") added the other two sites. Link: https://patch.msgid.link/20260802-nfsd-deleg-destroy-badhandle-v1-7-323aa7196055@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/state.h | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/fs/nfsd/state.h b/fs/nfsd/state.h index c4627dc91e2067..42d3320622eb9c 100644 --- a/fs/nfsd/state.h +++ b/fs/nfsd/state.h @@ -145,10 +145,13 @@ struct nfs4_stid { #define SC_TYPE_COPY BIT(4) unsigned short sc_type; -/* nn->deleg_lock protects sc_status for delegation stateids. - * ->cl_lock protects sc_status for open and lock stateids. - * ->st_mutex also protect sc_status for open stateids. - * ->ls_lock protects sc_status for layout stateids. +/* + * nn->deleg_lock protects sc_status for hashed delegation stateids. + * ->cl_lock protects the bits set as one is disposed of + * (SC_STATUS_CLOSED, SC_STATUS_FREEABLE, SC_STATUS_FREED) and + * sc_status for open and lock stateids. ->st_mutex also protects + * sc_status for open stateids. ->ls_lock protects sc_status for + * layout stateids. */ /* * For an open stateid kept around *only* to process close replays. From a95160a2537aff9c287d203c36d9b7caf6cb3281 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sun, 2 Aug 2026 13:04:35 -0400 Subject: [PATCH 0172/1352] NFSD: Destroy a recalled delegation the client does not hold A client that answers CB_RECALL with NFS4ERR_BADHANDLE or NFS4ERR_BAD_STATEID has no record of the delegation, so the FREE_STATEID that clears it from cl_revoked never arrives. Every later SEQUENCE reply carries SEQ4_STATUS_RECALLABLE_STATE_REVOKED, and the client loops issuing TEST_STATEID. Destroy such a delegation when it is reaped rather than revoking it onto cl_revoked. RFC 8881 Section 20.2.4 completes the recall at the reply when its status is neither NFS4_OK nor NFS4ERR_DELAY, so a rejected recall leaves nothing to revoke. An administrative revoke keeps that path, since NFS4ERR_ADMIN_REVOKED reports it. A destroyed stateid returns NFS4ERR_BAD_STATEID instead of NFS4ERR_DELEG_REVOKED. A client that rejects the recall but still holds the delegation gets no notice that its state was revoked. CB_RECALL can outrun the reply that granted the delegation, so honor a rejection only once the client has seen that grant. Per RFC 8881 Section 2.10.6.3, retirement of the slot that carried the grant is that proof; retry until then, and revoke when the retries lapse. Fixes: 3bd64a5ba171 ("nfsd4: implement SEQ4_STATUS_RECALLABLE_STATE_REVOKED") Cc: stable@vger.kernel.org # 6.14.x Link: https://patch.msgid.link/20260802-nfsd-deleg-destroy-badhandle-v1-8-323aa7196055@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfs4state.c | 198 +++++++++++++++++++++++++++++++++++++------- fs/nfsd/state.h | 13 ++- 2 files changed, 178 insertions(+), 33 deletions(-) diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c index 58e300c2a5ef72..3283b2225b7652 100644 --- a/fs/nfsd/nfs4state.c +++ b/fs/nfsd/nfs4state.c @@ -94,6 +94,8 @@ static void nfsd4_end_grace(struct nfsd_net *nn); static void _free_cpntf_state_locked(struct nfsd_net *nn, struct nfs4_cpntf_state *cps); static void nfsd4_file_hash_remove(struct nfs4_file *fi); static void deleg_reaper(struct nfsd_net *nn); +static void nfsd4_drop_revoked_stid(struct nfs4_stid *s) + __releases(&s->sc_client->cl_lock); static const struct lease_manager_operations nfsd_lease_mng_ops; @@ -1281,6 +1283,9 @@ __alloc_init_deleg(struct nfs4_client *clp, struct nfs4_file *fp, dp->dl_type = dl_type; dp->dl_retries = 1; dp->dl_recalled = false; + dp->dl_recall_rejected = false; + dp->dl_recall_grant.valid = false; + dp->dl_recall_grant.retired_at_send = false; get_nfs4_file(fp); dp->dl_stid.sc_file = fp; nfsd4_init_cb(&dp->dl_recall, dp->dl_stid.sc_client, @@ -1565,27 +1570,22 @@ static void destroy_delegation(struct nfs4_delegation *dp) } /** - * revoke_delegation - perform nfs4 delegation structure cleanup - * @dp: pointer to the delegation + * revoke_delegation - dispose of a delegation the server has revoked + * @dp: delegation to dispose of + * + * The caller holds a reference on @dp, which this function consumes. + * On NFSv4.1 and newer, @dp's sc_status must already carry + * SC_STATUS_REVOKED or SC_STATUS_ADMIN_REVOKED. * - * This function assumes that it's called either from the administrative - * interface (nfsd4_revoke_states()) that's revoking a specific delegation - * stateid or it's called from a laundromat thread (nfsd4_landromat()) that - * determined that this specific state has expired and needs to be revoked - * (both mark state with the appropriate stid sc_status mode). It is also - * assumed that a reference was taken on the @dp state. This function - * consumes that reference. + * @dp is parked on the client's cl_revoked list to await a FREE_STATEID. + * Where none can arrive, @dp is destroyed here instead: FREE_STATEID has + * already freed it, or the client rejected the recall with + * NFS4ERR_BADHANDLE or NFS4ERR_BAD_STATEID and holds no record of the + * delegation. NFS4ERR_ADMIN_REVOKED still prompts one, so an + * administrative revoke waits on cl_revoked. * - * If this function finds that the @dp state is SC_STATUS_FREED it means - * that a FREE_STATEID operation for this stateid has been processed and - * we can proceed to removing it from recalled list. However, if @dp state - * isn't marked SC_STATUS_FREED, it means we need place it on the cl_revoked - * list and wait for the FREE_STATEID to arrive from the client. At the same - * time, we need to mark it as SC_STATUS_FREEABLE to indicate to the - * nfsd4_free_stateid() function that this stateid has already been added - * to the cl_revoked list and that nfsd4_free_stateid() is now responsible - * for removing it from the list. Inspection of where the delegation state - * in the revocation process is protected by the clp->cl_lock. + * Context: Takes and releases the client's cl_lock; may sleep after + * dropping it. */ static void revoke_delegation(struct nfs4_delegation *dp) { @@ -1603,6 +1603,19 @@ static void revoke_delegation(struct nfs4_delegation *dp) list_del_init(&dp->dl_recall_lru); goto out; } + if (dp->dl_recall_rejected && + !(dp->dl_stid.sc_status & SC_STATUS_ADMIN_REVOKED)) { + /* + * SC_STATUS_CLOSED, set under cl_lock, makes a racing + * FREE_STATEID bail out rather than drop this reference + * too. The put releases what cl_revoked would have held. + */ + dp->dl_stid.sc_status |= SC_STATUS_CLOSED; + spin_unlock(&clp->cl_lock); + nfs4_put_stid(&dp->dl_stid); + destroy_unhashed_deleg(dp); + return; + } list_add(&dp->dl_recall_lru, &clp->cl_revoked); dp->dl_stid.sc_status |= SC_STATUS_FREEABLE; out: @@ -2897,11 +2910,18 @@ __destroy_client(struct nfs4_client *clp) list_del_init(&dp->dl_recall_lru); destroy_unhashed_deleg(dp); } + /* + * A CB_RECALL reply can release revoked delegations concurrently: + * nfsd4_shutdown_callback() has not run yet. + */ + spin_lock(&clp->cl_lock); while (!list_empty(&clp->cl_revoked)) { dp = list_entry(clp->cl_revoked.next, struct nfs4_delegation, dl_recall_lru); - list_del_init(&dp->dl_recall_lru); - nfs4_put_stid(&dp->dl_stid); + /* this function drops ->cl_lock */ + nfsd4_drop_revoked_stid(&dp->dl_stid); + spin_lock(&clp->cl_lock); } + spin_unlock(&clp->cl_lock); while (!list_empty(&clp->cl_openowners)) { oo = list_entry(clp->cl_openowners.next, struct nfs4_openowner, oo_perclient); nfs4_get_stateowner(&oo->oo_owner); @@ -6071,6 +6091,60 @@ bool nfsd_wait_for_delegreturn(struct svc_rqst *rqstp, struct inode *inode) return timeo > 0; } +static bool nfsd4_recall_grant_slot_retired(struct nfs4_delegation *dp) +{ + struct nfs4_client *clp = dp->dl_stid.sc_client; + struct nfsd_net *nn = net_generic(clp->net, nfsd_net_id); + struct nfsd4_session *ses; + struct nfsd4_sessionid sid; + bool retired = false; + void *entry; + + if (!dp->dl_recall_grant.valid) + return false; + + /* + * gen_sessionid() composes a sessionid from the client's clientid + * and a sequence counter, so the sequence alone identifies the + * granting session. + */ + sid.clientid = clp->cl_clientid; + sid.sequence = dp->dl_recall_grant.sessionid_seq; + sid.reserved = 0; + + /* + * A missing session does not prove the client saw the grant: a + * DESTROY_SESSION unhashes its own session before the reply to + * that compound is encoded. + */ + spin_lock(&nn->client_lock); + ses = __find_in_sessionid_hashtbl((struct nfs4_sessionid *)&sid, + clp->net); + entry = ses ? xa_load(&ses->se_slots, dp->dl_recall_grant.slotid) : NULL; + if (xa_is_value(entry)) { + /* + * A slot is freed only once the client has acknowledged + * the smaller slot table, which it cannot do while a + * request on that slot is outstanding. + */ + retired = true; + } else if (entry) { + struct nfsd4_slot *slot = entry; + + /* + * A reactivated slot was freed and rebuilt, so the same + * acknowledgment applies. The seqid test errs toward + * revoking: a rebuilt slot restarting at seqid 1 matches + * an old grant. + */ + retired = (slot->sl_flags & NFSD4_SLOT_REUSED) || + ((slot->sl_flags & NFSD4_SLOT_INITIALIZED) && + slot->sl_seqid != dp->dl_recall_grant.seqid); + } + spin_unlock(&nn->client_lock); + return retired; +} + static bool nfsd4_cb_recall_prepare(struct nfsd4_callback *cb) { struct nfs4_delegation *dp = cb_to_delegation(cb); @@ -6092,9 +6166,37 @@ static bool nfsd4_cb_recall_prepare(struct nfsd4_callback *cb) list_add_tail(&dp->dl_recall_lru, &nn->del_recall_lru); } spin_unlock(&nn->deleg_lock); + + dp->dl_recall_grant.retired_at_send = + nfsd4_recall_grant_slot_retired(dp); return true; } +/* + * cl_lock orders this against a laundromat reaping @dp: either + * revoke_delegation() observes dl_recall_rejected and destroys @dp, or + * it reached cl_revoked first and @dp is released here instead. + */ +static void nfsd4_deleg_recall_rejected(struct nfs4_delegation *dp) +{ + struct nfs4_client *clp = dp->dl_stid.sc_client; + + spin_lock(&clp->cl_lock); + if (dp->dl_stid.sc_status & (SC_STATUS_CLOSED | SC_STATUS_FREED | + SC_STATUS_ADMIN_REVOKED)) { + spin_unlock(&clp->cl_lock); + return; + } + if (dp->dl_stid.sc_status & SC_STATUS_FREEABLE) { + dp->dl_stid.sc_status |= SC_STATUS_CLOSED; + /* this function drops ->cl_lock */ + nfsd4_drop_revoked_stid(&dp->dl_stid); + return; + } + dp->dl_recall_rejected = true; + spin_unlock(&clp->cl_lock); +} + static int nfsd4_cb_recall_done(struct nfsd4_callback *cb, struct rpc_task *task) { @@ -6102,27 +6204,33 @@ static int nfsd4_cb_recall_done(struct nfsd4_callback *cb, trace_nfsd_cb_recall_done(&dp->dl_stid.sc_stateid, task); - if (dp->dl_stid.sc_status) - /* CLOSED or REVOKED */ - return 1; - switch (task->tk_status) { case 0: return 1; case -NFS4ERR_DELAY: + if (dp->dl_stid.sc_status) + /* CLOSED or REVOKED */ + return 1; rpc_delay(task, 2 * HZ); return 0; case -EBADHANDLE: case -NFS4ERR_BAD_STATEID: /* - * Race: client probably got cb_recall before open reply - * granting delegation. + * Retirement of the granting slot proves the client saw + * the grant. Trust the rejection only if the slot had + * retired when this recall was sent. */ - if (dp->dl_retries--) { + if (dp->dl_recall_grant.retired_at_send) { + nfsd4_deleg_recall_rejected(dp); + return 1; + } + if (!dp->dl_stid.sc_status && dp->dl_retries--) { + dp->dl_recall_grant.retired_at_send = + nfsd4_recall_grant_slot_retired(dp); rpc_delay(task, 2 * HZ); return 0; } - fallthrough; + return 1; default: return 1; } @@ -6716,9 +6824,25 @@ static bool nfsd4_want_deleg_timestamps(const struct nfsd4_open *open) return open->op_deleg_want & OPEN4_SHARE_ACCESS_WANT_DELEG_TIMESTAMPS; } +static void +nfs4_delegation_record_grant_slot(struct nfs4_delegation *dp, + const struct nfsd4_compound_state *cstate) +{ + const struct nfsd4_sessionid *sid; + + if (!cstate->session) + return; + sid = (struct nfsd4_sessionid *)cstate->session->se_sessionid.data; + dp->dl_recall_grant.sessionid_seq = sid->sequence; + dp->dl_recall_grant.slotid = cstate->slot->sl_index; + dp->dl_recall_grant.seqid = cstate->slot->sl_seqid; + dp->dl_recall_grant.valid = true; +} + static struct nfs4_delegation * -nfs4_set_delegation(struct nfsd4_open *open, struct nfs4_ol_stateid *stp, - struct svc_fh *parent) +nfs4_set_delegation(struct nfsd4_open *open, + const struct nfsd4_compound_state *cstate, + struct nfs4_ol_stateid *stp, struct svc_fh *parent) { bool deleg_ts = nfsd4_want_deleg_timestamps(open); struct nfs4_client *clp = stp->st_stid.sc_client; @@ -6808,6 +6932,14 @@ nfs4_set_delegation(struct nfsd4_open *open, struct nfs4_ol_stateid *stp, dp = alloc_init_deleg(clp, fp, odstate, dl_type); if (!dp) goto out_delegees; + + /* + * Record the granting slot before kernel_setlease() makes @dp + * visible to lease breakers. A conflicting open can drive + * CB_RECALL to completion from that point on. + */ + nfs4_delegation_record_grant_slot(dp, cstate); + if (stp->st_stid.sc_export) dp->dl_stid.sc_export = exp_get(stp->st_stid.sc_export); @@ -6972,6 +7104,7 @@ nfs4_open_delegation(struct svc_rqst *rqstp, struct nfsd4_open *open, struct nfs4_ol_stateid *stp, struct svc_fh *currentfh, struct svc_fh *fh) { + struct nfsd4_compoundres *resp = rqstp->rq_resp; struct nfs4_openowner *oo = openowner(stp->st_stateowner); bool deleg_ts = nfsd4_want_deleg_timestamps(open); struct nfs4_client *clp = stp->st_stid.sc_client; @@ -7008,7 +7141,7 @@ nfs4_open_delegation(struct svc_rqst *rqstp, struct nfsd4_open *open, default: goto out_no_deleg; } - dp = nfs4_set_delegation(open, stp, parent); + dp = nfs4_set_delegation(open, &resp->cstate, stp, parent); if (IS_ERR(dp)) goto out_no_deleg; @@ -10302,6 +10435,7 @@ nfsd_get_dir_deleg(struct nfsd4_compound_state *cstate, dp = alloc_init_dir_deleg(clp, fp); if (!dp) goto out_delegees; + nfs4_delegation_record_grant_slot(dp, cstate); if (cstate->current_fh.fh_export) dp->dl_stid.sc_export = exp_get(cstate->current_fh.fh_export); diff --git a/fs/nfsd/state.h b/fs/nfsd/state.h index 42d3320622eb9c..ff1c9fa731aa25 100644 --- a/fs/nfsd/state.h +++ b/fs/nfsd/state.h @@ -292,7 +292,8 @@ struct nfsd4_cb_notify { * If the server attempts to recall a delegation and the client doesn't do so * before a timeout, the server may also revoke the delegation. In that case, * the object will either be destroyed (v4.0) or moved to a per-client list of - * revoked delegations (v4.1+). + * revoked delegations (v4.1+). A v4.1+ client that rejects the recall holds + * no record of the delegation, so the object is destroyed rather than listed. * * This object is a superset of the nfs4_stid. */ @@ -308,9 +309,19 @@ struct nfs4_delegation { int dl_retries; struct nfsd4_callback dl_recall; bool dl_recalled; + bool dl_recall_rejected; bool dl_written; bool dl_setattr; + /* Forward-channel slot that carried the granting request */ + struct { + u32 sessionid_seq; + u32 slotid; + u32 seqid; + bool valid; + bool retired_at_send; + } dl_recall_grant; + union { /* for CB_GETATTR */ struct nfs4_cb_fattr dl_cb_fattr; From 745ae64ede37292b6874b4470305de16aa849af0 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sun, 2 Aug 2026 13:04:36 -0400 Subject: [PATCH 0173/1352] NFSD: Send referring calls with CB_RECALL When CB_RECALL races ahead of the reply that granted the delegation, the client has not yet recorded the delegation stateid and responds NFS4ERR_BADHANDLE or NFS4ERR_BAD_STATEID. The slot that carried the grant has not retired at that point, so NFSD cannot read the rejection as proof that the client never held the delegation. It retries the recall and, once the retries lapse, revokes a delegation the client is by then able to return. Remove the ambiguity with the referring call mechanism of RFC 8881 Section 2.10.6.3: until the slot that carried the grant retires, name that request as a referring call in the CB_SEQUENCE of each recall. A client that finds it still outstanding may respond NFS4ERR_DELAY, and the recall is retried until the client has processed the grant. A recall reuses one callback context across its retries, and ->prepare does not run on every send. The granting request does not change, so a send that inherits the previous list sends the right one. Retirement of the granting slot drops the list, and nfs4_free_deleg() releases what is left. Link: https://patch.msgid.link/20260802-nfsd-deleg-destroy-badhandle-v1-9-323aa7196055@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfs4callback.c | 7 ++++-- fs/nfsd/nfs4state.c | 54 ++++++++++++++++++++++++++++++++---------- 2 files changed, 47 insertions(+), 14 deletions(-) diff --git a/fs/nfsd/nfs4callback.c b/fs/nfsd/nfs4callback.c index 2939b5c6a5feac..a6b31d3f2bf66e 100644 --- a/fs/nfsd/nfs4callback.c +++ b/fs/nfsd/nfs4callback.c @@ -1530,12 +1530,14 @@ void nfsd41_cb_referring_call(struct nfsd4_callback *cb, /** * nfsd41_cb_destroy_referring_call_list - release referring call info - * @cb: context of a callback that has completed + * @cb: context of callback to release referring calls from * * Callers who allocate referring calls using nfsd41_cb_referring_call() must * release those resources by calling nfsd41_cb_destroy_referring_call_list. * - * Caller serializes access to @cb. + * Caller serializes access to @cb. No CB_COMPOUND for @cb may be in + * flight, because encode_cb_sequence4args() walks this list as it + * encodes. */ void nfsd41_cb_destroy_referring_call_list(struct nfsd4_callback *cb) { @@ -1557,6 +1559,7 @@ void nfsd41_cb_destroy_referring_call_list(struct nfsd4_callback *cb) list_del(&rcl->__list); kfree(rcl); } + cb->cb_nr_referring_call_list = 0; } static void nfsd4_cb_prepare(struct rpc_task *task, void *calldata) diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c index 3283b2225b7652..0bd694390b8a26 100644 --- a/fs/nfsd/nfs4state.c +++ b/fs/nfsd/nfs4state.c @@ -1167,6 +1167,8 @@ static void nfs4_free_deleg(struct nfs4_stid *stid) WARN_ON_ONCE(!list_empty(&dp->dl_perfile)); WARN_ON_ONCE(!list_empty(&dp->dl_perclnt)); WARN_ON_ONCE(!list_empty(&dp->dl_recall_lru)); + /* The list outlives one recall, so ->release() cannot free it. */ + nfsd41_cb_destroy_referring_call_list(&dp->dl_recall); kmem_cache_free(deleg_slab, stid); atomic_long_dec(&num_delegations); } @@ -6091,6 +6093,18 @@ bool nfsd_wait_for_delegreturn(struct svc_rqst *rqstp, struct inode *inode) return timeo > 0; } +/* + * gen_sessionid() composes a sessionid from the client's clientid and a + * sequence counter, so the sequence alone identifies the granting session. + */ +static void nfsd4_recall_grant_sessionid(const struct nfs4_delegation *dp, + struct nfsd4_sessionid *sid) +{ + sid->clientid = dp->dl_stid.sc_client->cl_clientid; + sid->sequence = dp->dl_recall_grant.sessionid_seq; + sid->reserved = 0; +} + static bool nfsd4_recall_grant_slot_retired(struct nfs4_delegation *dp) { struct nfs4_client *clp = dp->dl_stid.sc_client; @@ -6103,14 +6117,7 @@ static bool nfsd4_recall_grant_slot_retired(struct nfs4_delegation *dp) if (!dp->dl_recall_grant.valid) return false; - /* - * gen_sessionid() composes a sessionid from the client's clientid - * and a sequence counter, so the sequence alone identifies the - * granting session. - */ - sid.clientid = clp->cl_clientid; - sid.sequence = dp->dl_recall_grant.sessionid_seq; - sid.reserved = 0; + nfsd4_recall_grant_sessionid(dp, &sid); /* * A missing session does not prove the client saw the grant: a @@ -6145,6 +6152,21 @@ static bool nfsd4_recall_grant_slot_retired(struct nfs4_delegation *dp) return retired; } +/* + * ->prepare does not run on every send: nfsd4_run_cb_work() skips it + * on a requeue, and a retry via rpc_restart_call_prepare() re-enters + * the RPC layer beneath it. The granting request does not change, so + * a send inherits a correct list. Retirement is the one transition + * the list has to follow. + */ +static void nfsd4_refresh_recall_grant(struct nfs4_delegation *dp) +{ + dp->dl_recall_grant.retired_at_send = + nfsd4_recall_grant_slot_retired(dp); + if (dp->dl_recall_grant.retired_at_send) + nfsd41_cb_destroy_referring_call_list(&dp->dl_recall); +} + static bool nfsd4_cb_recall_prepare(struct nfsd4_callback *cb) { struct nfs4_delegation *dp = cb_to_delegation(cb); @@ -6167,8 +6189,17 @@ static bool nfsd4_cb_recall_prepare(struct nfsd4_callback *cb) } spin_unlock(&nn->deleg_lock); - dp->dl_recall_grant.retired_at_send = - nfsd4_recall_grant_slot_retired(dp); + nfsd4_refresh_recall_grant(dp); + + if (dp->dl_recall_grant.valid && !dp->dl_recall_grant.retired_at_send) { + struct nfsd4_sessionid sid; + + nfsd4_recall_grant_sessionid(dp, &sid); + nfsd41_cb_referring_call(&dp->dl_recall, + (struct nfs4_sessionid *)&sid, + dp->dl_recall_grant.slotid, + dp->dl_recall_grant.seqid); + } return true; } @@ -6225,8 +6256,7 @@ static int nfsd4_cb_recall_done(struct nfsd4_callback *cb, return 1; } if (!dp->dl_stid.sc_status && dp->dl_retries--) { - dp->dl_recall_grant.retired_at_send = - nfsd4_recall_grant_slot_retired(dp); + nfsd4_refresh_recall_grant(dp); rpc_delay(task, 2 * HZ); return 0; } From aea6b5681baf7aac4a84c236bd9dc6531ccfb360 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Tue, 4 Aug 2026 14:46:30 -0400 Subject: [PATCH 0174/1352] NFSD: Point contributors and sashiko.dev to the nfsd-testing branch Scripting and automation is sensitive to branch names in the subsystem entries in MAINTAINERS. Rather than pulling from cel.git/master, we really want CI to pull from cel.git/nfsd-testing. Link: https://patch.msgid.link/20260804184630.1395002-1-cel@kernel.org Signed-off-by: Chuck Lever --- MAINTAINERS | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/MAINTAINERS b/MAINTAINERS index cc3cae2e378b34..8eb8b069d117ef 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -14187,7 +14187,8 @@ L: linux-nfs@vger.kernel.org S: Supported P: Documentation/filesystems/nfs/nfsd-maintainer-entry-profile.rst B: https://bugzilla.kernel.org -T: git git://git.kernel.org/pub/scm/linux/kernel/git/cel/linux.git +T: git git://git.kernel.org/pub/scm/linux/kernel/git/cel/linux.git nfsd-testing +T: git git://git.kernel.org/pub/scm/linux/kernel/git/cel/linux.git nfsd-next F: Documentation/filesystems/nfs/ F: fs/lockd/ F: fs/nfs_common/ From a76d52a01792f315834a6273a3ee66b795a6b7b3 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sat, 8 Aug 2026 11:40:09 -0400 Subject: [PATCH 0175/1352] SUNRPC: Do not credit control-record octets to the RPC stream svc_tcp_sock_recv_cmsg() receives up to two octets into a local buffer, and returns that count for any record type other than TLS_RECORD_TYPE_ALERT. Nothing reached the caller's buffer, but svc_tcp_read_marker() adds the count to sk_tcplen and svc_tcp_read_msg()'s caller adds it to sk_datalen. The RPC stream advances over octets it never received. The fragment marker is assembled from stale sk_marker octets. The message body comes from pages nothing wrote. A conforming client reaches this. RFC 8446 Section 4.6.3 lets either peer send KeyUpdate once it has sent its Finished, and svcsock has no rekey path. kTLS leaves the partially consumed record on ctx->rx_list, so the body drains two octets per svc_tcp_recvfrom() call. Each pair is credited the same way. Return -EAGAIN for a record that is not an alert. That is what svc_tcp_sock_process_cmsg()'s default arm returned before the receive moved into a local buffer. Fixes: bee47cb026e7 ("sunrpc: fix handling of server side tls alerts") Cc: stable@vger.kernel.org Link: https://patch.msgid.link/20260808-svcsock-cmsg-fixes-v3-1-62d9a631c880@kernel.org Signed-off-by: Chuck Lever --- net/sunrpc/svcsock.c | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/net/sunrpc/svcsock.c b/net/sunrpc/svcsock.c index 50e5e7f5b762de..8e1009302e3b81 100644 --- a/net/sunrpc/svcsock.c +++ b/net/sunrpc/svcsock.c @@ -289,8 +289,13 @@ svc_tcp_sock_recv_cmsg(struct socket *sock, unsigned int *msg_flags) iov_iter_kvec(&msg.msg_iter, ITER_DEST, &alert_kvec, 1, alert_kvec.iov_len); ret = sock_recvmsg(sock, &msg, MSG_DONTWAIT); - if (ret > 0 && - tls_get_record_type(sock->sk, &u.cmsg) == TLS_RECORD_TYPE_ALERT) { + if (ret > 0) { + /* Returning the count would credit the RPC stream with + * octets that never reached the caller's buffer. + */ + if (tls_get_record_type(sock->sk, &u.cmsg) != + TLS_RECORD_TYPE_ALERT) + return -EAGAIN; iov_iter_revert(&msg.msg_iter, ret); ret = svc_tcp_sock_process_cmsg(sock, &msg, &u.cmsg, -EAGAIN); } From e263b5d674fcfe054549084e2c301892d0bf6010 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sat, 8 Aug 2026 11:40:10 -0400 Subject: [PATCH 0176/1352] SUNRPC: Reject a TLS alert record that is not two octets tls_alert_recv() reads two octets from the kvec it is handed and does not check the length (net/handshake/alert.c). svc_tcp_sock_recv_cmsg() calls it for any positive receive, and the alert[] buffer it supplies carries no initializer. A one-octet alert body leaves the description read from uninitialized stack and reported through trace_tls_alert_recv(). The peer controls that length. Neither tls_rx_msg_size() nor tls_rx_one_record() enforces the two-octet Alert payload. A TLS 1.3 record carrying only the inner content-type octet decrypts to a zero-length payload. RFC 8446 Section 5.1 requires a record with an Alert type to carry exactly one message, so any other length is malformed. Require exactly two octets before parsing and return -EBADMSG otherwise. That closes the transport rather than acting on a partly uninitialized alert. Gate the path on a control message rather than a positive count so that a zero-length record reaches the check. Fixes: bee47cb026e7 ("sunrpc: fix handling of server side tls alerts") Cc: stable@vger.kernel.org Link: https://patch.msgid.link/20260808-svcsock-cmsg-fixes-v3-2-62d9a631c880@kernel.org Signed-off-by: Chuck Lever --- net/sunrpc/svcsock.c | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/net/sunrpc/svcsock.c b/net/sunrpc/svcsock.c index 8e1009302e3b81..2e5107a1fb898c 100644 --- a/net/sunrpc/svcsock.c +++ b/net/sunrpc/svcsock.c @@ -289,13 +289,23 @@ svc_tcp_sock_recv_cmsg(struct socket *sock, unsigned int *msg_flags) iov_iter_kvec(&msg.msg_iter, ITER_DEST, &alert_kvec, 1, alert_kvec.iov_len); ret = sock_recvmsg(sock, &msg, MSG_DONTWAIT); - if (ret > 0) { + /* put_cmsg() shrinks msg_controllen, so a short one means + * kTLS filled in u.cmsg. + */ + if (ret >= 0 && msg.msg_controllen < sizeof(u)) { /* Returning the count would credit the RPC stream with * octets that never reached the caller's buffer. */ if (tls_get_record_type(sock->sk, &u.cmsg) != TLS_RECORD_TYPE_ALERT) return -EAGAIN; + /* An Alert record carries exactly one two-octet message + * (RFC 8446 Section 5.1). alert_kvec caps the receive at two, + * so a longer record produces the same count. MSG_EOR appears + * only once kTLS has drained the whole record. + */ + if (ret != sizeof(alert) || !(msg.msg_flags & MSG_EOR)) + return -EBADMSG; iov_iter_revert(&msg.msg_iter, ret); ret = svc_tcp_sock_process_cmsg(sock, &msg, &u.cmsg, -EAGAIN); } From ebcbe4a9a77797c0cc221bb76f0b246bc813fc51 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sat, 8 Aug 2026 11:40:11 -0400 Subject: [PATCH 0177/1352] SUNRPC: Treat every TLS error alert as fatal svc_tcp_sock_process_cmsg() decides whether an alert ends the session by reading the alert's level octet. RFC 8446 Section 6 retired that field. The severity is implicit in the description, and a receiver treats every alert listed in Section 6.2 as an error alert "regardless of the AlertLevel in the message". A peer that aborts with unexpected_message but leaves the legacy octet set to warning makes the server return -EAGAIN. svc_tcp_recvfrom() then leaves a dead TLS session attached to an open transport. NFSD keeps polling it. Decide from the alert description instead. close_notify and user_canceled are the closure alerts (RFC 8446 Section 6.1). Every other description ends the session, including one this kernel does not recognize. Fixes: 39067dda1d86 ("SUNRPC: Use new helpers to handle TLS Alerts") Cc: stable@vger.kernel.org Link: https://patch.msgid.link/20260808-svcsock-cmsg-fixes-v3-3-62d9a631c880@kernel.org Signed-off-by: Chuck Lever --- net/sunrpc/svcsock.c | 13 +++++++++++-- 1 file changed, 11 insertions(+), 2 deletions(-) diff --git a/net/sunrpc/svcsock.c b/net/sunrpc/svcsock.c index 2e5107a1fb898c..68db26522ddfb4 100644 --- a/net/sunrpc/svcsock.c +++ b/net/sunrpc/svcsock.c @@ -257,8 +257,17 @@ svc_tcp_sock_process_cmsg(struct socket *sock, struct msghdr *msg, break; case TLS_RECORD_TYPE_ALERT: tls_alert_recv(sock->sk, msg, &level, &description); - ret = (level == TLS_ALERT_LEVEL_FATAL) ? - -ENOTCONN : -EAGAIN; + /* RFC 8446 Section 6: every alert but a closure alert is + * an error alert. + */ + switch (description) { + case TLS_ALERT_DESC_CLOSE_NOTIFY: + case TLS_ALERT_DESC_USER_CANCELED: + ret = -EAGAIN; + break; + default: + ret = -ENOTCONN; + } break; default: /* discard this record type */ From 8f765d820c590ce68b65724e54f6fd2be37cfead Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sat, 8 Aug 2026 11:40:12 -0400 Subject: [PATCH 0178/1352] SUNRPC: Resume receiving after a TLS control record A TLS control record delivers no payload to the RPC layer. svc_tcp_recvfrom() clears XPT_DATA before the receive, and svc_tcp_sock_recv_cmsg() returns -EAGAIN for the record it consumed. Nothing marks the transport ready again. kTLS raises data_ready for arriving TCP segments, not for records it has already decrypted. An RPC Call queued behind an alert or a KeyUpdate waits until the client sends more. The client blocks until its RPC timeout expires. The receive takes only the first two octets of the record. kTLS holds the remainder on its receive list, where each later receive takes two octets more. Drain a record that is not an alert, then mark the transport ready once a control record has been consumed. Fixes: 5e052dda121e ("SUNRPC: Recognize control messages in server-side TCP socket code") Cc: stable@vger.kernel.org Link: https://patch.msgid.link/20260808-svcsock-cmsg-fixes-v3-4-62d9a631c880@kernel.org Signed-off-by: Chuck Lever --- net/sunrpc/svcsock.c | 55 +++++++++++++++++++++++++++++++++++++++++--- 1 file changed, 52 insertions(+), 3 deletions(-) diff --git a/net/sunrpc/svcsock.c b/net/sunrpc/svcsock.c index 68db26522ddfb4..7a423e9ee74d4a 100644 --- a/net/sunrpc/svcsock.c +++ b/net/sunrpc/svcsock.c @@ -238,6 +238,39 @@ static int svc_one_sock_name(struct svc_sock *svsk, char *buf, int remaining) return len; } +/* + * kTLS delivers a record only up to the caller's buffer and keeps + * the remainder on its receive list, where no further data_ready + * announces it. Consume the whole record. + */ +static void +svc_tcp_sock_drain_record(struct socket *sock) +{ + union { + struct cmsghdr cmsg; + u8 buf[CMSG_SPACE(sizeof(u8))]; + } u; + u8 discard[64]; + struct kvec discard_kvec = { + .iov_base = discard, + .iov_len = sizeof(discard), + }; + + for (;;) { + struct msghdr msg = { + .msg_control = &u, + .msg_controllen = sizeof(u), + }; + + iov_iter_kvec(&msg.msg_iter, ITER_DEST, &discard_kvec, 1, + discard_kvec.iov_len); + if (sock_recvmsg(sock, &msg, MSG_DONTWAIT) <= 0) + break; + if (msg.msg_flags & MSG_EOR) + break; + } +} + static int svc_tcp_sock_process_cmsg(struct socket *sock, struct msghdr *msg, struct cmsghdr *cmsg, int ret) @@ -302,12 +335,20 @@ svc_tcp_sock_recv_cmsg(struct socket *sock, unsigned int *msg_flags) * kTLS filled in u.cmsg. */ if (ret >= 0 && msg.msg_controllen < sizeof(u)) { + u8 content_type = tls_get_record_type(sock->sk, &u.cmsg); + /* Returning the count would credit the RPC stream with * octets that never reached the caller's buffer. */ - if (tls_get_record_type(sock->sk, &u.cmsg) != - TLS_RECORD_TYPE_ALERT) + if (content_type != TLS_RECORD_TYPE_ALERT) { + /* An application data record carries RPC payload. + * Draining one breaks RPC fragment framing. + */ + if (content_type != TLS_RECORD_TYPE_DATA && + !(msg.msg_flags & MSG_EOR)) + svc_tcp_sock_drain_record(sock); return -EAGAIN; + } /* An Alert record carries exactly one two-octet message * (RFC 8446 Section 5.1). alert_kvec caps the receive at two, * so a longer record produces the same count. MSG_EOR appears @@ -330,8 +371,16 @@ svc_tcp_sock_recvmsg(struct svc_sock *svsk, struct msghdr *msg) ret = sock_recvmsg(sock, msg, MSG_DONTWAIT); if (msg->msg_flags & MSG_CTRUNC) { msg->msg_flags &= ~(MSG_CTRUNC | MSG_EOR); - if (ret == 0 || ret == -EIO) + if (ret == 0 || ret == -EIO) { ret = svc_tcp_sock_recv_cmsg(sock, &msg->msg_flags); + /* A control record delivers nothing to the caller, + * and kTLS announces no data_ready for records it + * already holds. Mark the transport ready so that + * the records behind this one can be received. + */ + if (ret == -EAGAIN) + set_bit(XPT_DATA, &svsk->sk_xprt.xpt_flags); + } } return ret; } From 2827c44f148c7540f2982233364cde198b76660f Mon Sep 17 00:00:00 2001 From: Suraj Kandpal Date: Mon, 21 Sep 2026 07:50:17 +0530 Subject: [PATCH 0179/1352] drm/i915/xe3plpd: Map AUX power domains to DC_off From Xe2Lpd we removed the step to disable DC5/6 because as per Bspec for SNPS PHY Aux will be always up. But from Xe3p LT PHY was introduced which powers down the PHY. Which means we need to make sure "Disable DC5/6" of AUX Programming sequence is honored. Add the AUX power domains to the DC_off power well on Xe3p_LPD+, so that DC5/6 stays disabled for as long as an AUX power domain is held. Bspec: 68851, 68967 Suggested-by: Imre Deak Co-developed-by: Ramanaidu Naladala Signed-off-by: Ramanaidu Naladala Signed-off-by: Suraj Kandpal Reviewed-by: Arun R Murthy Link: https://patch.msgid.link/20260921022016.1525322-1-suraj.kandpal@intel.com --- .../i915/display/intel_display_power_map.c | 42 ++++++++++++++++++- 1 file changed, 41 insertions(+), 1 deletion(-) diff --git a/drivers/gpu/drm/i915/display/intel_display_power_map.c b/drivers/gpu/drm/i915/display/intel_display_power_map.c index 3400080d78d2a3..e18eb49983fe88 100644 --- a/drivers/gpu/drm/i915/display/intel_display_power_map.c +++ b/drivers/gpu/drm/i915/display/intel_display_power_map.c @@ -1759,6 +1759,44 @@ static const struct i915_power_well_desc_list wcl_power_wells[] = { I915_PW_DESCRIPTORS(wcl_power_wells_aux), }; +I915_DECL_PW_DOMAINS(xe3plpd_pwdoms_dc_off, + POWER_DOMAIN_DC_OFF, + XE3LPD_PW_2_POWER_DOMAINS, + XE3LPD_PW_C_POWER_DOMAINS, + XE3LPD_PW_D_POWER_DOMAINS, + POWER_DOMAIN_AUDIO_MMIO, + POWER_DOMAIN_AUDIO_PLAYBACK, + POWER_DOMAIN_AUX_A, + POWER_DOMAIN_AUX_B, + POWER_DOMAIN_AUX_USBC1, + POWER_DOMAIN_AUX_USBC2, + POWER_DOMAIN_AUX_USBC3, + POWER_DOMAIN_AUX_USBC4, + POWER_DOMAIN_AUX_TBT1, + POWER_DOMAIN_AUX_TBT2, + POWER_DOMAIN_AUX_TBT3, + POWER_DOMAIN_AUX_TBT4, + POWER_DOMAIN_INIT); + +static const struct i915_power_well_desc xe3plpd_power_wells_dcoff[] = { + { + .instances = &I915_PW_INSTANCES( + I915_PW("DC_off", &xe3plpd_pwdoms_dc_off, + .id = SKL_DISP_DC_OFF), + ), + .ops = &gen9_dc_off_power_well_ops, + }, +}; + +static const struct i915_power_well_desc_list xe3plpd_power_wells[] = { + I915_PW_DESCRIPTORS(i9xx_power_wells_always_on), + I915_PW_DESCRIPTORS(icl_power_wells_pw_1), + I915_PW_DESCRIPTORS(xe3plpd_power_wells_dcoff), + I915_PW_DESCRIPTORS(xe3lpd_power_wells_main), + I915_PW_DESCRIPTORS(xe2lpd_power_wells_pica), + I915_PW_DESCRIPTORS(xelpdp_power_wells_aux), +}; + static void init_power_well_domains(const struct i915_power_well_instance *inst, struct i915_power_well *power_well) { @@ -1865,7 +1903,9 @@ int intel_display_power_map_init(struct i915_power_domains *power_domains) return 0; } - if (DISPLAY_VERx100(display) == 3002) + if (DISPLAY_VER(display) >= 35) + return set_power_wells(power_domains, xe3plpd_power_wells); + else if (DISPLAY_VERx100(display) == 3002) return set_power_wells(power_domains, wcl_power_wells); else if (DISPLAY_VER(display) >= 30) return set_power_wells(power_domains, xe3lpd_power_wells); From 26f42375404e5558dc9fa8acc28fa74279394232 Mon Sep 17 00:00:00 2001 From: Kefeng Wang Date: Mon, 21 Sep 2026 15:10:30 +0800 Subject: [PATCH 0180/1352] xfs: fix NOFS state corruption in btree split worker xfs_btree_split_worker() calls xfs_trans_set_context() and xfs_trans_clear_context() on the caller's transaction, overwriting tp->t_pflags with the worker's NOFS state. When the caller already has PF_MEMALLOC_NOFS set (e.g. xfs_end_ioend_write, xfs_dio_write_end_io), the corrupted tp->t_pflags causes xfs_trans_free() to erroneously clear the caller's NOFS protection. Use memalloc_nofs_save/restore with a local variable instead so tp->t_pflags is never touched. Closes: https://sashiko.dev/#/patchset/20260902131653.1338227-1-wangkefeng.wang@huawei.com Fixes: 756b1c343333 ("xfs: use current->journal_info for detecting transaction recursion") Signed-off-by: Kefeng Wang Reviewed-by: Christoph Hellwig Reviewed-by: Brian Foster Reviewed-by: Carlos Maiolino Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_btree.c | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/fs/xfs/libxfs/xfs_btree.c b/fs/xfs/libxfs/xfs_btree.c index 60ef7f08b1d300..7effebe297ea0b 100644 --- a/fs/xfs/libxfs/xfs_btree.c +++ b/fs/xfs/libxfs/xfs_btree.c @@ -3010,6 +3010,7 @@ xfs_btree_split_worker( struct xfs_btree_split_args, work); unsigned long pflags; unsigned long new_pflags = 0; + unsigned int nofs_flags; /* * we are in a transaction context here, but may also be doing work @@ -3021,12 +3022,17 @@ xfs_btree_split_worker( new_pflags |= PF_MEMALLOC | PF_KSWAPD; current_set_flags_nested(&pflags, new_pflags); - xfs_trans_set_context(args->cur->bc_tp); + + /* + * Don't use xfs_trans_set_context() here: it would overwrite the + * caller's saved NOFS state in tp->t_pflags. Use a local scope. + */ + nofs_flags = memalloc_nofs_save(); args->result = __xfs_btree_split(args->cur, args->level, args->ptrp, args->key, args->curp, args->stat); - xfs_trans_clear_context(args->cur->bc_tp); + memalloc_nofs_restore(nofs_flags); current_restore_flags_nested(&pflags, new_pflags); /* From 3d79ea5703c580fa38b4418c5a4fe8cbe9418b53 Mon Sep 17 00:00:00 2001 From: Lalit Shankar Chowdhury Date: Fri, 18 Sep 2026 04:23:18 +0000 Subject: [PATCH 0181/1352] drm/i915: use string choice helpers Prefer the str_high_low() and str_read_write() helper functions over ternary operators with hardcoded strings. No functional changes. Signed-off-by: Lalit Shankar Chowdhury Reviewed-by: Jani Nikula Link: https://patch.msgid.link/20260918042325.17700-1-lalitshankarch@gmail.com Signed-off-by: Jani Nikula --- drivers/gpu/drm/i915/display/intel_cx0_phy.c | 5 +++-- drivers/gpu/drm/i915/gvt/aperture_gm.c | 4 +++- drivers/gpu/drm/i915/vlv_iosf_sb.c | 6 ++++-- 3 files changed, 10 insertions(+), 5 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_cx0_phy.c b/drivers/gpu/drm/i915/display/intel_cx0_phy.c index dbebd721084849..b83f0afaa9b523 100644 --- a/drivers/gpu/drm/i915/display/intel_cx0_phy.c +++ b/drivers/gpu/drm/i915/display/intel_cx0_phy.c @@ -5,6 +5,7 @@ #include #include +#include #include @@ -189,7 +190,7 @@ int intel_cx0_wait_for_ack(struct intel_encoder *encoder, drm_dbg_kms(display->drm, "PHY %c Error occurred during %s command. Status: 0x%x\n", phy_name(phy), - command == XELPDP_PORT_P2M_COMMAND_READ_ACK ? "read" : "write", *val); + str_read_write(command == XELPDP_PORT_P2M_COMMAND_READ_ACK), *val); intel_cx0_bus_reset(encoder, lane); return -EINVAL; } @@ -198,7 +199,7 @@ int intel_cx0_wait_for_ack(struct intel_encoder *encoder, drm_dbg_kms(display->drm, "PHY %c Not a %s response. MSGBUS Status: 0x%x.\n", phy_name(phy), - command == XELPDP_PORT_P2M_COMMAND_READ_ACK ? "read" : "write", *val); + str_read_write(command == XELPDP_PORT_P2M_COMMAND_READ_ACK), *val); intel_cx0_bus_reset(encoder, lane); return -EINVAL; } diff --git a/drivers/gpu/drm/i915/gvt/aperture_gm.c b/drivers/gpu/drm/i915/gvt/aperture_gm.c index 253b41789be910..794fb50e48dddb 100644 --- a/drivers/gpu/drm/i915/gvt/aperture_gm.c +++ b/drivers/gpu/drm/i915/gvt/aperture_gm.c @@ -34,6 +34,8 @@ * */ +#include + #include #include "gt/intel_ggtt_fencing.h" @@ -76,7 +78,7 @@ static int alloc_gm(struct intel_vgpu *vgpu, bool high_gm) mutex_unlock(>->ggtt->vm.mutex); if (ret) gvt_err("fail to alloc %s gm space from host\n", - high_gm ? "high" : "low"); + str_high_low(high_gm)); return ret; } diff --git a/drivers/gpu/drm/i915/vlv_iosf_sb.c b/drivers/gpu/drm/i915/vlv_iosf_sb.c index 1f0332b4ad0df5..fede1735bc0835 100644 --- a/drivers/gpu/drm/i915/vlv_iosf_sb.c +++ b/drivers/gpu/drm/i915/vlv_iosf_sb.c @@ -3,6 +3,8 @@ * Copyright © 2013-2021 Intel Corporation */ +#include + #include #include @@ -101,7 +103,7 @@ static int vlv_sideband_rw(struct drm_i915_private *i915, VLV_IOSF_DOORBELL_REQ, IOSF_SB_BUSY, 0, 5)) { drm_dbg(&i915->drm, "IOSF sideband idle wait (%s) timed out\n", - is_read ? "read" : "write"); + str_read_write(is_read)); return -EAGAIN; } @@ -125,7 +127,7 @@ static int vlv_sideband_rw(struct drm_i915_private *i915, err = 0; } else { drm_dbg(&i915->drm, "IOSF sideband finish wait (%s) timed out\n", - is_read ? "read" : "write"); + str_read_write(is_read)); err = -ETIMEDOUT; } From fd2e337ba66f6de39abad98a656fa6e5ff0c1609 Mon Sep 17 00:00:00 2001 From: Juasheem Sultan Date: Tue, 2 Jun 2026 13:54:02 -0700 Subject: [PATCH 0182/1352] drm/i915/dp: Handle VSC SDP revision 7 in unpack VSC SDP revision 7 (Panel Replay + Pixel Encoding/Colorimetry Format) unpacking is missing in intel_dp_vsc_sdp_unpack(). This causes pipe state mismatches during state readout because the VSC SDP state is not properly recovered when Panel Replay is active with colorimetry. Add the missing case for revision 7 to intel_dp_vsc_sdp_unpack() so that the state is correctly recovered. v2: Rebase onto latest upstream v3: Fix git commit description Cc: Jani Nikula Cc: Chaitanya Borah Cc: Gil Dekel Signed-off-by: Juasheem Sultan Reviewed-by: Chaitanya Kumar Borah Signed-off-by: Ankit Nautiyal Link: https://patch.msgid.link/20260602205422.1792007-1-jdsultan@google.com --- drivers/gpu/drm/i915/display/intel_dp.c | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/drivers/gpu/drm/i915/display/intel_dp.c b/drivers/gpu/drm/i915/display/intel_dp.c index 50ed615cf0f606..2480b8ce3bec71 100644 --- a/drivers/gpu/drm/i915/display/intel_dp.c +++ b/drivers/gpu/drm/i915/display/intel_dp.c @@ -5475,11 +5475,15 @@ static int intel_dp_vsc_sdp_unpack(struct drm_dp_vsc_sdp *vsc, * VSC SDP supporting 3D stereo + Panel Replay. */ return 0; - } else if (sdp->sdp_header.HB2 == 0x5 && sdp->sdp_header.HB3 == 0x13) { + } else if ((sdp->sdp_header.HB2 == 0x5 || sdp->sdp_header.HB2 == 0x7) && + sdp->sdp_header.HB3 == 0x13) { /* * - HB2 = 0x5, HB3 = 0x13 * VSC SDP supporting 3D stereo + PSR2 + Pixel Encoding/Colorimetry * Format. + * - HB2 = 0x7, HB3 = 0x13 + * VSC SDP supporting 3D stereo + Panel Replay + Pixel Encoding/Colorimetry + * Format. */ vsc->pixelformat = (sdp->db[16] >> 4) & 0xf; vsc->colorimetry = sdp->db[16] & 0xf; From d09f85ed8db24c7a56a5735ccc03277d88d5ab99 Mon Sep 17 00:00:00 2001 From: Pranjal Shrivastava Date: Fri, 14 Aug 2026 14:32:51 +0000 Subject: [PATCH 0183/1352] nfs: make nfs_page pin-aware Modernizing the NFS Direct I/O path to use iov_iter_extract_pages() introduces page pinning (GUP) instead of standard page referencing. To handle this correctly, nfs_page must track whether it holds a pin or a standard reference. Introduce a new flag, PG_PINNED, to struct nfs_page. Update the creation path (nfs_page_create_from_page and nfs_page_create_from_folio) to accept a pinned bool and set the flag accordingly. If the page is pinned, we skip the existing reference increment (get_page/folio_get) as the pin itself acts as a reference. Update nfs_clear_request() & nfs_direct_release_pages() to use unpin_user_page() or unpin_user_folio() instead of only refcount decrement (put_page) when PG_PINNED flag is set. Finally, ensure subrequests inherit the pinning status from their parent request. Signed-off-by: Pranjal Shrivastava Reviewed-by: Christoph Hellwig Signed-off-by: Anna Schumaker --- fs/nfs/direct.c | 22 +++++++++++++------- fs/nfs/pagelist.c | 44 +++++++++++++++++++++++++++++++--------- fs/nfs/read.c | 2 +- fs/nfs/write.c | 2 +- include/linux/nfs_page.h | 3 +++ 5 files changed, 54 insertions(+), 19 deletions(-) diff --git a/fs/nfs/direct.c b/fs/nfs/direct.c index ccafdc1ce64d48..6a7f487fd43e13 100644 --- a/fs/nfs/direct.c +++ b/fs/nfs/direct.c @@ -145,11 +145,17 @@ static void nfs_direct_file_adjust_size_locked(struct inode *inode, } } -static void nfs_direct_release_pages(struct page **pages, unsigned int npages) +static void nfs_direct_release_pages(struct page **pages, unsigned int npages, + bool pinned) { unsigned int i; - for (i = 0; i < npages; i++) - put_page(pages[i]); + + if (pinned) { + unpin_user_pages(pages, npages); + } else { + for (i = 0; i < npages; i++) + put_page(pages[i]); + } } void nfs_init_cinfo_from_dreq(struct nfs_commit_info *cinfo, @@ -351,7 +357,8 @@ static ssize_t nfs_direct_read_schedule_iovec(struct nfs_direct_req *dreq, unsigned int req_len = min_t(size_t, bytes, PAGE_SIZE - pgbase); /* XXX do we need to do the eof zeroing found in async_filler? */ req = nfs_page_create_from_page(dreq->ctx, pagevec[i], - pgbase, pos, req_len); + false, pgbase, pos, + req_len); if (IS_ERR(req)) { result = PTR_ERR(req); break; @@ -366,7 +373,7 @@ static ssize_t nfs_direct_read_schedule_iovec(struct nfs_direct_req *dreq, requested_bytes += req_len; pos += req_len; } - nfs_direct_release_pages(pagevec, npages); + nfs_direct_release_pages(pagevec, npages, false); kvfree(pagevec); if (result < 0) break; @@ -887,7 +894,8 @@ static ssize_t nfs_direct_write_schedule_iovec(struct nfs_direct_req *dreq, unsigned int req_len = min_t(size_t, bytes, PAGE_SIZE - pgbase); req = nfs_page_create_from_page(dreq->ctx, pagevec[i], - pgbase, pos, req_len); + false, pgbase, pos, + req_len); if (IS_ERR(req)) { result = PTR_ERR(req); break; @@ -930,7 +938,7 @@ static ssize_t nfs_direct_write_schedule_iovec(struct nfs_direct_req *dreq, desc.pg_error = 0; defer = true; } - nfs_direct_release_pages(pagevec, npages); + nfs_direct_release_pages(pagevec, npages, false); kvfree(pagevec); if (result < 0) break; diff --git a/fs/nfs/pagelist.c b/fs/nfs/pagelist.c index 7dd478ffc2fab4..a562cfe2a126d1 100644 --- a/fs/nfs/pagelist.c +++ b/fs/nfs/pagelist.c @@ -404,20 +404,28 @@ static struct nfs_page *nfs_page_create(struct nfs_lock_context *l_ctx, return req; } -static void nfs_page_assign_folio(struct nfs_page *req, struct folio *folio) +static void nfs_page_assign_folio(struct nfs_page *req, struct folio *folio, + bool pinned) { if (folio != NULL) { req->wb_folio = folio; - folio_get(folio); + if (pinned) + set_bit(PG_PINNED, &req->wb_flags); + else + folio_get(folio); set_bit(PG_FOLIO, &req->wb_flags); } } -static void nfs_page_assign_page(struct nfs_page *req, struct page *page) +static void nfs_page_assign_page(struct nfs_page *req, struct page *page, + bool pinned) { if (page != NULL) { req->wb_page = page; - get_page(page); + if (pinned) + set_bit(PG_PINNED, &req->wb_flags); + else + get_page(page); } } @@ -425,6 +433,7 @@ static void nfs_page_assign_page(struct nfs_page *req, struct page *page) * nfs_page_create_from_page - Create an NFS read/write request. * @ctx: open context to use * @page: page to write + * @pinned: true if page is pinned * @pgbase: starting offset within the page for the write * @offset: file offset for the write * @count: number of bytes to read/write @@ -435,6 +444,7 @@ static void nfs_page_assign_page(struct nfs_page *req, struct page *page) */ struct nfs_page *nfs_page_create_from_page(struct nfs_open_context *ctx, struct page *page, + bool pinned, unsigned int pgbase, loff_t offset, unsigned int count) { @@ -446,7 +456,7 @@ struct nfs_page *nfs_page_create_from_page(struct nfs_open_context *ctx, ret = nfs_page_create(l_ctx, pgbase, offset >> PAGE_SHIFT, offset_in_page(offset), count); if (!IS_ERR(ret)) { - nfs_page_assign_page(ret, page); + nfs_page_assign_page(ret, page, pinned); nfs_page_group_init(ret, NULL); } nfs_put_lock_context(l_ctx); @@ -457,6 +467,7 @@ struct nfs_page *nfs_page_create_from_page(struct nfs_open_context *ctx, * nfs_page_create_from_folio - Create an NFS read/write request. * @ctx: open context to use * @folio: folio to write + * @pinned: true if folio is pinned * @offset: starting offset within the folio for the write * @count: number of bytes to read/write * @@ -466,6 +477,7 @@ struct nfs_page *nfs_page_create_from_page(struct nfs_open_context *ctx, */ struct nfs_page *nfs_page_create_from_folio(struct nfs_open_context *ctx, struct folio *folio, + bool pinned, unsigned int offset, unsigned int count) { @@ -476,7 +488,7 @@ struct nfs_page *nfs_page_create_from_folio(struct nfs_open_context *ctx, return ERR_CAST(l_ctx); ret = nfs_page_create(l_ctx, offset, folio->index, offset, count); if (!IS_ERR(ret)) { - nfs_page_assign_folio(ret, folio); + nfs_page_assign_folio(ret, folio, pinned); nfs_page_group_init(ret, NULL); } nfs_put_lock_context(l_ctx); @@ -498,9 +510,11 @@ nfs_create_subreq(struct nfs_page *req, offset, count); if (!IS_ERR(ret)) { if (folio) - nfs_page_assign_folio(ret, folio); + nfs_page_assign_folio(ret, folio, + test_bit(PG_PINNED, &req->wb_flags)); else - nfs_page_assign_page(ret, page); + nfs_page_assign_page(ret, page, + test_bit(PG_PINNED, &req->wb_flags)); /* find the last request */ for (last = req->wb_head; last->wb_this_page != req->wb_head; @@ -552,11 +566,21 @@ static void nfs_clear_request(struct nfs_page *req) struct nfs_open_context *ctx; if (folio != NULL) { - folio_put(folio); + if (test_and_clear_bit(PG_PINNED, &req->wb_flags)) { + if (req == req->wb_head) + unpin_user_folio(folio, 1); + } else { + folio_put(folio); + } req->wb_folio = NULL; clear_bit(PG_FOLIO, &req->wb_flags); } else if (page != NULL) { - put_page(page); + if (test_and_clear_bit(PG_PINNED, &req->wb_flags)) { + if (req == req->wb_head) + unpin_user_page(page); + } else { + put_page(page); + } req->wb_page = NULL; } if (l_ctx != NULL) { diff --git a/fs/nfs/read.c b/fs/nfs/read.c index 2b70bd2b934b9a..e7497b029d6cee 100644 --- a/fs/nfs/read.c +++ b/fs/nfs/read.c @@ -324,7 +324,7 @@ int nfs_read_add_folio(struct nfs_pageio_descriptor *pgio, aligned_len = min_t(unsigned int, ALIGN(len, rsize), fsize); - new = nfs_page_create_from_folio(ctx, folio, 0, aligned_len); + new = nfs_page_create_from_folio(ctx, folio, false, 0, aligned_len); if (IS_ERR(new)) { error = PTR_ERR(new); if (nfs_netfs_folio_unlock(folio)) diff --git a/fs/nfs/write.c b/fs/nfs/write.c index 623e7ef1f73d57..7f173dbbeb9d9a 100644 --- a/fs/nfs/write.c +++ b/fs/nfs/write.c @@ -1089,7 +1089,7 @@ static struct nfs_page *nfs_setup_write_request(struct nfs_open_context *ctx, req = nfs_try_to_update_request(folio, offset, bytes); if (req != NULL) goto out; - req = nfs_page_create_from_folio(ctx, folio, offset, bytes); + req = nfs_page_create_from_folio(ctx, folio, false, offset, bytes); if (IS_ERR(req)) goto out; nfs_inode_add_request(req); diff --git a/include/linux/nfs_page.h b/include/linux/nfs_page.h index 4b9a35dbc062da..fd7aafe7cb54de 100644 --- a/include/linux/nfs_page.h +++ b/include/linux/nfs_page.h @@ -38,6 +38,7 @@ enum { PG_REMOVE, /* page group sync bit in write path */ PG_CONTENDED1, /* Is someone waiting for a lock? */ PG_CONTENDED2, /* Is someone waiting for a lock? */ + PG_PINNED, /* page is pinned by GUP */ }; struct nfs_inode; @@ -125,11 +126,13 @@ struct nfs_pageio_descriptor { extern struct nfs_page *nfs_page_create_from_page(struct nfs_open_context *ctx, struct page *page, + bool pinned, unsigned int pgbase, loff_t offset, unsigned int count); extern struct nfs_page *nfs_page_create_from_folio(struct nfs_open_context *ctx, struct folio *folio, + bool pinned, unsigned int offset, unsigned int count); extern void nfs_release_request(struct nfs_page *); From 6b387ac65185e5824a9b8c47c2a2701422d452e7 Mon Sep 17 00:00:00 2001 From: Pranjal Shrivastava Date: Fri, 14 Aug 2026 14:32:52 +0000 Subject: [PATCH 0184/1352] nfs: track number of pinned pages in nfs_page Track the number of pinned pages in nfs_page to handle unpinning correctly, ensuring that only primary requests perform the final unpinning operation, preventing subrequests from incorrectly performing unpinning on behalf of their parent requests. Add wb_nr_pinned to struct nfs_page to store the count of pinned pages owned by the request. Update request creation and cleanup helpers to initialize and use wb_nr_pinned for primary requests. Use the nfs_page_array_len() helper to calculate the number of pages spanned by a request's offset and length. Signed-off-by: Pranjal Shrivastava Reviewed-by: Christoph Hellwig Signed-off-by: Anna Schumaker --- fs/nfs/pagelist.c | 13 +++++++++---- include/linux/nfs_page.h | 1 + 2 files changed, 10 insertions(+), 4 deletions(-) diff --git a/fs/nfs/pagelist.c b/fs/nfs/pagelist.c index a562cfe2a126d1..b9ccf2a87e3c95 100644 --- a/fs/nfs/pagelist.c +++ b/fs/nfs/pagelist.c @@ -457,6 +457,8 @@ struct nfs_page *nfs_page_create_from_page(struct nfs_open_context *ctx, offset_in_page(offset), count); if (!IS_ERR(ret)) { nfs_page_assign_page(ret, page, pinned); + if (pinned) + ret->wb_nr_pinned = 1; nfs_page_group_init(ret, NULL); } nfs_put_lock_context(l_ctx); @@ -489,6 +491,9 @@ struct nfs_page *nfs_page_create_from_folio(struct nfs_open_context *ctx, ret = nfs_page_create(l_ctx, offset, folio->index, offset, count); if (!IS_ERR(ret)) { nfs_page_assign_folio(ret, folio, pinned); + if (pinned) + ret->wb_nr_pinned = nfs_page_array_len(offset_in_page(offset), + count); nfs_page_group_init(ret, NULL); } nfs_put_lock_context(l_ctx); @@ -567,8 +572,8 @@ static void nfs_clear_request(struct nfs_page *req) if (folio != NULL) { if (test_and_clear_bit(PG_PINNED, &req->wb_flags)) { - if (req == req->wb_head) - unpin_user_folio(folio, 1); + if (req->wb_nr_pinned > 0) + unpin_user_folio(folio, req->wb_nr_pinned); } else { folio_put(folio); } @@ -576,8 +581,8 @@ static void nfs_clear_request(struct nfs_page *req) clear_bit(PG_FOLIO, &req->wb_flags); } else if (page != NULL) { if (test_and_clear_bit(PG_PINNED, &req->wb_flags)) { - if (req == req->wb_head) - unpin_user_page(page); + if (req->wb_nr_pinned > 0) + unpin_user_pages(&page, req->wb_nr_pinned); } else { put_page(page); } diff --git a/include/linux/nfs_page.h b/include/linux/nfs_page.h index fd7aafe7cb54de..080fa3e2358092 100644 --- a/include/linux/nfs_page.h +++ b/include/linux/nfs_page.h @@ -59,6 +59,7 @@ struct nfs_page { struct nfs_page *wb_this_page; /* list of reqs for this page */ struct nfs_page *wb_head; /* head pointer for req list */ unsigned short wb_nio; /* Number of I/O attempts */ + unsigned int wb_nr_pinned; /* Number of pinned pages */ }; struct nfs_pgio_mirror; From 77dc84f2a230154f37f542552e98a42a711965c5 Mon Sep 17 00:00:00 2001 From: Pranjal Shrivastava Date: Fri, 14 Aug 2026 14:32:53 +0000 Subject: [PATCH 0185/1352] nfs: introduce nfs_release_request_list helper Introduce a centralized helper, nfs_release_request_list, to handle the bulk release of nfs_page requests from a list. This serves as a preparatory step for two upcoming improvements: 1. Pin-Aware Cleanup: As we migrate to iov_iter_extract_* API, requests will hold pins (GUP) instead of standard references. The helper ensures that the correct unpinning logic gets applied consistently across all requests in a list. 2. Folio Support: In subsequent patches where nfs_page structures will cover multi-page folios, this helper provides a clean infrastructure to unlock these larger units of I/O in bulk during completion, similar to the pattern in bio_release_pages. Additionally, refactor nfs_read_sync_pgio_error() to utilize this new helper. Signed-off-by: Pranjal Shrivastava Reviewed-by: Shivaji Kant Reviewed-by: Christoph Hellwig Signed-off-by: Anna Schumaker --- fs/nfs/direct.c | 8 +------- fs/nfs/pagelist.c | 17 +++++++++++++++++ include/linux/nfs_page.h | 4 ++-- 3 files changed, 20 insertions(+), 9 deletions(-) diff --git a/fs/nfs/direct.c b/fs/nfs/direct.c index 6a7f487fd43e13..8b094f8ef08778 100644 --- a/fs/nfs/direct.c +++ b/fs/nfs/direct.c @@ -294,13 +294,7 @@ static void nfs_direct_read_completion(struct nfs_pgio_header *hdr) static void nfs_read_sync_pgio_error(struct list_head *head, int error) { - struct nfs_page *req; - - while (!list_empty(head)) { - req = nfs_list_entry(head->next); - nfs_list_remove_request(req); - nfs_release_request(req); - } + nfs_release_request_list(head); } static void nfs_direct_pgio_init(struct nfs_pgio_header *hdr) diff --git a/fs/nfs/pagelist.c b/fs/nfs/pagelist.c index b9ccf2a87e3c95..71f0ce2bc4eabe 100644 --- a/fs/nfs/pagelist.c +++ b/fs/nfs/pagelist.c @@ -628,6 +628,23 @@ void nfs_release_request(struct nfs_page *req) } EXPORT_SYMBOL_GPL(nfs_release_request); +/* + * nfs_release_request_list - Release a list of NFS read/write requests + * @head: list of requests to release + * + * Removes each request from the list and drops it's refcount. + */ +void nfs_release_request_list(struct list_head *head) +{ + while (!list_empty(head)) { + struct nfs_page *req = nfs_list_entry(head->next); + + nfs_list_remove_request(req); + nfs_release_request(req); + } +} +EXPORT_SYMBOL_GPL(nfs_release_request_list); + /* * nfs_generic_pg_test - determine if requests can be coalesced * @desc: pointer to descriptor diff --git a/include/linux/nfs_page.h b/include/linux/nfs_page.h index 080fa3e2358092..c38e4b380be5fd 100644 --- a/include/linux/nfs_page.h +++ b/include/linux/nfs_page.h @@ -136,8 +136,8 @@ extern struct nfs_page *nfs_page_create_from_folio(struct nfs_open_context *ctx, bool pinned, unsigned int offset, unsigned int count); -extern void nfs_release_request(struct nfs_page *); - +void nfs_release_request(struct nfs_page *req); +void nfs_release_request_list(struct list_head *head); extern void nfs_pageio_init(struct nfs_pageio_descriptor *desc, struct inode *inode, From cb70adb9de019e631ec8106947056f7a09254879 Mon Sep 17 00:00:00 2001 From: Pranjal Shrivastava Date: Fri, 14 Aug 2026 14:32:54 +0000 Subject: [PATCH 0186/1352] nfs: migrate direct I/O to iov_iter_extract_pages Migrate the NFS Direct I/O path away from the legacy iov_iter_get_pages_alloc2() API to the modern iov_iter_extract_pages API. The transition aligns NFS with the modern VFS extraction model and serves as a preparatory step for supporting requirements such as page pinning via GUP for DMA. The migration fixes a bug in the Direct I/O loop where pages were being unpinned immediately after request creation. With the new extraction model, pins are held until the I/O is complete. Manual release in the loop is correspondingly updated to only clean up failed pages. Signed-off-by: Pranjal Shrivastava Reviewed-by: Shivaji Kant Reviewed-by: Christoph Hellwig Signed-off-by: Anna Schumaker --- fs/nfs/direct.c | 37 +++++++++++++++++++------------------ 1 file changed, 19 insertions(+), 18 deletions(-) diff --git a/fs/nfs/direct.c b/fs/nfs/direct.c index 8b094f8ef08778..59642e886a364d 100644 --- a/fs/nfs/direct.c +++ b/fs/nfs/direct.c @@ -148,14 +148,8 @@ static void nfs_direct_file_adjust_size_locked(struct inode *inode, static void nfs_direct_release_pages(struct page **pages, unsigned int npages, bool pinned) { - unsigned int i; - - if (pinned) { + if (pinned) unpin_user_pages(pages, npages); - } else { - for (i = 0; i < npages; i++) - put_page(pages[i]); - } } void nfs_init_cinfo_from_dreq(struct nfs_commit_info *cinfo, @@ -334,16 +328,17 @@ static ssize_t nfs_direct_read_schedule_iovec(struct nfs_direct_req *dreq, inode_dio_begin(inode); while (iov_iter_count(iter)) { - struct page **pagevec; + struct page **pagevec = NULL; size_t bytes; size_t pgbase; unsigned npages, i; + bool pinned = iov_iter_extract_will_pin(iter); - result = iov_iter_get_pages_alloc2(iter, &pagevec, - rsize, &pgbase); + result = iov_iter_extract_pages(iter, &pagevec, + rsize, ~0U, 0, &pgbase); if (result < 0) break; - + bytes = result; npages = (result + pgbase + PAGE_SIZE - 1) / PAGE_SIZE; for (i = 0; i < npages; i++) { @@ -351,7 +346,7 @@ static ssize_t nfs_direct_read_schedule_iovec(struct nfs_direct_req *dreq, unsigned int req_len = min_t(size_t, bytes, PAGE_SIZE - pgbase); /* XXX do we need to do the eof zeroing found in async_filler? */ req = nfs_page_create_from_page(dreq->ctx, pagevec[i], - false, pgbase, pos, + pinned, pgbase, pos, req_len); if (IS_ERR(req)) { result = PTR_ERR(req); @@ -360,6 +355,7 @@ static ssize_t nfs_direct_read_schedule_iovec(struct nfs_direct_req *dreq, if (!nfs_pageio_add_request(&desc, req)) { result = desc.pg_error; nfs_release_request(req); + i++; break; } pgbase = 0; @@ -367,7 +363,8 @@ static ssize_t nfs_direct_read_schedule_iovec(struct nfs_direct_req *dreq, requested_bytes += req_len; pos += req_len; } - nfs_direct_release_pages(pagevec, npages, false); + if (i < npages) + nfs_direct_release_pages(pagevec + i, npages - i, pinned); kvfree(pagevec); if (result < 0) break; @@ -871,13 +868,14 @@ static ssize_t nfs_direct_write_schedule_iovec(struct nfs_direct_req *dreq, NFS_I(inode)->write_io += iov_iter_count(iter); while (iov_iter_count(iter)) { - struct page **pagevec; + struct page **pagevec = NULL; size_t bytes; size_t pgbase; unsigned npages, i; + bool pinned = iov_iter_extract_will_pin(iter); - result = iov_iter_get_pages_alloc2(iter, &pagevec, - wsize, &pgbase); + result = iov_iter_extract_pages(iter, &pagevec, + wsize, ~0U, 0, &pgbase); if (result < 0) break; @@ -888,7 +886,7 @@ static ssize_t nfs_direct_write_schedule_iovec(struct nfs_direct_req *dreq, unsigned int req_len = min_t(size_t, bytes, PAGE_SIZE - pgbase); req = nfs_page_create_from_page(dreq->ctx, pagevec[i], - false, pgbase, pos, + pinned, pgbase, pos, req_len); if (IS_ERR(req)) { result = PTR_ERR(req); @@ -898,6 +896,7 @@ static ssize_t nfs_direct_write_schedule_iovec(struct nfs_direct_req *dreq, if (desc.pg_error < 0) { nfs_free_request(req); result = desc.pg_error; + i++; break; } @@ -919,6 +918,7 @@ static ssize_t nfs_direct_write_schedule_iovec(struct nfs_direct_req *dreq, if (desc.pg_error < 0 && desc.pg_error != -EAGAIN) { result = desc.pg_error; nfs_unlock_and_release_request(req); + i++; break; } @@ -932,7 +932,8 @@ static ssize_t nfs_direct_write_schedule_iovec(struct nfs_direct_req *dreq, desc.pg_error = 0; defer = true; } - nfs_direct_release_pages(pagevec, npages, false); + if (i < npages) + nfs_direct_release_pages(pagevec + i, npages - i, pinned); kvfree(pagevec); if (result < 0) break; From c364a07cc8747f0e4e76c01ae5c725206652a1cd Mon Sep 17 00:00:00 2001 From: Pranjal Shrivastava Date: Fri, 14 Aug 2026 14:32:55 +0000 Subject: [PATCH 0187/1352] nfs: introduce nfs_direct_extract_pages helper Introduce nfs_direct_extract_pages() in direct.c to centralize page extraction and request creation for the Direct I/O path. The helper manages extraction from the iters and builds a list of nfs_page requests Refactor nfs_direct_read_schedule_iovec() and nfs_direct_write_schedule_iovec() to utilize the new helper, unifying the extraction logic on both paths. Signed-off-by: Pranjal Shrivastava Reviewed-by: Shivaji Kant Reviewed-by: Christoph Hellwig Signed-off-by: Anna Schumaker --- fs/nfs/direct.c | 130 ++++++++++++++++++++++++------------------------ 1 file changed, 64 insertions(+), 66 deletions(-) diff --git a/fs/nfs/direct.c b/fs/nfs/direct.c index 59642e886a364d..a3c6e8f4ea06a6 100644 --- a/fs/nfs/direct.c +++ b/fs/nfs/direct.c @@ -152,6 +152,50 @@ static void nfs_direct_release_pages(struct page **pages, unsigned int npages, unpin_user_pages(pages, npages); } +static ssize_t nfs_direct_extract_pages(struct nfs_direct_req *dreq, + struct iov_iter *iter, + size_t size, loff_t *pos, + struct list_head *list) +{ + bool pinned = iov_iter_extract_will_pin(iter); + struct page **pagevec = NULL; + ssize_t result, bytes = 0; + int err = 0; + unsigned int npages, i; + size_t pgbase; + + result = iov_iter_extract_pages(iter, &pagevec, size, ~0U, 0, &pgbase); + if (result <= 0) + return result; + + npages = (result + pgbase + PAGE_SIZE - 1) >> PAGE_SHIFT; + for (i = 0; i < npages; i++) { + struct nfs_page *req; + unsigned int req_len = min_t(size_t, result - bytes, PAGE_SIZE - pgbase); + + req = nfs_page_create_from_page(dreq->ctx, pagevec[i], + pinned, pgbase, *pos, + req_len); + if (IS_ERR(req)) { + err = PTR_ERR(req); + break; + } + + list_add_tail(&req->wb_list, list); + pgbase = 0; + bytes += req_len; + *pos += req_len; + } + + if (i < npages) { + iov_iter_revert(iter, result - bytes); + nfs_direct_release_pages(pagevec + i, npages - i, pinned); + } + + kvfree(pagevec); + return bytes ? bytes : err; +} + void nfs_init_cinfo_from_dreq(struct nfs_commit_info *cinfo, struct nfs_direct_req *dreq) { @@ -320,6 +364,7 @@ static ssize_t nfs_direct_read_schedule_iovec(struct nfs_direct_req *dreq, ssize_t result = -EINVAL; size_t requested_bytes = 0; size_t rsize = max_t(size_t, NFS_SERVER(inode)->rsize, PAGE_SIZE); + LIST_HEAD(nfs_page_list); nfs_pageio_init_read(&desc, dreq->inode, false, &nfs_direct_read_completion_ops); @@ -328,44 +373,23 @@ static ssize_t nfs_direct_read_schedule_iovec(struct nfs_direct_req *dreq, inode_dio_begin(inode); while (iov_iter_count(iter)) { - struct page **pagevec = NULL; - size_t bytes; - size_t pgbase; - unsigned npages, i; - bool pinned = iov_iter_extract_will_pin(iter); - - result = iov_iter_extract_pages(iter, &pagevec, - rsize, ~0U, 0, &pgbase); + result = nfs_direct_extract_pages(dreq, iter, rsize, &pos, &nfs_page_list); if (result < 0) break; - bytes = result; - npages = (result + pgbase + PAGE_SIZE - 1) / PAGE_SIZE; - for (i = 0; i < npages; i++) { - struct nfs_page *req; - unsigned int req_len = min_t(size_t, bytes, PAGE_SIZE - pgbase); - /* XXX do we need to do the eof zeroing found in async_filler? */ - req = nfs_page_create_from_page(dreq->ctx, pagevec[i], - pinned, pgbase, pos, - req_len); - if (IS_ERR(req)) { - result = PTR_ERR(req); - break; - } + while (!list_empty(&nfs_page_list)) { + struct nfs_page *req = nfs_list_entry(nfs_page_list.next); + size_t req_len = req->wb_bytes; + + nfs_list_remove_request(req); if (!nfs_pageio_add_request(&desc, req)) { result = desc.pg_error; nfs_release_request(req); - i++; + nfs_release_request_list(&nfs_page_list); break; } - pgbase = 0; - bytes -= req_len; requested_bytes += req_len; - pos += req_len; } - if (i < npages) - nfs_direct_release_pages(pagevec + i, npages - i, pinned); - kvfree(pagevec); if (result < 0) break; } @@ -856,6 +880,7 @@ static ssize_t nfs_direct_write_schedule_iovec(struct nfs_direct_req *dreq, ssize_t result = 0; size_t requested_bytes = 0; size_t wsize = max_t(size_t, NFS_SERVER(inode)->wsize, PAGE_SIZE); + LIST_HEAD(nfs_page_list); bool defer = false; trace_nfs_direct_write_schedule_iovec(dreq); @@ -868,57 +893,32 @@ static ssize_t nfs_direct_write_schedule_iovec(struct nfs_direct_req *dreq, NFS_I(inode)->write_io += iov_iter_count(iter); while (iov_iter_count(iter)) { - struct page **pagevec = NULL; - size_t bytes; - size_t pgbase; - unsigned npages, i; - bool pinned = iov_iter_extract_will_pin(iter); - - result = iov_iter_extract_pages(iter, &pagevec, - wsize, ~0U, 0, &pgbase); + result = nfs_direct_extract_pages(dreq, iter, wsize, &pos, &nfs_page_list); if (result < 0) break; - bytes = result; - npages = (result + pgbase + PAGE_SIZE - 1) / PAGE_SIZE; - for (i = 0; i < npages; i++) { - struct nfs_page *req; - unsigned int req_len = min_t(size_t, bytes, PAGE_SIZE - pgbase); - - req = nfs_page_create_from_page(dreq->ctx, pagevec[i], - pinned, pgbase, pos, - req_len); - if (IS_ERR(req)) { - result = PTR_ERR(req); - break; - } - - if (desc.pg_error < 0) { - nfs_free_request(req); - result = desc.pg_error; - i++; - break; - } - - pgbase = 0; - bytes -= req_len; - requested_bytes += req_len; - pos += req_len; + while (!list_empty(&nfs_page_list)) { + struct nfs_page *req = nfs_list_entry(nfs_page_list.next); + size_t req_len = req->wb_bytes; + nfs_list_remove_request(req); if (defer) { nfs_mark_request_commit(req, NULL, &cinfo, 0); + requested_bytes += req_len; continue; } nfs_lock_request(req); - if (nfs_pageio_add_request(&desc, req)) + if (nfs_pageio_add_request(&desc, req)) { + requested_bytes += req_len; continue; + } /* Exit on hard errors */ if (desc.pg_error < 0 && desc.pg_error != -EAGAIN) { result = desc.pg_error; nfs_unlock_and_release_request(req); - i++; + nfs_release_request_list(&nfs_page_list); break; } @@ -929,12 +929,10 @@ static ssize_t nfs_direct_write_schedule_iovec(struct nfs_direct_req *dreq, spin_unlock(&dreq->lock); nfs_unlock_request(req); nfs_mark_request_commit(req, NULL, &cinfo, 0); + requested_bytes += req_len; desc.pg_error = 0; defer = true; } - if (i < npages) - nfs_direct_release_pages(pagevec + i, npages - i, pinned); - kvfree(pagevec); if (result < 0) break; } From 43788010d223de8569f682e5f6b638b0f9e41126 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Mon, 21 Sep 2026 16:15:29 +0200 Subject: [PATCH 0188/1352] file: let dup_fd() leave the punched hole behind close_range(CLOSE_RANGE_UNSHARE) passed the range it is about to close to dup_fd() so the clone is done without that range. This only works when the last open descriptor falls into the range. A range in the middle of the table is copied like everything else. That's wasteful. Don't copy them. Descriptors in a skipped range stay open in the source fdtable. They are never copied into the new table and no reference is taken on them. Figuring out the size of the table follows the same rule. That drops the special-case it has now. This also means we stop calling ->flush() on fds from a table they were never part of. With the range at the top of the table that was already mostly the case. Link: https://patch.msgid.link/20260921-work-file-close_range_except-v1-1-c20d0b49270d@kernel.org Reviewed-by: Jann Horn Signed-off-by: Christian Brauner (Amutable) --- fs/file.c | 60 ++++++++++++++++++++++++++++++++++++++++++------------- 1 file changed, 46 insertions(+), 14 deletions(-) diff --git a/fs/file.c b/fs/file.c index 628ca07dc4b179..95dbdfee555a6e 100644 --- a/fs/file.c +++ b/fs/file.c @@ -352,27 +352,51 @@ static inline bool fd_is_open(unsigned int fd, const struct fdtable *fdt) return test_bit(fd, fdt->open_fds); } +/* Bits of [range->from, range->to] that fall into word @i of a bitmap. */ +static unsigned long fd_range_word(struct fd_range *range, unsigned int i) +{ + unsigned int first = i * BITS_PER_LONG; + unsigned int last = first + BITS_PER_LONG - 1; + + if (range->to < first || range->from > last) + return 0; + return GENMASK(min(range->to, last) - first, + max(range->from, first) - first); +} + +/* Bits of word @i that dup_fd() leaves behind. */ +static unsigned long dup_fd_dropped_word(unsigned int i, struct fd_range *punch_hole) +{ + if (!punch_hole) + return 0; + return fd_range_word(punch_hole, i); +} + /* * Note that a sane fdtable size always has to be a multiple of * BITS_PER_LONG, since we have bitmaps that are sized by this. * * punch_hole is optional - when close_range() is asked to unshare - * and close, we don't need to copy descriptors in that range, so - * a smaller cloned descriptor table might suffice if the last - * currently opened descriptor falls into that range. + * and close, dup_fd() leaves the descriptors in that range behind, + * so the cloned table only has to reach the last open descriptor + * outside of it. */ static unsigned int sane_fdtable_size(struct fdtable *fdt, struct fd_range *punch_hole) { unsigned int last = find_last_bit(fdt->open_fds, fdt->max_fds); + unsigned int i; if (last == fdt->max_fds) return NR_OPEN_DEFAULT; - if (punch_hole && punch_hole->to >= last && punch_hole->from <= last) { - last = find_last_bit(fdt->open_fds, punch_hole->from); - if (last == punch_hole->from) - return NR_OPEN_DEFAULT; + /* Only words up to the last open descriptor can hold a kept one. */ + i = last / BITS_PER_LONG + 1; + while (i--) { + unsigned long dropped = dup_fd_dropped_word(i, punch_hole); + + if (fdt->open_fds[i] & ~dropped) + return (i + 1) * BITS_PER_LONG; } - return ALIGN(last + 1, BITS_PER_LONG); + return NR_OPEN_DEFAULT; } /* @@ -384,7 +408,8 @@ struct files_struct *dup_fd(struct files_struct *oldf, struct fd_range *punch_ho { struct files_struct *newf; struct file **old_fds, **new_fds; - unsigned int open_files, i; + unsigned int open_files, fd; + unsigned long dropped = 0; struct fdtable *old_fdt, *new_fdt; newf = kmem_cache_alloc(files_cachep, GFP_KERNEL); @@ -451,13 +476,18 @@ struct files_struct *dup_fd(struct files_struct *oldf, struct fd_range *punch_ho * * Instead of trying to placate userspace racing with itself, we * ref the file if we see it and mark the fd slot as unused otherwise. + * Descriptors dup_fd() is asked to leave behind get the same treatment. */ - for (i = open_files; i != 0; i--) { + for (fd = 0; fd < open_files; fd++) { struct file *f = rcu_dereference_raw(*old_fds++); - if (f) { + + if (!(fd % BITS_PER_LONG)) + dropped = dup_fd_dropped_word(fd / BITS_PER_LONG, punch_hole); + if (f && !(dropped & BIT_MASK(fd))) { get_file(f); } else { - __clear_open_fd(open_files - i, new_fdt); + f = NULL; + __clear_open_fd(fd, new_fdt); } rcu_assign_pointer(*new_fds++, f); } @@ -848,10 +878,12 @@ SYSCALL_DEFINE3(close_range, unsigned int, fd, unsigned int, max_fd, swap(cur_fds, fds); } - if (flags & CLOSE_RANGE_CLOEXEC) + if (flags & CLOSE_RANGE_CLOEXEC) { __range_cloexec(cur_fds, fd, max_fd); - else + } else if (!fds) { + /* If we unshared, dup_fd() left the range behind already. */ __range_close(cur_fds, fd, max_fd); + } if (fds) { /* From 47f2a98a6bcfb7046dd7a53e8abb8d10fac4901e Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Mon, 21 Sep 2026 16:15:30 +0200 Subject: [PATCH 0189/1352] selftests/core: test the hole CLOSE_RANGE_UNSHARE leaves behind Cover a range in the middle of a full word: - the descriptors in it are gone from the clone and the ones around it are still there - the slots are handed out again, the word is not left marked full - the table the child cloned from is untouched Link: https://patch.msgid.link/20260921-work-file-close_range_except-v1-2-c20d0b49270d@kernel.org Reviewed-by: Jann Horn Signed-off-by: Christian Brauner (Amutable) --- .../testing/selftests/core/close_range_test.c | 44 +++++++++++++++++++ 1 file changed, 44 insertions(+) diff --git a/tools/testing/selftests/core/close_range_test.c b/tools/testing/selftests/core/close_range_test.c index f14eca63f20c40..afae3462502d4e 100644 --- a/tools/testing/selftests/core/close_range_test.c +++ b/tools/testing/selftests/core/close_range_test.c @@ -236,6 +236,50 @@ TEST(close_range_unshare_capped) EXPECT_EQ(0, WEXITSTATUS(status)); } +TEST(close_range_unshare_hole) +{ + int i, status; + pid_t pid; + struct __clone_args args = { + .flags = CLONE_FILES, + .exit_signal = SIGCHLD, + }; + + /* Fill the first two words of the table. */ + for (i = 3; i < 128; i++) + ASSERT_GE(dup2(0, i), 0); + + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + /* Punch a hole into the second word, behind a full first one. */ + if (sys_close_range(70, 80, CLOSE_RANGE_UNSHARE)) + exit(EXIT_FAILURE); + + for (i = 3; i < 128; i++) { + bool closed = i >= 70 && i <= 80; + + if (closed == (fcntl(i, F_GETFD) != -1)) + exit(EXIT_FAILURE); + } + + /* A stale full bit on word 1 would hand out 128, not 70. */ + if (dup(0) != 70) + exit(EXIT_FAILURE); + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + /* The shared table the child unshared from is untouched. */ + for (i = 3; i < 128; i++) + EXPECT_NE(-1, fcntl(i, F_GETFD)); +} + TEST(close_range_cloexec) { int i, ret; From d38c8cc5f085347ac8ec130c1c42df4e810b7e27 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Mon, 21 Sep 2026 16:15:31 +0200 Subject: [PATCH 0190/1352] file: rename dup_fd()'s punch_hole to range dup_fd() will be taught what to do with the range that was passed to it. It won't always be dropped. Rename it to a plain "range" from "punch_hole". close_range() keeps a pointer named drop for the one case it has today. No functional changes. Link: https://patch.msgid.link/20260921-work-file-close_range_except-v1-3-c20d0b49270d@kernel.org Reviewed-by: Jann Horn Signed-off-by: Christian Brauner (Amutable) --- fs/file.c | 28 ++++++++++++++-------------- 1 file changed, 14 insertions(+), 14 deletions(-) diff --git a/fs/file.c b/fs/file.c index 95dbdfee555a6e..6234548be88bfb 100644 --- a/fs/file.c +++ b/fs/file.c @@ -365,23 +365,23 @@ static unsigned long fd_range_word(struct fd_range *range, unsigned int i) } /* Bits of word @i that dup_fd() leaves behind. */ -static unsigned long dup_fd_dropped_word(unsigned int i, struct fd_range *punch_hole) +static unsigned long dup_fd_dropped_word(unsigned int i, struct fd_range *range) { - if (!punch_hole) + if (!range) return 0; - return fd_range_word(punch_hole, i); + return fd_range_word(range, i); } /* * Note that a sane fdtable size always has to be a multiple of * BITS_PER_LONG, since we have bitmaps that are sized by this. * - * punch_hole is optional - when close_range() is asked to unshare + * range is optional - when close_range() is asked to unshare * and close, dup_fd() leaves the descriptors in that range behind, * so the cloned table only has to reach the last open descriptor * outside of it. */ -static unsigned int sane_fdtable_size(struct fdtable *fdt, struct fd_range *punch_hole) +static unsigned int sane_fdtable_size(struct fdtable *fdt, struct fd_range *range) { unsigned int last = find_last_bit(fdt->open_fds, fdt->max_fds); unsigned int i; @@ -391,7 +391,7 @@ static unsigned int sane_fdtable_size(struct fdtable *fdt, struct fd_range *punc /* Only words up to the last open descriptor can hold a kept one. */ i = last / BITS_PER_LONG + 1; while (i--) { - unsigned long dropped = dup_fd_dropped_word(i, punch_hole); + unsigned long dropped = dup_fd_dropped_word(i, range); if (fdt->open_fds[i] & ~dropped) return (i + 1) * BITS_PER_LONG; @@ -402,9 +402,9 @@ static unsigned int sane_fdtable_size(struct fdtable *fdt, struct fd_range *punc /* * Allocate a new descriptor table and copy contents from the passed in * instance. Returns a pointer to cloned table on success, ERR_PTR() - * on failure. For 'punch_hole' see sane_fdtable_size(). + * on failure. For 'range' see sane_fdtable_size(). */ -struct files_struct *dup_fd(struct files_struct *oldf, struct fd_range *punch_hole) +struct files_struct *dup_fd(struct files_struct *oldf, struct fd_range *range) { struct files_struct *newf; struct file **old_fds, **new_fds; @@ -431,7 +431,7 @@ struct files_struct *dup_fd(struct files_struct *oldf, struct fd_range *punch_ho spin_lock(&oldf->file_lock); old_fdt = files_fdtable(oldf); - open_files = sane_fdtable_size(old_fdt, punch_hole); + open_files = sane_fdtable_size(old_fdt, range); /* * Check whether we need to allocate a larger fd array and fd set. @@ -455,7 +455,7 @@ struct files_struct *dup_fd(struct files_struct *oldf, struct fd_range *punch_ho */ spin_lock(&oldf->file_lock); old_fdt = files_fdtable(oldf); - open_files = sane_fdtable_size(old_fdt, punch_hole); + open_files = sane_fdtable_size(old_fdt, range); } copy_fd_bitmaps(new_fdt, old_fdt, open_files / BITS_PER_LONG); @@ -482,7 +482,7 @@ struct files_struct *dup_fd(struct files_struct *oldf, struct fd_range *punch_ho struct file *f = rcu_dereference_raw(*old_fds++); if (!(fd % BITS_PER_LONG)) - dropped = dup_fd_dropped_word(fd / BITS_PER_LONG, punch_hole); + dropped = dup_fd_dropped_word(fd / BITS_PER_LONG, range); if (f && !(dropped & BIT_MASK(fd))) { get_file(f); } else { @@ -858,7 +858,7 @@ SYSCALL_DEFINE3(close_range, unsigned int, fd, unsigned int, max_fd, return -EINVAL; if ((flags & CLOSE_RANGE_UNSHARE) && atomic_read(&cur_fds->count) > 1) { - struct fd_range range = {fd, max_fd}, *punch_hole = ⦥ + struct fd_range range = {fd, max_fd}, *drop = ⦥ /* * If the caller requested all fds to be made cloexec we always @@ -866,9 +866,9 @@ SYSCALL_DEFINE3(close_range, unsigned int, fd, unsigned int, max_fd, * use them. */ if (flags & CLOSE_RANGE_CLOEXEC) - punch_hole = NULL; + drop = NULL; - fds = dup_fd(cur_fds, punch_hole); + fds = dup_fd(cur_fds, drop); if (IS_ERR(fds)) return PTR_ERR(fds); /* From 21a08478fa64abec2dc227d9a06c57a80d2c25fd Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Mon, 21 Sep 2026 16:15:32 +0200 Subject: [PATCH 0191/1352] file: let dup_fd() drop everything outside of the range Add FD_RANGE_EXCEPT. It turns the meaning of range around. Instead of indicating that the descriptors in the range are the ones that are left out of the copy they indicate the range that makes it into the copy. All other files are left behind. The clone only has to reach the last open descriptor inside the range. Nothing passes the flag yet. Link: https://patch.msgid.link/20260921-work-file-close_range_except-v1-4-c20d0b49270d@kernel.org Reviewed-by: Jann Horn Signed-off-by: Christian Brauner (Amutable) --- fs/file.c | 15 ++++++++++----- include/linux/fdtable.h | 6 ++++++ 2 files changed, 16 insertions(+), 5 deletions(-) diff --git a/fs/file.c b/fs/file.c index 6234548be88bfb..ae7d021398dda5 100644 --- a/fs/file.c +++ b/fs/file.c @@ -367,19 +367,24 @@ static unsigned long fd_range_word(struct fd_range *range, unsigned int i) /* Bits of word @i that dup_fd() leaves behind. */ static unsigned long dup_fd_dropped_word(unsigned int i, struct fd_range *range) { + unsigned long dropped; + if (!range) return 0; - return fd_range_word(range, i); + dropped = fd_range_word(range, i); + if (range->flags & FD_RANGE_EXCEPT) + dropped = ~dropped; + return dropped; } /* * Note that a sane fdtable size always has to be a multiple of * BITS_PER_LONG, since we have bitmaps that are sized by this. * - * range is optional - when close_range() is asked to unshare - * and close, dup_fd() leaves the descriptors in that range behind, - * so the cloned table only has to reach the last open descriptor - * outside of it. + * range is optional. When close_range() is asked to unshare dup_fd() + * will leave any files behind according to the range and its flags. The + * cloned table only has to reach the last open descriptor that is + * carried over. */ static unsigned int sane_fdtable_size(struct fdtable *fdt, struct fd_range *range) { diff --git a/include/linux/fdtable.h b/include/linux/fdtable.h index c45306a9f00723..d6c6c7a3400d94 100644 --- a/include/linux/fdtable.h +++ b/include/linux/fdtable.h @@ -101,8 +101,14 @@ struct task_struct; void put_files_struct(struct files_struct *fs); int unshare_files(void); +enum fd_range_flags { + /* Leave behind all descriptors outside of the specified range. */ + FD_RANGE_EXCEPT = (1U << 0), +}; + struct fd_range { unsigned int from, to; + enum fd_range_flags flags; }; struct files_struct *dup_fd(struct files_struct *, struct fd_range *) __latent_entropy; void do_close_on_exec(struct files_struct *); From ee2efd2c822cf4548470cebce6ab5ba605a11fd8 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Mon, 21 Sep 2026 16:15:33 +0200 Subject: [PATCH 0192/1352] close_range: turn the flags into an enum Turn the close_range() flags into an enum. Makes debugging a lot easier. No functional changes. Link: https://patch.msgid.link/20260921-work-file-close_range_except-v1-5-c20d0b49270d@kernel.org Reviewed-by: Jann Horn Signed-off-by: Christian Brauner (Amutable) --- include/uapi/linux/close_range.h | 21 +++++++++++++++++---- 1 file changed, 17 insertions(+), 4 deletions(-) diff --git a/include/uapi/linux/close_range.h b/include/uapi/linux/close_range.h index 2d804281554c5c..b1b637d74deb2b 100644 --- a/include/uapi/linux/close_range.h +++ b/include/uapi/linux/close_range.h @@ -2,11 +2,24 @@ #ifndef _UAPI_LINUX_CLOSE_RANGE_H #define _UAPI_LINUX_CLOSE_RANGE_H -/* Unshare the file descriptor table before closing file descriptors. */ -#define CLOSE_RANGE_UNSHARE (1U << 1) +/* + * A macro of one of these names defined before this header is parsed, by + * a libc or by a program's own fallback, would replace the enumerator. + */ +#undef CLOSE_RANGE_UNSHARE +#undef CLOSE_RANGE_CLOEXEC -/* Set the FD_CLOEXEC bit instead of closing the file descriptor. */ -#define CLOSE_RANGE_CLOEXEC (1U << 2) +enum close_range_flags { + /* Unshare the file descriptor table before closing file descriptors. */ + CLOSE_RANGE_UNSHARE = (1U << 1), + + /* Set the FD_CLOEXEC bit instead of closing the file descriptor. */ + CLOSE_RANGE_CLOEXEC = (1U << 2), +}; + +/* Keep #ifdef working and let glibc skip its own definitions. */ +#define CLOSE_RANGE_UNSHARE CLOSE_RANGE_UNSHARE +#define CLOSE_RANGE_CLOEXEC CLOSE_RANGE_CLOEXEC #endif /* _UAPI_LINUX_CLOSE_RANGE_H */ From 48236a7b735d781bba6642b78f3b45a9c3ec81b5 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Mon, 21 Sep 2026 16:15:34 +0200 Subject: [PATCH 0193/1352] file: add CLOSE_RANGE_EXCEPT close_range() operates on [fd, max_fd]. Add CLOSE_RANGE_EXCEPT. This new flag instructs it to operate on all file descriptors outside of the range instead. Without any other flags that means it closes everything except the file descriptors in the specified range. Together with CLOSE_RANGE_CLOEXEC it marks every file descriptor except the ones in the range as close-on-exec. With CLOSE_RANGE_UNSHARE dup_fd() never takes a reference on what is dropped. A task between clone(CLONE_FILES | CLONE_VM | CLONE_VFORK) and execve() that has the descriptors for the child in one window can shed the rest of the shared table in one call: close_range(lo, hi, CLOSE_RANGE_UNSHARE | CLOSE_RANGE_EXCEPT) Nothing outside of the window is referenced by the child at any point. Link: https://patch.msgid.link/20260921-work-file-close_range_except-v1-6-c20d0b49270d@kernel.org Reviewed-by: Jann Horn Signed-off-by: Christian Brauner (Amutable) --- fs/file.c | 76 +++++++++++++++++++++++++------- include/uapi/linux/close_range.h | 5 +++ 2 files changed, 64 insertions(+), 17 deletions(-) diff --git a/fs/file.c b/fs/file.c index ae7d021398dda5..39651c7a992e5b 100644 --- a/fs/file.c +++ b/fs/file.c @@ -794,34 +794,67 @@ static inline unsigned last_fd(struct fdtable *fdt) } static inline void __range_cloexec(struct files_struct *cur_fds, - unsigned int fd, unsigned int max_fd) + struct fd_range *range) { struct fdtable *fdt; + unsigned int last; - /* make sure we're using the correct maximum value */ spin_lock(&cur_fds->file_lock); fdt = files_fdtable(cur_fds); - max_fd = min(last_fd(fdt), max_fd); - if (fd <= max_fd) - bitmap_set(fdt->close_on_exec, fd, max_fd - fd + 1); + /* make sure we're using the correct maximum value */ + last = last_fd(fdt); + if (!(range->flags & FD_RANGE_EXCEPT)) { + if (range->from <= last) + bitmap_set(fdt->close_on_exec, range->from, + min(range->to, last) - range->from + 1); + } else { + if (range->from > 0) + bitmap_set(fdt->close_on_exec, 0, + min(range->from - 1, last) + 1); + if (range->to < last) + bitmap_set(fdt->close_on_exec, range->to + 1, + last - range->to); + } spin_unlock(&cur_fds->file_lock); } -static inline void __range_close(struct files_struct *files, unsigned int fd, - unsigned int max_fd) +/* Next open descriptor in [fd, max_fd] that @range selects. */ +static inline unsigned int next_fd_to_close(struct fdtable *fdt, + unsigned int fd, unsigned int max_fd, + struct fd_range *range) +{ + fd = find_next_bit(fdt->open_fds, max_fd + 1, fd); + /* Hop over the window the range keeps. */ + if ((range->flags & FD_RANGE_EXCEPT) && + fd >= range->from && fd <= range->to) { + if (range->to >= max_fd) + return max_fd + 1; + fd = find_next_bit(fdt->open_fds, max_fd + 1, range->to + 1); + } + return fd; +} + +static inline void __range_close(struct files_struct *files, + struct fd_range *range) { struct file *file; struct fdtable *fdt; - unsigned n; + unsigned int fd, max_fd; spin_lock(&files->file_lock); fdt = files_fdtable(files); - n = last_fd(fdt); - max_fd = min(max_fd, n); + if (range->flags & FD_RANGE_EXCEPT) { + /* Outside of the range means the whole table. */ + fd = 0; + max_fd = last_fd(fdt); + } else { + fd = range->from; + max_fd = min(range->to, last_fd(fdt)); + } - for (fd = find_next_bit(fdt->open_fds, max_fd + 1, fd); + for (fd = next_fd_to_close(fdt, fd, max_fd, range); fd <= max_fd; - fd = find_next_bit(fdt->open_fds, max_fd + 1, fd + 1)) { + fd = next_fd_to_close(fdt, fd + 1, max_fd, range)) { file = file_close_fd_locked(files, fd); if (file) { spin_unlock(&files->file_lock); @@ -849,21 +882,30 @@ static inline void __range_close(struct files_struct *files, unsigned int fd, * This closes a range of file descriptors. All file descriptors * from @fd up to and including @max_fd are closed. * Currently, errors to close a given file descriptor are ignored. + * + * With CLOSE_RANGE_EXCEPT the range names what to leave alone instead: + * every open file descriptor outside of [@fd, @max_fd] is closed, or + * marked close-on-exec with CLOSE_RANGE_CLOEXEC. */ SYSCALL_DEFINE3(close_range, unsigned int, fd, unsigned int, max_fd, unsigned int, flags) { struct task_struct *me = current; struct files_struct *cur_fds = me->files, *fds = NULL; + struct fd_range range = {fd, max_fd}; - if (flags & ~(CLOSE_RANGE_UNSHARE | CLOSE_RANGE_CLOEXEC)) + if (flags & ~(CLOSE_RANGE_UNSHARE | CLOSE_RANGE_CLOEXEC | + CLOSE_RANGE_EXCEPT)) return -EINVAL; if (fd > max_fd) return -EINVAL; + if (flags & CLOSE_RANGE_EXCEPT) + range.flags |= FD_RANGE_EXCEPT; + if ((flags & CLOSE_RANGE_UNSHARE) && atomic_read(&cur_fds->count) > 1) { - struct fd_range range = {fd, max_fd}, *drop = ⦥ + struct fd_range *drop = ⦥ /* * If the caller requested all fds to be made cloexec we always @@ -884,10 +926,10 @@ SYSCALL_DEFINE3(close_range, unsigned int, fd, unsigned int, max_fd, } if (flags & CLOSE_RANGE_CLOEXEC) { - __range_cloexec(cur_fds, fd, max_fd); + __range_cloexec(cur_fds, &range); } else if (!fds) { - /* If we unshared, dup_fd() left the range behind already. */ - __range_close(cur_fds, fd, max_fd); + /* If we unshared, dup_fd() already left behind what we'd close. */ + __range_close(cur_fds, &range); } if (fds) { diff --git a/include/uapi/linux/close_range.h b/include/uapi/linux/close_range.h index b1b637d74deb2b..39eddb3ab6138f 100644 --- a/include/uapi/linux/close_range.h +++ b/include/uapi/linux/close_range.h @@ -8,6 +8,7 @@ */ #undef CLOSE_RANGE_UNSHARE #undef CLOSE_RANGE_CLOEXEC +#undef CLOSE_RANGE_EXCEPT enum close_range_flags { /* Unshare the file descriptor table before closing file descriptors. */ @@ -15,11 +16,15 @@ enum close_range_flags { /* Set the FD_CLOEXEC bit instead of closing the file descriptor. */ CLOSE_RANGE_CLOEXEC = (1U << 2), + + /* Act on every file descriptor outside of the given range instead. */ + CLOSE_RANGE_EXCEPT = (1U << 3), }; /* Keep #ifdef working and let glibc skip its own definitions. */ #define CLOSE_RANGE_UNSHARE CLOSE_RANGE_UNSHARE #define CLOSE_RANGE_CLOEXEC CLOSE_RANGE_CLOEXEC +#define CLOSE_RANGE_EXCEPT CLOSE_RANGE_EXCEPT #endif /* _UAPI_LINUX_CLOSE_RANGE_H */ From a73e3ccf294bf3c5044498fb8aa50010234085d1 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Mon, 21 Sep 2026 16:15:35 +0200 Subject: [PATCH 0194/1352] selftests/core: test CLOSE_RANGE_EXCEPT Cover the inverted range: - with plain close the window stays and everything outside of it goes, stdio included - a window at the top, one that cannot hold a descriptor and one of a single descriptor keep just what they name, and so does the unshare form on a table that is not shared - with CLOSE_RANGE_CLOEXEC everything outside of the window is marked and the window is not, in place and in a clone, for a window in the middle, at the bottom, at the top and above the table - with CLOSE_RANGE_UNSHARE the table the child cloned from is untouched, a window near the top of the table comes back whole, one at the bottom keeps stdio, one that cannot hold a descriptor keeps nothing, and the slots left behind are handed out again from the bottom - the bounds are checked before the range is turned around The extra cases came out of walking the window positions that __range_close(), __range_cloexec() and dup_fd() tell apart: at the bottom, in the middle, at the top, above the table, a single descriptor and none, on a shared and on a private table. Link: https://patch.msgid.link/20260921-work-file-close_range_except-v1-7-c20d0b49270d@kernel.org Reviewed-by: Jann Horn Signed-off-by: Christian Brauner (Amutable) --- .../testing/selftests/core/close_range_test.c | 426 ++++++++++++++++++ 1 file changed, 426 insertions(+) diff --git a/tools/testing/selftests/core/close_range_test.c b/tools/testing/selftests/core/close_range_test.c index afae3462502d4e..caeb2f1ea800a5 100644 --- a/tools/testing/selftests/core/close_range_test.c +++ b/tools/testing/selftests/core/close_range_test.c @@ -36,6 +36,14 @@ static inline int sys_close_range(unsigned int fd, unsigned int max_fd, return syscall(__NR_close_range, fd, max_fd, flags); } +static void clear_cloexec(const int *fds, size_t n) +{ + size_t i; + + for (i = 0; i < n; i++) + fcntl(fds[i], F_SETFD, 0); +} + TEST(core_close_range) { int i, ret; @@ -637,6 +645,424 @@ TEST(close_range_cloexec_unshare_syzbot) EXPECT_EQ(close(fd3), 0); } +TEST(close_range_except) +{ + int i, ret, status; + pid_t pid; + int open_fds[101]; + struct __clone_args args = { + .exit_signal = SIGCHLD, + }; + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + int fd; + + fd = open("/dev/null", O_RDONLY); + ASSERT_GE(fd, 0) { + if (errno == ENOENT) + SKIP(return, "Skipping test since /dev/null does not exist"); + } + + open_fds[i] = fd; + } + + /* A range covering everything keeps everything. */ + ret = sys_close_range(0, UINT_MAX, CLOSE_RANGE_EXCEPT); + if (ret < 0) { + if (errno == ENOSYS) + SKIP(return, "close_range() syscall not supported"); + if (errno == EINVAL) + SKIP(return, "close_range() doesn't support CLOSE_RANGE_EXCEPT"); + } + ASSERT_EQ(0, ret); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_NE(-1, fcntl(open_fds[i], F_GETFD)); + + /* The bounds are checked before the range is turned around. */ + EXPECT_EQ(-1, sys_close_range(open_fds[20], open_fds[10], + CLOSE_RANGE_EXCEPT)); + EXPECT_EQ(EINVAL, errno); + + /* Everything above open_fds[50] goes. */ + ASSERT_EQ(0, sys_close_range(0, open_fds[50], CLOSE_RANGE_EXCEPT)); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_EQ(i <= 50, fcntl(open_fds[i], F_GETFD) != -1); + + /* A window in the middle takes stdio with it, so do that in a fork. */ + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + ret = sys_close_range(open_fds[10], open_fds[20], + CLOSE_RANGE_EXCEPT); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i <= 50; i++) { + bool kept = i >= 10 && i <= 20; + + if (kept != (fcntl(open_fds[i], F_GETFD) != -1)) + exit(EXIT_FAILURE); + } + + if (fcntl(STDERR_FILENO, F_GETFD) != -1) + exit(EXIT_FAILURE); + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + /* The fork had a table of its own. */ + for (i = 0; i <= 50; i++) + EXPECT_NE(-1, fcntl(open_fds[i], F_GETFD)); +} + +TEST(close_range_except_cloexec) +{ + int i, ret; + int open_fds[101]; + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + int fd; + + fd = open("/dev/null", O_RDONLY); + ASSERT_GE(fd, 0) { + if (errno == ENOENT) + SKIP(return, "Skipping test since /dev/null does not exist"); + } + + open_fds[i] = fd; + } + + ret = sys_close_range(open_fds[10], open_fds[20], + CLOSE_RANGE_CLOEXEC | CLOSE_RANGE_EXCEPT); + if (ret < 0) { + if (errno == ENOSYS) + SKIP(return, "close_range() syscall not supported"); + if (errno == EINVAL) + SKIP(return, "close_range() doesn't support CLOSE_RANGE_EXCEPT"); + } + ASSERT_EQ(0, ret); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + bool inside = i >= 10 && i <= 20; + int flags = fcntl(open_fds[i], F_GETFD); + + EXPECT_NE(-1, flags); + EXPECT_EQ(inside ? 0 : FD_CLOEXEC, flags & FD_CLOEXEC); + } + + /* stdio sits outside of the window too. */ + EXPECT_EQ(FD_CLOEXEC, fcntl(STDERR_FILENO, F_GETFD) & FD_CLOEXEC); + + /* A window that starts at 0 marks only what lies above it. */ + clear_cloexec(open_fds, ARRAY_SIZE(open_fds)); + ASSERT_EQ(0, fcntl(STDERR_FILENO, F_SETFD, 0)); + ASSERT_EQ(0, sys_close_range(0, open_fds[20], + CLOSE_RANGE_CLOEXEC | CLOSE_RANGE_EXCEPT)); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_EQ(i <= 20 ? 0 : FD_CLOEXEC, + fcntl(open_fds[i], F_GETFD) & FD_CLOEXEC); + EXPECT_EQ(0, fcntl(STDERR_FILENO, F_GETFD) & FD_CLOEXEC); + + /* One at the top marks only what lies below it, stdio included. */ + clear_cloexec(open_fds, ARRAY_SIZE(open_fds)); + ASSERT_EQ(0, sys_close_range(open_fds[80], UINT_MAX, + CLOSE_RANGE_CLOEXEC | CLOSE_RANGE_EXCEPT)); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_EQ(i < 80 ? FD_CLOEXEC : 0, + fcntl(open_fds[i], F_GETFD) & FD_CLOEXEC); + EXPECT_EQ(FD_CLOEXEC, fcntl(STDERR_FILENO, F_GETFD) & FD_CLOEXEC); + + /* One that cannot hold a descriptor marks everything. */ + clear_cloexec(open_fds, ARRAY_SIZE(open_fds)); + ASSERT_EQ(0, fcntl(STDERR_FILENO, F_SETFD, 0)); + ASSERT_EQ(0, sys_close_range(UINT_MAX, UINT_MAX, + CLOSE_RANGE_CLOEXEC | CLOSE_RANGE_EXCEPT)); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_EQ(FD_CLOEXEC, fcntl(open_fds[i], F_GETFD) & FD_CLOEXEC); + EXPECT_EQ(FD_CLOEXEC, fcntl(STDERR_FILENO, F_GETFD) & FD_CLOEXEC); +} + +TEST(close_range_except_cloexec_unshare) +{ + int i, ret, status; + pid_t pid; + int open_fds[101]; + struct __clone_args args = { + .flags = CLONE_FILES, + .exit_signal = SIGCHLD, + }; + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + int fd; + + fd = open("/dev/null", O_RDONLY); + ASSERT_GE(fd, 0) { + if (errno == ENOENT) + SKIP(return, "Skipping test since /dev/null does not exist"); + } + + open_fds[i] = fd; + } + + /* A range covering everything marks nothing. */ + ret = sys_close_range(0, UINT_MAX, + CLOSE_RANGE_CLOEXEC | CLOSE_RANGE_EXCEPT); + if (ret < 0) { + if (errno == ENOSYS) + SKIP(return, "close_range() syscall not supported"); + if (errno == EINVAL) + SKIP(return, "close_range() doesn't support CLOSE_RANGE_EXCEPT"); + } + ASSERT_EQ(0, ret); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_EQ(0, fcntl(open_fds[i], F_GETFD) & FD_CLOEXEC); + + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + ret = sys_close_range(open_fds[10], open_fds[20], + CLOSE_RANGE_UNSHARE | CLOSE_RANGE_CLOEXEC | + CLOSE_RANGE_EXCEPT); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + bool inside = i >= 10 && i <= 20; + int flags = fcntl(open_fds[i], F_GETFD); + + if (flags == -1) + exit(EXIT_FAILURE); + if ((flags & FD_CLOEXEC) != (inside ? 0 : FD_CLOEXEC)) + exit(EXIT_FAILURE); + } + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + /* The shared table the child unshared from is untouched. */ + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_EQ(0, fcntl(open_fds[i], F_GETFD) & FD_CLOEXEC); +} + +TEST(close_range_except_bounds) +{ + int i, c, ret, status; + pid_t pid; + int open_fds[101]; + struct __clone_args args = { + .exit_signal = SIGCHLD, + }; + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + int fd; + + fd = open("/dev/null", O_RDONLY); + ASSERT_GE(fd, 0) { + if (errno == ENOENT) + SKIP(return, "Skipping test since /dev/null does not exist"); + } + + open_fds[i] = fd; + } + + /* A range covering everything keeps everything. */ + ret = sys_close_range(0, UINT_MAX, CLOSE_RANGE_EXCEPT); + if (ret < 0) { + if (errno == ENOSYS) + SKIP(return, "close_range() syscall not supported"); + if (errno == EINVAL) + SKIP(return, "close_range() doesn't support CLOSE_RANGE_EXCEPT"); + } + ASSERT_EQ(0, ret); + + struct { + unsigned int fd, max_fd, flags; + } cases[] = { + /* A window at the top drops everything below it. */ + { open_fds[50], UINT_MAX, CLOSE_RANGE_EXCEPT }, + /* One that cannot hold a descriptor keeps nothing. */ + { UINT_MAX, UINT_MAX, CLOSE_RANGE_EXCEPT }, + /* One of a single descriptor keeps just that. */ + { open_fds[30], open_fds[30], CLOSE_RANGE_EXCEPT }, + /* The unshare form on a table that is not shared acts in place. */ + { open_fds[10], open_fds[20], + CLOSE_RANGE_UNSHARE | CLOSE_RANGE_EXCEPT }, + }; + + /* Each of them takes stdio with it, so do that in a fork. */ + for (c = 0; c < ARRAY_SIZE(cases); c++) { + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + ret = sys_close_range(cases[c].fd, cases[c].max_fd, + cases[c].flags); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + unsigned int fd = open_fds[i]; + bool kept = fd >= cases[c].fd && + fd <= cases[c].max_fd; + + if (kept != (fcntl(fd, F_GETFD) != -1)) + exit(EXIT_FAILURE); + } + + if (fcntl(STDERR_FILENO, F_GETFD) != -1) + exit(EXIT_FAILURE); + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + } + + /* Each fork had a table of its own. */ + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_NE(-1, fcntl(open_fds[i], F_GETFD)); +} + +TEST(close_range_except_unshare) +{ + int i, ret, status; + pid_t pid; + int open_fds[200]; + struct __clone_args args = { + .flags = CLONE_FILES, + .exit_signal = SIGCHLD, + }; + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + int fd; + + /* Odd slots are close-on-exec, which makes no difference here. */ + fd = open("/dev/null", O_RDONLY | (i % 2 ? O_CLOEXEC : 0)); + ASSERT_GE(fd, 0) { + if (errno == ENOENT) + SKIP(return, "Skipping test since /dev/null does not exist"); + } + + open_fds[i] = fd; + } + + /* A range covering everything keeps everything. */ + ret = sys_close_range(0, UINT_MAX, CLOSE_RANGE_EXCEPT); + if (ret < 0) { + if (errno == ENOSYS) + SKIP(return, "close_range() syscall not supported"); + if (errno == EINVAL) + SKIP(return, "close_range() doesn't support CLOSE_RANGE_EXCEPT"); + } + ASSERT_EQ(0, ret); + + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + /* The window sits near the top, so the clone is sized off its end. */ + ret = sys_close_range(open_fds[150], open_fds[160], + CLOSE_RANGE_UNSHARE | CLOSE_RANGE_EXCEPT); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + bool kept = i >= 150 && i <= 160; + + if (kept != (fcntl(open_fds[i], F_GETFD) != -1)) + exit(EXIT_FAILURE); + } + + if (fcntl(STDERR_FILENO, F_GETFD) != -1) + exit(EXIT_FAILURE); + + /* What was left behind is handed out again, from the bottom. */ + if (dup(open_fds[150]) != 0) + exit(EXIT_FAILURE); + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + /* A window at the bottom keeps just that, stdio included. */ + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + ret = sys_close_range(0, open_fds[10], + CLOSE_RANGE_UNSHARE | CLOSE_RANGE_EXCEPT); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + if ((i <= 10) != (fcntl(open_fds[i], F_GETFD) != -1)) + exit(EXIT_FAILURE); + } + + if (fcntl(STDERR_FILENO, F_GETFD) == -1) + exit(EXIT_FAILURE); + + /* The first slot left behind is the next one handed out. */ + if (dup(0) != open_fds[10] + 1) + exit(EXIT_FAILURE); + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + /* A window that cannot hold a descriptor keeps nothing. */ + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + ret = sys_close_range(UINT_MAX, UINT_MAX, + CLOSE_RANGE_UNSHARE | CLOSE_RANGE_EXCEPT); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + if (fcntl(open_fds[i], F_GETFD) != -1) + exit(EXIT_FAILURE); + + if (fcntl(STDERR_FILENO, F_GETFD) != -1) + exit(EXIT_FAILURE); + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + /* The shared table the child unshared from is untouched. */ + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_NE(-1, fcntl(open_fds[i], F_GETFD)); +} + TEST(close_range_bitmap_corruption) { pid_t pid; From 558cf21f0f2777022cb07ac84977d0c01878942f Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Mon, 21 Sep 2026 16:15:36 +0200 Subject: [PATCH 0195/1352] file: let dup_fd() drop only close-on-exec descriptors Add FD_RANGE_CLOEXEC_ONLY. When set dup_fd() leaves everything behind except for fds that are close-on-exec. Without FD_RANGE_EXCEPT the clone loses the close-on-exec descriptors in the range. With it the clone keeps the range and loses the close-on-exec descriptors everywhere else. Descriptors without the flag are carried over either way. Nothing passes the flag yet. Link: https://patch.msgid.link/20260921-work-file-close_range_except-v1-8-c20d0b49270d@kernel.org Reviewed-by: Jann Horn Signed-off-by: Christian Brauner (Amutable) --- fs/file.c | 43 ++++++++++++++++++++++++++++++++--------- include/linux/fdtable.h | 3 +++ 2 files changed, 37 insertions(+), 9 deletions(-) diff --git a/fs/file.c b/fs/file.c index 39651c7a992e5b..8952efd2cf3aff 100644 --- a/fs/file.c +++ b/fs/file.c @@ -365,7 +365,8 @@ static unsigned long fd_range_word(struct fd_range *range, unsigned int i) } /* Bits of word @i that dup_fd() leaves behind. */ -static unsigned long dup_fd_dropped_word(unsigned int i, struct fd_range *range) +static unsigned long dup_fd_dropped_word(struct fdtable *fdt, unsigned int i, + struct fd_range *range) { unsigned long dropped; @@ -374,6 +375,8 @@ static unsigned long dup_fd_dropped_word(unsigned int i, struct fd_range *range) dropped = fd_range_word(range, i); if (range->flags & FD_RANGE_EXCEPT) dropped = ~dropped; + if (range->flags & FD_RANGE_CLOEXEC_ONLY) + dropped &= fdt->close_on_exec[i]; return dropped; } @@ -393,15 +396,37 @@ static unsigned int sane_fdtable_size(struct fdtable *fdt, struct fd_range *rang if (last == fdt->max_fds) return NR_OPEN_DEFAULT; - /* Only words up to the last open descriptor can hold a kept one. */ - i = last / BITS_PER_LONG + 1; - while (i--) { - unsigned long dropped = dup_fd_dropped_word(i, range); + if (!range) + return ALIGN(last + 1, BITS_PER_LONG); + + if (range->flags & FD_RANGE_CLOEXEC_ONLY) { + /* The close-on-exec bits decide what is dropped, walk the words. */ + i = last / BITS_PER_LONG + 1; + while (i--) { + unsigned long dropped = dup_fd_dropped_word(fdt, i, range); + + if (fdt->open_fds[i] & ~dropped) + return (i + 1) * BITS_PER_LONG; + } + return NR_OPEN_DEFAULT; + } - if (fdt->open_fds[i] & ~dropped) - return (i + 1) * BITS_PER_LONG; + if (range->flags & FD_RANGE_EXCEPT) { + /* Only the range is carried over. */ + if (last > range->to) { + last = find_last_bit(fdt->open_fds, range->to + 1); + if (last > range->to) + return NR_OPEN_DEFAULT; + } + if (last < range->from) + return NR_OPEN_DEFAULT; + } else if (last >= range->from && last <= range->to) { + /* The last open descriptor goes, the kept ones sit below the range. */ + last = find_last_bit(fdt->open_fds, range->from); + if (last == range->from) + return NR_OPEN_DEFAULT; } - return NR_OPEN_DEFAULT; + return ALIGN(last + 1, BITS_PER_LONG); } /* @@ -487,7 +512,7 @@ struct files_struct *dup_fd(struct files_struct *oldf, struct fd_range *range) struct file *f = rcu_dereference_raw(*old_fds++); if (!(fd % BITS_PER_LONG)) - dropped = dup_fd_dropped_word(fd / BITS_PER_LONG, range); + dropped = dup_fd_dropped_word(old_fdt, fd / BITS_PER_LONG, range); if (f && !(dropped & BIT_MASK(fd))) { get_file(f); } else { diff --git a/include/linux/fdtable.h b/include/linux/fdtable.h index d6c6c7a3400d94..afdfaad381df16 100644 --- a/include/linux/fdtable.h +++ b/include/linux/fdtable.h @@ -104,6 +104,9 @@ int unshare_files(void); enum fd_range_flags { /* Leave behind all descriptors outside of the specified range. */ FD_RANGE_EXCEPT = (1U << 0), + + /* Only select descriptors that have close-on-exec set. */ + FD_RANGE_CLOEXEC_ONLY = (1U << 1), }; struct fd_range { From 5031e416f79ad4f087201d0b76e8ea05b75fa83f Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Mon, 21 Sep 2026 16:15:37 +0200 Subject: [PATCH 0196/1352] file: add CLOSE_RANGE_CLOEXEC_ONLY CLOSE_RANGE_CLOEXEC marks a range close-on-exec. Nothing happens to these fds unless an exec happens. Add CLOSE_RANGE_CLOEXEC_ONLY which closes the close-on-exec file descriptors in the range. When combined with CLOSE_RANGE_EXCEPT it names the close-on-exec file descriptors that are supposed to survive. Every close-on-exec fd outside of the range is closed. File descriptors without that flag are left alone. Whatever the caller deliberately passes down, stdio, LISTEN_FDS, an inherited pipe, remains where it is. The handful of close-on-exec descriptors the caller still needs for the exec such as the executable, an error pipe, sit in a specific range. That is what a task between clone(CLONE_FILES) and execve() actually wants: clone(CLONE_FILES | CLONE_VM | CLONE_VFORK) child: close_range(lo, hi, CLOSE_RANGE_UNSHARE | CLOSE_RANGE_CLOEXEC_ONLY | CLOSE_RANGE_EXCEPT) child: rearrange descriptors in the now private table child: execve() Between the clone and the close_range() the child holds no reference of its own on any file because copy_files() only bumps the fdtable refcount. So a close() in the parent takes effect immediately. The unshare also never takes a reference on the fds it leaves behind either. The range is expressed in the parent's numbering. So a caller that cannot name the file descriptors it keeps contiguously picks a range wide enough to cover them, unshares, and tidies up with a second close_range() on the table it now owns alone. That one is cheap. To keep nothing, name a range that cannot hold an open descriptor, e.g. close_range(~0U, ~0U, ...). CLOSE_RANGE_CLOEXEC and CLOSE_RANGE_CLOEXEC_ONLY are mutually exclusive. Link: https://patch.msgid.link/20260921-work-file-close_range_except-v1-9-c20d0b49270d@kernel.org Reviewed-by: Jann Horn Signed-off-by: Christian Brauner (Amutable) --- fs/file.c | 30 +++++++++++++++++++++++++++--- include/uapi/linux/close_range.h | 5 +++++ 2 files changed, 32 insertions(+), 3 deletions(-) diff --git a/fs/file.c b/fs/file.c index 8952efd2cf3aff..37f0ba740c393b 100644 --- a/fs/file.c +++ b/fs/file.c @@ -843,18 +843,29 @@ static inline void __range_cloexec(struct files_struct *cur_fds, spin_unlock(&cur_fds->file_lock); } +/* Next open descriptor in [fd, max_fd], or the next close-on-exec one. */ +static inline unsigned int next_open_fd(struct fdtable *fdt, unsigned int fd, + unsigned int max_fd, + struct fd_range *range) +{ + if (range->flags & FD_RANGE_CLOEXEC_ONLY) + return find_next_and_bit(fdt->open_fds, fdt->close_on_exec, + max_fd + 1, fd); + return find_next_bit(fdt->open_fds, max_fd + 1, fd); +} + /* Next open descriptor in [fd, max_fd] that @range selects. */ static inline unsigned int next_fd_to_close(struct fdtable *fdt, unsigned int fd, unsigned int max_fd, struct fd_range *range) { - fd = find_next_bit(fdt->open_fds, max_fd + 1, fd); + fd = next_open_fd(fdt, fd, max_fd, range); /* Hop over the window the range keeps. */ if ((range->flags & FD_RANGE_EXCEPT) && fd >= range->from && fd <= range->to) { if (range->to >= max_fd) return max_fd + 1; - fd = find_next_bit(fdt->open_fds, max_fd + 1, range->to + 1); + fd = next_open_fd(fdt, range->to + 1, max_fd, range); } return fd; } @@ -911,6 +922,12 @@ static inline void __range_close(struct files_struct *files, * With CLOSE_RANGE_EXCEPT the range names what to leave alone instead: * every open file descriptor outside of [@fd, @max_fd] is closed, or * marked close-on-exec with CLOSE_RANGE_CLOEXEC. + * + * With CLOSE_RANGE_CLOEXEC_ONLY only file descriptors that have + * close-on-exec set are closed. Together with CLOSE_RANGE_EXCEPT the + * range names the close-on-exec file descriptors to keep. To keep none + * of them, name a range that cannot hold an open file descriptor, e.g. + * close_range(~0U, ~0U, ...). */ SYSCALL_DEFINE3(close_range, unsigned int, fd, unsigned int, max_fd, unsigned int, flags) @@ -920,7 +937,12 @@ SYSCALL_DEFINE3(close_range, unsigned int, fd, unsigned int, max_fd, struct fd_range range = {fd, max_fd}; if (flags & ~(CLOSE_RANGE_UNSHARE | CLOSE_RANGE_CLOEXEC | - CLOSE_RANGE_EXCEPT)) + CLOSE_RANGE_EXCEPT | CLOSE_RANGE_CLOEXEC_ONLY)) + return -EINVAL; + + /* One marks close-on-exec, the other closes what is marked. */ + if (hweight32(flags & (CLOSE_RANGE_CLOEXEC | + CLOSE_RANGE_CLOEXEC_ONLY)) > 1) return -EINVAL; if (fd > max_fd) @@ -928,6 +950,8 @@ SYSCALL_DEFINE3(close_range, unsigned int, fd, unsigned int, max_fd, if (flags & CLOSE_RANGE_EXCEPT) range.flags |= FD_RANGE_EXCEPT; + if (flags & CLOSE_RANGE_CLOEXEC_ONLY) + range.flags |= FD_RANGE_CLOEXEC_ONLY; if ((flags & CLOSE_RANGE_UNSHARE) && atomic_read(&cur_fds->count) > 1) { struct fd_range *drop = ⦥ diff --git a/include/uapi/linux/close_range.h b/include/uapi/linux/close_range.h index 39eddb3ab6138f..7da9ed95258a2b 100644 --- a/include/uapi/linux/close_range.h +++ b/include/uapi/linux/close_range.h @@ -9,6 +9,7 @@ #undef CLOSE_RANGE_UNSHARE #undef CLOSE_RANGE_CLOEXEC #undef CLOSE_RANGE_EXCEPT +#undef CLOSE_RANGE_CLOEXEC_ONLY enum close_range_flags { /* Unshare the file descriptor table before closing file descriptors. */ @@ -19,12 +20,16 @@ enum close_range_flags { /* Act on every file descriptor outside of the given range instead. */ CLOSE_RANGE_EXCEPT = (1U << 3), + + /* Only close file descriptors that have the FD_CLOEXEC bit set. */ + CLOSE_RANGE_CLOEXEC_ONLY = (1U << 4), }; /* Keep #ifdef working and let glibc skip its own definitions. */ #define CLOSE_RANGE_UNSHARE CLOSE_RANGE_UNSHARE #define CLOSE_RANGE_CLOEXEC CLOSE_RANGE_CLOEXEC #define CLOSE_RANGE_EXCEPT CLOSE_RANGE_EXCEPT +#define CLOSE_RANGE_CLOEXEC_ONLY CLOSE_RANGE_CLOEXEC_ONLY #endif /* _UAPI_LINUX_CLOSE_RANGE_H */ From a637f53f90f51cfdfadf026fbd8aaecfaa8aca26 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Mon, 21 Sep 2026 16:15:38 +0200 Subject: [PATCH 0197/1352] selftests/core: test CLOSE_RANGE_CLOEXEC_ONLY Cover closing what is marked: - close-on-exec descriptors in the range go, the ones outside stay, and a range above the table closes nothing - with CLOSE_RANGE_EXCEPT it is the other way around, and the kept ones keep their flag, for a window in the middle, at the top and at the bottom - descriptors without close-on-exec are never touched, in or out of the range - a range that cannot hold an open descriptor keeps nothing, one that covers everything keeps everything, in place and in a clone - the unshare form leaves the table it was cloned from alone, with and without CLOSE_RANGE_EXCEPT - the unshare form keeps the descriptors without the flag that sit in the range it drops from, and hands the dropped slots out again - a kept range above the last descriptor without close-on-exec still comes back, so the clone is sized off the range too Also check that asking for CLOSE_RANGE_CLOEXEC at the same time is refused, whatever else is asked for, and that the bounds are still checked. The extra cases came out of the same walk with the close-on-exec mask on top: a marked and an unmarked descriptor on each side of every window position, and the size of the clone when only unmarked ones are left in the range it drops from. Link: https://patch.msgid.link/20260921-work-file-close_range_except-v1-10-c20d0b49270d@kernel.org Reviewed-by: Jann Horn Signed-off-by: Christian Brauner (Amutable) --- .../testing/selftests/core/close_range_test.c | 481 ++++++++++++++++++ 1 file changed, 481 insertions(+) diff --git a/tools/testing/selftests/core/close_range_test.c b/tools/testing/selftests/core/close_range_test.c index caeb2f1ea800a5..20ecb65e529b01 100644 --- a/tools/testing/selftests/core/close_range_test.c +++ b/tools/testing/selftests/core/close_range_test.c @@ -1063,6 +1063,487 @@ TEST(close_range_except_unshare) EXPECT_NE(-1, fcntl(open_fds[i], F_GETFD)); } +TEST(close_range_cloexec_only) +{ + int i, ret; + int open_fds[101]; + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + int fd; + + /* Odd slots are close-on-exec, even ones are not. */ + fd = open("/dev/null", O_RDONLY | (i % 2 ? O_CLOEXEC : 0)); + ASSERT_GE(fd, 0) { + if (errno == ENOENT) + SKIP(return, "Skipping test since /dev/null does not exist"); + } + + open_fds[i] = fd; + } + + ret = sys_close_range(open_fds[10], open_fds[20], + CLOSE_RANGE_CLOEXEC_ONLY); + if (ret < 0) { + if (errno == ENOSYS) + SKIP(return, "close_range() syscall not supported"); + if (errno == EINVAL) + SKIP(return, "close_range() doesn't support CLOSE_RANGE_CLOEXEC_ONLY"); + } + ASSERT_EQ(0, ret); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + bool closed = i % 2 && i >= 10 && i <= 20; + + EXPECT_EQ(!closed, fcntl(open_fds[i], F_GETFD) != -1); + } + + /* A range above the table closes nothing. */ + ASSERT_EQ(0, sys_close_range(UINT_MAX, UINT_MAX, + CLOSE_RANGE_CLOEXEC_ONLY)); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + bool closed = i % 2 && i >= 10 && i <= 20; + + EXPECT_EQ(!closed, fcntl(open_fds[i], F_GETFD) != -1); + } + + /* Do what an exec would do to the rest. */ + ASSERT_EQ(0, sys_close_range(0, UINT_MAX, CLOSE_RANGE_CLOEXEC_ONLY)); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_EQ(!(i % 2), fcntl(open_fds[i], F_GETFD) != -1); +} + +TEST(close_range_cloexec_only_except) +{ + int i, ret; + int open_fds[101]; + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + int fd; + + fd = open("/dev/null", O_RDONLY | (i % 2 ? O_CLOEXEC : 0)); + ASSERT_GE(fd, 0) { + if (errno == ENOENT) + SKIP(return, "Skipping test since /dev/null does not exist"); + } + + open_fds[i] = fd; + } + + ret = sys_close_range(open_fds[10], open_fds[20], + CLOSE_RANGE_CLOEXEC_ONLY | CLOSE_RANGE_EXCEPT); + if (ret < 0) { + if (errno == ENOSYS) + SKIP(return, "close_range() syscall not supported"); + if (errno == EINVAL) + SKIP(return, "close_range() doesn't support CLOSE_RANGE_CLOEXEC_ONLY"); + } + ASSERT_EQ(0, ret); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + bool kept = !(i % 2) || (i >= 10 && i <= 20); + int flags = i % 2 ? FD_CLOEXEC : 0; + + /* The kept ones keep their flag, so exec still drops them. */ + EXPECT_EQ(kept ? flags : -1, fcntl(open_fds[i], F_GETFD)); + } + + /* A range that cannot hold an open descriptor keeps nothing. */ + ASSERT_EQ(0, sys_close_range(UINT_MAX, UINT_MAX, + CLOSE_RANGE_CLOEXEC_ONLY | + CLOSE_RANGE_EXCEPT)); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_EQ(!(i % 2), fcntl(open_fds[i], F_GETFD) != -1); +} + +TEST(close_range_cloexec_only_except_bounds) +{ + int i, c, ret, status; + pid_t pid; + int open_fds[101]; + struct __clone_args args = { + .exit_signal = SIGCHLD, + }; + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + int fd; + + fd = open("/dev/null", O_RDONLY | (i % 2 ? O_CLOEXEC : 0)); + ASSERT_GE(fd, 0) { + if (errno == ENOENT) + SKIP(return, "Skipping test since /dev/null does not exist"); + } + + open_fds[i] = fd; + } + + /* A range covering everything keeps everything. */ + ret = sys_close_range(0, UINT_MAX, + CLOSE_RANGE_CLOEXEC_ONLY | CLOSE_RANGE_EXCEPT); + if (ret < 0) { + if (errno == ENOSYS) + SKIP(return, "close_range() syscall not supported"); + if (errno == EINVAL) + SKIP(return, "close_range() doesn't support CLOSE_RANGE_CLOEXEC_ONLY"); + } + ASSERT_EQ(0, ret); + + struct { + unsigned int fd, max_fd; + } cases[] = { + /* A window at the top keeps the marked ones in it. */ + { open_fds[80], UINT_MAX }, + /* One at the bottom keeps the marked ones in it. */ + { 0, open_fds[20] }, + /* One that cannot hold a descriptor keeps none of them. */ + { UINT_MAX, UINT_MAX }, + }; + + for (c = 0; c < ARRAY_SIZE(cases); c++) { + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + ret = sys_close_range(cases[c].fd, cases[c].max_fd, + CLOSE_RANGE_CLOEXEC_ONLY | + CLOSE_RANGE_EXCEPT); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + unsigned int fd = open_fds[i]; + bool kept = !(i % 2) || (fd >= cases[c].fd && + fd <= cases[c].max_fd); + int flags = i % 2 ? FD_CLOEXEC : 0; + + if (fcntl(fd, F_GETFD) != (kept ? flags : -1)) + exit(EXIT_FAILURE); + } + + /* stdio is neither marked nor gone. */ + if (fcntl(STDERR_FILENO, F_GETFD) & FD_CLOEXEC) + exit(EXIT_FAILURE); + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + } + + /* Each fork had a table of its own. */ + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_NE(-1, fcntl(open_fds[i], F_GETFD)); +} + +TEST(close_range_cloexec_only_unshare) +{ + int i, ret, status; + pid_t pid; + int open_fds[101]; + struct __clone_args args = { + .flags = CLONE_FILES, + .exit_signal = SIGCHLD, + }; + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + int fd; + + fd = open("/dev/null", O_RDONLY | (i % 2 ? O_CLOEXEC : 0)); + ASSERT_GE(fd, 0) { + if (errno == ENOENT) + SKIP(return, "Skipping test since /dev/null does not exist"); + } + + open_fds[i] = fd; + } + + /* A range covering everything keeps everything. */ + ret = sys_close_range(0, UINT_MAX, + CLOSE_RANGE_CLOEXEC_ONLY | CLOSE_RANGE_EXCEPT); + if (ret < 0) { + if (errno == ENOSYS) + SKIP(return, "close_range() syscall not supported"); + if (errno == EINVAL) + SKIP(return, "close_range() doesn't support CLOSE_RANGE_CLOEXEC_ONLY"); + } + ASSERT_EQ(0, ret); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + ASSERT_NE(-1, fcntl(open_fds[i], F_GETFD)); + + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + ret = sys_close_range(open_fds[10], open_fds[20], + CLOSE_RANGE_UNSHARE | + CLOSE_RANGE_CLOEXEC_ONLY); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + bool closed = i % 2 && i >= 10 && i <= 20; + + if (closed == (fcntl(open_fds[i], F_GETFD) != -1)) + exit(EXIT_FAILURE); + } + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + /* A range at the top keeps the descriptors without the flag in it. */ + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + ret = sys_close_range(open_fds[50], UINT_MAX, + CLOSE_RANGE_UNSHARE | + CLOSE_RANGE_CLOEXEC_ONLY); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + bool closed = i % 2 && i >= 50; + + if (closed == (fcntl(open_fds[i], F_GETFD) != -1)) + exit(EXIT_FAILURE); + } + + /* The first slot left behind is the next one handed out. */ + if (dup(0) != open_fds[51]) + exit(EXIT_FAILURE); + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + /* The shared table the child unshared from is untouched. */ + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_NE(-1, fcntl(open_fds[i], F_GETFD)); +} + +TEST(close_range_cloexec_only_except_unshare) +{ + int i, ret, status; + pid_t pid; + int open_fds[101]; + struct __clone_args args = { + .flags = CLONE_FILES, + .exit_signal = SIGCHLD, + }; + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + int fd; + + fd = open("/dev/null", O_RDONLY | (i % 2 ? O_CLOEXEC : 0)); + ASSERT_GE(fd, 0) { + if (errno == ENOENT) + SKIP(return, "Skipping test since /dev/null does not exist"); + } + + open_fds[i] = fd; + } + + /* A range covering everything keeps everything. */ + ret = sys_close_range(0, UINT_MAX, + CLOSE_RANGE_CLOEXEC_ONLY | CLOSE_RANGE_EXCEPT); + if (ret < 0) { + if (errno == ENOSYS) + SKIP(return, "close_range() syscall not supported"); + if (errno == EINVAL) + SKIP(return, "close_range() doesn't support CLOSE_RANGE_CLOEXEC_ONLY"); + } + ASSERT_EQ(0, ret); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + ASSERT_NE(-1, fcntl(open_fds[i], F_GETFD)); + + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + ret = sys_close_range(open_fds[10], open_fds[20], + CLOSE_RANGE_UNSHARE | + CLOSE_RANGE_CLOEXEC_ONLY | + CLOSE_RANGE_EXCEPT); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + bool kept = !(i % 2) || (i >= 10 && i <= 20); + int flags = i % 2 ? FD_CLOEXEC : 0; + + if (fcntl(open_fds[i], F_GETFD) != (kept ? flags : -1)) + exit(EXIT_FAILURE); + } + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + /* A window that cannot hold a descriptor keeps none of the marked. */ + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + ret = sys_close_range(UINT_MAX, UINT_MAX, + CLOSE_RANGE_UNSHARE | + CLOSE_RANGE_CLOEXEC_ONLY | + CLOSE_RANGE_EXCEPT); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + if ((i % 2) == (fcntl(open_fds[i], F_GETFD) != -1)) + exit(EXIT_FAILURE); + + if (fcntl(STDERR_FILENO, F_GETFD) == -1) + exit(EXIT_FAILURE); + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + /* One that covers everything keeps everything, in a clone too. */ + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + ret = sys_close_range(0, UINT_MAX, + CLOSE_RANGE_UNSHARE | + CLOSE_RANGE_CLOEXEC_ONLY | + CLOSE_RANGE_EXCEPT); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + if (fcntl(open_fds[i], F_GETFD) != (i % 2 ? FD_CLOEXEC : 0)) + exit(EXIT_FAILURE); + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + /* The shared table the child unshared from is untouched. */ + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_NE(-1, fcntl(open_fds[i], F_GETFD)); +} + +TEST(close_range_cloexec_only_except_unshare_sizing) +{ + int i, ret, status; + pid_t pid; + int open_fds[200]; + struct __clone_args args = { + .flags = CLONE_FILES, + .exit_signal = SIGCHLD, + }; + + /* All close-on-exec, so the kept range alone sizes the clone. */ + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + int fd; + + fd = open("/dev/null", O_RDONLY | O_CLOEXEC); + ASSERT_GE(fd, 0) { + if (errno == ENOENT) + SKIP(return, "Skipping test since /dev/null does not exist"); + } + + open_fds[i] = fd; + } + + ret = sys_close_range(0, UINT_MAX, + CLOSE_RANGE_CLOEXEC_ONLY | CLOSE_RANGE_EXCEPT); + if (ret < 0) { + if (errno == ENOSYS) + SKIP(return, "close_range() syscall not supported"); + if (errno == EINVAL) + SKIP(return, "close_range() doesn't support CLOSE_RANGE_CLOEXEC_ONLY"); + } + ASSERT_EQ(0, ret); + + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + ret = sys_close_range(open_fds[150], open_fds[160], + CLOSE_RANGE_UNSHARE | + CLOSE_RANGE_CLOEXEC_ONLY | + CLOSE_RANGE_EXCEPT); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + bool kept = i >= 150 && i <= 160; + + if (kept != (fcntl(open_fds[i], F_GETFD) != -1)) + exit(EXIT_FAILURE); + } + + /* Nothing set close-on-exec on stdio. */ + if (fcntl(STDERR_FILENO, F_GETFD) == -1) + exit(EXIT_FAILURE); + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); +} + +TEST(close_range_cloexec_only_einval) +{ + int ret; + + /* A range covering everything keeps everything, so this only probes. */ + ret = sys_close_range(0, UINT_MAX, + CLOSE_RANGE_CLOEXEC_ONLY | CLOSE_RANGE_EXCEPT); + if (ret < 0) { + if (errno == ENOSYS) + SKIP(return, "close_range() syscall not supported"); + if (errno == EINVAL) + SKIP(return, "close_range() doesn't support CLOSE_RANGE_CLOEXEC_ONLY"); + } + ASSERT_EQ(0, ret); + + EXPECT_EQ(-1, sys_close_range(3, UINT_MAX, CLOSE_RANGE_CLOEXEC | + CLOSE_RANGE_CLOEXEC_ONLY)); + EXPECT_EQ(EINVAL, errno); + + /* The other flags do not make the pair acceptable. */ + EXPECT_EQ(-1, sys_close_range(3, UINT_MAX, CLOSE_RANGE_UNSHARE | + CLOSE_RANGE_CLOEXEC | + CLOSE_RANGE_CLOEXEC_ONLY | + CLOSE_RANGE_EXCEPT)); + EXPECT_EQ(EINVAL, errno); + + /* The bounds are checked with the new flag too. */ + EXPECT_EQ(-1, sys_close_range(4, 3, CLOSE_RANGE_CLOEXEC_ONLY | + CLOSE_RANGE_EXCEPT)); + EXPECT_EQ(EINVAL, errno); +} + TEST(close_range_bitmap_corruption) { pid_t pid; From d1ef78f0581e856c4238c751e9ae2884ce58c275 Mon Sep 17 00:00:00 2001 From: Mitul Golani Date: Thu, 17 Sep 2026 13:13:22 +0530 Subject: [PATCH 0198/1352] drm/i915/vrr: Disable DC balance by default Disable VRR DC balance by default due to timing issues observed on some panel/TCON combinations. Keep the module parameter to enable DC balance during debugging and to isolate DC balance effects from underlying VRR/display timing issues. --v2: - Make enable_dc_balance a bool and keep it disabled by default; fix the parameter type/value mismatch and correct the description (Chaitanya Kumar Borah, Jani Nikula) - Explain in the commit message why the feature is gated and why a module parameter is used (Jani Nikula) --v3: - Commit message update (Jani Nikula) Fixes: 555819270707 ("drm/i915/vrr: Enable DC Balance") Cc: # v7.0+ Signed-off-by: Mitul Golani Reviewed-by: Ankit Nautiyal Signed-off-by: Ankit Nautiyal Link: https://patch.msgid.link/20260917074322.2606738-1-mitulkumar.ajitkumar.golani@intel.com --- drivers/gpu/drm/i915/display/intel_display_params.c | 4 ++++ drivers/gpu/drm/i915/display/intel_display_params.h | 1 + drivers/gpu/drm/i915/display/intel_vrr.c | 4 +++- 3 files changed, 8 insertions(+), 1 deletion(-) diff --git a/drivers/gpu/drm/i915/display/intel_display_params.c b/drivers/gpu/drm/i915/display/intel_display_params.c index 2aed110c5b090b..ca0ef466bb1039 100644 --- a/drivers/gpu/drm/i915/display/intel_display_params.c +++ b/drivers/gpu/drm/i915/display/intel_display_params.c @@ -120,6 +120,10 @@ intel_display_param_named_unsafe(enable_psr, int, 0400, "(0=disabled, 1=enable up to PSR1, 2=enable up to PSR2) " "Default: -1 (use per-chip default)"); +intel_display_param_named_unsafe(enable_dc_balance, bool, 0400, + "Enable VRR DC balance (0=disabled, 1=enabled). " + "Default: 0 (disabled)"); + intel_display_param_named_unsafe(enable_panel_replay, int, 0400, "Enable Panel Replay (0=disabled, 1=enabled). Default: -1 (use per-chip default)"); diff --git a/drivers/gpu/drm/i915/display/intel_display_params.h b/drivers/gpu/drm/i915/display/intel_display_params.h index ba01aeaf894407..5c5a1a1358c32a 100644 --- a/drivers/gpu/drm/i915/display/intel_display_params.h +++ b/drivers/gpu/drm/i915/display/intel_display_params.h @@ -46,6 +46,7 @@ struct drm_printer; param(bool, enable_dp_mst, true, 0600) \ param(int, enable_fbc, -1, 0600) \ param(int, enable_psr, -1, 0600) \ + param(bool, enable_dc_balance, false, 0600) \ param(int, enable_panel_replay, -1, 0600) \ param(bool, psr_safest_params, false, 0400) \ param(bool, enable_psr2_sel_fetch, true, 0400) \ diff --git a/drivers/gpu/drm/i915/display/intel_vrr.c b/drivers/gpu/drm/i915/display/intel_vrr.c index a516ba34e0d02a..d43a06907d18a9 100644 --- a/drivers/gpu/drm/i915/display/intel_vrr.c +++ b/drivers/gpu/drm/i915/display/intel_vrr.c @@ -441,10 +441,12 @@ static bool intel_vrr_dc_balance_possible(const struct intel_crtc_state *crtc_st static void intel_vrr_dc_balance_compute_config(struct intel_crtc_state *crtc_state) { + struct intel_display *display = to_intel_display(crtc_state); int guardband_usec, adjustment_usec; struct drm_display_mode *adjusted_mode = &crtc_state->hw.adjusted_mode; - if (!intel_vrr_dc_balance_possible(crtc_state) || !crtc_state->vrr.enable) + if (!intel_vrr_dc_balance_possible(crtc_state) || + !crtc_state->vrr.enable || !display->params.enable_dc_balance) return; crtc_state->vrr.dc_balance.vmax = crtc_state->vrr.vmax; From 4ffea6c107c835264577bc4ffc30f4fd8debc313 Mon Sep 17 00:00:00 2001 From: Vinod Govindapillai Date: Wed, 16 Sep 2026 22:56:02 +0300 Subject: [PATCH 0199/1352] drm/i915/fbc: remove uint16 from supported fbc formats in xe3plpd As UINT16 pixel formats are not listed as supported formats in any of the display versions so far, there is no point is having these formats listed as supported for FBC. So remove the UINT16 formats from the supported formats for FBC. v2: removed an unused variable Signed-off-by: Vinod Govindapillai Reviewed-by: Mika Kahola Link: https://patch.msgid.link/20260916195602.939242-1-vinod.govindapillai@intel.com --- drivers/gpu/drm/i915/display/intel_fbc.c | 12 +----------- 1 file changed, 1 insertion(+), 11 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_fbc.c b/drivers/gpu/drm/i915/display/intel_fbc.c index f61b4a218d6ef4..5694d20773f83c 100644 --- a/drivers/gpu/drm/i915/display/intel_fbc.c +++ b/drivers/gpu/drm/i915/display/intel_fbc.c @@ -1196,23 +1196,13 @@ xe3p_lpd_fbc_fp16_format_is_valid(const struct intel_plane_state *plane_state) static bool xe3p_lpd_fbc_pixel_format_is_valid(const struct intel_plane_state *plane_state) { - const struct drm_framebuffer *fb = plane_state->hw.fb; - if (lnl_fbc_pixel_format_is_valid(plane_state)) return true; if (xe3p_lpd_fbc_fp16_format_is_valid(plane_state)) return true; - switch (fb->format->format) { - case DRM_FORMAT_XRGB16161616: - case DRM_FORMAT_XBGR16161616: - case DRM_FORMAT_ARGB16161616: - case DRM_FORMAT_ABGR16161616: - return true; - default: - return false; - } + return false; } bool intel_fbc_need_pixel_normalizer(const struct intel_plane_state *plane_state) From 432fce6b2965b420b5319e7224a72d2116629b05 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Mon, 10 Aug 2026 10:18:59 -0400 Subject: [PATCH 0200/1352] NFSD: Replace NFS3_ACCESS_FULL in nfsd4_access() Clean up: Remove an NFSv3 constant (NFS3_ACCESS_FULL) used inside an NFSv4 code path. After this patch is applied, the nfs3.h header is no longer an implicit dependency of nfs4proc.c for this value. The definition itself stays in nfs3.h to avoid a kernel-user space API regression. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260810141900.33846-1-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 33dc92d48ce0f7..1f3f13357266b2 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -843,7 +843,9 @@ nfsd4_access(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate, struct nfsd4_access *access = &u->access; u32 access_full; - access_full = NFS3_ACCESS_FULL; + access_full = NFS4_ACCESS_READ | NFS4_ACCESS_LOOKUP | + NFS4_ACCESS_MODIFY | NFS4_ACCESS_EXTEND | + NFS4_ACCESS_DELETE | NFS4_ACCESS_EXECUTE; if (cstate->minorversion >= 2) access_full |= NFS4_ACCESS_XALIST | NFS4_ACCESS_XAREAD | NFS4_ACCESS_XAWRITE; From 7ca811f1c4b76bab711bacba2c34f22c0a2419e2 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Mon, 10 Aug 2026 10:19:00 -0400 Subject: [PATCH 0201/1352] NFSD: Move version-specific ACCESS maps into per-version code nfsd_access() owns three static tables that map on-the-wire ACCESS bits to NFSD_MAY flags, and every NFS version shares them. That puts protocol-version specifics in the version-agnostic VFS layer. The NFSv4.2 extended-attribute bits are wedged into the NFSv3 tables under CONFIG_NFSD_V4. Give each version its own tables in its proc code and pass the matching set into nfsd_access() as a new argument. Splitting the tables also stops the NFSv2-ACL and NFSv3 ACCESS paths from answering for the NFSv4.2 extended-attribute bits. Both use the NFSv4-augmented table whenever CONFIG_NFSD_V4 is set, so an NFSv3 request that sets an xattr bit has it echoed back in the reply even though those bits are undefined for v3. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260810141900.33846-2-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfs2acl.c | 46 ++++++++++++++++++++++++++- fs/nfsd/nfs3proc.c | 41 +++++++++++++++++++++++- fs/nfsd/nfs4proc.c | 42 +++++++++++++++++++++++-- fs/nfsd/vfs.c | 78 ++++++++++------------------------------------ fs/nfsd/vfs.h | 15 ++++++++- 5 files changed, 155 insertions(+), 67 deletions(-) diff --git a/fs/nfsd/nfs2acl.c b/fs/nfsd/nfs2acl.c index aba69dd278a136..33610deda3b0a7 100644 --- a/fs/nfsd/nfs2acl.c +++ b/fs/nfsd/nfs2acl.c @@ -16,6 +16,48 @@ #define NFSDDBG_FACILITY NFSDDBG_PROC +/* + * These maps are identical to the NFSv3 maps (nfs3proc.c). This enables + * the behavior of the two versions to diverge if needed. + */ +static const struct nfsd_access_map nfsd2_regaccess[] = { + { NFS3_ACCESS_READ, NFSD_MAY_READ }, + { NFS3_ACCESS_EXECUTE, NFSD_MAY_EXEC }, + { NFS3_ACCESS_MODIFY, NFSD_MAY_WRITE|NFSD_MAY_TRUNC }, + { NFS3_ACCESS_EXTEND, NFSD_MAY_WRITE }, + { 0, 0 } +}; + +static const struct nfsd_access_map nfsd2_diraccess[] = { + { NFS3_ACCESS_READ, NFSD_MAY_READ }, + { NFS3_ACCESS_LOOKUP, NFSD_MAY_EXEC }, + { NFS3_ACCESS_MODIFY, NFSD_MAY_EXEC|NFSD_MAY_WRITE|NFSD_MAY_TRUNC }, + { NFS3_ACCESS_EXTEND, NFSD_MAY_EXEC|NFSD_MAY_WRITE }, + { NFS3_ACCESS_DELETE, NFSD_MAY_REMOVE }, + { 0, 0 } +}; + +/* + * Some clients - Solaris 2.6 at least, make an access call to the NFS + * server to check for access for things like /dev/null (which really, + * NFSD doesn't care about). So NFSD provides simple access checking + * for those objects, looking mainly at mode bits, ignoring read-only + * filesystem checks. + */ +static const struct nfsd_access_map nfsd2_otheraccess[] = { + { NFS3_ACCESS_READ, NFSD_MAY_READ }, + { NFS3_ACCESS_EXECUTE, NFSD_MAY_EXEC }, + { NFS3_ACCESS_MODIFY, NFSD_MAY_WRITE|NFSD_MAY_LOCAL_ACCESS }, + { NFS3_ACCESS_EXTEND, NFSD_MAY_WRITE|NFSD_MAY_LOCAL_ACCESS }, + { 0, 0 } +}; + +static const struct nfsd_access_maps nfsd2_access_maps = { + .regular = nfsd2_regaccess, + .directory = nfsd2_diraccess, + .other = nfsd2_otheraccess, +}; + /* * NULL call. */ @@ -181,7 +223,9 @@ static __be32 nfsacld_proc_access(struct svc_rqst *rqstp) fh_copy(&resp->fh, &argp->fh); resp->access = argp->access; - resp->status = nfsd_access(rqstp, &resp->fh, &resp->access, NULL); + + resp->status = nfsd_access(rqstp, &resp->fh, &nfsd2_access_maps, + &resp->access, NULL); if (resp->status != nfs_ok) goto out; resp->status = fh_getattr(&resp->fh, &resp->stat); diff --git a/fs/nfsd/nfs3proc.c b/fs/nfsd/nfs3proc.c index 19ab0a713d8212..17bbe5d13f1833 100644 --- a/fs/nfsd/nfs3proc.c +++ b/fs/nfsd/nfs3proc.c @@ -49,6 +49,44 @@ static bool nfsd3_time_in_range(const struct iattr *iap) return true; } +static const struct nfsd_access_map nfsd3_regaccess[] = { + { NFS3_ACCESS_READ, NFSD_MAY_READ }, + { NFS3_ACCESS_EXECUTE, NFSD_MAY_EXEC }, + { NFS3_ACCESS_MODIFY, NFSD_MAY_WRITE|NFSD_MAY_TRUNC }, + { NFS3_ACCESS_EXTEND, NFSD_MAY_WRITE }, + { 0, 0 } +}; + +static const struct nfsd_access_map nfsd3_diraccess[] = { + { NFS3_ACCESS_READ, NFSD_MAY_READ }, + { NFS3_ACCESS_LOOKUP, NFSD_MAY_EXEC }, + { NFS3_ACCESS_MODIFY, NFSD_MAY_EXEC|NFSD_MAY_WRITE|NFSD_MAY_TRUNC }, + { NFS3_ACCESS_EXTEND, NFSD_MAY_EXEC|NFSD_MAY_WRITE }, + { NFS3_ACCESS_DELETE, NFSD_MAY_REMOVE }, + { 0, 0 } +}; + +/* + * Some clients - Solaris 2.6 at least, make an access call to the NFS + * server to check for access for things like /dev/null (which really, + * NFSD doesn't care about). So NFSD provides simple access checking + * for those objects, looking mainly at mode bits, ignoring read-only + * filesystem checks. + */ +static const struct nfsd_access_map nfsd3_otheraccess[] = { + { NFS3_ACCESS_READ, NFSD_MAY_READ }, + { NFS3_ACCESS_EXECUTE, NFSD_MAY_EXEC }, + { NFS3_ACCESS_MODIFY, NFSD_MAY_WRITE|NFSD_MAY_LOCAL_ACCESS }, + { NFS3_ACCESS_EXTEND, NFSD_MAY_WRITE|NFSD_MAY_LOCAL_ACCESS }, + { 0, 0 } +}; + +static const struct nfsd_access_maps nfsd3_access_maps = { + .regular = nfsd3_regaccess, + .directory = nfsd3_diraccess, + .other = nfsd3_otheraccess, +}; + static int nfsd3_iocb_flags(enum nfs3_stable_how how) { switch (how) { @@ -186,7 +224,8 @@ nfsd3_proc_access(struct svc_rqst *rqstp) fh_copy(&resp->fh, &argp->fh); resp->access = argp->access; - resp->status = nfsd_access(rqstp, &resp->fh, &resp->access, NULL); + resp->status = nfsd_access(rqstp, &resp->fh, &nfsd3_access_maps, + &resp->access, NULL); resp->status = nfsd3_map_status(resp->status); return rpc_success; } diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 1f3f13357266b2..52fc5a305d55db 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -72,6 +72,43 @@ MODULE_PARM_DESC(nfsd4_ssc_umount_timeout, #define NFSDDBG_FACILITY NFSDDBG_PROC +static const struct nfsd_access_map nfsd4_regaccess[] = { + { NFS4_ACCESS_READ, NFSD_MAY_READ }, + { NFS4_ACCESS_EXECUTE, NFSD_MAY_EXEC }, + { NFS4_ACCESS_MODIFY, NFSD_MAY_WRITE|NFSD_MAY_TRUNC }, + { NFS4_ACCESS_EXTEND, NFSD_MAY_WRITE }, + { NFS4_ACCESS_XAREAD, NFSD_MAY_READ }, + { NFS4_ACCESS_XAWRITE, NFSD_MAY_WRITE }, + { NFS4_ACCESS_XALIST, NFSD_MAY_READ }, + { 0, 0 } +}; + +static const struct nfsd_access_map nfsd4_diraccess[] = { + { NFS4_ACCESS_READ, NFSD_MAY_READ }, + { NFS4_ACCESS_LOOKUP, NFSD_MAY_EXEC }, + { NFS4_ACCESS_MODIFY, NFSD_MAY_EXEC|NFSD_MAY_WRITE|NFSD_MAY_TRUNC }, + { NFS4_ACCESS_EXTEND, NFSD_MAY_EXEC|NFSD_MAY_WRITE }, + { NFS4_ACCESS_DELETE, NFSD_MAY_REMOVE }, + { NFS4_ACCESS_XAREAD, NFSD_MAY_READ }, + { NFS4_ACCESS_XAWRITE, NFSD_MAY_WRITE }, + { NFS4_ACCESS_XALIST, NFSD_MAY_READ }, + { 0, 0 } +}; + +static const struct nfsd_access_map nfsd4_otheraccess[] = { + { NFS4_ACCESS_READ, NFSD_MAY_READ }, + { NFS4_ACCESS_EXECUTE, NFSD_MAY_EXEC }, + { NFS4_ACCESS_MODIFY, NFSD_MAY_WRITE|NFSD_MAY_LOCAL_ACCESS }, + { NFS4_ACCESS_EXTEND, NFSD_MAY_WRITE|NFSD_MAY_LOCAL_ACCESS }, + { 0, 0 } +}; + +static const struct nfsd_access_maps nfsd4_access_maps = { + .regular = nfsd4_regaccess, + .directory = nfsd4_diraccess, + .other = nfsd4_otheraccess, +}; + static int nfsd4_iocb_flags(enum stable_how4 how) { switch (how) { @@ -852,10 +889,9 @@ nfsd4_access(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate, if (access->ac_req_access & ~access_full) return nfserr_inval; - access->ac_resp_access = access->ac_req_access; - return nfsd_access(rqstp, &cstate->current_fh, &access->ac_resp_access, - &access->ac_supported); + return nfsd_access(rqstp, &cstate->current_fh, &nfsd4_access_maps, + &access->ac_resp_access, &access->ac_supported); } static __be32 diff --git a/fs/nfsd/vfs.c b/fs/nfsd/vfs.c index 807e09521e0c3e..6d865f4f9ba3de 100644 --- a/fs/nfsd/vfs.c +++ b/fs/nfsd/vfs.c @@ -782,64 +782,21 @@ __be32 nfsd4_vfs_fallocate(struct svc_rqst *rqstp, struct svc_fh *fhp, } #endif /* defined(CONFIG_NFSD_V4) */ -/* - * Check server access rights to a file system object +/** + * nfsd_access - Check caller's access rights to a file system object + * @rqstp: RPC transaction context + * @fhp: target NFS filehandle + * @maps: tables mapping on-the-wire access bits to NFSD_MAY flags + * @access: requested access bits on entry, permitted bits on return + * @supported: optional output of the access bits the server supports + * + * Return: nfs_ok on success, otherwise an nfserr status code */ -struct accessmap { - u32 access; - int how; -}; -static struct accessmap nfs3_regaccess[] = { - { NFS3_ACCESS_READ, NFSD_MAY_READ }, - { NFS3_ACCESS_EXECUTE, NFSD_MAY_EXEC }, - { NFS3_ACCESS_MODIFY, NFSD_MAY_WRITE|NFSD_MAY_TRUNC }, - { NFS3_ACCESS_EXTEND, NFSD_MAY_WRITE }, - -#ifdef CONFIG_NFSD_V4 - { NFS4_ACCESS_XAREAD, NFSD_MAY_READ }, - { NFS4_ACCESS_XAWRITE, NFSD_MAY_WRITE }, - { NFS4_ACCESS_XALIST, NFSD_MAY_READ }, -#endif - - { 0, 0 } -}; - -static struct accessmap nfs3_diraccess[] = { - { NFS3_ACCESS_READ, NFSD_MAY_READ }, - { NFS3_ACCESS_LOOKUP, NFSD_MAY_EXEC }, - { NFS3_ACCESS_MODIFY, NFSD_MAY_EXEC|NFSD_MAY_WRITE|NFSD_MAY_TRUNC}, - { NFS3_ACCESS_EXTEND, NFSD_MAY_EXEC|NFSD_MAY_WRITE }, - { NFS3_ACCESS_DELETE, NFSD_MAY_REMOVE }, - -#ifdef CONFIG_NFSD_V4 - { NFS4_ACCESS_XAREAD, NFSD_MAY_READ }, - { NFS4_ACCESS_XAWRITE, NFSD_MAY_WRITE }, - { NFS4_ACCESS_XALIST, NFSD_MAY_READ }, -#endif - - { 0, 0 } -}; - -static struct accessmap nfs3_anyaccess[] = { - /* Some clients - Solaris 2.6 at least, make an access call - * to the server to check for access for things like /dev/null - * (which really, the server doesn't care about). So - * We provide simple access checking for them, looking - * mainly at mode bits, and we make sure to ignore read-only - * filesystem checks - */ - { NFS3_ACCESS_READ, NFSD_MAY_READ }, - { NFS3_ACCESS_EXECUTE, NFSD_MAY_EXEC }, - { NFS3_ACCESS_MODIFY, NFSD_MAY_WRITE|NFSD_MAY_LOCAL_ACCESS }, - { NFS3_ACCESS_EXTEND, NFSD_MAY_WRITE|NFSD_MAY_LOCAL_ACCESS }, - - { 0, 0 } -}; - -__be32 -nfsd_access(struct svc_rqst *rqstp, struct svc_fh *fhp, u32 *access, u32 *supported) +__be32 nfsd_access(struct svc_rqst *rqstp, struct svc_fh *fhp, + const struct nfsd_access_maps *maps, + u32 *access, u32 *supported) { - struct accessmap *map; + const struct nfsd_access_map *map; struct svc_export *export; struct dentry *dentry; u32 query, result = 0, sresult = 0; @@ -853,12 +810,11 @@ nfsd_access(struct svc_rqst *rqstp, struct svc_fh *fhp, u32 *access, u32 *suppor dentry = fhp->fh_dentry; if (d_is_reg(dentry)) - map = nfs3_regaccess; + map = maps->regular; else if (d_is_dir(dentry)) - map = nfs3_diraccess; + map = maps->directory; else - map = nfs3_anyaccess; - + map = maps->other; query = *access; for (; map->access; map++) { @@ -868,7 +824,7 @@ nfsd_access(struct svc_rqst *rqstp, struct svc_fh *fhp, u32 *access, u32 *suppor sresult |= map->access; err2 = nfsd_permission(&rqstp->rq_cred, export, - dentry, map->how); + dentry, map->may); switch (err2) { case nfs_ok: result |= map->access; diff --git a/fs/nfsd/vfs.h b/fs/nfsd/vfs.h index aa7679d4c54adf..3aa4522ca0a4ec 100644 --- a/fs/nfsd/vfs.h +++ b/fs/nfsd/vfs.h @@ -37,6 +37,17 @@ #define NFSD_MAY_CREATE (NFSD_MAY_EXEC|NFSD_MAY_WRITE) #define NFSD_MAY_REMOVE (NFSD_MAY_EXEC|NFSD_MAY_WRITE|NFSD_MAY_TRUNC) +struct nfsd_access_map { + u32 access; + int may; +}; + +struct nfsd_access_maps { + const struct nfsd_access_map *regular; + const struct nfsd_access_map *directory; + const struct nfsd_access_map *other; +}; + struct nfsd_file; /* @@ -100,7 +111,9 @@ __be32 nfsd_create_locked(struct svc_rqst *, struct svc_fh *, __be32 nfsd_create(struct svc_rqst *, struct svc_fh *, char *name, int len, struct nfsd_attrs *attrs, int type, dev_t rdev, struct svc_fh *res); -__be32 nfsd_access(struct svc_rqst *, struct svc_fh *, u32 *, u32 *); +__be32 nfsd_access(struct svc_rqst *rqstp, struct svc_fh *fhp, + const struct nfsd_access_maps *maps, + u32 *access, u32 *supported); __be32 nfsd_create_setattr(struct svc_rqst *rqstp, struct svc_fh *fhp, struct svc_fh *resfhp, struct nfsd_attrs *iap); __be32 nfsd_commit(struct svc_rqst *rqst, struct svc_fh *fhp, From 00ec97cc31df49e622bdf70b7f118f5a43c851cb Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Wed, 12 Aug 2026 10:24:34 -0400 Subject: [PATCH 0202/1352] NFSD: Make the write verifier reset helper available outside vfs.c A subsequent patch moves the NFSv4-specific portions of nfsd4_clone_file_range() out of fs/nfsd/vfs.c and into its caller in fs/nfsd/nfs4proc.c. One of those portions resets the write verifier when the post-clone sync fails. netns.h already exposes nfsd_reset_write_verifier(), but that is the unconditional reset. commit_reset_write_verifier() wraps it with the policy that decides which errors warrant a reset: -EAGAIN and -ESTALE do not indicate a problem with durable storage, so they leave the verifier alone. A caller outside vfs.c has to apply the same policy, so make the wrapper visible rather than duplicate its switch. Rename it to nfsd_maybe_reset_write_verifier() on the way out. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260812142436.35042-2-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/vfs.c | 29 ++++++++++++++++++++--------- fs/nfsd/vfs.h | 5 +++++ 2 files changed, 25 insertions(+), 9 deletions(-) diff --git a/fs/nfsd/vfs.c b/fs/nfsd/vfs.c index 6d865f4f9ba3de..fad51a899062da 100644 --- a/fs/nfsd/vfs.c +++ b/fs/nfsd/vfs.c @@ -352,9 +352,20 @@ nfsd_lookup(struct svc_rqst *rqstp, struct svc_fh *fhp, const char *name, return err; } -static void -commit_reset_write_verifier(struct nfsd_net *nn, struct svc_rqst *rqstp, - int err) +/** + * nfsd_maybe_reset_write_verifier - Reset the write verifier after an I/O error + * @nn: nfsd namespace holding the write verifier + * @rqstp: RPC transaction context + * @err: errno reported by the failed operation + * + * A write verifier reset tells clients that unstable data the server has + * already acknowledged might have been lost. Client response is to resend + * in-flight dirty data. + * + * Context: Process context. + */ +void nfsd_maybe_reset_write_verifier(struct nfsd_net *nn, + struct svc_rqst *rqstp, int err) { switch (err) { case -EAGAIN: @@ -735,7 +746,7 @@ __be32 nfsd4_clone_file_range(struct svc_rqst *rqstp, &nfsd4_get_cstate(rqstp)->current_fh, dst_pos, count, status); - commit_reset_write_verifier(nn, rqstp, status); + nfsd_maybe_reset_write_verifier(nn, rqstp, status); ret = nfserrno(status); } } @@ -1472,21 +1483,21 @@ nfsd_vfs_write(struct svc_rqst *rqstp, struct svc_fh *fhp, break; } if (host_err < 0) { - commit_reset_write_verifier(nn, rqstp, host_err); + nfsd_maybe_reset_write_verifier(nn, rqstp, host_err); goto out_nfserr; } nfsd_stats_io_write_add(nn, exp, *cnt); fsnotify_modify(file); host_err = filemap_check_wb_err(file->f_mapping, since); if (host_err < 0) { - commit_reset_write_verifier(nn, rqstp, host_err); + nfsd_maybe_reset_write_verifier(nn, rqstp, host_err); goto out_nfserr; } if (iocb_flags && fhp->fh_use_wgather) { host_err = wait_for_concurrent_writes(file); if (host_err < 0) - commit_reset_write_verifier(nn, rqstp, host_err); + nfsd_maybe_reset_write_verifier(nn, rqstp, host_err); } out_nfserr: @@ -1662,14 +1673,14 @@ nfsd_commit(struct svc_rqst *rqstp, struct svc_fh *fhp, struct nfsd_file *nf, err2 = filemap_check_wb_err(nf->nf_file->f_mapping, since); if (err2 < 0) - commit_reset_write_verifier(nn, rqstp, err2); + nfsd_maybe_reset_write_verifier(nn, rqstp, err2); err = nfserrno(err2); break; case -EINVAL: err = nfserr_notsupp; break; default: - commit_reset_write_verifier(nn, rqstp, err2); + nfsd_maybe_reset_write_verifier(nn, rqstp, err2); err = nfserrno(err2); } } else diff --git a/fs/nfsd/vfs.h b/fs/nfsd/vfs.h index 3aa4522ca0a4ec..18171ccc6d161f 100644 --- a/fs/nfsd/vfs.h +++ b/fs/nfsd/vfs.h @@ -86,7 +86,12 @@ static inline bool nfsd_attrs_valid(struct nfsd_attrs *attrs) attrs->na_pacl || attrs->na_dpacl); } +struct nfsd_net; + __be32 nfserrno (int errno); +void nfsd_maybe_reset_write_verifier(struct nfsd_net *nn, + struct svc_rqst *rqstp, + int err); __be32 nfsd_cross_mnt(struct svc_rqst *rqstp, struct dentry **dpp, struct svc_export **expp); __be32 nfsd_lookup(struct svc_rqst *, struct svc_fh *, From 1412977082437c21f58cd0c50aeecdbcb00f7b93 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Wed, 12 Aug 2026 10:24:35 -0400 Subject: [PATCH 0203/1352] NFSD: Move NFSv4-specific CLONE logic into nfsd4_clone() nfsd4_clone_file_range() lives in fs/nfsd/vfs.c but reaches into the NFSv4 compound reply buffer: nfsd4_get_cstate() casts rq_resp to a struct nfsd4_compoundres to recover the saved and current file handles a tracepoint wants. That is the only reference to the NFSv4 XDR definitions left in vfs.c, and it puts knowledge of the compound reply layout in the VFS layer. Refactor nfsd4_clone_file_range() to remove NFSv4-specific componentry from fs/nfsd/vfs.c. Splitting nfsd_clone_file_range() and nfsd_clone_sync_range() lets nfsd4_clone() distinguish a clone failure from a sync failure, which it has to do because only the latter invalidates the write verifier. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260812142436.35042-3-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 27 +++++++++++-- fs/nfsd/vfs.c | 99 +++++++++++++++++++++++++--------------------- fs/nfsd/vfs.h | 9 +++-- 3 files changed, 84 insertions(+), 51 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 52fc5a305d55db..88385a161b4d04 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -1524,16 +1524,37 @@ nfsd4_clone(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate, { struct nfsd4_clone *clone = &u->clone; struct nfsd_file *src, *dst; + bool sync_failed = false; + errseq_t since; __be32 status; + int host_err; status = nfsd4_verify_copy(rqstp, cstate, &clone->cl_src_stateid, &src, &clone->cl_dst_stateid, &dst); if (status) goto out; - status = nfsd4_clone_file_range(rqstp, src, clone->cl_src_pos, - dst, clone->cl_dst_pos, clone->cl_count, - EX_ISSYNC(cstate->current_fh.fh_export)); + host_err = nfsd_clone_file_range(src->nf_file, clone->cl_src_pos, + dst->nf_file, clone->cl_dst_pos, + clone->cl_count, &since); + if (!host_err && EX_ISSYNC(cstate->current_fh.fh_export)) { + host_err = nfsd_clone_sync_range(src->nf_file, dst->nf_file, + clone->cl_dst_pos, + clone->cl_count, since); + sync_failed = host_err < 0; + } + if (host_err < 0) { + trace_nfsd_clone_file_range_err(rqstp, &cstate->save_fh, + clone->cl_src_pos, &cstate->current_fh, + clone->cl_dst_pos, clone->cl_count, host_err); + if (sync_failed) { + struct nfsd_net *nn = net_generic(dst->nf_net, + nfsd_net_id); + + nfsd_maybe_reset_write_verifier(nn, rqstp, host_err); + } + } + status = nfserrno(host_err); if (!status && (READ_ONCE(dst->nf_file->f_mode) & FMODE_NOCMTIME) != 0) nfsd_update_cmtime_attr(dst->nf_file, 0); diff --git a/fs/nfsd/vfs.c b/fs/nfsd/vfs.c index fad51a899062da..27683f48360fd8 100644 --- a/fs/nfsd/vfs.c +++ b/fs/nfsd/vfs.c @@ -702,56 +702,67 @@ int nfsd4_is_junction(struct dentry *dentry) return 1; } -static struct nfsd4_compound_state *nfsd4_get_cstate(struct svc_rqst *rqstp) +/** + * nfsd_clone_file_range - Clone a range of one file into another + * @src: file the range is cloned from + * @src_pos: offset in @src where the source range begins + * @dst: file the range is cloned into + * @dst_pos: offset in @dst where the destination range begins + * @count: length of the range, or zero to clone through end-of-file + * @since: receives @dst's writeback error state, sampled before the clone + * + * A caller that has to place the cloned data on durable storage passes + * @since to nfsd_clone_sync_range() once this call succeeds. Sampling + * happens here because a writeback error raised by the clone's own + * dirty pages has to fall inside the sampled interval. + * + * Context: Process context. + * Return: zero on success, or a negative errno + */ +int nfsd_clone_file_range(struct file *src, u64 src_pos, struct file *dst, + u64 dst_pos, u64 count, errseq_t *since) { - return &((struct nfsd4_compoundres *)rqstp->rq_resp)->cstate; + loff_t cloned; + + *since = READ_ONCE(dst->f_wb_err); + cloned = vfs_clone_file_range(src, src_pos, dst, dst_pos, count, 0); + if (cloned < 0) + return cloned; + if (count && cloned != count) + return -EINVAL; + return 0; } -__be32 nfsd4_clone_file_range(struct svc_rqst *rqstp, - struct nfsd_file *nf_src, u64 src_pos, - struct nfsd_file *nf_dst, u64 dst_pos, - u64 count, bool sync) +/** + * nfsd_clone_sync_range - Commit a cloned range to durable storage + * @src: file the range was cloned from, whose metadata is committed too + * @dst: file the range was cloned into + * @dst_pos: offset in @dst where the cloned range begins + * @count: length of the range, or zero if the clone ran to end-of-file + * @since: @dst's writeback error state as sampled by + * nfsd_clone_file_range() + * + * Context: Process context. + * Return: zero on success, or a negative errno + */ +int nfsd_clone_sync_range(struct file *src, struct file *dst, u64 dst_pos, + u64 count, errseq_t since) { - struct file *src = nf_src->nf_file; - struct file *dst = nf_dst->nf_file; - errseq_t since; - loff_t cloned; - __be32 ret = 0; + loff_t dst_end = count ? dst_pos + count - 1 : LLONG_MAX; + int status; - since = READ_ONCE(dst->f_wb_err); - cloned = vfs_clone_file_range(src, src_pos, dst, dst_pos, count, 0); - if (cloned < 0) { - ret = nfserrno(cloned); - goto out_err; - } - if (count && cloned != count) { - ret = nfserrno(-EINVAL); - goto out_err; - } - if (sync) { - loff_t dst_end = count ? dst_pos + count - 1 : LLONG_MAX; - int status = vfs_fsync_range(dst, dst_pos, dst_end, 0); - - if (!status) - status = filemap_check_wb_err(dst->f_mapping, since); - if (!status) - status = commit_inode_metadata(file_inode(src)); - if (status < 0) { - struct nfsd_net *nn = net_generic(nf_dst->nf_net, - nfsd_net_id); - - trace_nfsd_clone_file_range_err(rqstp, - &nfsd4_get_cstate(rqstp)->save_fh, - src_pos, - &nfsd4_get_cstate(rqstp)->current_fh, - dst_pos, - count, status); - nfsd_maybe_reset_write_verifier(nn, rqstp, status); - ret = nfserrno(status); - } + status = vfs_fsync_range(dst, dst_pos, dst_end, 0); + if (!status) + status = filemap_check_wb_err(dst->f_mapping, since); + if (!status) { + /* + * A reflink marks extents shared in the source inode too, + * so the source's metadata has to reach durable storage + * even though its data is untouched. + */ + status = commit_inode_metadata(file_inode(src)); } -out_err: - return ret; + return status; } ssize_t nfsd_copy_file_range(struct file *src, u64 src_pos, struct file *dst, diff --git a/fs/nfsd/vfs.h b/fs/nfsd/vfs.h index 18171ccc6d161f..f0cb184643f2f4 100644 --- a/fs/nfsd/vfs.h +++ b/fs/nfsd/vfs.h @@ -105,10 +105,11 @@ int nfsd_mountpoint(struct dentry *, struct svc_export *); #ifdef CONFIG_NFSD_V4 __be32 nfsd4_vfs_fallocate(struct svc_rqst *, struct svc_fh *, struct file *, loff_t, loff_t, int); -__be32 nfsd4_clone_file_range(struct svc_rqst *rqstp, - struct nfsd_file *nf_src, u64 src_pos, - struct nfsd_file *nf_dst, u64 dst_pos, - u64 count, bool sync); +int nfsd_clone_file_range(struct file *src, u64 src_pos, + struct file *dst, u64 dst_pos, + u64 count, errseq_t *since); +int nfsd_clone_sync_range(struct file *src, struct file *dst, + u64 dst_pos, u64 count, errseq_t since); #endif /* CONFIG_NFSD_V4 */ __be32 nfsd_create_locked(struct svc_rqst *, struct svc_fh *, struct nfsd_attrs *attrs, int type, dev_t rdev, From dc6962ba6d8279e2ff3cd48cffbc390ae44e7442 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Wed, 12 Aug 2026 10:24:36 -0400 Subject: [PATCH 0204/1352] NFSD: Remove xdr-related headers from fs/nfsd/vfs.c Clean up: Nothing in fs/nfsd/vfs.c references an NFSv3 XDR definition. The preceding patch moved the last NFSv4 reference, a struct nfsd4_compoundres dereference that recovered a pair of file handles for a tracepoint, into fs/nfsd/nfs4proc.c. Remove both includes so that the VFS layer no longer names on-the-wire types. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260812142436.35042-4-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/vfs.c | 3 --- 1 file changed, 3 deletions(-) diff --git a/fs/nfsd/vfs.c b/fs/nfsd/vfs.c index 27683f48360fd8..f9131827d391ea 100644 --- a/fs/nfsd/vfs.c +++ b/fs/nfsd/vfs.c @@ -34,12 +34,9 @@ #include #include -#include "xdr3.h" - #ifdef CONFIG_NFSD_V4 #include "acl.h" #include "idmap.h" -#include "xdr4.h" #endif /* CONFIG_NFSD_V4 */ #include "nfsd.h" From ce5ac909517e5c71a0d31397a999d2f4c664e966 Mon Sep 17 00:00:00 2001 From: Jeff Layton Date: Wed, 12 Aug 2026 14:08:14 -0400 Subject: [PATCH 0205/1352] nfsd: pass caller-provided attrmask storage into nfsd4_setup_notify_entry4() nfsd4_setup_notify_entry4() stole 3 words from the xdr stream via xdr_reserve_space() to hold the host-order bmval[3] attrmask that nfsd4_encode_attr_vals() consumes and ne_attrs.attrmask.element points at. Stashing host-endian scratch in an XDR stream buffer is fragile: the buffer layout is not guaranteed by sunrpc, and it blocks moving the encoder to pages or xdrgen. The attrmask only needs to live until the enclosing encode call serializes the notify_entry4, so hand it caller-provided stack storage instead. The two callers keep the words on the stack: up to three concurrent entries for a rename in nfsd4_encode_notify_event(), one in nfsd4_encode_dir_attr_change(). No wire change: the reserved words were never emitted; attr_vals.data/len are still captured relative to xdr->p. Assisted-by: LLM Signed-off-by: Jeff Layton Link: https://patch.msgid.link/20260812-dir-deleg-v1-1-411faa713068@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfs4xdr.c | 28 ++++++++++++++-------------- 1 file changed, 14 insertions(+), 14 deletions(-) diff --git a/fs/nfsd/nfs4xdr.c b/fs/nfsd/nfs4xdr.c index a47eb544b99f6f..7d1b2d6f57f206 100644 --- a/fs/nfsd/nfs4xdr.c +++ b/fs/nfsd/nfs4xdr.c @@ -4378,21 +4378,16 @@ setup_notify_fhandle(struct dentry *dentry, struct nfs4_delegation *dp, static bool nfsd4_setup_notify_entry4(struct notify_entry4 *ne, struct xdr_stream *xdr, struct dentry *dentry, struct nfs4_delegation *dp, - struct nfsd_file *nf, char *name, u32 namelen) + struct nfsd_file *nf, char *name, u32 namelen, + u32 *attrmask) { struct path path = nf->nf_file->f_path; struct nfsd4_fattr_args args = { }; const u32 *reqmask; - uint32_t *attrmask; __be32 status; bool parent; int ret; - /* Reserve space for attrmask */ - attrmask = xdr_reserve_space(xdr, 3 * sizeof(uint32_t)); - if (!attrmask) - return false; - ne->ne_file.data = name; ne->ne_file.len = namelen; ne->ne_attrs.attrmask.element = attrmask; @@ -4476,6 +4471,7 @@ u8 *nfsd4_encode_notify_event(struct xdr_stream *xdr, struct nfsd_notify_event * struct nfs4_delegation *dp, struct nfsd_file *nf, u32 *notify_mask) { + u32 attrmask[3][3] = { }; u8 *p = NULL; *notify_mask = 0; @@ -4484,7 +4480,8 @@ u8 *nfsd4_encode_notify_event(struct xdr_stream *xdr, struct nfsd_notify_event * struct notify_remove4 nr = { }; if (!nfsd4_setup_notify_entry4(&nr.nrm_old_entry, xdr, nne->ne_dentry, dp, - nf, nne->ne_name, nne->ne_namelen)) + nf, nne->ne_name, nne->ne_namelen, + attrmask[0])) goto out_err; p = (u8 *)xdr->p; if (!xdrgen_encode_notify_remove4(xdr, &nr)) @@ -4495,14 +4492,16 @@ u8 *nfsd4_encode_notify_event(struct xdr_stream *xdr, struct nfsd_notify_event * struct notify_remove4 old = { }; if (!nfsd4_setup_notify_entry4(&na.nad_new_entry, xdr, nne->ne_dentry, dp, - nf, nne->ne_name, nne->ne_namelen)) + nf, nne->ne_name, nne->ne_namelen, + attrmask[0])) goto out_err; /* If a file was overwritten, report it in nad_old_entry */ if (nne->ne_target) { if (!nfsd4_setup_notify_entry4(&old.nrm_old_entry, xdr, NULL, dp, nf, - nne->ne_name, nne->ne_namelen)) + nne->ne_name, nne->ne_namelen, + attrmask[1])) goto out_err; na.nad_old_entry.count = 1; na.nad_old_entry.element = &old; @@ -4521,19 +4520,19 @@ u8 *nfsd4_encode_notify_event(struct xdr_stream *xdr, struct nfsd_notify_event * /* Don't send any attributes in the old_entry since they're the same in new */ if (!nfsd4_setup_notify_entry4(&nr.nrn_old_entry.nrm_old_entry, xdr, NULL, dp, nf, nne->ne_name, - nne->ne_namelen)) + nne->ne_namelen, attrmask[0])) goto out_err; if (!nfsd4_setup_notify_entry4(&nr.nrn_new_entry.nad_new_entry, xdr, nne->ne_dentry, dp, nf, newname, - nne->ne_newnamelen)) + nne->ne_newnamelen, attrmask[1])) goto out_err; /* If a file was overwritten, report it in nad_old_entry */ if (nne->ne_target) { if (!nfsd4_setup_notify_entry4(&old.nrm_old_entry, xdr, NULL, dp, nf, newname, - nne->ne_newnamelen)) + nne->ne_newnamelen, attrmask[2])) goto out_err; nr.nrn_new_entry.nad_old_entry.count = 1; nr.nrn_new_entry.nad_old_entry.element = &old; @@ -4569,11 +4568,12 @@ u8 *nfsd4_encode_dir_attr_change(struct xdr_stream *xdr, struct nfs4_delegation { struct dentry *dentry = nf->nf_file->f_path.dentry; struct notify_attr4 na = { }; + u32 attrmask[3] = { }; u8 *p; /* RFC 8881 s10.4.3: ne_file must be a zero-length string for dir attrs */ if (!nfsd4_setup_notify_entry4(&na.na_changed_entry, xdr, - dentry, dp, nf, "", 0)) + dentry, dp, nf, "", 0, attrmask)) return ERR_PTR(-ENOBUFS); /* No requested attributes to report; omit the event */ From a64b1d5bbb29e01c0a65da4b5bc3153a9f195501 Mon Sep 17 00:00:00 2001 From: Jeff Layton Date: Wed, 12 Aug 2026 14:08:15 -0400 Subject: [PATCH 0206/1352] nfsd: back CB_NOTIFY notify_mask words with per-delegation storage nfsd4_cb_notify_prepare() reserved a word from the encoding xdr stream for each notify4's host-order notify_mask, storing the pointer in ncn_nf[].notify_mask.element. That element must survive until the RPC encode re-reads ncn_nf, so the host-endian word lived inside the XDR staging buffer for the whole callback lifetime - the same fragile pattern as the attrmask, and one that keeps host-order bytes in a buffer meant to hold big-endian XDR. ncn_nf is a bounded per-delegation array reused across every CB_NOTIFY, so give it a parallel ncn_masks array with the same lifetime: - allocate/free ncn_masks alongside ncn_nf in alloc_init_dir_deleg() / nfs4_free_dir_deleg() - point notify_mask.element at &ncn_masks[i] (events) and &ncn_masks[count] (dir attr change) The mask backing is now pre-allocated, so the per-word NULL checks in prepare go away. The staging stream holds only encoded XDR. Assisted-by: LLM Signed-off-by: Jeff Layton Link: https://patch.msgid.link/20260812-dir-deleg-v1-2-411faa713068@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfs4state.c | 20 +++++++++----------- fs/nfsd/state.h | 1 + 2 files changed, 10 insertions(+), 11 deletions(-) diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c index 0bd694390b8a26..06e4192bc6938c 100644 --- a/fs/nfsd/nfs4state.c +++ b/fs/nfsd/nfs4state.c @@ -1323,6 +1323,7 @@ static void nfs4_free_dir_deleg(struct nfs4_stid *stid) for (i = 0; i < ncn->ncn_evt_cnt; ++i) nfsd_notify_event_put(ncn->ncn_evt[i]); kfree(ncn->ncn_nf); + kfree(ncn->ncn_masks); for (i = 0; i < NOTIFY4_PAGE_ARRAY_SIZE; i++) { if (!ncn->ncn_pages[i]) break; @@ -1355,6 +1356,11 @@ alloc_init_dir_deleg(struct nfs4_client *clp, struct nfs4_file *fp) nfs4_put_stid(&dp->dl_stid); return NULL; } + ncn->ncn_masks = kcalloc(NOTIFY4_EVENT_QUEUE_SIZE, sizeof(*ncn->ncn_masks), GFP_KERNEL); + if (!ncn->ncn_masks) { + nfs4_put_stid(&dp->dl_stid); + return NULL; + } spin_lock_init(&ncn->ncn_lock); nfsd4_init_cb(&ncn->ncn_cb, dp->dl_stid.sc_client, &nfsd4_cb_notify_ops, NFSPROC4_CLNT_CB_NOTIFY); @@ -3767,14 +3773,9 @@ nfsd4_cb_notify_prepare(struct nfsd4_callback *cb) struct nfsd_notify_event *nne = events[i]; if (!error) { - u32 *maskp = (u32 *)xdr_reserve_space(&stream, sizeof(*maskp)); + u32 *maskp = &ncn->ncn_masks[i]; u8 *p; - if (!maskp) { - error = true; - goto put_event; - } - p = nfsd4_encode_notify_event(&stream, nne, dp, nf, maskp); if (!p) { pr_notice("Could not generate CB_NOTIFY from fsnotify mask 0x%x\n", @@ -3792,13 +3793,10 @@ nfsd4_cb_notify_prepare(struct nfsd4_callback *cb) nfsd_notify_event_put(nne); } if (!error && (dp->dl_notify_mask & BIT(NOTIFY4_CHANGE_DIR_ATTRS))) { - u32 *maskp = (u32 *)xdr_reserve_space(&stream, sizeof(*maskp)); + u32 *maskp = &ncn->ncn_masks[count]; u8 *p; - if (maskp) - p = nfsd4_encode_dir_attr_change(&stream, dp, nf); - else - p = ERR_PTR(-ENOBUFS); + p = nfsd4_encode_dir_attr_change(&stream, dp, nf); if (IS_ERR(p)) { /* diff --git a/fs/nfsd/state.h b/fs/nfsd/state.h index ff1c9fa731aa25..c65b604e29f1be 100644 --- a/fs/nfsd/state.h +++ b/fs/nfsd/state.h @@ -271,6 +271,7 @@ struct nfsd4_cb_notify { struct nfsd_notify_event *ncn_evt[NOTIFY4_EVENT_QUEUE_SIZE]; // list of events struct page *ncn_pages[NOTIFY4_PAGE_ARRAY_SIZE]; // for encoding struct notify4 *ncn_nf; // array of notify4's to be sent + u32 *ncn_masks; // host-order notify_mask backing for ncn_nf[] bool ncn_encode_err; // did encoding fail? struct nfsd4_callback ncn_cb; // notify4 callback }; From 6b5a866533e834e4142948cb2ab70e22c7c88175 Mon Sep 17 00:00:00 2001 From: Ameer Hamza Date: Sat, 15 Aug 2026 03:19:52 +0500 Subject: [PATCH 0207/1352] sunrpc: treat empty auth.unix.gid replies as negative entries When rpc.mountd cannot resolve a uid (getpwuid() or getgrouplist() failure, e.g. while winbind or sssd is briefly unreachable), it answers the auth.unix.gid upcall with zero groups. unix_gid_parse() installs that as a valid positive entry, and svcauth_unix_set_client() then replaces the credential's group list with the empty one on every request, RPCSEC_GSS included via svcauth_gss_set_client(). One failed lookup strips that uid of all supplementary groups on every export for up to mountd's configured TTL (30 minutes by default), long after the NSS backend has recovered. mountd cannot send an empty list for a successful lookup, since getgrouplist(3) always includes at least the user's primary group, so a zero-group reply can only mean the lookup failed. Record it as a negative entry: unix_gid_find() then returns -ENOENT and svcauth_unix_set_client() keeps the groups the RPC credential already carries. This is the fallback that commit 3fc605a2aa38 ("[PATCH] knfsd: allow the server to provide a gid list when using AUTH_UNIX authentication") promised when no answer is available, and the same state try_to_negate_entry() already creates when no listener holds the channel open. Fixes: 3fc605a2aa38 ("[PATCH] knfsd: allow the server to provide a gid list when using AUTH_UNIX authentication") Cc: stable@vger.kernel.org Assisted-by: Claude:claude-fable-5 Signed-off-by: Ameer Hamza Link: https://patch.msgid.link/20260814221953.108837-2-ameer.hamza@truenas.com Signed-off-by: Chuck Lever --- net/sunrpc/svcauth_unix.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/net/sunrpc/svcauth_unix.c b/net/sunrpc/svcauth_unix.c index 31a1bc60a5f6d3..1b5616a0b918c8 100644 --- a/net/sunrpc/svcauth_unix.c +++ b/net/sunrpc/svcauth_unix.c @@ -540,6 +540,13 @@ static int unix_gid_parse(struct cache_detail *cd, if (ugp) { struct cache_head *ch; ug.h.flags = 0; + /* + * mountd sends at least the user's primary group on + * success, so an empty list can only mean the lookup + * failed. Keep the credential's own groups instead. + */ + if (gids == 0) + set_bit(CACHE_NEGATIVE, &ug.h.flags); ug.h.expiry_time = expiry; ch = sunrpc_cache_update(cd, &ug.h, &ugp->h, From 1e7c049d40f35ab70bc7c951f706863ead177fa9 Mon Sep 17 00:00:00 2001 From: Ameer Hamza Date: Sat, 15 Aug 2026 03:19:53 +0500 Subject: [PATCH 0208/1352] sunrpc: honor the netlink unix_gid NEGATIVE flag The unix_gid netlink upcall protocol has an explicit SUNRPC_A_UNIX_GID_NEGATIVE attribute for failed group lookups, and mountd sends it, but sunrpc_nl_parse_one_unix_gid() only allocates an empty group list for a flagged reply without propagating the flag into the entry, so it installs a valid positive entry with zero groups. Such an entry strips the uid of all supplementary groups until it is refreshed or expires: the same defect the previous patch fixes on the classic channel, on a transport that can say "lookup failed" explicitly. Set CACHE_NEGATIVE for flagged replies, as the netlink ip_map path already does for its negative flag. An empty GIDS list without the flag remains a positive entry. Fixes: 0850e8603cd7 ("sunrpc: add netlink upcall for the auth.unix.gid cache") Cc: stable@vger.kernel.org Assisted-by: Claude:claude-fable-5 Signed-off-by: Ameer Hamza Link: https://patch.msgid.link/20260814221953.108837-3-ameer.hamza@truenas.com Signed-off-by: Chuck Lever --- net/sunrpc/svcauth_unix.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/net/sunrpc/svcauth_unix.c b/net/sunrpc/svcauth_unix.c index 1b5616a0b918c8..a68fe44d1f6259 100644 --- a/net/sunrpc/svcauth_unix.c +++ b/net/sunrpc/svcauth_unix.c @@ -737,6 +737,8 @@ static int sunrpc_nl_parse_one_unix_gid(struct cache_detail *cd, boot.tv_sec; if (tb[SUNRPC_A_UNIX_GID_NEGATIVE]) { + /* failed lookup: keep the credential's own groups */ + set_bit(CACHE_NEGATIVE, &ug.h.flags); ug.gi = groups_alloc(0); if (!ug.gi) return -ENOMEM; From da3b19b4d257bc5ffddf0d18be424d1f8277b339 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sat, 15 Aug 2026 12:28:44 -0400 Subject: [PATCH 0209/1352] SUNRPC: Reject a socket that already has an svc_sock attached Writing the same socket descriptor to /proc/fs/nfsd/portlist twice attaches a second svc_sock to one socket. svc_setup_socket() saves the socket's callbacks before installing its own, so the second attach records svc_write_space() as the old write_space callback. svc_udp_init() invokes that callback by way of svc_sock_setbufsize(), and svc_write_space() then calls itself until the kernel stack is exhausted: BUG: TASK stack guard page was hit at ffffc900037d7ff8 svc_write_space+0x90/0x2b0 net/sunrpc/svcsock.c:429 svc_write_space+0xe6/0x2b0 net/sunrpc/svcsock.c:430 ... 700 more ... svc_sock_setbufsize+0x18d/0x220 net/sunrpc/svcsock.c:386 svc_udp_init net/sunrpc/svcsock.c:854 [inline] svc_setup_socket+0xb2f/0x1090 net/sunrpc/svcsock.c:1498 svc_addsock+0x2fd/0x760 net/sunrpc/svcsock.c:1547 __write_ports_addfd fs/nfsd/nfsctl.c:742 [inline] write_ports+0xa5b/0xcc0 fs/nfsd/nfsctl.c:861 nfsctl_transaction_write+0x106/0x1a0 fs/nfsd/nfsctl.c:112 svc_data_ready() and svc_tcp_state_change() chain through their saved callbacks the same way, so a TCP descriptor added twice recurses on the next incoming segment instead. Reaching any of this takes a writer on portlist, and the nfsd filesystem sets no FS_USERNS_MOUNT, so the reproducer needs CAP_SYS_ADMIN in the initial user namespace. Reject a socket that already carries sk_user_data. svc_setup_socket() overwrites that field unconditionally, so a socket some other consumer has claimed is one NFSD would corrupt whether or not the callbacks recurse. Fixes: b41b66d63c73 ("[PATCH] knfsd: allow sockets to be passed to nfsd via 'portlist'") Cc: stable@vger.kernel.org Reported-by: syzbot+54cdc566f64abf51b7f1@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=54cdc566f64abf51b7f1 Link: https://patch.msgid.link/20260815162844.8219-1-cel@kernel.org Reviewed-by: Jeff Layton Signed-off-by: Chuck Lever --- net/sunrpc/svcsock.c | 3 +++ 1 file changed, 3 insertions(+) diff --git a/net/sunrpc/svcsock.c b/net/sunrpc/svcsock.c index 7a423e9ee74d4a..5a2d52284d7514 100644 --- a/net/sunrpc/svcsock.c +++ b/net/sunrpc/svcsock.c @@ -1614,6 +1614,9 @@ int svc_addsock(struct svc_serv *serv, struct net *net, const int fd, err = -EISCONN; if (so->state > SS_UNCONNECTED) goto out; + err = -EBUSY; + if (so->sk->sk_user_data) + goto out; err = -ENOENT; if (!try_module_get(THIS_MODULE)) goto out; From 2daf0f3b10e37275f7a1ba2f1552cbd99548bd42 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sun, 16 Aug 2026 13:19:14 -0400 Subject: [PATCH 0210/1352] NFSD: Fail a pool_threads read whose reply does not fit The reply to a pool_threads read is the list of per-pool thread counts, formatted into a buffer of SIMPLE_TRANSACTION_LIMIT bytes. snprintf() truncates its last write and strlen() measures only what fit, so a reply too long for that buffer ends mid-number with no terminating newline. A pool running 4096 threads is reported as 40. Nothing marks the reply as incomplete, so an administrator reads a plausible but wrong count. Take snprintf()'s return value, which reports the truncation strlen() cannot see, and fail the read with -ENAMETOOLONG when the list does not fit. That is the errno svc_one_xprt_name() already returns for the same condition. Suggested-by: David Laight Fixes: eed2965af1ba ("[PATCH] knfsd: allow admin to set nthreads per node") Cc: stable@vger.kernel.org Link: https://patch.msgid.link/20260812193349.13347-1-david.laight.linux@gmail.com Signed-off-by: Chuck Lever --- fs/nfsd/nfsctl.c | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/fs/nfsd/nfsctl.c b/fs/nfsd/nfsctl.c index c0f10517470be7..efde963d909ad6 100644 --- a/fs/nfsd/nfsctl.c +++ b/fs/nfsd/nfsctl.c @@ -479,7 +479,7 @@ static ssize_t write_pool_threads(struct file *file, char *buf, size_t size) char *mesg = buf; int i; int rv; - int len; + size_t len; int npools; int *nthreads; struct net *net = netns(file); @@ -533,9 +533,13 @@ static ssize_t write_pool_threads(struct file *file, char *buf, size_t size) mesg = buf; size = SIMPLE_TRANSACTION_LIMIT; - for (i = 0; i < npools && size > 0; i++) { - snprintf(mesg, size, "%d%c", nthreads[i], (i == npools-1 ? '\n' : ' ')); - len = strlen(mesg); + for (i = 0; i < npools; i++) { + len = snprintf(mesg, size, "%d%c", nthreads[i], + (i == npools - 1 ? '\n' : ' ')); + if (len >= size) { + rv = -ENAMETOOLONG; + goto out_free; + } size -= len; mesg += len; } From e69f33b9232bf5af67152e4e616abdb028fbd3ab Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sat, 8 Aug 2026 11:40:13 -0400 Subject: [PATCH 0211/1352] SUNRPC: reject a client-side TLS alert record that is not two octets tls_alert_recv() reads two octets from the kvec it is handed and does not check the length (net/handshake/alert.c). xs_sock_process_cmsg() calls it for any alert record, and the alert[] buffer that xs_sock_recv_cmsg() supplies carries no initializer. A one-octet alert body leaves the description read from uninitialized stack and reported through trace_tls_alert_recv(). The peer controls that length. Neither tls_rx_msg_size() nor tls_rx_one_record() enforces the two-octet Alert payload. A TLS 1.3 record carrying only the inner content-type octet decrypts to a zero-length payload. RFC 8446 Section 5.1 requires a record with an Alert type to carry exactly one message, so any other length is malformed. RFC 9289 Section 5 bars RPC-with-TLS from negotiating a version below TLS 1.3, so no other alert framing applies. Require exactly two octets before parsing and return -EACCES otherwise. xs_stream_data_receive() already treats -EACCES as a fatal alert and reports it to the pending tasks. Gate the path on a control message rather than a positive count so that a zero-length record reaches the check. Fixes: cc5d59081fa2 ("sunrpc: fix client side handling of tls alerts") Signed-off-by: Chuck Lever Signed-off-by: Anna Schumaker --- net/sunrpc/xprtsock.c | 20 ++++++++++++++++++-- 1 file changed, 18 insertions(+), 2 deletions(-) diff --git a/net/sunrpc/xprtsock.c b/net/sunrpc/xprtsock.c index 7f60723fa64d84..8274752969219b 100644 --- a/net/sunrpc/xprtsock.c +++ b/net/sunrpc/xprtsock.c @@ -407,9 +407,25 @@ xs_sock_recv_cmsg(struct socket *sock, unsigned int *msg_flags, int flags) iov_iter_kvec(&msg.msg_iter, ITER_DEST, &alert_kvec, 1, alert_kvec.iov_len); ret = sock_recvmsg(sock, &msg, flags); - if (ret > 0) { - if (tls_get_record_type(sock->sk, &u.cmsg) == TLS_RECORD_TYPE_ALERT) + /* put_cmsg() shrinks msg_controllen, so a short one means + * kTLS filled in u.cmsg. + */ + if (ret >= 0 && msg.msg_controllen < sizeof(u)) { + if (tls_get_record_type(sock->sk, &u.cmsg) == + TLS_RECORD_TYPE_ALERT) { + /* RFC 8446 Section 5.1 requires a record with an + * Alert type to carry exactly one message. An alert + * is two octets. tls_alert_recv() reads both without + * checking the length. alert_kvec caps the count at + * two, so a longer record fills it as well. kTLS + * sets MSG_EOR only once the record has been + * drained. + */ + if (ret != sizeof(alert) || + !(msg.msg_flags & MSG_EOR)) + return -EACCES; iov_iter_revert(&msg.msg_iter, ret); + } ret = xs_sock_process_cmsg(sock, &msg, msg_flags, &u.cmsg, -EAGAIN); } From 03faaee7e0811816173c514cd5bfe507bfa3cbe7 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sat, 8 Aug 2026 11:40:14 -0400 Subject: [PATCH 0212/1352] SUNRPC: treat every client-side TLS error alert as fatal xs_sock_process_cmsg() decides whether an alert ends the session by reading the alert's level octet. RFC 8446 Section 6 retired that field. The severity is implicit in the description, and a receiver treats every alert listed in Section 6.2 as an error alert "regardless of the AlertLevel in the message". A peer that aborts with unexpected_message but leaves the legacy octet set to warning makes the client return -EAGAIN. xs_stream_data_receive() wakes no pending task for that error, so RPC Calls queued on a dead TLS session wait for their timeouts to expire. Decide from the alert description instead. close_notify and user_canceled are the closure alerts (RFC 8446 Section 6.1). Every other description ends the session, including one this kernel does not recognize. Fixes: 39067dda1d86 ("SUNRPC: Use new helpers to handle TLS Alerts") Signed-off-by: Chuck Lever Signed-off-by: Anna Schumaker --- net/sunrpc/xprtsock.c | 14 ++++++++++++-- 1 file changed, 12 insertions(+), 2 deletions(-) diff --git a/net/sunrpc/xprtsock.c b/net/sunrpc/xprtsock.c index 8274752969219b..963e46a514217a 100644 --- a/net/sunrpc/xprtsock.c +++ b/net/sunrpc/xprtsock.c @@ -375,8 +375,18 @@ xs_sock_process_cmsg(struct socket *sock, struct msghdr *msg, break; case TLS_RECORD_TYPE_ALERT: tls_alert_recv(sock->sk, msg, &level, &description); - ret = (level == TLS_ALERT_LEVEL_FATAL) ? - -EACCES : -EAGAIN; + /* RFC 8446 Section 6: every alert but a closure alert is + * an error alert, whatever the legacy AlertLevel octet + * says. + */ + switch (description) { + case TLS_ALERT_DESC_CLOSE_NOTIFY: + case TLS_ALERT_DESC_USER_CANCELED: + ret = -EAGAIN; + break; + default: + ret = -EACCES; + } break; default: /* discard this record type */ From 706a92da0dee5870553a3bd96d8d3526e0ecbbe6 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sat, 8 Aug 2026 11:40:15 -0400 Subject: [PATCH 0213/1352] SUNRPC: fold xs_sock_process_cmsg() into its only caller xs_sock_process_cmsg() switches on the TLS record type, and every arm but TLS_RECORD_TYPE_ALERT returns the -EAGAIN its caller passed in. The DATA arm clears MSG_EOR in the caller's msghdr, but xs_sock_recvmsg() has already cleared that flag before the call. Deriving the record type a second time inside the helper also fires trace_tls_contenttype() twice for every alert. Move the alert handling into xs_sock_recv_cmsg() and delete the helper. Every other record type still returns -EAGAIN. The DATA arm's account of MSG_EOR moves to xs_sock_recvmsg(), where the flag is cleared. Signed-off-by: Chuck Lever Signed-off-by: Anna Schumaker --- net/sunrpc/xprtsock.c | 85 +++++++++++++++---------------------------- 1 file changed, 30 insertions(+), 55 deletions(-) diff --git a/net/sunrpc/xprtsock.c b/net/sunrpc/xprtsock.c index 963e46a514217a..f5a5136327ba8f 100644 --- a/net/sunrpc/xprtsock.c +++ b/net/sunrpc/xprtsock.c @@ -356,45 +356,6 @@ xs_alloc_sparse_pages(struct xdr_buf *buf, size_t want, gfp_t gfp) return want; } -static int -xs_sock_process_cmsg(struct socket *sock, struct msghdr *msg, - unsigned int *msg_flags, struct cmsghdr *cmsg, int ret) -{ - u8 content_type = tls_get_record_type(sock->sk, cmsg); - u8 level, description; - - switch (content_type) { - case 0: - break; - case TLS_RECORD_TYPE_DATA: - /* TLS sets EOR at the end of each application data - * record, even though there might be more frames - * waiting to be decrypted. - */ - *msg_flags &= ~MSG_EOR; - break; - case TLS_RECORD_TYPE_ALERT: - tls_alert_recv(sock->sk, msg, &level, &description); - /* RFC 8446 Section 6: every alert but a closure alert is - * an error alert, whatever the legacy AlertLevel octet - * says. - */ - switch (description) { - case TLS_ALERT_DESC_CLOSE_NOTIFY: - case TLS_ALERT_DESC_USER_CANCELED: - ret = -EAGAIN; - break; - default: - ret = -EACCES; - } - break; - default: - /* discard this record type */ - ret = -EAGAIN; - } - return ret; -} - static int xs_sock_recv_cmsg(struct socket *sock, unsigned int *msg_flags, int flags) { @@ -412,6 +373,7 @@ xs_sock_recv_cmsg(struct socket *sock, unsigned int *msg_flags, int flags) .msg_control = &u, .msg_controllen = sizeof(u), }; + u8 level, description; int ret; iov_iter_kvec(&msg.msg_iter, ITER_DEST, &alert_kvec, 1, @@ -421,23 +383,32 @@ xs_sock_recv_cmsg(struct socket *sock, unsigned int *msg_flags, int flags) * kTLS filled in u.cmsg. */ if (ret >= 0 && msg.msg_controllen < sizeof(u)) { - if (tls_get_record_type(sock->sk, &u.cmsg) == - TLS_RECORD_TYPE_ALERT) { - /* RFC 8446 Section 5.1 requires a record with an - * Alert type to carry exactly one message. An alert - * is two octets. tls_alert_recv() reads both without - * checking the length. alert_kvec caps the count at - * two, so a longer record fills it as well. kTLS - * sets MSG_EOR only once the record has been - * drained. - */ - if (ret != sizeof(alert) || - !(msg.msg_flags & MSG_EOR)) - return -EACCES; - iov_iter_revert(&msg.msg_iter, ret); + if (tls_get_record_type(sock->sk, &u.cmsg) != + TLS_RECORD_TYPE_ALERT) + return -EAGAIN; + /* RFC 8446 Section 5.1: a record with an Alert type carries + * exactly one message, and an alert is two octets. + * tls_alert_recv() reads both without checking the length. + * alert_kvec caps the count at two, so a longer record + * fills it as well. kTLS sets MSG_EOR only once the + * record has been drained. + */ + if (ret != sizeof(alert) || !(msg.msg_flags & MSG_EOR)) + return -EACCES; + iov_iter_revert(&msg.msg_iter, ret); + tls_alert_recv(sock->sk, &msg, &level, &description); + /* RFC 8446 Section 6: every alert but a closure alert is + * an error alert, whatever the legacy AlertLevel octet + * says. + */ + switch (description) { + case TLS_ALERT_DESC_CLOSE_NOTIFY: + case TLS_ALERT_DESC_USER_CANCELED: + ret = -EAGAIN; + break; + default: + ret = -EACCES; } - ret = xs_sock_process_cmsg(sock, &msg, msg_flags, &u.cmsg, - -EAGAIN); } return ret; } @@ -451,6 +422,10 @@ xs_sock_recvmsg(struct socket *sock, struct msghdr *msg, int flags, size_t seek) ret = sock_recvmsg(sock, msg, flags); /* Handle TLS inband control message lazily */ if (msg->msg_flags & MSG_CTRUNC) { + /* TLS sets EOR at the end of each application data + * record, even though there might be more frames + * waiting to be decrypted. + */ msg->msg_flags &= ~(MSG_CTRUNC | MSG_EOR); if (ret == 0 || ret == -EIO) ret = xs_sock_recv_cmsg(sock, &msg->msg_flags, flags); From 3ae43b6f769a3d09f34cb89f2310358e06304442 Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Fri, 4 Sep 2026 08:56:02 -0400 Subject: [PATCH 0214/1352] NFSv4.1/pnfs: suspend pNFS on NFS4ERR_TOOSMALL from LAYOUTGET If the layout for the requested range is larger than the size the client advertised in loga_maxcount, RFC 8881 Section 18.43.3 has the metadata server return NFS4ERR_TOOSMALL. The client caps loga_maxcount at a single page, so a flexfiles server that stripes a layout segment across several dozen data servers produces this error today. The client has no handling for it: nfs4_stat_to_errno() maps the error to -ETOOSMALL during decode, which nothing in the layoutget path recognizes and nfs_error_is_fatal() does not consider fatal, so pnfs_update_layout() clears the layout fail bit and returns no segment. The I/O falls back to the MDS, but because no fail bit was set, every subsequent pageio attempt sends another LAYOUTGET that is doomed to the same NFS4ERR_TOOSMALL. Files whose layouts do not fit the reply buffer never use pNFS and pay an extra round trip on every pageio. Map -ETOOSMALL to -EMSGSIZE in the layoutget exception handler and have pnfs_update_layout() treat it like NFS4ERR_LAYOUTUNAVAILABLE: mark the layout mode as failed and fall back to I/O through the MDS. Fixes: d600ad1f2bdb ("NFS41: pop some layoutget errors to application") Cc: stable@vger.kernel.org Assisted-by: Claude:claude-opus-4-8 Signed-off-by: Benjamin Coddington Reviewed-by: Jeff Layton Signed-off-by: Anna Schumaker --- fs/nfs/nfs4proc.c | 8 ++++++++ fs/nfs/pnfs.c | 2 ++ 2 files changed, 10 insertions(+) diff --git a/fs/nfs/nfs4proc.c b/fs/nfs/nfs4proc.c index 04b1987115d5da..e41c792a2725b6 100644 --- a/fs/nfs/nfs4proc.c +++ b/fs/nfs/nfs4proc.c @@ -9680,6 +9680,14 @@ nfs4_layoutget_handle_exception(struct rpc_task *task, case -NFS4ERR_BADLAYOUT: status = -EOVERFLOW; goto out; + /* + * NFS4ERR_TOOSMALL means the layout for the requested range + * exceeds what the client advertised in loga_maxcount (see + * RFC8881 section 18.43.3). + */ + case -ETOOSMALL: + status = -EMSGSIZE; + goto out; /* * NFS4ERR_LAYOUTTRYLATER is a conflict with another client * (or clients) writing to the same RAID stripe except when diff --git a/fs/nfs/pnfs.c b/fs/nfs/pnfs.c index 4f9c0f639014f4..c5d1951d540018 100644 --- a/fs/nfs/pnfs.c +++ b/fs/nfs/pnfs.c @@ -2331,6 +2331,8 @@ pnfs_update_layout(struct inode *ino, break; case -ENODATA: /* The server returned NFS4ERR_LAYOUTUNAVAILABLE */ + case -EMSGSIZE: + /* The layout exceeded loga_maxcount (NFS4ERR_TOOSMALL) */ pnfs_layout_set_fail_bit( lo, pnfs_iomode_to_fail_bit(iomode)); lseg = NULL; From b0282d0ccee74adc2f193acd0ae7cb3b910b9eb4 Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Fri, 4 Sep 2026 08:56:03 -0400 Subject: [PATCH 0215/1352] NFSv4.1/pnfs: derive loga_maxcount from the LAYOUTGET reply buffer The client advertises a fixed PNFS_LAYOUT_MAXSIZE (4096) in loga_maxcount no matter what reply buffer it actually allocated. The two only happen to agree for layout drivers that cap max_layoutget_response at a single page. Block and SCSI layouts allocate a session-sized reply buffer, yet still tell the server it must not send more than 4KB of layout: a server whose extent list encodes larger than a page returns NFS4ERR_TOOSMALL even though the client has ample room to receive it. Set loga_maxcount from the size of the page array allocated for the reply. The client never advertises more than it can receive, and any future change to the reply buffer sizing automatically carries its wire advertisement with it. Assisted-by: Claude:claude-opus-4-8 Signed-off-by: Benjamin Coddington Reviewed-by: Jeff Layton Signed-off-by: Anna Schumaker --- fs/nfs/pnfs.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/nfs/pnfs.c b/fs/nfs/pnfs.c index c5d1951d540018..9bef2997f512fe 100644 --- a/fs/nfs/pnfs.c +++ b/fs/nfs/pnfs.c @@ -1210,7 +1210,7 @@ pnfs_alloc_init_layoutget_args(struct inode *ino, lgp->args.minlength = i_size - range->offset; } } - lgp->args.maxcount = PNFS_LAYOUT_MAXSIZE; + lgp->args.maxcount = lgp->args.layout.pglen; pnfs_copy_range(&lgp->args.range, range); lgp->args.type = server->pnfs_curr_ld->id; lgp->args.inode = ino; From 3d63416b0cf2194b11535991041fb98fc8c0f009 Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Fri, 4 Sep 2026 08:56:04 -0400 Subject: [PATCH 0216/1352] NFSv4.1/pnfs: retry LAYOUTGET with a larger reply buffer on NFS4ERR_TOOSMALL Layout drivers cap the LAYOUTGET reply buffer with max_layoutget_response; the flexfiles driver caps it at a single page, which limits a striped layout segment to roughly 28 stripes. A server striping wider than that returns NFS4ERR_TOOSMALL, and the client falls back to I/O through the MDS -- pNFS never engages for those files. Instead of failing over to the MDS on the first NFS4ERR_TOOSMALL, retry the LAYOUTGET once with the reply buffer raised to the session's maximum response size, the same bound GETDEVICEINFO already uses. The server offers no size hint in the TOOSMALL error, and the page array is transient (freed when the RPC completes), so a single jump to the ceiling is preferred over incremental growth. If the layout does not fit even the session-sized buffer, fall back to the MDS as before. The common path is unchanged: the first LAYOUTGET for a layout is still sent with the driver's default reply buffer, and larger buffers are only ever allocated against servers that actually hand out wide layouts. Assisted-by: Claude:claude-opus-4-8 Signed-off-by: Benjamin Coddington Reviewed-by: Jeff Layton Signed-off-by: Anna Schumaker --- fs/nfs/pnfs.c | 45 ++++++++++++++++++++++++++++++++++++++------- 1 file changed, 38 insertions(+), 7 deletions(-) diff --git a/fs/nfs/pnfs.c b/fs/nfs/pnfs.c index 9bef2997f512fe..0c938d17b6c673 100644 --- a/fs/nfs/pnfs.c +++ b/fs/nfs/pnfs.c @@ -1167,11 +1167,12 @@ pnfs_alloc_init_layoutget_args(struct inode *ino, struct nfs_open_context *ctx, const nfs4_stateid *stateid, const struct pnfs_layout_range *range, - gfp_t gfp_flags) + size_t min_reply_sz, gfp_t gfp_flags) { struct nfs_server *server = pnfs_find_server(ino, ctx); size_t max_reply_sz = server->pnfs_curr_ld->max_layoutget_response; - size_t max_pages = max_response_pages(server); + size_t session_pages = max_response_pages(server); + size_t max_pages = session_pages; struct nfs4_layoutget *lgp; dprintk("--> %s\n", __func__); @@ -1186,6 +1187,17 @@ pnfs_alloc_init_layoutget_args(struct inode *ino, max_pages = npages; } + /* + * A previous LAYOUTGET for this layout did not fit the reply + * buffer: raise the layout driver's default up to the session's + * maximum response size. + */ + if (min_reply_sz) { + size_t npages = (min_reply_sz + PAGE_SIZE - 1) >> PAGE_SHIFT; + if (npages > max_pages) + max_pages = min(npages, session_pages); + } + lgp->args.layout.pages = nfs4_alloc_pages(max_pages, gfp_flags); if (!lgp->args.layout.pages) { kfree(lgp); @@ -2145,6 +2157,7 @@ pnfs_update_layout(struct inode *ino, .inode = ino, }; unsigned long giveup = jiffies + (clp->cl_lease_time << 1); + size_t reply_sz = 0; bool first; if (!pnfs_enabled_sb(NFS_SERVER(ino))) { @@ -2304,7 +2317,8 @@ pnfs_update_layout(struct inode *ino, if (arg.length != NFS4_MAX_UINT64) arg.length = PAGE_ALIGN(arg.length); - lgp = pnfs_alloc_init_layoutget_args(ino, ctx, &stateid, &arg, gfp_flags); + lgp = pnfs_alloc_init_layoutget_args(ino, ctx, &stateid, &arg, reply_sz, + gfp_flags); if (!lgp) { lseg = ERR_PTR(-ENOMEM); trace_pnfs_update_layout(ino, pos, count, iomode, lo, NULL, @@ -2331,12 +2345,29 @@ pnfs_update_layout(struct inode *ino, break; case -ENODATA: /* The server returned NFS4ERR_LAYOUTUNAVAILABLE */ - case -EMSGSIZE: - /* The layout exceeded loga_maxcount (NFS4ERR_TOOSMALL) */ pnfs_layout_set_fail_bit( lo, pnfs_iomode_to_fail_bit(iomode)); lseg = NULL; goto out_put_layout_hdr; + case -EMSGSIZE: { + /* + * The layout exceeded loga_maxcount (NFS4ERR_TOOSMALL): + * retry once with the reply buffer raised to the + * session's maximum response size before falling back + * to I/O through the MDS. + */ + size_t max = max_response_pages(server) << PAGE_SHIFT; + + if (reply_sz < max) { + reply_sz = max; + exception.retry = 1; + break; + } + pnfs_layout_set_fail_bit( + lo, pnfs_iomode_to_fail_bit(iomode)); + lseg = NULL; + goto out_put_layout_hdr; + } default: if (!nfs_error_is_fatal(PTR_ERR(lseg))) { pnfs_layout_clear_fail_bit(lo, pnfs_iomode_to_fail_bit(iomode)); @@ -2450,7 +2481,7 @@ static void _lgopen_prepare_attached(struct nfs4_opendata *data, lo = _pnfs_grab_empty_layout(ino, ctx); if (!lo) return; - lgp = pnfs_alloc_init_layoutget_args(ino, ctx, ¤t_stateid, &rng, + lgp = pnfs_alloc_init_layoutget_args(ino, ctx, ¤t_stateid, &rng, 0, nfs_io_gfp_mask()); if (!lgp) { clear_and_wake_up_bit(NFS_LAYOUT_FIRST_LAYOUTGET, &lo->plh_flags); @@ -2476,7 +2507,7 @@ static void _lgopen_prepare_floating(struct nfs4_opendata *data, }; struct nfs4_layoutget *lgp; - lgp = pnfs_alloc_init_layoutget_args(ino, ctx, ¤t_stateid, &rng, + lgp = pnfs_alloc_init_layoutget_args(ino, ctx, ¤t_stateid, &rng, 0, nfs_io_gfp_mask()); if (!lgp) return; From 39e19b9f5c5110e39825b843127babdc373dde76 Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Fri, 4 Sep 2026 08:56:05 -0400 Subject: [PATCH 0217/1352] NFSv4.1/pnfs: treat an oversized LAYOUTGET reply as -EMSGSIZE decode_layoutget() already detects a server that sends a layout body larger than the reply buffer we provided ("server cheating in layoutget reply") but maps it to -EINVAL, which pnfs_update_layout() treats as a transient error: the client falls back to the MDS for this I/O only and re-sends a doomed LAYOUTGET on every subsequent pageio attempt. A server that overruns the reply buffer has ignored loga_maxcount, but the client can recover the same way it recovers from a conformant server's NFS4ERR_TOOSMALL: return -EMSGSIZE from exactly this check so that the layoutget path retries once with a session-sized reply buffer and otherwise marks the layout mode failed and falls back to the MDS. Assisted-by: Claude:claude-opus-4-8 Signed-off-by: Benjamin Coddington Reviewed-by: Jeff Layton Signed-off-by: Anna Schumaker --- fs/nfs/nfs4xdr.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/nfs/nfs4xdr.c b/fs/nfs/nfs4xdr.c index fc049ce4ba8aca..8b3d96b4a0f0f2 100644 --- a/fs/nfs/nfs4xdr.c +++ b/fs/nfs/nfs4xdr.c @@ -6187,7 +6187,7 @@ static int decode_layoutget(struct xdr_stream *xdr, struct rpc_rqst *req, dprintk("NFS: server cheating in layoutget reply: " "layout len %u > recvd %u\n", res->layoutp->len, recvd); - status = -EINVAL; + status = -EMSGSIZE; goto out; } From ee4d94ca9f8e7ad1d4bbad4e0c9a70490f30a395 Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Fri, 4 Sep 2026 08:56:06 -0400 Subject: [PATCH 0218/1352] NFSv4.1/pnfs: remember when a server needs a larger LAYOUTGET reply buffer When a LAYOUTGET only succeeds after escalating the reply buffer, every layout fetched from that server is likely to need the larger buffer: remember the escalated size on the nfs_server and use it as the floor for subsequent LAYOUTGET reply buffers, skipping the doomed attempt at the layout driver's default size. This also lets the LAYOUTGET attached to OPEN benefit: the lgopen path is best-effort with no retry of its own, so without the learned size it would fail with NFS4ERR_TOOSMALL at every open against a server handing out wide layouts, and layouts would only ever be acquired by the I/O path's retry. The field is a hint: reads and writes are racy by design, the value only ever grows toward the session's maximum response size, and a stale-low read merely costs one escalation round trip. Assisted-by: Claude:claude-opus-4-8 Signed-off-by: Benjamin Coddington Reviewed-by: Jeff Layton Signed-off-by: Anna Schumaker --- fs/nfs/pnfs.c | 15 ++++++++++++--- include/linux/nfs_fs_sb.h | 4 ++++ 2 files changed, 16 insertions(+), 3 deletions(-) diff --git a/fs/nfs/pnfs.c b/fs/nfs/pnfs.c index 0c938d17b6c673..3e8d1a1fd827cd 100644 --- a/fs/nfs/pnfs.c +++ b/fs/nfs/pnfs.c @@ -1188,10 +1188,12 @@ pnfs_alloc_init_layoutget_args(struct inode *ino, } /* - * A previous LAYOUTGET for this layout did not fit the reply - * buffer: raise the layout driver's default up to the session's - * maximum response size. + * A previous LAYOUTGET on this layout or on this server did not + * fit the reply buffer: raise the layout driver's default up to + * the session's maximum response size. */ + if (!min_reply_sz) + min_reply_sz = READ_ONCE(server->lg_reply_sz); if (min_reply_sz) { size_t npages = (min_reply_sz + PAGE_SIZE - 1) >> PAGE_SHIFT; if (npages > max_pages) @@ -2387,6 +2389,13 @@ pnfs_update_layout(struct inode *ino, goto lookup_again; } } else { + /* + * A LAYOUTGET that only succeeded with an escalated reply + * buffer: remember the size so that future LAYOUTGETs to + * this server skip the attempt at the driver's default. + */ + if (reply_sz) + WRITE_ONCE(server->lg_reply_sz, reply_sz); pnfs_layout_clear_fail_bit(lo, pnfs_iomode_to_fail_bit(iomode)); } diff --git a/include/linux/nfs_fs_sb.h b/include/linux/nfs_fs_sb.h index 34d294774f8cec..3e6bae7e52218f 100644 --- a/include/linux/nfs_fs_sb.h +++ b/include/linux/nfs_fs_sb.h @@ -248,6 +248,10 @@ struct nfs_server { that are supported on this filesystem */ struct pnfs_layoutdriver_type *pnfs_curr_ld; /* Active layout driver */ + unsigned int lg_reply_sz; /* Learned LAYOUTGET reply + buffer size, when the layout + driver's default has proved + too small */ struct rpc_wait_queue roc_rpcwaitq; /* the following fields are protected by nfs_client->cl_lock */ From ad6fcc902bb1e2649b4e7715286006d5af9b1649 Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Fri, 4 Sep 2026 08:56:07 -0400 Subject: [PATCH 0219/1352] NFSv4/flexfiles: allocate the per-mirror stripe array with kvzalloc_objs Each mirror's stripe array is a single contiguous allocation of dss_count * sizeof(struct nfs4_ff_layout_ds_stripe) -- roughly 300 bytes per stripe. With the LAYOUTGET reply buffer no longer capped at a single page, a wide striped layout can push this well past the high-order allocation comfort zone (a 2048-stripe mirror is a ~600KB contiguous allocation) where it can fail under memory fragmentation. Use kvzalloc_objs() so wide stripe arrays fall back to vmalloc. Note the vmalloc fallback is unavailable when the pageio path allocates under memalloc_noio (swap over pNFS); that case simply behaves as before. Assisted-by: Claude:claude-opus-4-8 Signed-off-by: Benjamin Coddington Reviewed-by: Jeff Layton Signed-off-by: Anna Schumaker --- fs/nfs/flexfilelayout/flexfilelayout.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/fs/nfs/flexfilelayout/flexfilelayout.c b/fs/nfs/flexfilelayout/flexfilelayout.c index 7fe8b91fa47c94..1267eab830fdae 100644 --- a/fs/nfs/flexfilelayout/flexfilelayout.c +++ b/fs/nfs/flexfilelayout/flexfilelayout.c @@ -285,8 +285,8 @@ static struct nfs4_ff_layout_mirror *ff_layout_alloc_mirror(u32 dss_count, mirror->dss_count = dss_count; mirror->dss = - kzalloc_objs(struct nfs4_ff_layout_ds_stripe, dss_count, - gfp_flags); + kvzalloc_objs(struct nfs4_ff_layout_ds_stripe, dss_count, + gfp_flags); if (mirror->dss == NULL) { kfree(mirror); return NULL; @@ -315,7 +315,7 @@ static void ff_layout_free_mirror(struct nfs4_ff_layout_mirror *mirror) nfs4_ff_layout_put_deviceid(mirror->dss[dss_id].mirror_ds); } - kfree(mirror->dss); + kvfree(mirror->dss); kfree(mirror); } From 728a4a60e86e871ba9fc0aa723a000098c251c0d Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:03 -0400 Subject: [PATCH 0220/1352] NFSv4/pnfs: Free the netid when draining a data-server address list nfs4_pnfs_ds_addr_free() releases all three of an address entry's allocations, but the four sites that drain a decoded list before it reaches a data server open-code the loop and free only da_remotestr and the entry, leaking da_netid. Both layout drivers drain that list when nfs4_pnfs_ds_add() returned a data server already in the cache, and again on their error exits. Lift the drain into nfs4_pnfs_ds_addr_list_free() so an address list has one owner. A striping mount reaches the cached case on nearly every GETDEVICEINFO, since many deviceIDs name the same data servers. Fixes: 4be78d26810b ("NFSv4/pNFS: Store the transport type in struct nfs4_pnfs_ds_addr") Cc: stable@vger.kernel.org Assisted-by: Claude:claude-opus-5 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/filelayout/filelayoutdev.c | 17 ++--------------- fs/nfs/flexfilelayout/flexfilelayoutdev.c | 17 ++--------------- fs/nfs/pnfs.h | 1 + fs/nfs/pnfs_nfs.c | 21 +++++++++++++-------- 4 files changed, 18 insertions(+), 38 deletions(-) diff --git a/fs/nfs/filelayout/filelayoutdev.c b/fs/nfs/filelayout/filelayoutdev.c index 88bc79ec345968..57e0654cd98c34 100644 --- a/fs/nfs/filelayout/filelayoutdev.c +++ b/fs/nfs/filelayout/filelayoutdev.c @@ -177,27 +177,14 @@ nfs4_fl_alloc_deviceid_node(struct nfs_server *server, struct pnfs_device *pdev, trace_fl_getdevinfo(server, &pdev->dev_id, dsaddr->ds_list[i]->ds_remotestr); /* If DS was already in cache, free ds addrs */ - while (!list_empty(&dsaddrs)) { - da = list_first_entry(&dsaddrs, - struct nfs4_pnfs_ds_addr, - da_node); - list_del_init(&da->da_node); - kfree(da->da_remotestr); - kfree(da); - } + nfs4_pnfs_ds_addr_list_free(&dsaddrs); } folio_put(scratch); return dsaddr; out_err_drain_dsaddrs: - while (!list_empty(&dsaddrs)) { - da = list_first_entry(&dsaddrs, struct nfs4_pnfs_ds_addr, - da_node); - list_del_init(&da->da_node); - kfree(da->da_remotestr); - kfree(da); - } + nfs4_pnfs_ds_addr_list_free(&dsaddrs); out_err_free_deviceid: nfs4_fl_free_deviceid(dsaddr); /* stripe_indicies was part of dsaddr */ diff --git a/fs/nfs/flexfilelayout/flexfilelayoutdev.c b/fs/nfs/flexfilelayout/flexfilelayoutdev.c index 5c0216bd5fcea1..6cd3859d4a243d 100644 --- a/fs/nfs/flexfilelayout/flexfilelayoutdev.c +++ b/fs/nfs/flexfilelayout/flexfilelayoutdev.c @@ -159,26 +159,13 @@ nfs4_ff_alloc_deviceid_node(struct nfs_server *server, struct pnfs_device *pdev, goto out_err_drain_dsaddrs; /* If DS was already in cache, free ds addrs */ - while (!list_empty(&dsaddrs)) { - da = list_first_entry(&dsaddrs, - struct nfs4_pnfs_ds_addr, - da_node); - list_del_init(&da->da_node); - kfree(da->da_remotestr); - kfree(da); - } + nfs4_pnfs_ds_addr_list_free(&dsaddrs); folio_put(scratch); return new_ds; out_err_drain_dsaddrs: - while (!list_empty(&dsaddrs)) { - da = list_first_entry(&dsaddrs, struct nfs4_pnfs_ds_addr, - da_node); - list_del_init(&da->da_node); - kfree(da->da_remotestr); - kfree(da); - } + nfs4_pnfs_ds_addr_list_free(&dsaddrs); kfree(ds_versions); out_scratch: diff --git a/fs/nfs/pnfs.h b/fs/nfs/pnfs.h index 70d20f7796787d..5725ea06efc2d8 100644 --- a/fs/nfs/pnfs.h +++ b/fs/nfs/pnfs.h @@ -427,6 +427,7 @@ int nfs4_pnfs_ds_connect(struct nfs_server *mds_srv, struct nfs4_pnfs_ds *ds, struct nfs4_pnfs_ds_addr *nfs4_decode_mp_ds_addr(struct net *net, struct xdr_stream *xdr, gfp_t gfp_flags); +void nfs4_pnfs_ds_addr_list_free(struct list_head *dsaddrs); void pnfs_layout_mark_request_commit(struct nfs_page *req, struct pnfs_layout_segment *lseg, struct nfs_commit_info *cinfo, diff --git a/fs/nfs/pnfs_nfs.c b/fs/nfs/pnfs_nfs.c index 93d63f75a355cd..5bd014b981bf2d 100644 --- a/fs/nfs/pnfs_nfs.c +++ b/fs/nfs/pnfs_nfs.c @@ -633,23 +633,28 @@ static void nfs4_pnfs_ds_addr_free(struct nfs4_pnfs_ds_addr *da) kfree(da); } -static void destroy_ds(struct nfs4_pnfs_ds *ds) +void nfs4_pnfs_ds_addr_list_free(struct list_head *dsaddrs) { struct nfs4_pnfs_ds_addr *da; + while (!list_empty(dsaddrs)) { + da = list_first_entry(dsaddrs, struct nfs4_pnfs_ds_addr, + da_node); + list_del_init(&da->da_node); + nfs4_pnfs_ds_addr_free(da); + } +} +EXPORT_SYMBOL_GPL(nfs4_pnfs_ds_addr_list_free); + +static void destroy_ds(struct nfs4_pnfs_ds *ds) +{ dprintk("--> %s\n", __func__); ifdebug(FACILITY) print_ds(ds); nfs_put_client(ds->ds_clp); - while (!list_empty(&ds->ds_addrs)) { - da = list_first_entry(&ds->ds_addrs, - struct nfs4_pnfs_ds_addr, - da_node); - list_del_init(&da->da_node); - nfs4_pnfs_ds_addr_free(da); - } + nfs4_pnfs_ds_addr_list_free(&ds->ds_addrs); kfree(ds->ds_remotestr); kfree(ds); From 80c8ffd94ebfd6903e53c1b81bd34cf70293804f Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:04 -0400 Subject: [PATCH 0221/1352] NFSv4/flexfiles: Use the full 64-bit stripe_unit ff_layout_alloc_lseg() decodes stripe_unit as the 64-bit value the protocol defines and struct nfs4_ff_layout_segment stores it as one, but both consumers narrow it back to 32 bits: nfs4_ff_layout_calc_dss_id() divides with do_div(), which casts the divisor, and ff_layout_pg_test() copies it into a u32 first. A stripe_unit that does not fit is silently truncated, so the client stripes on a unit the server did not specify -- or divides by zero, if the low 32 bits happen to be clear. Divide by the full value, with div64_u64() and div64_u64_rem(). Fixes: 20b1d75fb840 ("NFSv4/flexfiles: Add support for striped layouts") Cc: stable@vger.kernel.org Assisted-by: Claude:claude-fable-5 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/flexfilelayout/flexfilelayout.c | 12 ++++++------ fs/nfs/flexfilelayout/flexfilelayout.h | 2 +- 2 files changed, 7 insertions(+), 7 deletions(-) diff --git a/fs/nfs/flexfilelayout/flexfilelayout.c b/fs/nfs/flexfilelayout/flexfilelayout.c index 1267eab830fdae..5895824974b296 100644 --- a/fs/nfs/flexfilelayout/flexfilelayout.c +++ b/fs/nfs/flexfilelayout/flexfilelayout.c @@ -995,9 +995,9 @@ ff_layout_pg_test(struct nfs_pageio_descriptor *pgio, struct nfs_page *prev, { unsigned int size; u64 p_stripe, r_stripe; - u32 stripe_offset; + u64 stripe_offset; u64 segment_offset = pgio->pg_lseg->pls_range.offset; - u32 stripe_unit = FF_LAYOUT_LSEG(pgio->pg_lseg)->stripe_unit; + u64 stripe_unit = FF_LAYOUT_LSEG(pgio->pg_lseg)->stripe_unit; /* calls nfs_generic_pg_test */ size = pnfs_generic_pg_test(pgio, prev, req); @@ -1010,21 +1010,21 @@ ff_layout_pg_test(struct nfs_pageio_descriptor *pgio, struct nfs_page *prev, if (prev) { p_stripe = (u64)req_offset(prev) - segment_offset; r_stripe = (u64)req_offset(req) - segment_offset; - do_div(p_stripe, stripe_unit); - do_div(r_stripe, stripe_unit); + p_stripe = div64_u64(p_stripe, stripe_unit); + r_stripe = div64_u64(r_stripe, stripe_unit); if (p_stripe != r_stripe) return 0; } /* calculate remaining bytes in the current stripe */ - div_u64_rem((u64)req_offset(req) - segment_offset, + div64_u64_rem((u64)req_offset(req) - segment_offset, stripe_unit, &stripe_offset); WARN_ON_ONCE(stripe_offset > stripe_unit); if (stripe_offset >= stripe_unit) return 0; - return min(stripe_unit - (unsigned int)stripe_offset, size); + return min_t(u64, stripe_unit - stripe_offset, size); } static void diff --git a/fs/nfs/flexfilelayout/flexfilelayout.h b/fs/nfs/flexfilelayout/flexfilelayout.h index a5bd00f69e8242..ec69cd1c3ae963 100644 --- a/fs/nfs/flexfilelayout/flexfilelayout.h +++ b/fs/nfs/flexfilelayout/flexfilelayout.h @@ -220,7 +220,7 @@ nfs4_ff_layout_calc_dss_id(const u64 stripe_unit, const u32 dss_count, const lof if (dss_count == 1 || stripe_unit == 0) return 0; - do_div(tmp, stripe_unit); + tmp = div64_u64(tmp, stripe_unit); return do_div(tmp, dss_count); } From f4f6b27a21eee2f0b3306070a31160d41bfa8ff1 Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:05 -0400 Subject: [PATCH 0222/1352] NFSv4/pnfs: bound the CB_NOTIFY_DEVICEID array count before allocating decode_devicenotify_args() hands the server's notification count straight to kmalloc_objs() as the array length, without checking it against the message that carried it. Bound it the way nfs4xdr.c bounds an attribute length, against xdr_stream_remaining(). Fixes: 1be5683b03a7 ("pnfs: CB_NOTIFY_DEVICEID") Cc: stable@vger.kernel.org Assisted-by: Claude:claude-fable-5 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/callback_xdr.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/fs/nfs/callback_xdr.c b/fs/nfs/callback_xdr.c index 88e1af0fda01aa..b1bc2c02b5a6c1 100644 --- a/fs/nfs/callback_xdr.c +++ b/fs/nfs/callback_xdr.c @@ -271,6 +271,13 @@ __be32 decode_devicenotify_args(struct svc_rqst *rqstp, if (n == 0) goto out; + /* sanity check the count against the remaining stream */ + if (n > xdr_stream_remaining(xdr) / + ((4 * sizeof(uint32_t)) + NFS4_DEVICEID4_SIZE)) { + status = htonl(NFS4ERR_BADXDR); + goto out; + } + args->devs = kmalloc_objs(*args->devs, n); if (!args->devs) { status = htonl(NFS4ERR_DELAY); From ed4b6d87b42c9ff39f537cb471c287fec0fc7e65 Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:06 -0400 Subject: [PATCH 0223/1352] pNFS: Fix CB_NOTIFY_DEVICEID CHANGE to consume ndc_immediate decode_devicenotify_args() gated the trailing ndc_immediate on cbd_layout_type rather than cbd_notify_type, so the flag was never consumed and a multi-item cnda_changes<> misaligned after the first entry. Fixes: 1be5683b03a7 ("pnfs: CB_NOTIFY_DEVICEID") Cc: stable@vger.kernel.org Assisted-by: Claude:claude-opus-4-8 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/callback_xdr.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/nfs/callback_xdr.c b/fs/nfs/callback_xdr.c index b1bc2c02b5a6c1..5a1f054692689e 100644 --- a/fs/nfs/callback_xdr.c +++ b/fs/nfs/callback_xdr.c @@ -319,7 +319,7 @@ __be32 decode_devicenotify_args(struct svc_rqst *rqstp, memcpy(dev->cbd_dev_id.data, p, NFS4_DEVICEID4_SIZE); p += XDR_QUADLEN(NFS4_DEVICEID4_SIZE); - if (dev->cbd_layout_type == NOTIFY_DEVICEID4_CHANGE) { + if (dev->cbd_notify_type == NOTIFY_DEVICEID4_CHANGE) { p = xdr_inline_decode(xdr, sizeof(uint32_t)); if (unlikely(p == NULL)) { status = htonl(NFS4ERR_BADXDR); From daa2ba7cf6259cc7635d9f29251da4a72a5cbd7d Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:07 -0400 Subject: [PATCH 0224/1352] NFSv4/flexfiles: Use the full 64-bit offset for read DS selection The read data-server selection helpers declare their offset parameter as u32, truncating the file offset before nfs4_ff_layout_calc_dss_id() computes the stripe index. The write path already passes the full offset; make the read path match. Fixes: 4934ccbeaed3 ("NFSv4/flexfiles: Read path updates for striped layouts") Cc: stable@vger.kernel.org Assisted-by: Claude:claude-opus-4-8 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/flexfilelayout/flexfilelayout.c | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/fs/nfs/flexfilelayout/flexfilelayout.c b/fs/nfs/flexfilelayout/flexfilelayout.c index 5895824974b296..a36f29880e946e 100644 --- a/fs/nfs/flexfilelayout/flexfilelayout.c +++ b/fs/nfs/flexfilelayout/flexfilelayout.c @@ -878,7 +878,7 @@ ff_layout_mark_ds_reachable(struct pnfs_layout_segment *lseg, u32 idx, u32 dss_i static struct nfs4_pnfs_ds * ff_layout_choose_ds_for_read(struct pnfs_layout_segment *lseg, u32 start_idx, u32 *best_idx, - u32 offset, u32 *dss_id, + u64 offset, u32 *dss_id, bool check_device) { struct nfs4_ff_layout_segment *fls = FF_LAYOUT_LSEG(lseg); @@ -914,7 +914,7 @@ ff_layout_choose_ds_for_read(struct pnfs_layout_segment *lseg, static struct nfs4_pnfs_ds * ff_layout_choose_any_ds_for_read(struct pnfs_layout_segment *lseg, u32 start_idx, u32 *best_idx, - u32 offset, u32 *dss_id) + u64 offset, u32 *dss_id) { return ff_layout_choose_ds_for_read(lseg, start_idx, best_idx, offset, dss_id, false); @@ -923,7 +923,7 @@ ff_layout_choose_any_ds_for_read(struct pnfs_layout_segment *lseg, static struct nfs4_pnfs_ds * ff_layout_choose_valid_ds_for_read(struct pnfs_layout_segment *lseg, u32 start_idx, u32 *best_idx, - u32 offset, u32 *dss_id) + u64 offset, u32 *dss_id) { return ff_layout_choose_ds_for_read(lseg, start_idx, best_idx, offset, dss_id, true); @@ -932,7 +932,7 @@ ff_layout_choose_valid_ds_for_read(struct pnfs_layout_segment *lseg, static struct nfs4_pnfs_ds * ff_layout_choose_best_ds_for_read(struct pnfs_layout_segment *lseg, u32 start_idx, u32 *best_idx, - u32 offset, u32 *dss_id) + u64 offset, u32 *dss_id) { struct nfs4_pnfs_ds *ds; @@ -947,7 +947,7 @@ ff_layout_choose_best_ds_for_read(struct pnfs_layout_segment *lseg, static struct nfs4_pnfs_ds * ff_layout_get_ds_for_read(struct nfs_pageio_descriptor *pgio, u32 *best_idx, - u32 offset, + u64 offset, u32 *dss_id) { struct pnfs_layout_segment *lseg = pgio->pg_lseg; From 1f12be389861cef0052e4f2fa9f1cc129d7101d1 Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:08 -0400 Subject: [PATCH 0225/1352] NFSv4/flexfiles: Bound page coalescing on the absolute stripe offset ff_layout_pg_test() was copied from the files layout, which partitions data servers relative to pattern_offset and bounds coalescing with a segment-relative offset. Flexfiles has no pattern_offset: nfs4_ff_layout_calc_dss_id() partitions on the absolute file offset (RFC 8435 Section 6). So when a layout segment's offset is not stripe-unit-aligned, the segment-relative coalescing window is shifted off the absolute stripe grid and a coalesced I/O can straddle a stripe boundary, sending the bytes past it to the wrong data server. Bound coalescing on the absolute offset to match calc_dss_id(). Aligned segments, including all whole-file layouts, are unchanged. Fixes: 4934ccbeaed3 ("NFSv4/flexfiles: Read path updates for striped layouts") Cc: stable@vger.kernel.org Assisted-by: Claude:claude-opus-4-8 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/flexfilelayout/flexfilelayout.c | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/fs/nfs/flexfilelayout/flexfilelayout.c b/fs/nfs/flexfilelayout/flexfilelayout.c index a36f29880e946e..d2ea67dfdb9b84 100644 --- a/fs/nfs/flexfilelayout/flexfilelayout.c +++ b/fs/nfs/flexfilelayout/flexfilelayout.c @@ -996,7 +996,6 @@ ff_layout_pg_test(struct nfs_pageio_descriptor *pgio, struct nfs_page *prev, unsigned int size; u64 p_stripe, r_stripe; u64 stripe_offset; - u64 segment_offset = pgio->pg_lseg->pls_range.offset; u64 stripe_unit = FF_LAYOUT_LSEG(pgio->pg_lseg)->stripe_unit; /* calls nfs_generic_pg_test */ @@ -1008,8 +1007,8 @@ ff_layout_pg_test(struct nfs_pageio_descriptor *pgio, struct nfs_page *prev, /* see if req and prev are in the same stripe */ if (prev) { - p_stripe = (u64)req_offset(prev) - segment_offset; - r_stripe = (u64)req_offset(req) - segment_offset; + p_stripe = (u64)req_offset(prev); + r_stripe = (u64)req_offset(req); p_stripe = div64_u64(p_stripe, stripe_unit); r_stripe = div64_u64(r_stripe, stripe_unit); @@ -1018,7 +1017,7 @@ ff_layout_pg_test(struct nfs_pageio_descriptor *pgio, struct nfs_page *prev, } /* calculate remaining bytes in the current stripe */ - div64_u64_rem((u64)req_offset(req) - segment_offset, + div64_u64_rem((u64)req_offset(req), stripe_unit, &stripe_offset); WARN_ON_ONCE(stripe_offset > stripe_unit); From bdae5a44696faaf3cc4e18644ded720c7c6d9ee8 Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:09 -0400 Subject: [PATCH 0226/1352] NFSv4/filelayout: Anchor page coalescing on pattern_offset filelayout_pg_test() bounds page coalescing to a single stripe unit using an offset relative to pls_range.offset, but the data server is selected by nfs4_fl_calc_j_index() using an offset relative to pattern_offset. When a segment's pattern_offset and range offset are not congruent modulo the stripe unit, the coalescing window is shifted off the DS-selection grid, so a coalesced I/O can straddle a stripe boundary and send the bytes past it to the wrong data server. Anchor coalescing on pattern_offset to match nfs4_fl_calc_j_index(). Stripe-congruent segments, including the common whole-file pattern_offset 0 case, are unchanged. Fixes: c6194271f94b ("pnfs: filelayout: support non page aligned layouts") Cc: stable@vger.kernel.org Assisted-by: Claude:claude-opus-4-8 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/filelayout/filelayout.c | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/fs/nfs/filelayout/filelayout.c b/fs/nfs/filelayout/filelayout.c index 72e20b56fbc705..0d53277f972eb0 100644 --- a/fs/nfs/filelayout/filelayout.c +++ b/fs/nfs/filelayout/filelayout.c @@ -796,7 +796,7 @@ filelayout_pg_test(struct nfs_pageio_descriptor *pgio, struct nfs_page *prev, unsigned int size; u64 p_stripe, r_stripe; u32 stripe_offset; - u64 segment_offset = pgio->pg_lseg->pls_range.offset; + u64 pattern_offset = FILELAYOUT_LSEG(pgio->pg_lseg)->pattern_offset; u32 stripe_unit = FILELAYOUT_LSEG(pgio->pg_lseg)->stripe_unit; /* calls nfs_generic_pg_test */ @@ -808,8 +808,8 @@ filelayout_pg_test(struct nfs_pageio_descriptor *pgio, struct nfs_page *prev, /* see if req and prev are in the same stripe */ if (prev) { - p_stripe = (u64)req_offset(prev) - segment_offset; - r_stripe = (u64)req_offset(req) - segment_offset; + p_stripe = (u64)req_offset(prev) - pattern_offset; + r_stripe = (u64)req_offset(req) - pattern_offset; do_div(p_stripe, stripe_unit); do_div(r_stripe, stripe_unit); @@ -818,7 +818,7 @@ filelayout_pg_test(struct nfs_pageio_descriptor *pgio, struct nfs_page *prev, } /* calculate remaining bytes in the current stripe */ - div_u64_rem((u64)req_offset(req) - segment_offset, + div_u64_rem((u64)req_offset(req) - pattern_offset, stripe_unit, &stripe_offset); WARN_ON_ONCE(stripe_offset > stripe_unit); From 18264ecd4b0d761d67810011ca16e3be4d025875 Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:10 -0400 Subject: [PATCH 0227/1352] NFSv4/flexfiles: Reference the device node across DS setup The flexfiles I/O setup paths read the mirror's pinned per-stripe device node (mirror->dss[dss_id].mirror_ds) repeatedly and locklessly, relying on the mirror's lifetime pin. To prepare for re-resolving that pointer in place on CB_NOTIFY_DEVICEID CHANGE, readers must hold their own reference on the node they are using rather than trusting the pin. Replace ff_layout_init_mirror_ds() with ff_layout_get_mirror_ds(), which resolves on first use as before but returns the node with its own reference held. Thread the referenced node through nfs4_ff_layout_prepare_ds() and the DS selection helpers so each setup path snapshots the node once, and drop the reference when setup is done. No functional change: the pointer is still resolved once and pinned for the life of the mirror. Each reference this adds is released on every exit from the path that took it, error paths included. Assisted-by: Claude:claude-fable-5 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/flexfilelayout/flexfilelayout.c | 140 ++++++++++++++-------- fs/nfs/flexfilelayout/flexfilelayout.h | 16 ++- fs/nfs/flexfilelayout/flexfilelayoutdev.c | 74 +++++++----- 3 files changed, 141 insertions(+), 89 deletions(-) diff --git a/fs/nfs/flexfilelayout/flexfilelayout.c b/fs/nfs/flexfilelayout/flexfilelayout.c index d2ea67dfdb9b84..7a908faa41319b 100644 --- a/fs/nfs/flexfilelayout/flexfilelayout.c +++ b/fs/nfs/flexfilelayout/flexfilelayout.c @@ -875,7 +875,7 @@ ff_layout_mark_ds_reachable(struct pnfs_layout_segment *lseg, u32 idx, u32 dss_i nfs4_mark_deviceid_available(devid); } -static struct nfs4_pnfs_ds * +static struct nfs4_ff_layout_ds * ff_layout_choose_ds_for_read(struct pnfs_layout_segment *lseg, u32 start_idx, u32 *best_idx, u64 offset, u32 *dss_id, @@ -883,7 +883,9 @@ ff_layout_choose_ds_for_read(struct pnfs_layout_segment *lseg, { struct nfs4_ff_layout_segment *fls = FF_LAYOUT_LSEG(lseg); struct nfs4_ff_layout_mirror *mirror; - struct nfs4_pnfs_ds *ds = ERR_PTR(-EAGAIN); + struct nfs4_ff_layout_ds *mirror_ds; + struct nfs4_ff_layout_ds *ret = ERR_PTR(-EAGAIN); + struct nfs4_pnfs_ds *ds; u32 idx; /* mirrors are initially sorted by efficiency */ @@ -893,25 +895,32 @@ ff_layout_choose_ds_for_read(struct pnfs_layout_segment *lseg, fls->stripe_unit, fls->mirror_array[idx]->dss_count, offset); - ds = nfs4_ff_layout_prepare_ds(lseg, mirror, *dss_id, false); - if (IS_ERR(ds)) + mirror_ds = ff_layout_get_mirror_ds(lseg->pls_layout, mirror, + *dss_id); + ds = nfs4_ff_layout_prepare_ds(lseg, mirror, mirror_ds, + *dss_id, false); + if (IS_ERR(ds)) { + nfs4_ff_layout_put_deviceid(mirror_ds); + ret = ERR_CAST(ds); continue; + } if (check_device && - nfs4_test_deviceid_unavailable(&mirror->dss[*dss_id].mirror_ds->id_node)) { + nfs4_test_deviceid_unavailable(&mirror_ds->id_node)) { + nfs4_ff_layout_put_deviceid(mirror_ds); // reinitialize the error state in case if this is the last iteration - ds = ERR_PTR(-EINVAL); + ret = ERR_PTR(-EINVAL); continue; } *best_idx = idx; - break; + return mirror_ds; } - return ds; + return ret; } -static struct nfs4_pnfs_ds * +static struct nfs4_ff_layout_ds * ff_layout_choose_any_ds_for_read(struct pnfs_layout_segment *lseg, u32 start_idx, u32 *best_idx, u64 offset, u32 *dss_id) @@ -920,7 +929,7 @@ ff_layout_choose_any_ds_for_read(struct pnfs_layout_segment *lseg, offset, dss_id, false); } -static struct nfs4_pnfs_ds * +static struct nfs4_ff_layout_ds * ff_layout_choose_valid_ds_for_read(struct pnfs_layout_segment *lseg, u32 start_idx, u32 *best_idx, u64 offset, u32 *dss_id) @@ -929,34 +938,36 @@ ff_layout_choose_valid_ds_for_read(struct pnfs_layout_segment *lseg, offset, dss_id, true); } -static struct nfs4_pnfs_ds * +static struct nfs4_ff_layout_ds * ff_layout_choose_best_ds_for_read(struct pnfs_layout_segment *lseg, u32 start_idx, u32 *best_idx, u64 offset, u32 *dss_id) { - struct nfs4_pnfs_ds *ds; + struct nfs4_ff_layout_ds *mirror_ds; - ds = ff_layout_choose_valid_ds_for_read(lseg, start_idx, best_idx, - offset, dss_id); - if (!IS_ERR(ds)) - return ds; + mirror_ds = ff_layout_choose_valid_ds_for_read(lseg, start_idx, + best_idx, offset, + dss_id); + if (!IS_ERR(mirror_ds)) + return mirror_ds; return ff_layout_choose_any_ds_for_read(lseg, start_idx, best_idx, offset, dss_id); } -static struct nfs4_pnfs_ds * +static struct nfs4_ff_layout_ds * ff_layout_get_ds_for_read(struct nfs_pageio_descriptor *pgio, u32 *best_idx, u64 offset, u32 *dss_id) { struct pnfs_layout_segment *lseg = pgio->pg_lseg; - struct nfs4_pnfs_ds *ds; + struct nfs4_ff_layout_ds *mirror_ds; - ds = ff_layout_choose_best_ds_for_read(lseg, pgio->pg_mirror_idx, - best_idx, offset, dss_id); - if (!IS_ERR(ds) || !pgio->pg_mirror_idx) - return ds; + mirror_ds = ff_layout_choose_best_ds_for_read(lseg, + pgio->pg_mirror_idx, + best_idx, offset, dss_id); + if (!IS_ERR(mirror_ds) || !pgio->pg_mirror_idx) + return mirror_ds; return ff_layout_choose_best_ds_for_read(lseg, 0, best_idx, offset, dss_id); } @@ -1031,8 +1042,7 @@ ff_layout_pg_init_read(struct nfs_pageio_descriptor *pgio, struct nfs_page *req) { struct nfs_pgio_mirror *pgm; - struct nfs4_ff_layout_mirror *mirror; - struct nfs4_pnfs_ds *ds; + struct nfs4_ff_layout_ds *mirror_ds; u32 ds_idx, dss_id; if (NFS_SERVER(pgio->pg_inode)->flags & @@ -1054,9 +1064,9 @@ ff_layout_pg_init_read(struct nfs_pageio_descriptor *pgio, /* Reset wb_nio, since getting layout segment was successful */ req->wb_nio = 0; - ds = ff_layout_get_ds_for_read(pgio, &ds_idx, - req_offset(req), &dss_id); - if (IS_ERR(ds)) { + mirror_ds = ff_layout_get_ds_for_read(pgio, &ds_idx, + req_offset(req), &dss_id); + if (IS_ERR(mirror_ds)) { if (!ff_layout_no_fallback_to_mds(pgio->pg_lseg)) goto out_mds; pnfs_generic_pg_cleanup(pgio); @@ -1065,9 +1075,9 @@ ff_layout_pg_init_read(struct nfs_pageio_descriptor *pgio, goto retry; } - mirror = FF_LAYOUT_COMP(pgio->pg_lseg, ds_idx); pgm = &pgio->pg_mirrors[0]; - pgm->pg_bsize = mirror->dss[dss_id].mirror_ds->ds_versions[0].rsize; + pgm->pg_bsize = mirror_ds->ds_versions[0].rsize; + nfs4_ff_layout_put_deviceid(mirror_ds); pgio->pg_mirror_idx = ds_idx; return; @@ -1102,6 +1112,7 @@ ff_layout_pg_init_write(struct nfs_pageio_descriptor *pgio, struct nfs_page *req) { struct nfs4_ff_layout_mirror *mirror; + struct nfs4_ff_layout_ds *mirror_ds; struct nfs_pgio_mirror *pgm; struct nfs4_pnfs_ds *ds; u32 i, dss_id; @@ -1133,9 +1144,12 @@ ff_layout_pg_init_write(struct nfs_pageio_descriptor *pgio, FF_LAYOUT_LSEG(pgio->pg_lseg)->stripe_unit, mirror->dss_count, req_offset(req)); + mirror_ds = ff_layout_get_mirror_ds(pgio->pg_lseg->pls_layout, + mirror, dss_id); ds = nfs4_ff_layout_prepare_ds(pgio->pg_lseg, mirror, - dss_id, true); + mirror_ds, dss_id, true); if (IS_ERR(ds)) { + nfs4_ff_layout_put_deviceid(mirror_ds); if (!ff_layout_no_fallback_to_mds(pgio->pg_lseg)) goto out_mds; pnfs_generic_pg_cleanup(pgio); @@ -1144,7 +1158,8 @@ ff_layout_pg_init_write(struct nfs_pageio_descriptor *pgio, goto retry; } pgm = &pgio->pg_mirrors[i]; - pgm->pg_bsize = mirror->dss[dss_id].mirror_ds->ds_versions[0].wsize; + pgm->pg_bsize = mirror_ds->ds_versions[0].wsize; + nfs4_ff_layout_put_deviceid(mirror_ds); } if (NFS_SERVER(pgio->pg_inode)->flags & @@ -1276,14 +1291,16 @@ static void ff_layout_resend_pnfs_read(struct nfs_pgio_header *hdr) u32 idx = hdr->pgio_mirror_idx + 1; u32 new_idx = 0; u32 dss_id = 0; - struct nfs4_pnfs_ds *ds; + struct nfs4_ff_layout_ds *mirror_ds; - ds = ff_layout_choose_any_ds_for_read(hdr->lseg, idx, &new_idx, - hdr->args.offset, &dss_id); - if (IS_ERR(ds)) + mirror_ds = ff_layout_choose_any_ds_for_read(hdr->lseg, idx, &new_idx, + hdr->args.offset, &dss_id); + if (IS_ERR(mirror_ds)) { pnfs_error_mark_layout_for_return(hdr->inode, hdr->lseg); - else + } else { + nfs4_ff_layout_put_deviceid(mirror_ds); ff_layout_send_layouterror(hdr->lseg); + } pnfs_read_resend_pnfs(hdr, new_idx); } @@ -2166,6 +2183,7 @@ ff_layout_read_pagelist(struct nfs_pgio_header *hdr) struct rpc_clnt *ds_clnt; struct nfsd_file *localio; struct nfs4_ff_layout_mirror *mirror; + struct nfs4_ff_layout_ds *mirror_ds; const struct cred *ds_cred; loff_t offset = hdr->args.offset; u32 idx = hdr->pgio_mirror_idx; @@ -2183,22 +2201,24 @@ ff_layout_read_pagelist(struct nfs_pgio_header *hdr) FF_LAYOUT_LSEG(lseg)->stripe_unit, mirror->dss_count, offset); - ds = nfs4_ff_layout_prepare_ds(lseg, mirror, dss_id, false); + mirror_ds = ff_layout_get_mirror_ds(lseg->pls_layout, mirror, dss_id); + ds = nfs4_ff_layout_prepare_ds(lseg, mirror, mirror_ds, dss_id, false); if (IS_ERR(ds)) { ds_fatal_error = nfs_error_is_fatal(PTR_ERR(ds)); goto out_failed; } - ds_clnt = nfs4_ff_find_or_create_ds_client(mirror, ds->ds_clp, - hdr->inode, dss_id); + ds_clnt = nfs4_ff_find_or_create_ds_client(mirror_ds, ds->ds_clp, + hdr->inode); if (IS_ERR(ds_clnt)) goto out_failed; - ds_cred = ff_layout_get_ds_cred(mirror, &lseg->pls_range, hdr->cred, dss_id); + ds_cred = ff_layout_get_ds_cred(mirror, &lseg->pls_range, hdr->cred, + mirror_ds, dss_id); if (!ds_cred) goto out_failed; - vers = nfs4_ff_layout_ds_version(mirror, dss_id); + vers = nfs4_ff_layout_ds_version(mirror_ds); dprintk("%s USE DS: %s cl_count %d vers %d\n", __func__, ds->ds_remotestr, refcount_read(&ds->ds_clp->cl_count), vers); @@ -2210,7 +2230,8 @@ ff_layout_read_pagelist(struct nfs_pgio_header *hdr) if (fh) hdr->args.fh = fh; - nfs4_ff_layout_select_ds_stateid(mirror, dss_id, &hdr->args.stateid); + nfs4_ff_layout_select_ds_stateid(mirror, mirror_ds, dss_id, + &hdr->args.stateid); /* * Note that if we ever decide to split across DSes, @@ -2233,9 +2254,11 @@ ff_layout_read_pagelist(struct nfs_pgio_header *hdr) &ff_layout_read_call_ops_v4, 0, RPC_TASK_SOFTCONN, localio); put_cred(ds_cred); + nfs4_ff_layout_put_deviceid(mirror_ds); return PNFS_ATTEMPTED; out_failed: + nfs4_ff_layout_put_deviceid(mirror_ds); if (ff_layout_avoid_mds_available_ds(lseg) && !ds_fatal_error) return PNFS_TRY_AGAIN; if (ff_layout_no_fallback_to_mds(lseg)) { @@ -2261,6 +2284,7 @@ ff_layout_write_pagelist(struct nfs_pgio_header *hdr, int sync) struct rpc_clnt *ds_clnt; struct nfsd_file *localio; struct nfs4_ff_layout_mirror *mirror; + struct nfs4_ff_layout_ds *mirror_ds; const struct cred *ds_cred; loff_t offset = hdr->args.offset; int vers; @@ -2274,22 +2298,24 @@ ff_layout_write_pagelist(struct nfs_pgio_header *hdr, int sync) FF_LAYOUT_LSEG(lseg)->stripe_unit, mirror->dss_count, offset); - ds = nfs4_ff_layout_prepare_ds(lseg, mirror, dss_id, true); + mirror_ds = ff_layout_get_mirror_ds(lseg->pls_layout, mirror, dss_id); + ds = nfs4_ff_layout_prepare_ds(lseg, mirror, mirror_ds, dss_id, true); if (IS_ERR(ds)) { ds_fatal_error = nfs_error_is_fatal(PTR_ERR(ds)); goto out_failed; } - ds_clnt = nfs4_ff_find_or_create_ds_client(mirror, ds->ds_clp, - hdr->inode, dss_id); + ds_clnt = nfs4_ff_find_or_create_ds_client(mirror_ds, ds->ds_clp, + hdr->inode); if (IS_ERR(ds_clnt)) goto out_failed; - ds_cred = ff_layout_get_ds_cred(mirror, &lseg->pls_range, hdr->cred, dss_id); + ds_cred = ff_layout_get_ds_cred(mirror, &lseg->pls_range, hdr->cred, + mirror_ds, dss_id); if (!ds_cred) goto out_failed; - vers = nfs4_ff_layout_ds_version(mirror, dss_id); + vers = nfs4_ff_layout_ds_version(mirror_ds); dprintk("%s ino %llu sync %d req %zu@%llu DS: %s cl_count %d vers %d\n", __func__, hdr->inode->i_ino, sync, (size_t) hdr->args.count, @@ -2304,7 +2330,8 @@ ff_layout_write_pagelist(struct nfs_pgio_header *hdr, int sync) if (fh) hdr->args.fh = fh; - nfs4_ff_layout_select_ds_stateid(mirror, dss_id, &hdr->args.stateid); + nfs4_ff_layout_select_ds_stateid(mirror, mirror_ds, dss_id, + &hdr->args.stateid); /* * Note that if we ever decide to split across DSes, @@ -2326,9 +2353,11 @@ ff_layout_write_pagelist(struct nfs_pgio_header *hdr, int sync) &ff_layout_write_call_ops_v4, sync, RPC_TASK_SOFTCONN, localio); put_cred(ds_cred); + nfs4_ff_layout_put_deviceid(mirror_ds); return PNFS_ATTEMPTED; out_failed: + nfs4_ff_layout_put_deviceid(mirror_ds); if (ff_layout_avoid_mds_available_ds(lseg) && !ds_fatal_error) return PNFS_TRY_AGAIN; if (ff_layout_no_fallback_to_mds(lseg)) { @@ -2363,6 +2392,7 @@ static int ff_layout_initiate_commit(struct nfs_commit_data *data, int how) struct rpc_clnt *ds_clnt; struct nfsd_file *localio; struct nfs4_ff_layout_mirror *mirror; + struct nfs4_ff_layout_ds *mirror_ds = NULL; const struct cred *ds_cred; u32 idx, dss_id; int vers, ret; @@ -2375,20 +2405,22 @@ static int ff_layout_initiate_commit(struct nfs_commit_data *data, int how) idx = calc_mirror_idx_from_commit(lseg, data->ds_commit_index); mirror = FF_LAYOUT_COMP(lseg, idx); dss_id = calc_dss_id_from_commit(lseg, data->ds_commit_index); - ds = nfs4_ff_layout_prepare_ds(lseg, mirror, dss_id, true); + mirror_ds = ff_layout_get_mirror_ds(lseg->pls_layout, mirror, dss_id); + ds = nfs4_ff_layout_prepare_ds(lseg, mirror, mirror_ds, dss_id, true); if (IS_ERR(ds)) goto out_err; - ds_clnt = nfs4_ff_find_or_create_ds_client(mirror, ds->ds_clp, - data->inode, dss_id); + ds_clnt = nfs4_ff_find_or_create_ds_client(mirror_ds, ds->ds_clp, + data->inode); if (IS_ERR(ds_clnt)) goto out_err; - ds_cred = ff_layout_get_ds_cred(mirror, &lseg->pls_range, data->cred, dss_id); + ds_cred = ff_layout_get_ds_cred(mirror, &lseg->pls_range, data->cred, + mirror_ds, dss_id); if (!ds_cred) goto out_err; - vers = nfs4_ff_layout_ds_version(mirror, dss_id); + vers = nfs4_ff_layout_ds_version(mirror_ds); dprintk("%s ino %llu, how %d cl_count %d vers %d\n", __func__, data->inode->i_ino, how, refcount_read(&ds->ds_clp->cl_count), @@ -2414,8 +2446,10 @@ static int ff_layout_initiate_commit(struct nfs_commit_data *data, int how) &ff_layout_commit_call_ops_v4, how, RPC_TASK_SOFTCONN, localio); put_cred(ds_cred); + nfs4_ff_layout_put_deviceid(mirror_ds); return ret; out_err: + nfs4_ff_layout_put_deviceid(mirror_ds); pnfs_generic_prepare_to_resend_writes(data); pnfs_generic_commit_release(data); return -EAGAIN; diff --git a/fs/nfs/flexfilelayout/flexfilelayout.h b/fs/nfs/flexfilelayout/flexfilelayout.h index ec69cd1c3ae963..f9e491a0347e15 100644 --- a/fs/nfs/flexfilelayout/flexfilelayout.h +++ b/fs/nfs/flexfilelayout/flexfilelayout.h @@ -207,9 +207,9 @@ ff_layout_no_read_on_rw(struct pnfs_layout_segment *lseg) } static inline int -nfs4_ff_layout_ds_version(const struct nfs4_ff_layout_mirror *mirror, u32 dss_id) +nfs4_ff_layout_ds_version(const struct nfs4_ff_layout_ds *mirror_ds) { - return mirror->dss[dss_id].mirror_ds->ds_versions[0].version; + return mirror_ds->ds_versions[0].version; } static inline u32 @@ -245,23 +245,29 @@ struct nfs_fh * nfs4_ff_layout_select_ds_fh(struct nfs4_ff_layout_mirror *mirror, u32 dss_id); void nfs4_ff_layout_select_ds_stateid(const struct nfs4_ff_layout_mirror *mirror, + const struct nfs4_ff_layout_ds *mirror_ds, u32 dss_id, nfs4_stateid *stateid); +struct nfs4_ff_layout_ds * +ff_layout_get_mirror_ds(struct pnfs_layout_hdr *lo, + struct nfs4_ff_layout_mirror *mirror, + u32 dss_id); struct nfs4_pnfs_ds * nfs4_ff_layout_prepare_ds(struct pnfs_layout_segment *lseg, struct nfs4_ff_layout_mirror *mirror, + struct nfs4_ff_layout_ds *mirror_ds, u32 dss_id, bool fail_return); struct rpc_clnt * -nfs4_ff_find_or_create_ds_client(struct nfs4_ff_layout_mirror *mirror, +nfs4_ff_find_or_create_ds_client(const struct nfs4_ff_layout_ds *mirror_ds, struct nfs_client *ds_clp, - struct inode *inode, - u32 dss_id); + struct inode *inode); const struct cred *ff_layout_get_ds_cred(struct nfs4_ff_layout_mirror *mirror, const struct pnfs_layout_range *range, const struct cred *mdscred, + const struct nfs4_ff_layout_ds *mirror_ds, u32 dss_id); bool ff_layout_avoid_mds_available_ds(struct pnfs_layout_segment *lseg); bool ff_layout_avoid_read_on_rw(struct pnfs_layout_segment *lseg); diff --git a/fs/nfs/flexfilelayout/flexfilelayoutdev.c b/fs/nfs/flexfilelayout/flexfilelayoutdev.c index 6cd3859d4a243d..3712a47b20d08b 100644 --- a/fs/nfs/flexfilelayout/flexfilelayoutdev.c +++ b/fs/nfs/flexfilelayout/flexfilelayoutdev.c @@ -304,24 +304,33 @@ nfs4_ff_layout_select_ds_fh(struct nfs4_ff_layout_mirror *mirror, u32 dss_id) void nfs4_ff_layout_select_ds_stateid(const struct nfs4_ff_layout_mirror *mirror, + const struct nfs4_ff_layout_ds *mirror_ds, u32 dss_id, nfs4_stateid *stateid) { - if (nfs4_ff_layout_ds_version(mirror, dss_id) == 4) + if (nfs4_ff_layout_ds_version(mirror_ds) == 4) nfs4_stateid_copy(stateid, &mirror->dss[dss_id].stateid); } -static bool -ff_layout_init_mirror_ds(struct pnfs_layout_hdr *lo, - struct nfs4_ff_layout_mirror *mirror, - u32 dss_id) +/* + * Resolve the stripe's deviceid on first use and pin the node on the + * mirror. Returns a node the caller must put, or an ERR_PTR. + */ +struct nfs4_ff_layout_ds * +ff_layout_get_mirror_ds(struct pnfs_layout_hdr *lo, + struct nfs4_ff_layout_mirror *mirror, + u32 dss_id) { + struct nfs4_ff_layout_ds *mirror_ds; + if (mirror == NULL) - goto outerr; - if (mirror->dss[dss_id].mirror_ds == NULL) { + return ERR_PTR(-ENODEV); + + mirror_ds = mirror->dss[dss_id].mirror_ds; + if (mirror_ds == NULL) { struct nfs4_deviceid_node *node; - struct nfs4_ff_layout_ds *mirror_ds = ERR_PTR(-ENODEV); + mirror_ds = ERR_PTR(-ENODEV); node = nfs4_find_get_deviceid(NFS_SERVER(lo->plh_inode), &mirror->dss[dss_id].devid, lo->plh_lc_cred, GFP_KERNEL); @@ -332,20 +341,23 @@ ff_layout_init_mirror_ds(struct pnfs_layout_hdr *lo, if (cmpxchg(&mirror->dss[dss_id].mirror_ds, NULL, mirror_ds) && mirror_ds != ERR_PTR(-ENODEV)) nfs4_put_deviceid_node(node); - } - if (IS_ERR(mirror->dss[dss_id].mirror_ds)) - goto outerr; + mirror_ds = mirror->dss[dss_id].mirror_ds; + } - return true; -outerr: - return false; + if (IS_ERR(mirror_ds)) + return mirror_ds; + if (!atomic_inc_not_zero(&mirror_ds->id_node.ref)) + return ERR_PTR(-ENODEV); + return mirror_ds; } /** * nfs4_ff_layout_prepare_ds - prepare a DS connection for an RPC call * @lseg: the layout segment we're operating on * @mirror: layout mirror describing the DS to use + * @mirror_ds: referenced device node for the stripe, from + * ff_layout_get_mirror_ds() (may be an ERR_PTR) * @dss_id: DS stripe id to select stripe to use * @fail_return: return layout on connect failure? * @@ -363,6 +375,7 @@ ff_layout_init_mirror_ds(struct pnfs_layout_hdr *lo, struct nfs4_pnfs_ds * nfs4_ff_layout_prepare_ds(struct pnfs_layout_segment *lseg, struct nfs4_ff_layout_mirror *mirror, + struct nfs4_ff_layout_ds *mirror_ds, u32 dss_id, bool fail_return) { @@ -372,10 +385,10 @@ nfs4_ff_layout_prepare_ds(struct pnfs_layout_segment *lseg, unsigned int max_payload; int status = -EAGAIN; - if (!ff_layout_init_mirror_ds(lseg->pls_layout, mirror, dss_id)) + if (IS_ERR_OR_NULL(mirror_ds)) goto noconnect; - ds = mirror->dss[dss_id].mirror_ds->ds; + ds = mirror_ds->ds; if (READ_ONCE(ds->ds_clp)) goto out; /* matching smp_wmb() in _nfs4_pnfs_v3/4_ds_connect */ @@ -384,11 +397,11 @@ nfs4_ff_layout_prepare_ds(struct pnfs_layout_segment *lseg, /* FIXME: For now we assume the server sent only one version of NFS * to use for the DS. */ - status = nfs4_pnfs_ds_connect(s, ds, &mirror->dss[dss_id].mirror_ds->id_node, + status = nfs4_pnfs_ds_connect(s, ds, &mirror_ds->id_node, dataserver_timeo, dataserver_retrans, - mirror->dss[dss_id].mirror_ds->ds_versions[0].version, - mirror->dss[dss_id].mirror_ds->ds_versions[0].minor_version, - mirror->dss[dss_id].mirror_ds->ds_versions[0].tightly_coupled); + mirror_ds->ds_versions[0].version, + mirror_ds->ds_versions[0].minor_version, + mirror_ds->ds_versions[0].tightly_coupled); /* connect success, check rsize/wsize limit */ if (!status) { @@ -401,10 +414,10 @@ nfs4_ff_layout_prepare_ds(struct pnfs_layout_segment *lseg, max_payload = nfs_block_size(rpc_max_payload(ds->ds_clp->cl_rpcclient), NULL); - if (mirror->dss[dss_id].mirror_ds->ds_versions[0].rsize > max_payload) - mirror->dss[dss_id].mirror_ds->ds_versions[0].rsize = max_payload; - if (mirror->dss[dss_id].mirror_ds->ds_versions[0].wsize > max_payload) - mirror->dss[dss_id].mirror_ds->ds_versions[0].wsize = max_payload; + if (mirror_ds->ds_versions[0].rsize > max_payload) + mirror_ds->ds_versions[0].rsize = max_payload; + if (mirror_ds->ds_versions[0].wsize > max_payload) + mirror_ds->ds_versions[0].wsize = max_payload; goto out; } noconnect: @@ -424,11 +437,12 @@ const struct cred * ff_layout_get_ds_cred(struct nfs4_ff_layout_mirror *mirror, const struct pnfs_layout_range *range, const struct cred *mdscred, + const struct nfs4_ff_layout_ds *mirror_ds, u32 dss_id) { const struct cred *cred; - if (mirror && !mirror->dss[dss_id].mirror_ds->ds_versions[0].tightly_coupled) { + if (mirror && !mirror_ds->ds_versions[0].tightly_coupled) { cred = ff_layout_get_mirror_cred(mirror, range->iomode, dss_id); if (!cred) cred = get_cred(mdscred); @@ -440,20 +454,18 @@ ff_layout_get_ds_cred(struct nfs4_ff_layout_mirror *mirror, /** * nfs4_ff_find_or_create_ds_client - Find or create a DS rpc client - * @mirror: pointer to the mirror + * @mirror_ds: device node for the stripe * @ds_clp: nfs_client for the DS * @inode: pointer to inode - * @dss_id: DS stripe id * * Find or create a DS rpc client with th MDS server rpc client auth flavor * in the nfs_client cl_ds_clients list. */ struct rpc_clnt * -nfs4_ff_find_or_create_ds_client(struct nfs4_ff_layout_mirror *mirror, - struct nfs_client *ds_clp, struct inode *inode, - u32 dss_id) +nfs4_ff_find_or_create_ds_client(const struct nfs4_ff_layout_ds *mirror_ds, + struct nfs_client *ds_clp, struct inode *inode) { - switch (mirror->dss[dss_id].mirror_ds->ds_versions[0].version) { + switch (mirror_ds->ds_versions[0].version) { case 3: /* For NFSv3 DS, flavor is set when creating DS connections */ return ds_clp->cl_rpcclient; From 6fcb69e2fdee94b033b26ea0bc9d0217f4bfd5bb Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:11 -0400 Subject: [PATCH 0228/1352] NFSv4/flexfiles: Carry the device node reference across each I/O Rather than dropping the device node reference when DS setup completes, transfer it to the in-flight I/O: carry it on nfs_pgio_header and nfs_commit_data as ds_dev, and release it when the header or commit data is released, alongside the lseg reference. Convert the completion paths to use the carried node instead of re-reading the mirror's pinned pointer: DS error tracking, marking the deviceid available/unavailable, and deleting the deviceid on connection errors now act on the node the I/O was actually sent to. Once a CHANGE notification can re-point the pinned pointer mid-flight, this keeps error attribution on the old device rather than its replacement (and removes a NULL dereference had the pointer been reset to NULL). FF_LAYOUT_DEVID_NODE() is now unused; remove it. Assisted-by: Claude:claude-fable-5 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/flexfilelayout/flexfilelayout.c | 82 ++++++++++------------- fs/nfs/flexfilelayout/flexfilelayout.h | 15 +---- fs/nfs/flexfilelayout/flexfilelayoutdev.c | 11 +-- fs/nfs/pnfs.c | 2 + fs/nfs/pnfs.h | 8 +++ fs/nfs/pnfs_nfs.c | 1 + include/linux/nfs_xdr.h | 2 + 7 files changed, 58 insertions(+), 63 deletions(-) diff --git a/fs/nfs/flexfilelayout/flexfilelayout.c b/fs/nfs/flexfilelayout/flexfilelayout.c index 7a908faa41319b..928e691af9e72d 100644 --- a/fs/nfs/flexfilelayout/flexfilelayout.c +++ b/fs/nfs/flexfilelayout/flexfilelayout.c @@ -857,24 +857,6 @@ nfs4_ff_layout_stat_io_end_write(struct rpc_task *task, spin_unlock(&mirror->lock); } -static void -ff_layout_mark_ds_unreachable(struct pnfs_layout_segment *lseg, u32 idx, u32 dss_id) -{ - struct nfs4_deviceid_node *devid = FF_LAYOUT_DEVID_NODE(lseg, idx, dss_id); - - if (devid) - nfs4_mark_deviceid_unavailable(devid); -} - -static void -ff_layout_mark_ds_reachable(struct pnfs_layout_segment *lseg, u32 idx, u32 dss_id) -{ - struct nfs4_deviceid_node *devid = FF_LAYOUT_DEVID_NODE(lseg, idx, dss_id); - - if (devid) - nfs4_mark_deviceid_available(devid); -} - static struct nfs4_ff_layout_ds * ff_layout_choose_ds_for_read(struct pnfs_layout_segment *lseg, u32 start_idx, u32 *best_idx, @@ -1333,11 +1315,10 @@ static int ff_layout_async_handle_error_v4(struct rpc_task *task, struct nfs4_state *state, struct nfs_client *clp, struct pnfs_layout_segment *lseg, - u32 idx, u32 dss_id) + struct nfs4_deviceid_node *devid) { struct pnfs_layout_hdr *lo = lseg->pls_layout; struct inode *inode = lo->plh_inode; - struct nfs4_deviceid_node *devid = FF_LAYOUT_DEVID_NODE(lseg, idx, dss_id); struct nfs4_slot_table *tbl = nfs4_has_session(clp) ? &clp->cl_session->fc_slot_table : clp->cl_slot_tbl; @@ -1410,8 +1391,9 @@ static int ff_layout_async_handle_error_v4(struct rpc_task *task, case -ENODEV: dprintk("%s DS connection error %d\n", __func__, task->tk_status); - nfs4_delete_deviceid(devid->ld, devid->nfs_client, - &devid->deviceid); + if (devid) + nfs4_delete_deviceid(devid->ld, devid->nfs_client, + &devid->deviceid); rpc_wake_up(&tbl->slot_tbl_waitq); break; default: @@ -1435,9 +1417,8 @@ static int ff_layout_async_handle_error_v3(struct rpc_task *task, u32 op_status, struct nfs_client *clp, struct pnfs_layout_segment *lseg, - u32 idx, u32 dss_id) + struct nfs4_deviceid_node *devid) { - struct nfs4_deviceid_node *devid = FF_LAYOUT_DEVID_NODE(lseg, idx, dss_id); switch (op_status) { case NFS_OK: @@ -1483,8 +1464,9 @@ static int ff_layout_async_handle_error_v3(struct rpc_task *task, default: dprintk("%s DS connection error %d\n", __func__, task->tk_status); - nfs4_delete_deviceid(devid->ld, devid->nfs_client, - &devid->deviceid); + if (devid) + nfs4_delete_deviceid(devid->ld, devid->nfs_client, + &devid->deviceid); } out_reset_to_pnfs: /* FIXME: Need to prevent infinite looping here. */ @@ -1501,12 +1483,13 @@ static int ff_layout_async_handle_error(struct rpc_task *task, struct nfs4_state *state, struct nfs_client *clp, struct pnfs_layout_segment *lseg, - u32 idx, u32 dss_id) + struct nfs4_deviceid_node *devid) { int vers = clp->cl_nfs_mod->rpc_vers->number; if (task->tk_status >= 0) { - ff_layout_mark_ds_reachable(lseg, idx, dss_id); + if (devid) + nfs4_mark_deviceid_available(devid); return 0; } @@ -1517,10 +1500,10 @@ static int ff_layout_async_handle_error(struct rpc_task *task, switch (vers) { case 3: return ff_layout_async_handle_error_v3(task, op_status, clp, - lseg, idx, dss_id); + lseg, devid); case 4: return ff_layout_async_handle_error_v4(task, op_status, state, - clp, lseg, idx, dss_id); + clp, lseg, devid); default: /* should never happen */ WARN_ON_ONCE(1); @@ -1529,6 +1512,7 @@ static int ff_layout_async_handle_error(struct rpc_task *task, } static void ff_layout_io_track_ds_error(struct pnfs_layout_segment *lseg, + struct nfs4_deviceid_node *devid, u32 idx, u32 dss_id, u64 offset, u64 length, u32 *op_status, int opnum, int error) { @@ -1578,8 +1562,8 @@ static void ff_layout_io_track_ds_error(struct pnfs_layout_segment *lseg, mirror = FF_LAYOUT_COMP(lseg, idx); err = ff_layout_track_ds_error(FF_LAYOUT_FROM_HDR(lseg->pls_layout), - mirror, dss_id, offset, length, status, opnum, - nfs_io_gfp_mask()); + mirror, devid, dss_id, offset, length, + status, opnum, nfs_io_gfp_mask()); /* * I/O we cancelled ourselves to return a recalled or revoked layout @@ -1596,7 +1580,8 @@ static void ff_layout_io_track_ds_error(struct pnfs_layout_segment *lseg, case NFS4ERR_PERM: break; case NFS4ERR_NXIO: - ff_layout_mark_ds_unreachable(lseg, idx, dss_id); + if (devid) + nfs4_mark_deviceid_unavailable(devid); /* * Don't return the layout if this is a read and we still * have layouts to try @@ -1625,7 +1610,7 @@ static int ff_layout_read_done_cb(struct rpc_task *task, int err; if (task->tk_status < 0) { - ff_layout_io_track_ds_error(hdr->lseg, + ff_layout_io_track_ds_error(hdr->lseg, hdr->ds_dev, hdr->pgio_mirror_idx, dss_id, hdr->args.offset, hdr->args.count, &hdr->res.op_status, OP_READ, @@ -1636,8 +1621,7 @@ static int ff_layout_read_done_cb(struct rpc_task *task, err = ff_layout_async_handle_error(task, hdr->res.op_status, hdr->args.context->state, hdr->ds_clp, hdr->lseg, - hdr->pgio_mirror_idx, - dss_id); + hdr->ds_dev); trace_nfs4_pnfs_read(hdr, err); clear_bit(NFS_IOHDR_RESEND_PNFS, &hdr->flags); @@ -1830,7 +1814,7 @@ static int ff_layout_write_done_cb(struct rpc_task *task, int err; if (task->tk_status < 0) { - ff_layout_io_track_ds_error(hdr->lseg, + ff_layout_io_track_ds_error(hdr->lseg, hdr->ds_dev, hdr->pgio_mirror_idx, dss_id, hdr->args.offset, hdr->args.count, &hdr->res.op_status, OP_WRITE, @@ -1841,8 +1825,7 @@ static int ff_layout_write_done_cb(struct rpc_task *task, err = ff_layout_async_handle_error(task, hdr->res.op_status, hdr->args.context->state, hdr->ds_clp, hdr->lseg, - hdr->pgio_mirror_idx, - dss_id); + hdr->ds_dev); trace_nfs4_pnfs_write(hdr, err); clear_bit(NFS_IOHDR_RESEND_PNFS, &hdr->flags); @@ -1884,7 +1867,7 @@ static int ff_layout_commit_done_cb(struct rpc_task *task, u32 dss_id = calc_dss_id_from_commit(data->lseg, data->ds_commit_index); if (task->tk_status < 0) { - ff_layout_io_track_ds_error(data->lseg, idx, dss_id, + ff_layout_io_track_ds_error(data->lseg, data->ds_dev, idx, dss_id, data->args.offset, data->args.count, &data->res.op_status, OP_COMMIT, task->tk_status); @@ -1892,8 +1875,8 @@ static int ff_layout_commit_done_cb(struct rpc_task *task, } err = ff_layout_async_handle_error(task, data->res.op_status, - NULL, data->ds_clp, data->lseg, idx, - dss_id); + NULL, data->ds_clp, data->lseg, + data->ds_dev); trace_nfs4_pnfs_commit_ds(data, err); switch (err) { @@ -2248,13 +2231,16 @@ ff_layout_read_pagelist(struct nfs_pgio_header *hdr) ff_layout_read_record_layoutstats_start(&hdr->task, hdr); } + /* Transfer the device node reference to the I/O; put on release */ + pnfs_put_ds_dev(hdr->ds_dev); + hdr->ds_dev = &mirror_ds->id_node; + /* Perform an asynchronous read to ds */ nfs_initiate_pgio(ds_clnt, hdr, ds_cred, ds->ds_clp->rpc_ops, vers == 3 ? &ff_layout_read_call_ops_v3 : &ff_layout_read_call_ops_v4, 0, RPC_TASK_SOFTCONN, localio); put_cred(ds_cred); - nfs4_ff_layout_put_deviceid(mirror_ds); return PNFS_ATTEMPTED; out_failed: @@ -2347,13 +2333,16 @@ ff_layout_write_pagelist(struct nfs_pgio_header *hdr, int sync) ff_layout_write_record_layoutstats_start(&hdr->task, hdr); } + /* Transfer the device node reference to the I/O; put on release */ + pnfs_put_ds_dev(hdr->ds_dev); + hdr->ds_dev = &mirror_ds->id_node; + /* Perform an asynchronous write */ nfs_initiate_pgio(ds_clnt, hdr, ds_cred, ds->ds_clp->rpc_ops, vers == 3 ? &ff_layout_write_call_ops_v3 : &ff_layout_write_call_ops_v4, sync, RPC_TASK_SOFTCONN, localio); put_cred(ds_cred); - nfs4_ff_layout_put_deviceid(mirror_ds); return PNFS_ATTEMPTED; out_failed: @@ -2441,12 +2430,15 @@ static int ff_layout_initiate_commit(struct nfs_commit_data *data, int how) ff_layout_commit_record_layoutstats_start(&data->task, data); } + /* Transfer the device node reference to the commit; put on release */ + pnfs_put_ds_dev(data->ds_dev); + data->ds_dev = &mirror_ds->id_node; + ret = nfs_initiate_commit(ds_clnt, data, ds->ds_clp->rpc_ops, vers == 3 ? &ff_layout_commit_call_ops_v3 : &ff_layout_commit_call_ops_v4, how, RPC_TASK_SOFTCONN, localio); put_cred(ds_cred); - nfs4_ff_layout_put_deviceid(mirror_ds); return ret; out_err: nfs4_ff_layout_put_deviceid(mirror_ds); diff --git a/fs/nfs/flexfilelayout/flexfilelayout.h b/fs/nfs/flexfilelayout/flexfilelayout.h index f9e491a0347e15..7eb47ce09442d1 100644 --- a/fs/nfs/flexfilelayout/flexfilelayout.h +++ b/fs/nfs/flexfilelayout/flexfilelayout.h @@ -162,20 +162,6 @@ FF_LAYOUT_COMP(struct pnfs_layout_segment *lseg, u32 idx) return NULL; } -static inline struct nfs4_deviceid_node * -FF_LAYOUT_DEVID_NODE(struct pnfs_layout_segment *lseg, u32 idx, u32 dss_id) -{ - struct nfs4_ff_layout_mirror *mirror = FF_LAYOUT_COMP(lseg, idx); - - if (mirror != NULL) { - struct nfs4_ff_layout_ds *mirror_ds = mirror->dss[dss_id].mirror_ds; - - if (!IS_ERR_OR_NULL(mirror_ds)) - return &mirror_ds->id_node; - } - return NULL; -} - static inline u32 FF_LAYOUT_MIRROR_COUNT(struct pnfs_layout_segment *lseg) { @@ -232,6 +218,7 @@ void nfs4_ff_layout_put_deviceid(struct nfs4_ff_layout_ds *mirror_ds); void nfs4_ff_layout_free_deviceid(struct nfs4_ff_layout_ds *mirror_ds); int ff_layout_track_ds_error(struct nfs4_flexfile_layout *flo, struct nfs4_ff_layout_mirror *mirror, + const struct nfs4_deviceid_node *devid, u32 dss_id, u64 offset, u64 length, int status, enum nfs_opnum4 opnum, gfp_t gfp_flags); void ff_layout_send_layouterror(struct pnfs_layout_segment *lseg); diff --git a/fs/nfs/flexfilelayout/flexfilelayoutdev.c b/fs/nfs/flexfilelayout/flexfilelayoutdev.c index 3712a47b20d08b..ebcfba44787989 100644 --- a/fs/nfs/flexfilelayout/flexfilelayoutdev.c +++ b/fs/nfs/flexfilelayout/flexfilelayoutdev.c @@ -243,6 +243,7 @@ ff_layout_add_ds_error_locked(struct nfs4_flexfile_layout *flo, int ff_layout_track_ds_error(struct nfs4_flexfile_layout *flo, struct nfs4_ff_layout_mirror *mirror, + const struct nfs4_deviceid_node *devid, u32 dss_id, u64 offset, u64 length, int status, enum nfs_opnum4 opnum, gfp_t gfp_flags) { @@ -251,7 +252,7 @@ int ff_layout_track_ds_error(struct nfs4_flexfile_layout *flo, if (status == 0) return 0; - if (IS_ERR_OR_NULL(mirror->dss[dss_id].mirror_ds)) + if (devid == NULL) return -EINVAL; dserr = kmalloc_obj(*dserr, gfp_flags); @@ -264,8 +265,7 @@ int ff_layout_track_ds_error(struct nfs4_flexfile_layout *flo, dserr->status = status; dserr->opnum = opnum; nfs4_stateid_copy(&dserr->stateid, &mirror->dss[dss_id].stateid); - memcpy(&dserr->deviceid, &mirror->dss[dss_id].mirror_ds->id_node.deviceid, - NFS4_DEVICEID4_SIZE); + memcpy(&dserr->deviceid, &devid->deviceid, NFS4_DEVICEID4_SIZE); spin_lock(&flo->generic_hdr.plh_inode->i_lock); ff_layout_add_ds_error_locked(flo, dserr); @@ -422,7 +422,10 @@ nfs4_ff_layout_prepare_ds(struct pnfs_layout_segment *lseg, } noconnect: ff_layout_track_ds_error(FF_LAYOUT_FROM_HDR(lseg->pls_layout), - mirror, dss_id, lseg->pls_range.offset, + mirror, + IS_ERR_OR_NULL(mirror_ds) ? + NULL : &mirror_ds->id_node, + dss_id, lseg->pls_range.offset, lseg->pls_range.length, NFS4ERR_NXIO, OP_ILLEGAL, GFP_NOIO); ff_layout_send_layouterror(lseg); diff --git a/fs/nfs/pnfs.c b/fs/nfs/pnfs.c index 3e8d1a1fd827cd..a74ffdda5bcd5c 100644 --- a/fs/nfs/pnfs.c +++ b/fs/nfs/pnfs.c @@ -3145,6 +3145,7 @@ pnfs_do_write(struct nfs_pageio_descriptor *desc, static void pnfs_writehdr_free(struct nfs_pgio_header *hdr) { + pnfs_put_ds_dev(hdr->ds_dev); pnfs_put_lseg(hdr->lseg); nfs_pgio_header_free(hdr); } @@ -3290,6 +3291,7 @@ pnfs_do_read(struct nfs_pageio_descriptor *desc, struct nfs_pgio_header *hdr) static void pnfs_readhdr_free(struct nfs_pgio_header *hdr) { + pnfs_put_ds_dev(hdr->ds_dev); pnfs_put_lseg(hdr->lseg); nfs_pgio_header_free(hdr); } diff --git a/fs/nfs/pnfs.h b/fs/nfs/pnfs.h index 5725ea06efc2d8..a2c70573dda529 100644 --- a/fs/nfs/pnfs.h +++ b/fs/nfs/pnfs.h @@ -385,6 +385,14 @@ void nfs4_init_deviceid_node(struct nfs4_deviceid_node *, struct nfs_server *, const struct nfs4_deviceid *); bool nfs4_put_deviceid_node(struct nfs4_deviceid_node *); void nfs4_mark_deviceid_available(struct nfs4_deviceid_node *node); + +/* Put the device node reference carried by an in-flight I/O, if any */ +static inline void pnfs_put_ds_dev(struct nfs4_deviceid_node *dev) +{ + if (dev) + nfs4_put_deviceid_node(dev); +} + void nfs4_mark_deviceid_unavailable(struct nfs4_deviceid_node *node); bool nfs4_test_deviceid_unavailable(struct nfs4_deviceid_node *node); void nfs4_deviceid_purge_client(const struct nfs_client *); diff --git a/fs/nfs/pnfs_nfs.c b/fs/nfs/pnfs_nfs.c index 5bd014b981bf2d..e0e3fc7414e600 100644 --- a/fs/nfs/pnfs_nfs.c +++ b/fs/nfs/pnfs_nfs.c @@ -55,6 +55,7 @@ void pnfs_generic_commit_release(void *calldata) struct nfs_commit_data *data = calldata; data->completion_ops->completion(data); + pnfs_put_ds_dev(data->ds_dev); pnfs_put_lseg(data->lseg); nfs_put_client(data->ds_clp); nfs_commitdata_release(data); diff --git a/include/linux/nfs_xdr.h b/include/linux/nfs_xdr.h index 7ed8fdb930d664..c0e29b4dfa6261 100644 --- a/include/linux/nfs_xdr.h +++ b/include/linux/nfs_xdr.h @@ -1693,6 +1693,7 @@ struct nfs_pgio_header { struct nfs_client *ds_clp; /* pNFS data server */ u32 ds_commit_idx; /* ds index if ds_clp is set */ u32 pgio_mirror_idx;/* mirror index in pgio layer */ + struct nfs4_deviceid_node *ds_dev; /* device node ref held across the I/O */ }; struct nfs_mds_commit_info { @@ -1731,6 +1732,7 @@ struct nfs_commit_data { struct nfs_open_context *context; struct pnfs_layout_segment *lseg; struct nfs_client *ds_clp; /* pNFS data server */ + struct nfs4_deviceid_node *ds_dev; /* device node ref held across the commit */ int ds_commit_index; loff_t lwb; const struct rpc_call_ops *mds_ops; From b96fe993143eeeb4e3145d24fd8c53782a9c4bbe Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:12 -0400 Subject: [PATCH 0229/1352] NFSv4/flexfiles: Hold a device node reference for layoutstats encoding ff_layout_mirror_prepare_stats() records a stripe's device under i_lock, but the layoutstats/layoutreturn XDR encode runs later and dereferences the mirror's pinned device node (for the DS netaddr) with no reference of its own. Once a CHANGE notification can re-point the pinned node, that read becomes use-after-free. Take a node reference at prepare time and carry it with the stripe in a per-devinfo nfs4_ff_layoutstat_priv, released in ff_layout_free_layoutstats(). The priv entries live alongside the devinfo array: co-allocated for LAYOUTSTATS, an additional member of nfs4_flexfile_layoutreturn_args for LAYOUTRETURN. Assisted-by: Claude:claude-fable-5 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/flexfilelayout/flexfilelayout.c | 53 +++++++++++++++++--------- fs/nfs/flexfilelayout/flexfilelayout.h | 11 ++++++ 2 files changed, 47 insertions(+), 17 deletions(-) diff --git a/fs/nfs/flexfilelayout/flexfilelayout.c b/fs/nfs/flexfilelayout/flexfilelayout.c index 928e691af9e72d..ecdacaae80c6da 100644 --- a/fs/nfs/flexfilelayout/flexfilelayout.c +++ b/fs/nfs/flexfilelayout/flexfilelayout.c @@ -44,10 +44,11 @@ static void ff_layout_read_record_layoutstats_done(struct rpc_task *task, static int ff_layout_mirror_prepare_stats(struct pnfs_layout_hdr *lo, struct nfs42_layoutstat_devinfo *devinfo, + struct nfs4_ff_layoutstat_priv *priv, int dev_limit, enum nfs4_ff_op_type type); static void ff_layout_encode_ff_layoutupdate(struct xdr_stream *xdr, const struct nfs42_layoutstat_devinfo *devinfo, - struct nfs4_ff_layout_ds_stripe *dss_info); + struct nfs4_ff_layoutstat_priv *priv); static struct pnfs_layout_hdr * ff_layout_alloc_layout_hdr(struct inode *inode, gfp_t gfp_flags) @@ -2725,7 +2726,7 @@ ff_layout_prepare_layoutreturn(struct nfs4_layoutreturn_args *args) spin_lock(&args->inode->i_lock); ff_args->num_dev = ff_layout_mirror_prepare_stats( - &ff_layout->generic_hdr, &ff_args->devinfo[0], + &ff_layout->generic_hdr, &ff_args->devinfo[0], &ff_args->priv[0], ARRAY_SIZE(ff_args->devinfo), NFS4_FF_OP_LAYOUTRETURN); spin_unlock(&args->inode->i_lock); @@ -2900,10 +2901,11 @@ ff_layout_encode_io_latency(struct xdr_stream *xdr, static void ff_layout_encode_ff_layoutupdate(struct xdr_stream *xdr, const struct nfs42_layoutstat_devinfo *devinfo, - struct nfs4_ff_layout_ds_stripe *dss_info) + struct nfs4_ff_layoutstat_priv *priv) { + struct nfs4_ff_layout_ds_stripe *dss_info = priv->dss_info; struct nfs4_pnfs_ds_addr *da; - struct nfs4_pnfs_ds *ds = dss_info->mirror_ds->ds; + struct nfs4_pnfs_ds *ds = priv->mirror_ds->ds; struct nfs_fh *fh = &dss_info->fh_versions[0]; __be32 *p; @@ -2950,10 +2952,10 @@ ff_layout_encode_layoutstats(struct xdr_stream *xdr, const void *args, static void ff_layout_free_layoutstats(struct nfs4_xdr_opaque_data *opaque) { - struct nfs4_ff_layout_ds_stripe *dss_info = opaque->data; - struct nfs4_ff_layout_mirror *mirror = dss_info->mirror; + struct nfs4_ff_layoutstat_priv *priv = opaque->data; - ff_layout_put_mirror(mirror); + nfs4_ff_layout_put_deviceid(priv->mirror_ds); + ff_layout_put_mirror(priv->dss_info->mirror); } static const struct nfs4_xdr_opaque_ops layoutstat_ops = { @@ -2964,12 +2966,13 @@ static const struct nfs4_xdr_opaque_ops layoutstat_ops = { static int ff_layout_mirror_prepare_stats(struct pnfs_layout_hdr *lo, struct nfs42_layoutstat_devinfo *devinfo, + struct nfs4_ff_layoutstat_priv *priv, int dev_limit, enum nfs4_ff_op_type type) { struct nfs4_flexfile_layout *ff_layout = FF_LAYOUT_FROM_HDR(lo); struct nfs4_ff_layout_mirror *mirror; struct nfs4_ff_layout_ds_stripe *dss_info; - struct nfs4_deviceid_node *dev; + struct nfs4_ff_layout_ds *mirror_ds; int i = 0, dss_id; list_for_each_entry(mirror, &ff_layout->mirrors, mirrors) { @@ -2977,7 +2980,8 @@ ff_layout_mirror_prepare_stats(struct pnfs_layout_hdr *lo, dss_info = &mirror->dss[dss_id]; if (i >= dev_limit) break; - if (IS_ERR_OR_NULL(dss_info->mirror_ds)) + mirror_ds = dss_info->mirror_ds; + if (IS_ERR_OR_NULL(mirror_ds)) continue; if (!test_and_clear_bit(NFS4_FF_MIRROR_STAT_AVAIL, &mirror->flags) && @@ -2986,9 +2990,14 @@ ff_layout_mirror_prepare_stats(struct pnfs_layout_hdr *lo, /* mirror refcount put in cleanup_layoutstats */ if (!refcount_inc_not_zero(&mirror->ref)) continue; - dev = &dss_info->mirror_ds->id_node; + /* + * The mirror's pin holds the node while we're under + * i_lock; take a reference for the encode, put in + * ff_layout_free_layoutstats(). + */ + atomic_inc(&mirror_ds->id_node.ref); memcpy(&devinfo->dev_id, - &dev->deviceid, + &mirror_ds->id_node.deviceid, NFS4_DEVICEID4_SIZE); devinfo->offset = 0; devinfo->length = NFS4_MAX_UINT64; @@ -3004,9 +3013,12 @@ ff_layout_mirror_prepare_stats(struct pnfs_layout_hdr *lo, spin_unlock(&mirror->lock); devinfo->layout_type = LAYOUT_FLEX_FILES; devinfo->ld_private.ops = &layoutstat_ops; - devinfo->ld_private.data = &mirror->dss[dss_id]; + priv->dss_info = dss_info; + priv->mirror_ds = mirror_ds; + devinfo->ld_private.data = priv; devinfo++; + priv++; i++; } } @@ -3017,21 +3029,28 @@ static int ff_layout_prepare_layoutstats(struct nfs42_layoutstat_args *args) { struct pnfs_layout_hdr *lo; struct nfs4_flexfile_layout *ff_layout; + struct nfs4_ff_layoutstat_priv *priv; const int dev_count = PNFS_LAYOUTSTATS_MAXDEV; - /* For now, send at most PNFS_LAYOUTSTATS_MAXDEV statistics */ - args->devinfo = kmalloc_objs(*args->devinfo, dev_count, - nfs_io_gfp_mask()); + /* + * For now, send at most PNFS_LAYOUTSTATS_MAXDEV statistics. + * The per-devinfo private entries are co-allocated after the + * devinfo array and freed along with it. + */ + args->devinfo = kmalloc(dev_count * (sizeof(*args->devinfo) + + sizeof(*priv)), + nfs_io_gfp_mask()); if (!args->devinfo) return -ENOMEM; + priv = (struct nfs4_ff_layoutstat_priv *)&args->devinfo[dev_count]; spin_lock(&args->inode->i_lock); lo = NFS_I(args->inode)->layout; if (lo && pnfs_layout_is_valid(lo)) { ff_layout = FF_LAYOUT_FROM_HDR(lo); args->num_dev = ff_layout_mirror_prepare_stats( - &ff_layout->generic_hdr, &args->devinfo[0], dev_count, - NFS4_FF_OP_LAYOUTSTATS); + &ff_layout->generic_hdr, &args->devinfo[0], priv, + dev_count, NFS4_FF_OP_LAYOUTSTATS); } else args->num_dev = 0; spin_unlock(&args->inode->i_lock); diff --git a/fs/nfs/flexfilelayout/flexfilelayout.h b/fs/nfs/flexfilelayout/flexfilelayout.h index 7eb47ce09442d1..8b42a98a2c4c95 100644 --- a/fs/nfs/flexfilelayout/flexfilelayout.h +++ b/fs/nfs/flexfilelayout/flexfilelayout.h @@ -124,9 +124,20 @@ struct nfs4_flexfile_layout { unsigned long flags; }; +/* + * Per-devinfo private data for a layoutstats/layoutreturn encode: the + * stripe the stats describe plus a reference on its device node so the + * node (and its DS addresses) stay valid until the XDR encode runs. + */ +struct nfs4_ff_layoutstat_priv { + struct nfs4_ff_layout_ds_stripe *dss_info; + struct nfs4_ff_layout_ds *mirror_ds; +}; + struct nfs4_flexfile_layoutreturn_args { struct list_head errors; struct nfs42_layoutstat_devinfo devinfo[FF_LAYOUTSTATS_MAXDEV]; + struct nfs4_ff_layoutstat_priv priv[FF_LAYOUTSTATS_MAXDEV]; unsigned int num_errors; unsigned int num_dev; struct page *pages[1]; From a716422fb73d157548429c1ccb2d8675f525cdbd Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:13 -0400 Subject: [PATCH 0230/1352] NFSv4/flexfiles: Make the pinned device node pointer RCU-managed Annotate mirror->dss[dss_id].mirror_ds as __rcu and convert the remaining readers, completing the preparation for re-pointing the pinned node while I/O is in flight: - ff_layout_get_mirror_ds() takes its reference under rcu_read_lock() with atomic_inc_not_zero(), retrying if it races a reset. The resolve path takes the caller's reference before publishing the node with cmpxchg(), so a concurrent reset cannot drop the last reference under the caller: one reference is held for the installed pointer and one for the caller, and the pointer's reference is released elsewhere by xchg + put. - The availability scans hold rcu_read_lock() across the walk; they only test flags on the RCU-freed node. - ff_layout_cancel_io() holds a reference across the cancel and disconnect calls. Both of its callers hold i_lock and nothing on that path sleeps, so the reference is not about blocking: it is what keeps the node alive between the RCU-protected read and the use of mirror_ds->ds. The final put is never reached here, because the mirror's own pin outlives the loop -- which matters, since that put ends in nfs_put_client() and cannot run under a spinlock. - ff_layout_mirror_prepare_stats() reads the pointer with rcu_dereference() under rcu_read_lock(). i_lock, which both callers hold, is what excludes the re-pointing walk added later in this series; it does not exclude the resolve path's cmpxchg(), which runs from I/O submission, so the read cannot claim i_lock as its update- side lock. A non-NULL pointer read there is stable regardless: the resolve path only ever installs over NULL. - ff_layout_free_mirror() tears down the last reference; no concurrency. The pointer is still only ever set once per mirror lifetime, so there is no behavior change; this commit makes the subsequent in-place re-resolve on CB_NOTIFY_DEVICEID CHANGE safe to introduce. Assisted-by: Claude:claude-fable-5 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/flexfilelayout/flexfilelayout.c | 35 ++++--- fs/nfs/flexfilelayout/flexfilelayout.h | 2 +- fs/nfs/flexfilelayout/flexfilelayoutdev.c | 110 ++++++++++++++-------- 3 files changed, 93 insertions(+), 54 deletions(-) diff --git a/fs/nfs/flexfilelayout/flexfilelayout.c b/fs/nfs/flexfilelayout/flexfilelayout.c index ecdacaae80c6da..6edebc0e73a2b0 100644 --- a/fs/nfs/flexfilelayout/flexfilelayout.c +++ b/fs/nfs/flexfilelayout/flexfilelayout.c @@ -313,7 +313,9 @@ static void ff_layout_free_mirror(struct nfs4_ff_layout_mirror *mirror) cred = rcu_access_pointer(mirror->dss[dss_id].rw_cred); put_cred(cred); nfs_close_local_fh(&mirror->dss[dss_id].nfl); - nfs4_ff_layout_put_deviceid(mirror->dss[dss_id].mirror_ds); + /* the last reference to the mirror is gone; no concurrency */ + nfs4_ff_layout_put_deviceid(rcu_dereference_protected( + mirror->dss[dss_id].mirror_ds, 1)); } kvfree(mirror->dss); @@ -2498,22 +2500,29 @@ static void ff_layout_cancel_io(struct pnfs_layout_segment *lseg) for (idx = 0; idx < flseg->mirror_array_cnt; idx++) { mirror = flseg->mirror_array[idx]; for (dss_id = 0; dss_id < mirror->dss_count; dss_id++) { - mirror_ds = mirror->dss[dss_id].mirror_ds; - if (IS_ERR_OR_NULL(mirror_ds)) + rcu_read_lock(); + mirror_ds = rcu_dereference(mirror->dss[dss_id].mirror_ds); + if (IS_ERR_OR_NULL(mirror_ds) || + !atomic_inc_not_zero(&mirror_ds->id_node.ref)) { + rcu_read_unlock(); continue; - ds = mirror->dss[dss_id].mirror_ds->ds; + } + rcu_read_unlock(); + ds = mirror_ds->ds; if (!ds) - continue; + goto next; ds_clp = ds->ds_clp; if (!ds_clp) - continue; + goto next; clnt = ds_clp->cl_rpcclient; if (!clnt) - continue; + goto next; if (!rpc_cancel_tasks(clnt, -ECANCELED, ff_layout_match_io, lseg)) - continue; + goto next; rpc_clnt_disconnect(clnt); +next: + nfs4_ff_layout_put_deviceid(mirror_ds); } } } @@ -2975,12 +2984,13 @@ ff_layout_mirror_prepare_stats(struct pnfs_layout_hdr *lo, struct nfs4_ff_layout_ds *mirror_ds; int i = 0, dss_id; + rcu_read_lock(); list_for_each_entry(mirror, &ff_layout->mirrors, mirrors) { for (dss_id = 0; dss_id < mirror->dss_count; ++dss_id) { dss_info = &mirror->dss[dss_id]; if (i >= dev_limit) break; - mirror_ds = dss_info->mirror_ds; + mirror_ds = rcu_dereference(dss_info->mirror_ds); if (IS_ERR_OR_NULL(mirror_ds)) continue; if (!test_and_clear_bit(NFS4_FF_MIRROR_STAT_AVAIL, @@ -2990,10 +3000,8 @@ ff_layout_mirror_prepare_stats(struct pnfs_layout_hdr *lo, /* mirror refcount put in cleanup_layoutstats */ if (!refcount_inc_not_zero(&mirror->ref)) continue; - /* - * The mirror's pin holds the node while we're under - * i_lock; take a reference for the encode, put in - * ff_layout_free_layoutstats(). + /* The pin holds a reference; it is exchanged out only + * under i_lock. Put in ff_layout_free_layoutstats(). */ atomic_inc(&mirror_ds->id_node.ref); memcpy(&devinfo->dev_id, @@ -3022,6 +3030,7 @@ ff_layout_mirror_prepare_stats(struct pnfs_layout_hdr *lo, i++; } } + rcu_read_unlock(); return i; } diff --git a/fs/nfs/flexfilelayout/flexfilelayout.h b/fs/nfs/flexfilelayout/flexfilelayout.h index 8b42a98a2c4c95..72b11034851a63 100644 --- a/fs/nfs/flexfilelayout/flexfilelayout.h +++ b/fs/nfs/flexfilelayout/flexfilelayout.h @@ -79,7 +79,7 @@ struct nfs4_ff_layout_ds_stripe { struct nfs4_ff_layout_mirror *mirror; struct nfs4_deviceid devid; u32 efficiency; - struct nfs4_ff_layout_ds *mirror_ds; + struct nfs4_ff_layout_ds __rcu *mirror_ds; u32 fh_versions_cnt; struct nfs_fh *fh_versions; nfs4_stateid stateid; diff --git a/fs/nfs/flexfilelayout/flexfilelayoutdev.c b/fs/nfs/flexfilelayout/flexfilelayoutdev.c index ebcfba44787989..5cb09e5e2138f2 100644 --- a/fs/nfs/flexfilelayout/flexfilelayoutdev.c +++ b/fs/nfs/flexfilelayout/flexfilelayoutdev.c @@ -321,35 +321,54 @@ ff_layout_get_mirror_ds(struct pnfs_layout_hdr *lo, struct nfs4_ff_layout_mirror *mirror, u32 dss_id) { - struct nfs4_ff_layout_ds *mirror_ds; + struct nfs4_ff_layout_ds *mirror_ds, *old; + struct nfs4_deviceid_node *node; if (mirror == NULL) return ERR_PTR(-ENODEV); - mirror_ds = mirror->dss[dss_id].mirror_ds; - if (mirror_ds == NULL) { - struct nfs4_deviceid_node *node; - +retry: + rcu_read_lock(); + mirror_ds = rcu_dereference(mirror->dss[dss_id].mirror_ds); + if (mirror_ds && !IS_ERR(mirror_ds) && + atomic_inc_not_zero(&mirror_ds->id_node.ref)) { + rcu_read_unlock(); + return mirror_ds; + } + rcu_read_unlock(); + if (IS_ERR(mirror_ds)) + return mirror_ds; + if (mirror_ds != NULL) + /* raced with a reset; the field is being re-pointed */ + goto retry; + + node = nfs4_find_get_deviceid(NFS_SERVER(lo->plh_inode), + &mirror->dss[dss_id].devid, lo->plh_lc_cred, + GFP_KERNEL); + if (node) { + mirror_ds = FF_LAYOUT_MIRROR_DS(node); + /* + * Take the caller's reference before the pointer becomes + * visible below, so a concurrent reset of the installed + * pointer cannot drop the last reference under us. + */ + atomic_inc(&node->ref); + } else { mirror_ds = ERR_PTR(-ENODEV); - node = nfs4_find_get_deviceid(NFS_SERVER(lo->plh_inode), - &mirror->dss[dss_id].devid, lo->plh_lc_cred, - GFP_KERNEL); - if (node) - mirror_ds = FF_LAYOUT_MIRROR_DS(node); - - /* check for race with another call to this function */ - if (cmpxchg(&mirror->dss[dss_id].mirror_ds, NULL, mirror_ds) && - mirror_ds != ERR_PTR(-ENODEV)) - nfs4_put_deviceid_node(node); - - mirror_ds = mirror->dss[dss_id].mirror_ds; } - if (IS_ERR(mirror_ds)) + /* check for race with another call to this function */ + old = unrcu_pointer(cmpxchg(&mirror->dss[dss_id].mirror_ds, + NULL, RCU_INITIALIZER(mirror_ds))); + if (old == NULL) return mirror_ds; - if (!atomic_inc_not_zero(&mirror_ds->id_node.ref)) - return ERR_PTR(-ENODEV); - return mirror_ds; + + /* lost the race; use the winner's node instead */ + if (node) { + nfs4_put_deviceid_node(node); + nfs4_put_deviceid_node(node); + } + goto retry; } /** @@ -573,49 +592,60 @@ unsigned int ff_layout_fetch_ds_ioerr(struct pnfs_layout_hdr *lo, static bool ff_read_layout_has_available_ds(struct pnfs_layout_segment *lseg) { struct nfs4_ff_layout_mirror *mirror; - struct nfs4_deviceid_node *devid; + struct nfs4_ff_layout_ds *mirror_ds; + bool ret = false; u32 idx, dss_id; + rcu_read_lock(); for (idx = 0; idx < FF_LAYOUT_MIRROR_COUNT(lseg); idx++) { mirror = FF_LAYOUT_COMP(lseg, idx); if (!mirror) continue; for (dss_id = 0; dss_id < mirror->dss_count; dss_id++) { - if (!mirror->dss[dss_id].mirror_ds) - return true; - if (IS_ERR(mirror->dss[dss_id].mirror_ds)) + mirror_ds = rcu_dereference(mirror->dss[dss_id].mirror_ds); + if (!mirror_ds) { + ret = true; + goto out; + } + if (IS_ERR(mirror_ds)) continue; - devid = &mirror->dss[dss_id].mirror_ds->id_node; - if (!nfs4_test_deviceid_unavailable(devid)) - return true; + if (!nfs4_test_deviceid_unavailable(&mirror_ds->id_node)) { + ret = true; + goto out; + } } } - - return false; +out: + rcu_read_unlock(); + return ret; } static bool ff_rw_layout_has_available_ds(struct pnfs_layout_segment *lseg) { struct nfs4_ff_layout_mirror *mirror; - struct nfs4_deviceid_node *devid; + struct nfs4_ff_layout_ds *mirror_ds; + bool ret = false; u32 idx, dss_id; + rcu_read_lock(); for (idx = 0; idx < FF_LAYOUT_MIRROR_COUNT(lseg); idx++) { mirror = FF_LAYOUT_COMP(lseg, idx); if (!mirror) - return false; + goto out; for (dss_id = 0; dss_id < mirror->dss_count; dss_id++) { - if (IS_ERR(mirror->dss[dss_id].mirror_ds)) - return false; - if (!mirror->dss[dss_id].mirror_ds) + mirror_ds = rcu_dereference(mirror->dss[dss_id].mirror_ds); + if (IS_ERR(mirror_ds)) + goto out; + if (!mirror_ds) continue; - devid = &mirror->dss[dss_id].mirror_ds->id_node; - if (nfs4_test_deviceid_unavailable(devid)) - return false; + if (nfs4_test_deviceid_unavailable(&mirror_ds->id_node)) + goto out; } } - - return FF_LAYOUT_MIRROR_COUNT(lseg) != 0; + ret = FF_LAYOUT_MIRROR_COUNT(lseg) != 0; +out: + rcu_read_unlock(); + return ret; } static bool ff_layout_has_available_ds(struct pnfs_layout_segment *lseg) From 7de6176926280f28dd9aab29d709aa1d495c69be Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:14 -0400 Subject: [PATCH 0231/1352] pNFS: Add a reresolve_deviceid layout driver hook RFC 8881 Section 12.2.10 lets a server change a deviceid's mapping under live layouts by sending CB_NOTIFY_DEVICEID CHANGE instead of recalling the layouts, but the client's only response today is to unhash the cached device, which never reaches references pinned inside a layout driver's segments. Add a reresolve_deviceid hook to pnfs_layoutdriver_type and a generic driver, pnfs_layout_reresolve_deviceid_byclid(), that walks every layout on every server of the client using that driver and invokes the hook under the layout inode's i_lock. Because the final put of a device node can sleep (it may tear down the DS nfs_client), the hook must not drop references itself: for each node it un-pins it allocates an nfs4_deviceid_put entry and queues it on a list, and the generic driver puts the node and frees the entry once all locks are dropped. A deviceid node is a shared, refcounted object, so one re-resolve pass can unpin the same node more than once (multiple stripes, or multiple layouts over a common data server); a per-reference entry expresses that, where a single list_head embedded in the node could not. The walk is deliberately not gated on pnfs_layout_is_valid(). A header with NFS_LAYOUT_INVALID_STID set can still carry lsegs whose mirrors pin the stale node -- pnfs_mark_layout_stateid_invalid() sets the bit and reports whether segments were left behind -- and by the time the walk runs the cached device has already been unhashed, so nothing would re-resolve that pin later. Worse, a subsequent LAYOUTGET on the same header can pick the surviving mirror back up (the driver dedups mirrors by deviceid and filehandle) and carry the old mapping into a fresh layout. Un-pinning a device node does not touch the layout stateid, so the hook has no need of a valid one; it walks only the driver's mirror list, which i_lock protects, and the header cannot be freed under the walk because the driver frees it with kfree_rcu(). Note this differs from the reference-collection walker added later in the series, which does take a layout header reference and therefore does depend on the validity check for its refcount argument. No driver implements the hook yet, and nothing calls the walker: no behavior change. Assisted-by: Claude:claude-fable-5 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/pnfs.c | 62 +++++++++++++++++++++++++++++++++++++++++++++++++++ fs/nfs/pnfs.h | 25 +++++++++++++++++++++ 2 files changed, 87 insertions(+) diff --git a/fs/nfs/pnfs.c b/fs/nfs/pnfs.c index a74ffdda5bcd5c..69ea9abe28c925 100644 --- a/fs/nfs/pnfs.c +++ b/fs/nfs/pnfs.c @@ -2917,6 +2917,68 @@ pnfs_layout_return_unused_byclid(struct nfs_client *clp, &range); } +struct pnfs_reresolve_deviceid_args { + const struct pnfs_layoutdriver_type *ld; + const struct nfs4_deviceid *devid; + bool immediate; + struct list_head put_list; +}; + +static int pnfs_layout_reresolve_deviceid_byserver(struct nfs_server *server, + void *data) +{ + struct pnfs_reresolve_deviceid_args *args = data; + struct pnfs_layout_hdr *lo; + struct inode *inode; + + if (server->pnfs_curr_ld != args->ld) + return 0; + + rcu_read_lock(); + list_for_each_entry_rcu(lo, &server->layouts, plh_layouts) { + inode = lo->plh_inode; + if (!inode) + continue; + spin_lock(&inode->i_lock); + args->ld->reresolve_deviceid(lo, args->devid, args->immediate, + &args->put_list); + spin_unlock(&inode->i_lock); + } + rcu_read_unlock(); + return 0; +} + +/* + * Invoke @ld's reresolve_deviceid hook for @devid on every layout of @clp's + * servers, then drain the put_list once the locks are dropped. + */ +void +pnfs_layout_reresolve_deviceid_byclid(struct nfs_client *clp, + const struct pnfs_layoutdriver_type *ld, + const struct nfs4_deviceid *devid, + bool immediate) +{ + struct pnfs_reresolve_deviceid_args args = { + .ld = ld, + .devid = devid, + .immediate = immediate, + .put_list = LIST_HEAD_INIT(args.put_list), + }; + struct nfs4_deviceid_put *put, *tmp; + + if (!ld->reresolve_deviceid) + return; + + nfs_client_for_each_server(clp, + pnfs_layout_reresolve_deviceid_byserver, &args); + + list_for_each_entry_safe(put, tmp, &args.put_list, node) { + list_del(&put->node); + nfs4_put_deviceid_node(put->dev); + kfree(put); + } +} + /* Check if we have we have a valid layout but if there isn't an intersection * between the request and the pgio->pg_lseg, put this pgio->pg_lseg away. */ diff --git a/fs/nfs/pnfs.h b/fs/nfs/pnfs.h index a2c70573dda529..bb1ca7ba0221f3 100644 --- a/fs/nfs/pnfs.h +++ b/fs/nfs/pnfs.h @@ -171,6 +171,19 @@ struct pnfs_layoutdriver_type { struct nfs4_deviceid_node * (*alloc_deviceid_node) (struct nfs_server *server, struct pnfs_device *pdev, gfp_t gfp_flags); + /* + * Re-resolve @lo's references to the changed deviceid @id. Called + * under @lo's inode i_lock inside an RCU read-side critical section: + * must not sleep, allocations are GFP_ATOMIC. Rather than put the + * references it gives up (the final put can sleep), the hook + * allocates an nfs4_deviceid_put per reference and queues it on + * @put_list for the caller to put and free. On allocation failure + * it must leave the reference in place. + */ + void (*reresolve_deviceid)(struct pnfs_layout_hdr *lo, + const struct nfs4_deviceid *id, + bool immediate, + struct list_head *put_list); int (*prepare_layoutreturn) (struct nfs4_layoutreturn_args *); @@ -354,6 +367,10 @@ void pnfs_error_mark_layout_for_return(struct inode *inode, struct pnfs_layout_segment *lseg); void pnfs_layout_return_unused_byclid(struct nfs_client *clp, enum pnfs_iomode iomode); +void pnfs_layout_reresolve_deviceid_byclid(struct nfs_client *clp, + const struct pnfs_layoutdriver_type *ld, + const struct nfs4_deviceid *devid, + bool immediate); int pnfs_layout_handle_reboot(struct nfs_client *clp); /* nfs4_deviceid_flags */ @@ -376,6 +393,14 @@ struct nfs4_deviceid_node { atomic_t ref; }; +/* One reference given up by reresolve_deviceid; nodes are shared, so a + * single pass can unpin the same node more than once. + */ +struct nfs4_deviceid_put { + struct list_head node; + struct nfs4_deviceid_node *dev; +}; + struct nfs4_deviceid_node * nfs4_find_get_deviceid(struct nfs_server *server, const struct nfs4_deviceid *id, const struct cred *cred, From c3325086f1d737eedded43f846da6a6d14fb9e34 Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:15 -0400 Subject: [PATCH 0232/1352] NFSv4/flexfiles: Implement in-place device re-resolve on CHANGE Implement the reresolve_deviceid hook: walk the layout's mirrors and, for every stripe whose raw deviceid matches the changed one, exchange the pinned device node out for NULL. In-flight I/O completes on the old node through its own reference; the next I/O to the stripe re-resolves via ff_layout_get_mirror_ds() and picks up the server's new mapping with a fresh GETDEVICEINFO. The un-pinned references are handed back on the walker's list to be put outside the locks. Matching uses the raw deviceid decoded from the layout (dss[].devid), so a stripe that was never resolved, or already exchanged, is left alone. A stripe whose resolution previously failed holds an ERR_PTR sentinel rather than a node; that is cleared too, so the stripe retries GETDEVICEINFO on its next I/O. Nothing dispatches CHANGE to the walker yet: no behavior change. Assisted-by: Claude:claude-fable-5 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/flexfilelayout/flexfilelayout.c | 40 ++++++++++++++++++++++++++ 1 file changed, 40 insertions(+) diff --git a/fs/nfs/flexfilelayout/flexfilelayout.c b/fs/nfs/flexfilelayout/flexfilelayout.c index 6edebc0e73a2b0..55057f8f9a7735 100644 --- a/fs/nfs/flexfilelayout/flexfilelayout.c +++ b/fs/nfs/flexfilelayout/flexfilelayout.c @@ -2527,6 +2527,45 @@ static void ff_layout_cancel_io(struct pnfs_layout_segment *lseg) } } +/* + * Un-pin every stripe node resolved from @id: in-flight I/O drains on the + * old node through its own reference, the next I/O re-resolves. + */ +static void ff_layout_reresolve_deviceid(struct pnfs_layout_hdr *lo, + const struct nfs4_deviceid *id, + bool immediate, + struct list_head *head) +{ + struct nfs4_flexfile_layout *flo = FF_LAYOUT_FROM_HDR(lo); + struct nfs4_ff_layout_mirror *mirror; + struct nfs4_ff_layout_ds *old; + struct nfs4_deviceid_put *put; + u32 dss_id; + + list_for_each_entry(mirror, &flo->mirrors, mirrors) { + for (dss_id = 0; dss_id < mirror->dss_count; dss_id++) { + if (memcmp(&mirror->dss[dss_id].devid, id, + sizeof(*id)) != 0) + continue; + /* Allocate before un-pinning: on failure the reference + * stays put rather than being dropped here, where the + * final put may not sleep. + */ + put = kzalloc_obj(*put, GFP_ATOMIC); + if (!put) + continue; + old = unrcu_pointer( + xchg(&mirror->dss[dss_id].mirror_ds, NULL)); + if (IS_ERR_OR_NULL(old)) { + kfree(put); + continue; + } + put->dev = &old->id_node; + list_add(&put->node, head); + } + } +} + static struct pnfs_ds_commit_info * ff_layout_get_ds_info(struct inode *inode) { @@ -3108,6 +3147,7 @@ static struct pnfs_layoutdriver_type flexfilelayout_type = { .pg_write_ops = &ff_layout_pg_write_ops, .get_ds_info = ff_layout_get_ds_info, .free_deviceid_node = ff_layout_free_deviceid_node, + .reresolve_deviceid = ff_layout_reresolve_deviceid, .read_pagelist = ff_layout_read_pagelist, .write_pagelist = ff_layout_write_pagelist, .alloc_deviceid_node = ff_layout_alloc_deviceid_node, From bc5f1b86d014647b47d9fa7c4d34ff9eb2f10754 Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:16 -0400 Subject: [PATCH 0233/1352] NFSv4: Dispatch CB_NOTIFY_DEVICEID CHANGE to an in-place refresh nfs4_callback_devicenotify() treated CHANGE identically to DELETE: both only unhashed the cached device, so references pinned under live layouts kept sending I/O to the old mapping until the layouts were freed. Per RFC 8881 Section 12.2.10, CHANGE exists precisely so a server can change a mapping without recalling the layouts. For CHANGE, after unhashing the stale cache entry (so re-resolution cannot re-pin it), invoke the layout driver's re-resolve walker. The walker does not install the new mapping itself: it un-pins the stale node so the next I/O to that stripe re-resolves and fetches the new one, while I/O already in flight completes on the old node through its own reference. DELETE is unchanged, and drivers without a reresolve_deviceid hook see no change. Re-resolution is best effort. A lookup that hit the cache just before the unhash can still install that node after the walk has passed the stripe, and the walk skips nothing else; such a stripe keeps the old mapping until the next notification. Nothing is left inconsistent by that -- the old mapping is a valid address the server published -- so the race is tolerated rather than serialised against. Assisted-by: Claude:claude-fable-5 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/callback_proc.c | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/fs/nfs/callback_proc.c b/fs/nfs/callback_proc.c index 3fb10c8e427187..01372d8548e1ea 100644 --- a/fs/nfs/callback_proc.c +++ b/fs/nfs/callback_proc.c @@ -391,7 +391,16 @@ __be32 nfs4_callback_devicenotify(void *argp, void *resp, if (!ld) continue; } + /* + * Unhash the cached device first so re-resolution cannot + * re-pin the stale node, then re-point any references + * pinned under live layouts (RFC 8881 Section 12.2.10). + */ nfs4_delete_deviceid(ld, cps->clp, &dev->cbd_dev_id); + if (dev->cbd_notify_type == NOTIFY_DEVICEID4_CHANGE) + pnfs_layout_reresolve_deviceid_byclid(cps->clp, ld, + &dev->cbd_dev_id, + dev->cbd_immediate); } pnfs_put_layoutdriver(ld); out: From 81493a431059084c8e17c55adda77a6323074021 Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:17 -0400 Subject: [PATCH 0234/1352] NFSv4/flexfiles: Honor ndc_immediate on CB_NOTIFY_DEVICEID CHANGE When a CHANGE notification carries ndc_immediate, RFC 8881 Section 20.12 says the change is enforced immediately and the client might not be able to complete pending I/O. In addition to un-pinning the stripe's device node, mark the old node unavailable. Marking does not recall the references already handed out. A write whose DS connection is already up keeps using the old node until that I/O errors: nfs4_ff_layout_prepare_ds() returns early on a live ds_clp, and the unavailable flag is only consulted when a connection is being established. What the mark does change is that a read skips the node while another mirror is usable, and that an IOMODE_RW segment still pinning it stops counting as fully available -- so I/O the server rejects falls back to the MDS rather than being retried against a mapping the server has already withdrawn. The walk can also exchange out a node that already carries the new mapping: a re-resolve that completed between the unhash and the walk reaching the stripe, raced by this walk or by the walk of a later CHANGE for the same deviceid. Ripping such a node out is harmless (it is still hashed, so the next I/O re-pins it from the cache), but it must not be marked unavailable. A superseded node is distinguishable by hashed-ness: it was unhashed before its walk began and is never re-inserted, while a fresh node is inserted before it is installed. Only mark nodes that are no longer hashed. That test is hlist_unhashed_lockless(): the hook runs under the layout inode's i_lock and rcu_read_lock(), but not under nfs4_deviceid_lock, which is what serializes the writers of node.pprev -- and __hlist_del() stores a neighbour's pprev with WRITE_ONCE(), so removing any other entry in the same bucket can write the field this test reads. Without ndc_immediate, pending I/O drains on the old mapping and only new I/O re-resolves, as before. Assisted-by: Claude:claude-fable-5 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/flexfilelayout/flexfilelayout.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/fs/nfs/flexfilelayout/flexfilelayout.c b/fs/nfs/flexfilelayout/flexfilelayout.c index 55057f8f9a7735..a5f4a1df298ea1 100644 --- a/fs/nfs/flexfilelayout/flexfilelayout.c +++ b/fs/nfs/flexfilelayout/flexfilelayout.c @@ -2560,6 +2560,13 @@ static void ff_layout_reresolve_deviceid(struct pnfs_layout_hdr *lo, kfree(put); continue; } + /* A node still hashed was fetched after the unhash + * and carries the new mapping; mark only the + * superseded ones. + */ + if (immediate && + hlist_unhashed_lockless(&old->id_node.node)) + nfs4_mark_deviceid_unavailable(&old->id_node); put->dev = &old->id_node; list_add(&put->node, head); } From b7af4bf5690761fcb3eacd92d99bee798f81386d Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:18 -0400 Subject: [PATCH 0235/1352] pNFS: Discard a GETDEVICEINFO reply that raced a CHANGE notification RFC 8881 Section 18.40.4: a GETDEVICEINFO reply in flight while the server changes the device mapping may carry the pre-change mapping; if it is inserted into the cache after the CHANGE notification unhashed the stale entry, the client re-caches stale data. Track a change epoch, bumped when a CHANGE notification is processed before the stale entry is unhashed. nfs4_find_get_deviceid() snapshots the epoch before issuing GETDEVICEINFO and, serialized against the unhash by nfs4_deviceid_lock at insert time, discards the reply and refetches if the epoch moved. A stale insert that instead precedes the unhash is removed by the unhash itself, so the cache does not retain the pre-change entry either way; a reference already handed to a caller in that ordering is dropped by the re-resolve walk instead. The refetch is bounded. The epoch is bumped once per CHANGE entry -- that is, at a rate the server chooses -- so an unbounded retry would let a server drive GETDEVICEINFO traffic without limit, and each discarded node can carry a DS client teardown and reconnect with it. After NFS4_DEVICEID_FETCH_RETRIES attempts the reply is accepted. That is safe because discarding is an optimisation rather than a correctness requirement: it avoids caching a mapping already known to be superseded, but before this patch the client cached whatever the reply carried, so the bounded case is no worse than the previous behaviour and a mapping that really is stale is corrected by the notification that follows. The epoch lives on the nfs_client, so a CHANGE delivered on one server's callback channel does not force an unrelated server's in-flight lookup to discard its reply and refetch. Mounts that share an nfs_client do share the counter; the deviceid cache is keyed per client ID, so that is the granularity the race is defined at. Assisted-by: Claude:claude-fable-5 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/callback_proc.c | 4 ++++ fs/nfs/pnfs.h | 1 + fs/nfs/pnfs_dev.c | 25 +++++++++++++++++++++++++ include/linux/nfs_fs_sb.h | 2 ++ 4 files changed, 32 insertions(+) diff --git a/fs/nfs/callback_proc.c b/fs/nfs/callback_proc.c index 01372d8548e1ea..ea8c558b07b6bc 100644 --- a/fs/nfs/callback_proc.c +++ b/fs/nfs/callback_proc.c @@ -395,7 +395,11 @@ __be32 nfs4_callback_devicenotify(void *argp, void *resp, * Unhash the cached device first so re-resolution cannot * re-pin the stale node, then re-point any references * pinned under live layouts (RFC 8881 Section 12.2.10). + * The epoch bump lets an in-flight GETDEVICEINFO detect + * that its reply may predate the change. */ + if (dev->cbd_notify_type == NOTIFY_DEVICEID4_CHANGE) + nfs4_deviceid_bump_change_epoch(cps->clp); nfs4_delete_deviceid(ld, cps->clp, &dev->cbd_dev_id); if (dev->cbd_notify_type == NOTIFY_DEVICEID4_CHANGE) pnfs_layout_reresolve_deviceid_byclid(cps->clp, ld, diff --git a/fs/nfs/pnfs.h b/fs/nfs/pnfs.h index bb1ca7ba0221f3..9a5b8070f595c9 100644 --- a/fs/nfs/pnfs.h +++ b/fs/nfs/pnfs.h @@ -406,6 +406,7 @@ nfs4_find_get_deviceid(struct nfs_server *server, const struct nfs4_deviceid *id, const struct cred *cred, gfp_t gfp_mask); void nfs4_delete_deviceid(const struct pnfs_layoutdriver_type *, const struct nfs_client *, const struct nfs4_deviceid *); +void nfs4_deviceid_bump_change_epoch(struct nfs_client *clp); void nfs4_init_deviceid_node(struct nfs4_deviceid_node *, struct nfs_server *, const struct nfs4_deviceid *); bool nfs4_put_deviceid_node(struct nfs4_deviceid_node *); diff --git a/fs/nfs/pnfs_dev.c b/fs/nfs/pnfs_dev.c index 274abdd6d5f3cf..a3b28409539ae1 100644 --- a/fs/nfs/pnfs_dev.c +++ b/fs/nfs/pnfs_dev.c @@ -181,6 +181,21 @@ __nfs4_find_get_deviceid(struct nfs_server *server, return d; } +/* + * Bumped before the stale entry is unhashed, so an insert serialised + * after the unhash by nfs4_deviceid_lock observes the new epoch. + */ +void +nfs4_deviceid_bump_change_epoch(struct nfs_client *clp) +{ + atomic_inc(&clp->cl_deviceid_change_epoch); +} + +/* Discarding a raced reply is an optimisation, not a correctness + * requirement, and the epoch moves at the server's rate: bound it. + */ +#define NFS4_DEVICEID_FETCH_RETRIES 3 + struct nfs4_deviceid_node * nfs4_find_get_deviceid(struct nfs_server *server, const struct nfs4_deviceid *id, const struct cred *cred, @@ -188,11 +203,14 @@ nfs4_find_get_deviceid(struct nfs_server *server, { long hash = nfs4_deviceid_hash(id); struct nfs4_deviceid_node *d, *new; + int epoch, tries = 0; +retry: d = __nfs4_find_get_deviceid(server, id, hash); if (d) goto found; + epoch = atomic_read(&server->nfs_client->cl_deviceid_change_epoch); new = nfs4_get_device_info(server, id, cred, gfp_mask); if (!new) { trace_nfs4_find_deviceid(server, id, -ENOENT); @@ -200,6 +218,13 @@ nfs4_find_get_deviceid(struct nfs_server *server, } spin_lock(&nfs4_deviceid_lock); + if (atomic_read(&server->nfs_client->cl_deviceid_change_epoch) != epoch && + ++tries <= NFS4_DEVICEID_FETCH_RETRIES) { + /* a mapping changed while we fetched; ours may be stale */ + spin_unlock(&nfs4_deviceid_lock); + server->pnfs_curr_ld->free_deviceid_node(new); + goto retry; + } d = __nfs4_find_get_deviceid(server, id, hash); if (d) { spin_unlock(&nfs4_deviceid_lock); diff --git a/include/linux/nfs_fs_sb.h b/include/linux/nfs_fs_sb.h index 3e6bae7e52218f..482daf1fa4b0cc 100644 --- a/include/linux/nfs_fs_sb.h +++ b/include/linux/nfs_fs_sb.h @@ -74,6 +74,8 @@ struct nfs_client { u64 cl_clientid; /* constant */ nfs4_verifier cl_confirm; /* Clientid verifier */ unsigned long cl_state; + /* bumped on each CB_NOTIFY_DEVICEID CHANGE for this client */ + atomic_t cl_deviceid_change_epoch; spinlock_t cl_lock; From e1156867c93501baa3af9ad37f4671ed50054d4e Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:19 -0400 Subject: [PATCH 0236/1352] pNFS: Add deviceid reference query and collection walkers The CB_NOTIFY_DEVICEID DELETE race recovery (RFC 8881 Section 18.40.4) needs to ask whether any live layout still references a deviceID, and to enumerate those layouts for TEST_STATEID. Add a layout_references_deviceid hook (sibling of reresolve_deviceid; the flexfiles implementation memcmps each mirror stripe's decoded devid, valid independent of the pinned device node) and two walkers over the byserver pattern: - pnfs_layout_deviceid_referenced_byclid(): boolean existence query, early-stopping, entirely under i_lock. - pnfs_layout_collect_deviceid_refs(): collects each matching layout with the hdr pinned, the inode grabbed with its superblock active (a pinned hdr does not hold its inode -- same discipline as the bulk-destroy walker), and the layout stateid and cred snapshotted under i_lock, so the caller can issue sleeping RPCs against the collection. The collection walker pins each matching header with a plain pnfs_get_layout_hdr(), which cannot resurrect a dying header because the walker holds i_lock and has checked NFS_I()->layout == lo. pnfs_put_layout_hdr() drops the last reference under i_lock -- refcount_dec_and_lock() only decrements one to zero once it holds the lock -- and clears NFS_I()->layout in that same critical section, before it unlocks and frees. So a header still installed on its inode cannot have reached a zero refcount. That argument deliberately does not rest on NFS_LAYOUT_INVALID_STID. A header can reach its final put while still valid: a full LAYOUTRETURN ends in pnfs_layoutreturn_free_lsegs(), which resets the layout stateid rather than invalidating it, and that is the ordinary end of life for a return-on-close layout. The validity check the walkers do apply is a policy filter, not a lifetime guarantee. Because pnfs_put_layout_hdr() can send a layoutreturn and sleep, a header pinned for a layout whose inode can no longer be grabbed is put after the RCU read-side critical section, not within it. Both ways the walk can end early report it. A GFP_ATOMIC allocation failure aborts with -ENOMEM, and an inode that can no longer be grabbed -- igrab() fails from I_FREEING on, while the layout may still be valid and still name the deviceID -- aborts with -EAGAIN. Either way the caller is told the collection is partial instead of receiving a short list it would read as "no references". No callers yet; no behavior change. Assisted-by: Claude:claude-fable-5 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/flexfilelayout/flexfilelayout.c | 17 +++ fs/nfs/pnfs.c | 164 +++++++++++++++++++++++++ fs/nfs/pnfs.h | 29 +++++ 3 files changed, 210 insertions(+) diff --git a/fs/nfs/flexfilelayout/flexfilelayout.c b/fs/nfs/flexfilelayout/flexfilelayout.c index a5f4a1df298ea1..2ea63fda0035b7 100644 --- a/fs/nfs/flexfilelayout/flexfilelayout.c +++ b/fs/nfs/flexfilelayout/flexfilelayout.c @@ -2527,6 +2527,22 @@ static void ff_layout_cancel_io(struct pnfs_layout_segment *lseg) } } +/* Called under @lo's inode i_lock. */ +static bool ff_layout_references_deviceid(struct pnfs_layout_hdr *lo, + const struct nfs4_deviceid *id) +{ + struct nfs4_flexfile_layout *flo = FF_LAYOUT_FROM_HDR(lo); + struct nfs4_ff_layout_mirror *mirror; + u32 dss_id; + + list_for_each_entry(mirror, &flo->mirrors, mirrors) + for (dss_id = 0; dss_id < mirror->dss_count; dss_id++) + if (memcmp(&mirror->dss[dss_id].devid, id, + sizeof(*id)) == 0) + return true; + return false; +} + /* * Un-pin every stripe node resolved from @id: in-flight I/O drains on the * old node through its own reference, the next I/O re-resolves. @@ -3155,6 +3171,7 @@ static struct pnfs_layoutdriver_type flexfilelayout_type = { .get_ds_info = ff_layout_get_ds_info, .free_deviceid_node = ff_layout_free_deviceid_node, .reresolve_deviceid = ff_layout_reresolve_deviceid, + .layout_references_deviceid = ff_layout_references_deviceid, .read_pagelist = ff_layout_read_pagelist, .write_pagelist = ff_layout_write_pagelist, .alloc_deviceid_node = ff_layout_alloc_deviceid_node, diff --git a/fs/nfs/pnfs.c b/fs/nfs/pnfs.c index 69ea9abe28c925..fbe4b58b5b7227 100644 --- a/fs/nfs/pnfs.c +++ b/fs/nfs/pnfs.c @@ -2979,6 +2979,170 @@ pnfs_layout_reresolve_deviceid_byclid(struct nfs_client *clp, } } +struct pnfs_deviceid_ref_args { + const struct pnfs_layoutdriver_type *ld; + const struct nfs4_deviceid *devid; + struct list_head *result; + bool found; +}; + +static int pnfs_layout_deviceid_referenced_byserver( + struct nfs_server *server, void *data) +{ + struct pnfs_deviceid_ref_args *args = data; + struct pnfs_layout_hdr *lo; + struct inode *inode; + + if (server->pnfs_curr_ld != args->ld) + return 0; + + rcu_read_lock(); + list_for_each_entry_rcu(lo, &server->layouts, plh_layouts) { + inode = lo->plh_inode; + if (!inode) + continue; + spin_lock(&inode->i_lock); + if (NFS_I(inode)->layout == lo && pnfs_layout_is_valid(lo) && + args->ld->layout_references_deviceid(lo, args->devid)) + args->found = true; + spin_unlock(&inode->i_lock); + if (args->found) + break; + } + rcu_read_unlock(); + return args->found; +} + +/* + * pnfs_layout_deviceid_referenced_byclid - does any live layout of + * @clp's servers using @ld still reference deviceid @devid? + */ +bool +pnfs_layout_deviceid_referenced_byclid(struct nfs_client *clp, + const struct pnfs_layoutdriver_type *ld, + const struct nfs4_deviceid *devid) +{ + struct pnfs_deviceid_ref_args args = { + .ld = ld, + .devid = devid, + }; + + if (!ld->layout_references_deviceid) + return false; + + nfs_client_for_each_server(clp, + pnfs_layout_deviceid_referenced_byserver, &args); + return args.found; +} + +static int pnfs_layout_collect_deviceid_refs_byserver( + struct nfs_server *server, void *data) +{ + struct pnfs_deviceid_ref_args *args = data; + struct nfs4_deviceid_ref *ref, *tmp; + struct pnfs_layout_hdr *lo; + struct inode *inode; + LIST_HEAD(putme); + bool matched; + int ret = 0; + + if (server->pnfs_curr_ld != args->ld) + return 0; + + rcu_read_lock(); + list_for_each_entry_rcu(lo, &server->layouts, plh_layouts) { + inode = lo->plh_inode; + if (!inode) + continue; + + spin_lock(&inode->i_lock); + matched = NFS_I(inode)->layout == lo && + pnfs_layout_is_valid(lo) && + args->ld->layout_references_deviceid(lo, args->devid); + if (!matched) { + spin_unlock(&inode->i_lock); + continue; + } + ref = kzalloc_obj(*ref, GFP_ATOMIC); + if (!ref) { + spin_unlock(&inode->i_lock); + ret = -ENOMEM; + break; + } + /* NFS_I()->layout == lo under i_lock means the refcount has + * not reached zero: pnfs_put_layout_hdr() decrements to zero + * and detaches in the same critical section. + */ + pnfs_get_layout_hdr(lo); + ref->lo = lo; + nfs4_stateid_copy(&ref->stateid, &lo->plh_stateid); + ref->cred = get_cred(lo->plh_lc_cred); + spin_unlock(&inode->i_lock); + + /* the pinned hdr does not hold the inode: grab it (and + * keep the superblock active) for use across RPCs + */ + ref->inode = nfs_igrab_and_active(inode); + if (!ref->inode) { + /* The layout may still name the deviceID, so report a + * partial list rather than silently shortening it. + * Defer the put: it can layoutreturn and sleep. + */ + list_add(&ref->node, &putme); + ret = -EAGAIN; + break; + } + list_add_tail(&ref->node, args->result); + } + rcu_read_unlock(); + + list_for_each_entry_safe(ref, tmp, &putme, node) { + list_del(&ref->node); + pnfs_put_layout_hdr(ref->lo); + put_cred(ref->cred); + kfree(ref); + } + return ret; +} + +/* + * Collect @clp's layouts referencing @devid onto @result as entries usable + * across sleeping RPCs; release with pnfs_layout_put_deviceid_refs(). + * A negative return means @result is only a partial set. + */ +int +pnfs_layout_collect_deviceid_refs(struct nfs_client *clp, + const struct pnfs_layoutdriver_type *ld, + const struct nfs4_deviceid *devid, + struct list_head *result) +{ + struct pnfs_deviceid_ref_args args = { + .ld = ld, + .devid = devid, + .result = result, + }; + + if (!ld->layout_references_deviceid) + return 0; + + return nfs_client_for_each_server(clp, + pnfs_layout_collect_deviceid_refs_byserver, &args); +} + +void +pnfs_layout_put_deviceid_refs(struct list_head *result) +{ + struct nfs4_deviceid_ref *ref, *tmp; + + list_for_each_entry_safe(ref, tmp, result, node) { + list_del(&ref->node); + put_cred(ref->cred); + pnfs_put_layout_hdr(ref->lo); + nfs_iput_and_deactive(ref->inode); + kfree(ref); + } +} + /* Check if we have we have a valid layout but if there isn't an intersection * between the request and the pgio->pg_lseg, put this pgio->pg_lseg away. */ diff --git a/fs/nfs/pnfs.h b/fs/nfs/pnfs.h index 9a5b8070f595c9..8dd892d875e3ab 100644 --- a/fs/nfs/pnfs.h +++ b/fs/nfs/pnfs.h @@ -184,6 +184,12 @@ struct pnfs_layoutdriver_type { const struct nfs4_deviceid *id, bool immediate, struct list_head *put_list); + /* + * Does @lo hold any reference to deviceid @id? Called under + * @lo's inode i_lock; must not sleep. + */ + bool (*layout_references_deviceid)(struct pnfs_layout_hdr *lo, + const struct nfs4_deviceid *id); int (*prepare_layoutreturn) (struct nfs4_layoutreturn_args *); @@ -371,6 +377,29 @@ void pnfs_layout_reresolve_deviceid_byclid(struct nfs_client *clp, const struct pnfs_layoutdriver_type *ld, const struct nfs4_deviceid *devid, bool immediate); +bool pnfs_layout_deviceid_referenced_byclid(struct nfs_client *clp, + const struct pnfs_layoutdriver_type *ld, + const struct nfs4_deviceid *devid); + +/* + * One live layout referencing a deviceID, collected for the + * CB_NOTIFY_DEVICEID DELETE recovery: the hdr is pinned, the inode + * igrab'd with its superblock active, and the layout stateid and + * cred snapshotted for TEST_STATEID. + */ +struct nfs4_deviceid_ref { + struct list_head node; + struct pnfs_layout_hdr *lo; + struct inode *inode; + nfs4_stateid stateid; + const struct cred *cred; +}; + +int pnfs_layout_collect_deviceid_refs(struct nfs_client *clp, + const struct pnfs_layoutdriver_type *ld, + const struct nfs4_deviceid *devid, + struct list_head *result); +void pnfs_layout_put_deviceid_refs(struct list_head *result); int pnfs_layout_handle_reboot(struct nfs_client *clp); /* nfs4_deviceid_flags */ From 4fd02d0776169a4ccdc6d49ce2d2a37eb2a1dc6b Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:20 -0400 Subject: [PATCH 0237/1352] NFSv4/pnfs: Recover revoked layouts on a deleted deviceID RFC 8881 Section 20.12 lets a server send CB_NOTIFY_DEVICEID DELETE for a deviceID once it has revoked every layout referring to it. Revocation is not announced, so the client can still be holding what it believes are live layouts on that deviceID. Section 18.40.4 resolves that: TEST_STATEID each referring layout and recover the ones that come back revoked -- mark the layout stateid invalid, free the lsegs, FREE_STATEID to acknowledge. The callback thread cannot issue fore-channel RPCs, so suspects are queued on the nfs_client (dedup'd, holding a layoutdriver reference) and resolved by a new state-manager step keyed on NFS4CLNT_DEVICEID_DELETE. The worker re-collects the referring layouts, so layouts returned or recalled in the meantime are skipped. Drop the cached device once the collected layouts account for the delete: every one of them was revoked here. Section 18.48.3 defines TEST_STATEID's answers, and NFS4ERR_OLD_STATEID says the layout exists and was not revoked -- only that it moved on after this stateid was snapshotted -- so it counts against the delete as NFS4_OK does. Any other answer leaves the revocation unresolved and keeps the device cached, as does a layout the server still considers valid (verifying that one with GETDEVICEINFO comes next). A layout counts as revoked only if it was invalidated here; a stateid that no longer matches its layout is a stale snapshot. Invalidating one is paired with nfs_commit_inode(), since pnfs_clear_lseg_state() drops only the VALID and LAYOUTCOMMIT references, and an lseg still held by a commit bucket would keep the layout -- and the device nodes this recovery is trying to release -- alive. If the walk collects no referring layouts, the device is unreferenced and the delete is carried out directly. If the collection could not be completed, recovery leaves the device cached for the next notification. Nothing enqueues suspects yet, so no behavior change. Assisted-by: Claude:claude-fable-5 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/nfs4_fs.h | 2 + fs/nfs/nfs4client.c | 2 + fs/nfs/nfs4proc.c | 83 +++++++++++++++++++++++++++++++++++++++ fs/nfs/nfs4state.c | 3 ++ fs/nfs/pnfs.c | 62 +++++++++++++++++++++++++++++ fs/nfs/pnfs.h | 19 +++++++++ include/linux/nfs_fs_sb.h | 2 + 7 files changed, 173 insertions(+) diff --git a/fs/nfs/nfs4_fs.h b/fs/nfs/nfs4_fs.h index b48e5b87cb2a9b..d642aca0adc3d4 100644 --- a/fs/nfs/nfs4_fs.h +++ b/fs/nfs/nfs4_fs.h @@ -52,6 +52,7 @@ enum nfs4_client_state { NFS4CLNT_RECALL_ANY_LAYOUT_READ, NFS4CLNT_RECALL_ANY_LAYOUT_RW, NFS4CLNT_DELEGRETURN_DELAYED, + NFS4CLNT_DEVICEID_DELETE, }; #define NFS4_RENEW_TIMEOUT 0x01 @@ -493,6 +494,7 @@ int nfs41_discover_server_trunking(struct nfs_client *clp, struct nfs_client **, const struct cred *); extern void nfs4_schedule_session_recovery(struct nfs4_session *, int); extern void nfs41_notify_server(struct nfs_client *); +extern void nfs4_deviceid_delete_recover_run(struct nfs_client *clp); bool nfs4_check_serverowner_major_id(struct nfs41_server_owner *o1, struct nfs41_server_owner *o2); diff --git a/fs/nfs/nfs4client.c b/fs/nfs/nfs4client.c index b661f446ea49e8..e6a58991366628 100644 --- a/fs/nfs/nfs4client.c +++ b/fs/nfs/nfs4client.c @@ -217,6 +217,7 @@ struct nfs_client *nfs4_alloc_client(const struct nfs_client_initdata *cl_init) clp->cl_last_renewal = jiffies; init_waitqueue_head(&clp->cl_lock_waitq); INIT_LIST_HEAD(&clp->pending_cb_stateids); + INIT_LIST_HEAD(&clp->cl_deviceid_deletes); if (cl_init->minorversion != 0) __set_bit(NFS_CS_INFINITE_SLOTS, &clp->cl_flags); @@ -286,6 +287,7 @@ static void nfs4_shutdown_client(struct nfs_client *clp) nfs4_kill_renewd(clp); clp->cl_mvops->shutdown_client(clp); nfs4_destroy_callback(clp); + pnfs_deviceid_delete_queue_free(clp); if (__test_and_clear_bit(NFS_CS_IDMAP, &clp->cl_res_state)) nfs_idmap_delete(clp); diff --git a/fs/nfs/nfs4proc.c b/fs/nfs/nfs4proc.c index e41c792a2725b6..2192875c168e72 100644 --- a/fs/nfs/nfs4proc.c +++ b/fs/nfs/nfs4proc.c @@ -10506,6 +10506,89 @@ static int nfs41_free_stateid(struct nfs_server *server, return ret; } +/* + * A DELETE for a deviceID we still hold layouts on implies the server + * revoked them: run the RFC 8881 Section 18.40.4 recovery. + */ +static void nfs4_deviceid_delete_recover(struct nfs_client *clp, + const struct pnfs_layoutdriver_type *ld, + const struct nfs4_deviceid *id) +{ + LIST_HEAD(layouts); + struct nfs4_deviceid_ref *ref; + bool revoked = false; + bool referenced = false; + bool inconclusive = false; + + if (pnfs_layout_collect_deviceid_refs(clp, ld, id, &layouts)) { + /* Only a partial list -- an allocation failed, or an inode is + * being evicted. Leave the device cached and recover on a + * later notification. + */ + pnfs_layout_put_deviceid_refs(&layouts); + return; + } + + if (list_empty(&layouts)) { + nfs4_delete_deviceid(ld, clp, id); + return; + } + + list_for_each_entry(ref, &layouts, node) { + struct pnfs_layout_hdr *lo = ref->lo; + struct inode *inode = ref->inode; + bool invalidated = false; + LIST_HEAD(head); + int status; + + status = nfs41_test_stateid(NFS_SERVER(inode), &ref->stateid, + ref->cred); + switch (status) { + case NFS_OK: + case -NFS4ERR_OLD_STATEID: + referenced = true; + break; + case -NFS4ERR_ADMIN_REVOKED: + case -NFS4ERR_DELEG_REVOKED: + case -NFS4ERR_EXPIRED: + case -NFS4ERR_BAD_STATEID: + spin_lock(&inode->i_lock); + if (pnfs_layout_is_valid(lo) && + nfs4_stateid_match_other(&ref->stateid, + &lo->plh_stateid)) { + pnfs_mark_layout_stateid_invalid(lo, &head); + revoked = true; + invalidated = true; + } + spin_unlock(&inode->i_lock); + pnfs_free_lseg_list(&head); + if (invalidated) + nfs_commit_inode(inode, 0); + nfs41_free_stateid(NFS_SERVER(inode), &ref->stateid, + ref->cred, true); + break; + default: + inconclusive = true; + break; + } + } + pnfs_layout_put_deviceid_refs(&layouts); + + if (revoked && !referenced && !inconclusive) + nfs4_delete_deviceid(ld, clp, id); +} + +void nfs4_deviceid_delete_recover_run(struct nfs_client *clp) +{ + struct nfs4_deviceid_delete *dd; + + while ((dd = pnfs_deviceid_delete_dequeue(clp)) != NULL) { + nfs4_deviceid_delete_recover(clp, dd->ld, &dd->id); + pnfs_put_layoutdriver(dd->ld); + kfree(dd); + } +} + static void nfs41_free_lock_state(struct nfs_server *server, struct nfs4_lock_state *lsp) { diff --git a/fs/nfs/nfs4state.c b/fs/nfs/nfs4state.c index a5dec0473e221a..1faf9dafd3314c 100644 --- a/fs/nfs/nfs4state.c +++ b/fs/nfs/nfs4state.c @@ -2669,6 +2669,9 @@ static void nfs4_state_manager(struct nfs_client *clp) set_bit(NFS4CLNT_RUN_MANAGER, &clp->cl_state); } nfs4_layoutreturn_any_run(clp); + if (test_and_clear_bit(NFS4CLNT_DEVICEID_DELETE, + &clp->cl_state)) + nfs4_deviceid_delete_recover_run(clp); clear_bit(NFS4CLNT_RECALL_RUNNING, &clp->cl_state); } diff --git a/fs/nfs/pnfs.c b/fs/nfs/pnfs.c index fbe4b58b5b7227..79f671355e5b48 100644 --- a/fs/nfs/pnfs.c +++ b/fs/nfs/pnfs.c @@ -3143,6 +3143,68 @@ pnfs_layout_put_deviceid_refs(struct list_head *result) } } +/* + * Queue @id for the state manager's Section 18.40.4 recovery, + * dropping duplicates of an already-queued suspect. + */ +void pnfs_deviceid_delete_mark(struct nfs_client *clp, + const struct pnfs_layoutdriver_type *ld, + const struct nfs4_deviceid *id) +{ + struct nfs4_deviceid_delete *dd, *new; + + new = kzalloc_obj(*new, GFP_KERNEL); + if (!new) + return; /* lost notification; recovery waits for the next */ + new->ld = pnfs_find_layoutdriver(ld->id); + if (!new->ld) { + kfree(new); + return; + } + memcpy(&new->id, id, sizeof(new->id)); + + spin_lock(&clp->cl_lock); + list_for_each_entry(dd, &clp->cl_deviceid_deletes, list) { + if (dd->ld == new->ld && + !memcmp(&dd->id, &new->id, sizeof(dd->id))) { + spin_unlock(&clp->cl_lock); + pnfs_put_layoutdriver(new->ld); + kfree(new); + return; + } + } + list_add_tail(&new->list, &clp->cl_deviceid_deletes); + spin_unlock(&clp->cl_lock); + + set_bit(NFS4CLNT_DEVICEID_DELETE, &clp->cl_state); + nfs4_schedule_state_manager(clp); +} + +struct nfs4_deviceid_delete *pnfs_deviceid_delete_dequeue( + struct nfs_client *clp) +{ + struct nfs4_deviceid_delete *dd = NULL; + + spin_lock(&clp->cl_lock); + if (!list_empty(&clp->cl_deviceid_deletes)) { + dd = list_first_entry(&clp->cl_deviceid_deletes, + struct nfs4_deviceid_delete, list); + list_del(&dd->list); + } + spin_unlock(&clp->cl_lock); + return dd; +} + +void pnfs_deviceid_delete_queue_free(struct nfs_client *clp) +{ + struct nfs4_deviceid_delete *dd; + + while ((dd = pnfs_deviceid_delete_dequeue(clp)) != NULL) { + pnfs_put_layoutdriver(dd->ld); + kfree(dd); + } +} + /* Check if we have we have a valid layout but if there isn't an intersection * between the request and the pgio->pg_lseg, put this pgio->pg_lseg away. */ diff --git a/fs/nfs/pnfs.h b/fs/nfs/pnfs.h index 8dd892d875e3ab..18ea4e8e0d859f 100644 --- a/fs/nfs/pnfs.h +++ b/fs/nfs/pnfs.h @@ -400,6 +400,25 @@ int pnfs_layout_collect_deviceid_refs(struct nfs_client *clp, const struct nfs4_deviceid *devid, struct list_head *result); void pnfs_layout_put_deviceid_refs(struct list_head *result); + +/* + * A CB_NOTIFY_DEVICEID DELETE naming a deviceID that live layouts + * still reference (RFC 8881 Section 18.40.4). Queued on + * nfs_client.cl_deviceid_deletes under cl_lock for the state manager + * to resolve; holds a layoutdriver reference. + */ +struct nfs4_deviceid_delete { + struct list_head list; + const struct pnfs_layoutdriver_type *ld; + struct nfs4_deviceid id; +}; + +void pnfs_deviceid_delete_mark(struct nfs_client *clp, + const struct pnfs_layoutdriver_type *ld, + const struct nfs4_deviceid *id); +struct nfs4_deviceid_delete *pnfs_deviceid_delete_dequeue( + struct nfs_client *clp); +void pnfs_deviceid_delete_queue_free(struct nfs_client *clp); int pnfs_layout_handle_reboot(struct nfs_client *clp); /* nfs4_deviceid_flags */ diff --git a/include/linux/nfs_fs_sb.h b/include/linux/nfs_fs_sb.h index 482daf1fa4b0cc..416c6f39f31ddb 100644 --- a/include/linux/nfs_fs_sb.h +++ b/include/linux/nfs_fs_sb.h @@ -103,6 +103,8 @@ struct nfs_client { /* The flags used for obtaining the clientid during EXCHANGE_ID */ u32 cl_exchange_flags; struct nfs4_session *cl_session; /* shared session */ + /* CB_NOTIFY_DEVICEID DELETE suspects, protected by cl_lock */ + struct list_head cl_deviceid_deletes; bool cl_preserve_clid; struct nfs41_server_owner *cl_serverowner; struct nfs41_server_scope *cl_serverscope; From a65c650fea3bf744d1f7499f9674382f537241ba Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:21 -0400 Subject: [PATCH 0238/1352] NFSv4/pnfs: Confirm a deviceID delete via GETDEVICEINFO RFC 8881 Section 18.40.4: if TEST_STATEID says at least one layout referring to the deleted deviceID is still valid, the delete cannot be trusted -- verify it with GETDEVICEINFO. The device really being gone while the server also considers a referring layout valid means the server is faulty; recover by re-establishing the client ID and drop the cached device. Any other answer -- including the device existing, i.e. an erroneous DELETE -- keeps the cached device and the layout intact. Re-establishing the client ID is nfs4_reset_all_state(), which sets NFS4CLNT_PURGE_STATE so the state manager runs nfs4_purge_lease(): a fresh EXCHANGE_ID, then state reclaim with no grace period. The grace-less reclaim is the point -- the server has not rebooted, so there is nothing to reclaim under CLAIM_PREVIOUS, and the new client ID orphans the state held under the old one. The obvious-looking nfs4_schedule_lease_recovery() is not the right call here: it sets NFS4CLNT_CHECK_LEASE, which the state manager turns into a lease renewal, and on a healthy session -- which this one is, the server having just answered TEST_STATEID and GETDEVICEINFO on it -- that renewal succeeds and no EXCHANGE_ID is ever sent. This is the only path on which a device notification can escalate to a full client-ID reset, and every open, lock and delegation on the client is reclaimed as a result. From userspace that is indistinguishable from a spontaneous lease expiry, so the escalation is announced with a rate-limited warning naming the server. The raw-status probe calls nfs4_proc_getdeviceinfo() directly because nfs4_get_device_info() swallows the RPC status and cannot distinguish NFS4ERR_NOENT from a transient failure. A one-page reply buffer is enough: a device too large for it fails with something other than -ENOENT, which still proves existence. Still nothing enqueues suspects; no behavior change. Assisted-by: Claude:claude-fable-5 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/nfs4_fs.h | 1 + fs/nfs/nfs4proc.c | 59 ++++++++++++++++++++++++++++++++++++++++------ fs/nfs/nfs4state.c | 2 +- 3 files changed, 54 insertions(+), 8 deletions(-) diff --git a/fs/nfs/nfs4_fs.h b/fs/nfs/nfs4_fs.h index d642aca0adc3d4..76dae699d4d781 100644 --- a/fs/nfs/nfs4_fs.h +++ b/fs/nfs/nfs4_fs.h @@ -511,6 +511,7 @@ extern void nfs_inode_find_state_and_recover(struct inode *inode, const nfs4_stateid *stateid); extern int nfs4_state_mark_reclaim_nograce(struct nfs_client *, struct nfs4_state *); extern void nfs4_schedule_lease_recovery(struct nfs_client *); +extern void nfs4_reset_all_state(struct nfs_client *); extern int nfs4_wait_clnt_recover(struct nfs_client *clp); extern int nfs4_client_recover_expired_lease(struct nfs_client *clp); extern void nfs4_schedule_state_manager(struct nfs_client *); diff --git a/fs/nfs/nfs4proc.c b/fs/nfs/nfs4proc.c index 2192875c168e72..518348e87dd874 100644 --- a/fs/nfs/nfs4proc.c +++ b/fs/nfs/nfs4proc.c @@ -10506,19 +10506,50 @@ static int nfs41_free_stateid(struct nfs_server *server, return ret; } +/* + * GETDEVICEINFO surfacing the raw status; nfs4_get_device_info() + * swallows it. A device too large for one page fails with something + * other than -ENOENT, which still proves existence. + */ +static int nfs4_deviceid_validate(struct nfs_server *server, + const struct pnfs_layoutdriver_type *ld, + const struct nfs4_deviceid *id, const struct cred *cred) +{ + struct pnfs_device pdev; + struct page *page; + int status; + + page = alloc_page(GFP_KERNEL); + if (!page) + return -ENOMEM; + + memset(&pdev, 0, sizeof(pdev)); + memcpy(&pdev.dev_id, id, sizeof(pdev.dev_id)); + pdev.layout_type = ld->id; + pdev.pages = &page; + pdev.pglen = PAGE_SIZE; + pdev.maxcount = PAGE_SIZE - nfs41_maxgetdevinfo_overhead; + + status = nfs4_proc_getdeviceinfo(server, &pdev, cred); + __free_page(page); + return status; +} + /* * A DELETE for a deviceID we still hold layouts on implies the server - * revoked them: run the RFC 8881 Section 18.40.4 recovery. + * revoked them: run the RFC 8881 Section 18.40.4 recovery. A layout the + * server still calls valid leaves the revocations unable to confirm the + * delete, so verify it with GETDEVICEINFO. */ static void nfs4_deviceid_delete_recover(struct nfs_client *clp, const struct pnfs_layoutdriver_type *ld, const struct nfs4_deviceid *id) { LIST_HEAD(layouts); - struct nfs4_deviceid_ref *ref; + struct nfs4_deviceid_ref *ref, *confirm = NULL; bool revoked = false; - bool referenced = false; bool inconclusive = false; + int status; if (pnfs_layout_collect_deviceid_refs(clp, ld, id, &layouts)) { /* Only a partial list -- an allocation failed, or an inode is @@ -10539,14 +10570,14 @@ static void nfs4_deviceid_delete_recover(struct nfs_client *clp, struct inode *inode = ref->inode; bool invalidated = false; LIST_HEAD(head); - int status; status = nfs41_test_stateid(NFS_SERVER(inode), &ref->stateid, ref->cred); switch (status) { case NFS_OK: case -NFS4ERR_OLD_STATEID: - referenced = true; + if (!confirm) + confirm = ref; break; case -NFS4ERR_ADMIN_REVOKED: case -NFS4ERR_DELEG_REVOKED: @@ -10572,10 +10603,24 @@ static void nfs4_deviceid_delete_recover(struct nfs_client *clp, break; } } - pnfs_layout_put_deviceid_refs(&layouts); - if (revoked && !referenced && !inconclusive) + if (confirm) { + status = nfs4_deviceid_validate(NFS_SERVER(confirm->inode), + ld, id, confirm->cred); + if (status == -ENOENT) { + /* Section 18.40.4 prescribes EXCHANGE_ID here; + * nfs4_schedule_lease_recovery() would only renew + * the existing lease. + */ + pr_warn_ratelimited("NFS: server %s deleted a deviceID referred to by a layout it still considers valid; re-establishing the client ID\n", + clp->cl_hostname); + nfs4_reset_all_state(clp); + nfs4_delete_deviceid(ld, clp, id); + } + } else if (revoked && !inconclusive) { nfs4_delete_deviceid(ld, clp, id); + } + pnfs_layout_put_deviceid_refs(&layouts); } void nfs4_deviceid_delete_recover_run(struct nfs_client *clp) diff --git a/fs/nfs/nfs4state.c b/fs/nfs/nfs4state.c index 1faf9dafd3314c..4c085c00abb1b5 100644 --- a/fs/nfs/nfs4state.c +++ b/fs/nfs/nfs4state.c @@ -2354,7 +2354,7 @@ void nfs41_notify_server(struct nfs_client *clp) nfs4_schedule_state_manager(clp); } -static void nfs4_reset_all_state(struct nfs_client *clp) +void nfs4_reset_all_state(struct nfs_client *clp) { if (test_and_set_bit(NFS4CLNT_LEASE_EXPIRED, &clp->cl_state) == 0) { set_bit(NFS4CLNT_PURGE_STATE, &clp->cl_state); From ce0c3011d6cef5b85d880d679cd41fe01abdd18b Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:22 -0400 Subject: [PATCH 0239/1352] NFSv4/pnfs: Dispatch CB_NOTIFY_DEVICEID DELETE to race recovery Turn on the RFC 8881 Section 18.40.4 DELETE handling: a DELETE for a deviceID that no live layout references keeps today's cheap behavior (drop the cached device, no state-manager wake). A DELETE for a deviceID that live layouts still reference is deferred to the state manager, which TEST_STATEIDs the referring layouts, recovers revoked ones, and confirms or refutes the delete with GETDEVICEINFO. The deferred case no longer unhashes the device immediately: if the recovery concludes the DELETE was erroneous (the deviceID still exists and a referring layout is still valid), the client keeps using the cached device. Re-arm the state manager afterwards, as the delegation return above it does. The recovery issues synchronous RPCs, and a manager thread that starts and exits while it runs clears NFS4CLNT_RUN_MANAGER on its way out. Bump the deviceid change epoch for both notification types rather than only for CHANGE. A GETDEVICEINFO whose reply is already in flight can otherwise re-cache a device the notification has just invalidated; that is as true of a delete as of a change, and Section 18.40.4 opens by describing the race for the delete case. Assisted-by: Claude:claude-fable-5 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/callback_proc.c | 24 +++++++++++++++--------- fs/nfs/nfs4state.c | 4 +++- 2 files changed, 18 insertions(+), 10 deletions(-) diff --git a/fs/nfs/callback_proc.c b/fs/nfs/callback_proc.c index ea8c558b07b6bc..c9b71dbae9ea65 100644 --- a/fs/nfs/callback_proc.c +++ b/fs/nfs/callback_proc.c @@ -392,19 +392,25 @@ __be32 nfs4_callback_devicenotify(void *argp, void *resp, continue; } /* - * Unhash the cached device first so re-resolution cannot - * re-pin the stale node, then re-point any references - * pinned under live layouts (RFC 8881 Section 12.2.10). - * The epoch bump lets an in-flight GETDEVICEINFO detect - * that its reply may predate the change. + * Bump the epoch before touching the cache so a + * GETDEVICEINFO already in flight can detect that it + * predates the notification. A referenced DELETE may be + * racing revocation, so defer it to the state manager -- + * this thread cannot issue fore-channel RPCs. */ - if (dev->cbd_notify_type == NOTIFY_DEVICEID4_CHANGE) - nfs4_deviceid_bump_change_epoch(cps->clp); - nfs4_delete_deviceid(ld, cps->clp, &dev->cbd_dev_id); - if (dev->cbd_notify_type == NOTIFY_DEVICEID4_CHANGE) + nfs4_deviceid_bump_change_epoch(cps->clp); + if (dev->cbd_notify_type == NOTIFY_DEVICEID4_CHANGE) { + nfs4_delete_deviceid(ld, cps->clp, &dev->cbd_dev_id); pnfs_layout_reresolve_deviceid_byclid(cps->clp, ld, &dev->cbd_dev_id, dev->cbd_immediate); + } else if (pnfs_layout_deviceid_referenced_byclid(cps->clp, + ld, &dev->cbd_dev_id)) { + pnfs_deviceid_delete_mark(cps->clp, ld, + &dev->cbd_dev_id); + } else { + nfs4_delete_deviceid(ld, cps->clp, &dev->cbd_dev_id); + } } pnfs_put_layoutdriver(ld); out: diff --git a/fs/nfs/nfs4state.c b/fs/nfs/nfs4state.c index 4c085c00abb1b5..b5d6daa2c41624 100644 --- a/fs/nfs/nfs4state.c +++ b/fs/nfs/nfs4state.c @@ -2670,8 +2670,10 @@ static void nfs4_state_manager(struct nfs_client *clp) } nfs4_layoutreturn_any_run(clp); if (test_and_clear_bit(NFS4CLNT_DEVICEID_DELETE, - &clp->cl_state)) + &clp->cl_state)) { nfs4_deviceid_delete_recover_run(clp); + set_bit(NFS4CLNT_RUN_MANAGER, &clp->cl_state); + } clear_bit(NFS4CLNT_RECALL_RUNNING, &clp->cl_state); } From 415117fabac8001f65ff44156bc1874ce18f1ebf Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:23 -0400 Subject: [PATCH 0240/1352] NFSv4/pnfs: Grow the deviceid cache hash table The global deviceid cache has 32 buckets shared by every server and layout type. Striping deployments initially anticipate device counts scaling to 1024 or more devices, which leaves those chains 32 entries deep for every resolution to walk. Grow to 256 buckets, four deep at that scale, for 2KB of BSS on 64-bit. Assisted-by: Claude:claude-fable-5 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/pnfs_dev.c | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/fs/nfs/pnfs_dev.c b/fs/nfs/pnfs_dev.c index a3b28409539ae1..e8143977965cae 100644 --- a/fs/nfs/pnfs_dev.c +++ b/fs/nfs/pnfs_dev.c @@ -40,8 +40,11 @@ /* * Device ID RCU cache. A device ID is unique per server and layout type. + * + * 256 buckets keeps the chains short at the 1024-or-more devices a + * striping deployment expects, for 2KB of BSS on 64-bit. */ -#define NFS4_DEVICE_ID_HASH_BITS 5 +#define NFS4_DEVICE_ID_HASH_BITS 8 #define NFS4_DEVICE_ID_HASH_SIZE (1 << NFS4_DEVICE_ID_HASH_BITS) #define NFS4_DEVICE_ID_HASH_MASK (NFS4_DEVICE_ID_HASH_SIZE - 1) From 77f2fec628b4d2781d978f42d8d82a710e58e036 Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:24 -0400 Subject: [PATCH 0241/1352] NFSv4/pnfs: Re-home the data-server cache onto hash buckets The per-net data-server cache is a single list, and every GETDEVICEINFO decode walks all of it looking for a match, so filling the cache costs O(n^2) in the number of data servers -- which a striping mount does in one burst, at the same scale the deviceid cache was just sized for. Key it by the DS address set instead. This patch is the mechanical half: nfs4_pnfs_ds.ds_node becomes an hlist_node, netns init and teardown cover every bucket, and removal uses hlist_del_init (which needs no bucket reference). Insertion still targets bucket 0 and lookup still scans every entry, so behavior is unchanged; the key comes next. Splitting it this way keeps a bisect able to tell a list-conversion bug from a hash-key bug. The buckets live in struct nfs_net, so this costs 2KB per network namespace on 64-bit, paid once nfs.ko is loaded whether or not that namespace ever mounts NFS. Assisted-by: Claude:claude-fable-5 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/client.c | 6 ++++-- fs/nfs/netns.h | 5 ++++- fs/nfs/pnfs.h | 2 +- fs/nfs/pnfs_nfs.c | 19 +++++++++++-------- 4 files changed, 20 insertions(+), 12 deletions(-) diff --git a/fs/nfs/client.c b/fs/nfs/client.c index 60386330aeeca2..dbb5131375f8dc 100644 --- a/fs/nfs/client.c +++ b/fs/nfs/client.c @@ -1295,7 +1295,8 @@ void nfs_clients_init(struct net *net) INIT_LIST_HEAD(&nn->nfs_volume_list); #if IS_ENABLED(CONFIG_NFS_V4) idr_init(&nn->cb_ident_idr); - INIT_LIST_HEAD(&nn->nfs4_data_server_cache); + for (int i = 0; i < NFS4_DS_CACHE_HASH_SIZE; i++) + INIT_HLIST_HEAD(&nn->nfs4_data_server_cache[i]); spin_lock_init(&nn->nfs4_data_server_lock); #endif /* CONFIG_NFS_V4 */ spin_lock_init(&nn->nfs_client_lock); @@ -1315,7 +1316,8 @@ void nfs_clients_exit(struct net *net) WARN_ON_ONCE(!list_empty(&nn->nfs_client_list)); WARN_ON_ONCE(!list_empty(&nn->nfs_volume_list)); #if IS_ENABLED(CONFIG_NFS_V4) - WARN_ON_ONCE(!list_empty(&nn->nfs4_data_server_cache)); + for (int i = 0; i < NFS4_DS_CACHE_HASH_SIZE; i++) + WARN_ON_ONCE(!hlist_empty(&nn->nfs4_data_server_cache[i])); #endif /* CONFIG_NFS_V4 */ } diff --git a/fs/nfs/netns.h b/fs/nfs/netns.h index 36658579100dbe..560fa95726b046 100644 --- a/fs/nfs/netns.h +++ b/fs/nfs/netns.h @@ -31,7 +31,10 @@ struct nfs_net { unsigned short nfs_callback_tcpport; unsigned short nfs_callback_tcpport6; int cb_users[NFS4_MAX_MINOR_VERSION + 1]; - struct list_head nfs4_data_server_cache; +#define NFS4_DS_CACHE_HASH_BITS 8 +#define NFS4_DS_CACHE_HASH_SIZE (1 << NFS4_DS_CACHE_HASH_BITS) + /* every entry is still in bucket 0 until the key is added */ + struct hlist_head nfs4_data_server_cache[NFS4_DS_CACHE_HASH_SIZE]; spinlock_t nfs4_data_server_lock; #endif /* CONFIG_NFS_V4 */ struct nfs_netns_client *nfs_client; diff --git a/fs/nfs/pnfs.h b/fs/nfs/pnfs.h index 18ea4e8e0d859f..3a1c012ba18e42 100644 --- a/fs/nfs/pnfs.h +++ b/fs/nfs/pnfs.h @@ -57,7 +57,7 @@ struct nfs4_pnfs_ds_addr { }; struct nfs4_pnfs_ds { - struct list_head ds_node; /* nfs4_pnfs_dev_hlist dev_dslist */ + struct hlist_node ds_node; /* nfs_net nfs4_data_server_cache */ char *ds_remotestr; /* comma sep list of addrs */ struct list_head ds_addrs; const struct net *ds_net; diff --git a/fs/nfs/pnfs_nfs.c b/fs/nfs/pnfs_nfs.c index e0e3fc7414e600..f88a9784988ac1 100644 --- a/fs/nfs/pnfs_nfs.c +++ b/fs/nfs/pnfs_nfs.c @@ -604,7 +604,7 @@ _same_data_server_addrs_locked(const struct list_head *dsaddrs1, } /* - * Lookup DS by addresses and NFS version. nfs4_ds_cache_lock is held + * Lookup DS by addresses and NFS version. nfs4_data_server_lock is held */ static struct nfs4_pnfs_ds * _data_server_lookup_locked(const struct nfs_net *nn, @@ -612,10 +612,13 @@ _data_server_lookup_locked(const struct nfs_net *nn, { struct nfs4_pnfs_ds *ds; - list_for_each_entry(ds, &nn->nfs4_data_server_cache, ds_node) - if (ds->ds_version == version && - _same_data_server_addrs_locked(&ds->ds_addrs, dsaddrs)) - return ds; + for (int i = 0; i < NFS4_DS_CACHE_HASH_SIZE; i++) + hlist_for_each_entry(ds, &nn->nfs4_data_server_cache[i], + ds_node) + if (ds->ds_version == version && + _same_data_server_addrs_locked(&ds->ds_addrs, + dsaddrs)) + return ds; return NULL; } @@ -666,7 +669,7 @@ void nfs4_pnfs_ds_put(struct nfs4_pnfs_ds *ds) struct nfs_net *nn = net_generic(ds->ds_net, nfs_net_id); if (refcount_dec_and_lock(&ds->ds_count, &nn->nfs4_data_server_lock)) { - list_del_init(&ds->ds_node); + hlist_del_init(&ds->ds_node); spin_unlock(&nn->nfs4_data_server_lock); destroy_ds(ds); } @@ -753,11 +756,11 @@ nfs4_pnfs_ds_add(const struct net *net, struct list_head *dsaddrs, u32 version, list_splice_init(dsaddrs, &ds->ds_addrs); ds->ds_remotestr = remotestr; refcount_set(&ds->ds_count, 1); - INIT_LIST_HEAD(&ds->ds_node); + INIT_HLIST_NODE(&ds->ds_node); ds->ds_net = net; ds->ds_clp = NULL; ds->ds_version = version; - list_add(&ds->ds_node, &nn->nfs4_data_server_cache); + hlist_add_head(&ds->ds_node, &nn->nfs4_data_server_cache[0]); dprintk("%s add new data server %s\n", __func__, ds->ds_remotestr); } else { From c35a25faa74cfd7d8ea9c4983577a46a343f49b4 Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:25 -0400 Subject: [PATCH 0242/1352] NFSv4/pnfs: Key the data-server cache by its address set and version Populating the data-server cache was quadratic: every GETDEVICEINFO decode scanned the whole per-net cache under one spinlock. At the anticipated scale of 1024 data servers a striping mount pays that in a burst at first access, and again on notification-driven re-resolution. Hash each DS to a bucket keyed on the whole cache key -- per-address jhash over exactly the fields same_sockaddr() compares, combined by addition so multipath ordering cannot change the bucket, then the NFS version folded in -- with the existing comparator as the in-bucket tiebreaker. Lookup and insert now touch one bucket; teardown is unchanged (hlist_del_init needs no bucket). Keying on the whole set requires the comparator to test set equality, so it is tightened from the subset test it did before -- a subset match would hash to a different bucket and simply never be found. That test was also wrong in a way worth naming. It answered "is dsaddrs1 a subset of dsaddrs2", and the caller passes the cached list first, so a cached data server whose address set was contained in an incoming one was returned for that incoming set. The aliasing was therefore one-directional: cache {A,B} first and an incoming {A} did not match, but cache {A} first and an incoming {A,B} did. The consequence was mild, which is why it went unnoticed: every address on one device's multipath list names the same data server, so a merged entry's addresses are all paths that server also advertised. The effect is lost path diversity and a truncated ds_remotestr (and the netaddr flexfiles reports in layoutstats), not I/O sent to the wrong server. Assisted-by: Claude:claude-fable-5 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/pnfs_nfs.c | 68 ++++++++++++++++++++++++++++++++++++++++------- 1 file changed, 58 insertions(+), 10 deletions(-) diff --git a/fs/nfs/pnfs_nfs.c b/fs/nfs/pnfs_nfs.c index f88a9784988ac1..c9bbcb765f4a84 100644 --- a/fs/nfs/pnfs_nfs.c +++ b/fs/nfs/pnfs_nfs.c @@ -15,6 +15,8 @@ #include "nfs4session.h" #include "internal.h" +#include +#include #include "pnfs.h" #include "netns.h" #include "nfs4trace.h" @@ -577,8 +579,8 @@ same_sockaddr(struct sockaddr *addr1, struct sockaddr *addr2) } /* - * Checks if 'dsaddrs1' contains a subset of 'dsaddrs2'. If it does, - * declare a match. + * Checks if 'dsaddrs1' and 'dsaddrs2' hold the same set of addresses. + * If they do, declare a match. */ static bool _same_data_server_addrs_locked(const struct list_head *dsaddrs1, @@ -588,6 +590,10 @@ _same_data_server_addrs_locked(const struct list_head *dsaddrs1, struct sockaddr *sa1, *sa2; bool match = false; + if (list_count_nodes((struct list_head *)dsaddrs1) != + list_count_nodes((struct list_head *)dsaddrs2)) + return false; + list_for_each_entry(da1, dsaddrs1, da_node) { sa1 = (struct sockaddr *)&da1->da_addr; match = false; @@ -603,6 +609,47 @@ _same_data_server_addrs_locked(const struct list_head *dsaddrs1, return match; } +/* Hash family, address bytes, and port - as same_sockaddr() */ +static u32 +nfs4_ds_addr_hash(const struct sockaddr *sa) +{ + u32 h = sa->sa_family; + + switch (sa->sa_family) { + case AF_INET: { + const struct sockaddr_in *a = (const struct sockaddr_in *)sa; + + h = jhash(&a->sin_addr.s_addr, sizeof(a->sin_addr.s_addr), h); + h = jhash(&a->sin_port, sizeof(a->sin_port), h); + break; + } + case AF_INET6: { + const struct sockaddr_in6 *a = (const struct sockaddr_in6 *)sa; + + h = jhash(&a->sin6_addr, sizeof(a->sin6_addr), h); + h = jhash(&a->sin6_port, sizeof(a->sin6_port), h); + break; + } + } + return h; +} + +/* + * Bucket index for a DS cache key. Per-address hashes combine by + * addition so the multipath list order cannot change the bucket, + * matching the order-independent set comparison above. + */ +static u32 +nfs4_ds_cache_hash(const struct list_head *dsaddrs, u32 version) +{ + const struct nfs4_pnfs_ds_addr *da; + u32 h = 0; + + list_for_each_entry(da, dsaddrs, da_node) + h += nfs4_ds_addr_hash((const struct sockaddr *)&da->da_addr); + return hash_32(jhash_1word(version, h), NFS4_DS_CACHE_HASH_BITS); +} + /* * Lookup DS by addresses and NFS version. nfs4_data_server_lock is held */ @@ -611,14 +658,12 @@ _data_server_lookup_locked(const struct nfs_net *nn, const struct list_head *dsaddrs, u32 version) { struct nfs4_pnfs_ds *ds; + u32 bucket = nfs4_ds_cache_hash(dsaddrs, version); - for (int i = 0; i < NFS4_DS_CACHE_HASH_SIZE; i++) - hlist_for_each_entry(ds, &nn->nfs4_data_server_cache[i], - ds_node) - if (ds->ds_version == version && - _same_data_server_addrs_locked(&ds->ds_addrs, - dsaddrs)) - return ds; + hlist_for_each_entry(ds, &nn->nfs4_data_server_cache[bucket], ds_node) + if (ds->ds_version == version && + _same_data_server_addrs_locked(&ds->ds_addrs, dsaddrs)) + return ds; return NULL; } @@ -735,6 +780,7 @@ nfs4_pnfs_ds_add(const struct net *net, struct list_head *dsaddrs, u32 version, { struct nfs_net *nn = net_generic(net, nfs_net_id); struct nfs4_pnfs_ds *tmp_ds, *ds = NULL; + struct hlist_head *bucket; char *remotestr; if (list_empty(dsaddrs)) { @@ -748,6 +794,8 @@ nfs4_pnfs_ds_add(const struct net *net, struct list_head *dsaddrs, u32 version, /* this is only used for debugging, so it's ok if its NULL */ remotestr = nfs4_pnfs_remotestr(dsaddrs, gfp_flags); + /* @dsaddrs is empty after the splice below. */ + bucket = &nn->nfs4_data_server_cache[nfs4_ds_cache_hash(dsaddrs, version)]; spin_lock(&nn->nfs4_data_server_lock); tmp_ds = _data_server_lookup_locked(nn, dsaddrs, version); @@ -760,7 +808,7 @@ nfs4_pnfs_ds_add(const struct net *net, struct list_head *dsaddrs, u32 version, ds->ds_net = net; ds->ds_clp = NULL; ds->ds_version = version; - hlist_add_head(&ds->ds_node, &nn->nfs4_data_server_cache[0]); + hlist_add_head(&ds->ds_node, bucket); dprintk("%s add new data server %s\n", __func__, ds->ds_remotestr); } else { From 3e68b72f03739cbf4e6fe7f5d2db989526c9c0bd Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Tue, 15 Sep 2026 08:22:26 -0400 Subject: [PATCH 0243/1352] NFSv4/flexfiles: Add a dataserver_nconnect cap Data-server clients inherit the MDS nconnect setting. A striping mount multiplies that by every distinct data server: at the anticipated scale of 1024 and nconnect=16 the client opens 16k sockets plus their sunrpc slot tables, and striping already spreads I/O across the data servers, so a high per-DS transport count buys little for that workload. Add a dataserver_nconnect module parameter to the flexfiles layout driver alongside its existing dataserver_timeo and dataserver_retrans knobs, and thread the value through nfs4_pnfs_ds_connect() to both the v3 and v4 data-server client setup paths. The default of 0 preserves today's inherit-from-MDS behavior, so the cap is opt-in and non-regressing. The files layout passes 0, unchanged. Assisted-by: Claude:claude-fable-5 Signed-off-by: Benjamin Coddington Signed-off-by: Anna Schumaker --- fs/nfs/filelayout/filelayoutdev.c | 2 +- fs/nfs/flexfilelayout/flexfilelayoutdev.c | 7 +++++++ fs/nfs/internal.h | 3 ++- fs/nfs/nfs3client.c | 9 +++++++-- fs/nfs/nfs4client.c | 6 +++++- fs/nfs/pnfs.h | 3 ++- fs/nfs/pnfs_nfs.c | 21 ++++++++++++++------- 7 files changed, 38 insertions(+), 13 deletions(-) diff --git a/fs/nfs/filelayout/filelayoutdev.c b/fs/nfs/filelayout/filelayoutdev.c index 57e0654cd98c34..5995f32f07eca0 100644 --- a/fs/nfs/filelayout/filelayoutdev.c +++ b/fs/nfs/filelayout/filelayoutdev.c @@ -267,7 +267,7 @@ nfs4_fl_prepare_ds(struct pnfs_layout_segment *lseg, u32 ds_idx) goto out_test_devid; status = nfs4_pnfs_ds_connect(s, ds, devid, dataserver_timeo, - dataserver_retrans, 4, + dataserver_retrans, 0, 4, s->nfs_client->cl_minorversion, true); if (status) { nfs4_mark_deviceid_unavailable(devid); diff --git a/fs/nfs/flexfilelayout/flexfilelayoutdev.c b/fs/nfs/flexfilelayout/flexfilelayoutdev.c index 5cb09e5e2138f2..d52f485d0650e6 100644 --- a/fs/nfs/flexfilelayout/flexfilelayoutdev.c +++ b/fs/nfs/flexfilelayout/flexfilelayoutdev.c @@ -20,6 +20,7 @@ static unsigned int dataserver_timeo = NFS_DEF_TCP_TIMEO; static unsigned int dataserver_retrans; +static unsigned int dataserver_nconnect; static bool ff_layout_has_available_ds(struct pnfs_layout_segment *lseg); @@ -418,6 +419,7 @@ nfs4_ff_layout_prepare_ds(struct pnfs_layout_segment *lseg, */ status = nfs4_pnfs_ds_connect(s, ds, &mirror_ds->id_node, dataserver_timeo, dataserver_retrans, + dataserver_nconnect, mirror_ds->ds_versions[0].version, mirror_ds->ds_versions[0].minor_version, mirror_ds->ds_versions[0].tightly_coupled); @@ -676,3 +678,8 @@ module_param(dataserver_timeo, uint, 0644); MODULE_PARM_DESC(dataserver_timeo, "The time (in tenths of a second) the " "NFSv4.1 client waits for a response from a " " data server before it retries an NFS request."); +module_param(dataserver_nconnect, uint, 0644); +MODULE_PARM_DESC(dataserver_nconnect, "The maximum number of connections " + "the NFSv4.1 client opens to each data server, " + "capping the value inherited from the MDS nconnect " + "mount option. 0 (default) applies no cap."); diff --git a/fs/nfs/internal.h b/fs/nfs/internal.h index abc81f5ae57802..48f7c0e25da144 100644 --- a/fs/nfs/internal.h +++ b/fs/nfs/internal.h @@ -251,6 +251,7 @@ extern struct nfs_client *nfs4_set_ds_client(struct nfs_server *mds_srv, int ds_addrlen, int ds_proto, unsigned int ds_timeo, unsigned int ds_retrans, + unsigned int ds_nconnect, u32 minor_version, bool tightly_coupled); extern struct rpc_clnt *nfs4_find_or_create_ds_client(struct nfs_client *, @@ -260,7 +261,7 @@ extern void nfs4_session_limit_xasize(struct nfs_server *server); extern struct nfs_client *nfs3_set_ds_client(struct nfs_server *mds_srv, const struct sockaddr_storage *ds_addr, int ds_addrlen, int ds_proto, unsigned int ds_timeo, - unsigned int ds_retrans); + unsigned int ds_retrans, unsigned int ds_nconnect); #ifdef CONFIG_PROC_FS extern int __init nfs_fs_proc_init(void); extern void nfs_fs_proc_exit(void); diff --git a/fs/nfs/nfs3client.c b/fs/nfs/nfs3client.c index 5d97c1d38bb62d..cf2f7be4b43515 100644 --- a/fs/nfs/nfs3client.c +++ b/fs/nfs/nfs3client.c @@ -84,7 +84,8 @@ struct nfs_server *nfs3_clone_server(struct nfs_server *source, */ struct nfs_client *nfs3_set_ds_client(struct nfs_server *mds_srv, const struct sockaddr_storage *ds_addr, int ds_addrlen, - int ds_proto, unsigned int ds_timeo, unsigned int ds_retrans) + int ds_proto, unsigned int ds_timeo, unsigned int ds_retrans, + unsigned int ds_nconnect) { struct rpc_timeout ds_timeout; unsigned long connect_timeout = ds_timeo * (ds_retrans + 1) * HZ / 10; @@ -124,8 +125,12 @@ struct nfs_client *nfs3_set_ds_client(struct nfs_server *mds_srv, fallthrough; case XPRT_TRANSPORT_RDMA: case XPRT_TRANSPORT_TCP: - if (mds_clp->cl_nconnect > 1) + if (mds_clp->cl_nconnect > 1) { cl_init.nconnect = mds_clp->cl_nconnect; + if (ds_nconnect) + cl_init.nconnect = min(cl_init.nconnect, + ds_nconnect); + } } if (mds_srv->flags & NFS_MOUNT_NORESVPORT) diff --git a/fs/nfs/nfs4client.c b/fs/nfs/nfs4client.c index e6a58991366628..fe779fb2ec726d 100644 --- a/fs/nfs/nfs4client.c +++ b/fs/nfs/nfs4client.c @@ -794,7 +794,8 @@ static int nfs4_set_client(struct nfs_server *server, struct nfs_client *nfs4_set_ds_client(struct nfs_server *mds_srv, const struct sockaddr_storage *ds_addr, int ds_addrlen, int ds_proto, unsigned int ds_timeo, unsigned int ds_retrans, - u32 minor_version, bool tightly_coupled) + unsigned int ds_nconnect, u32 minor_version, + bool tightly_coupled) { struct rpc_timeout ds_timeout; struct nfs_client *mds_clp = mds_srv->nfs_client; @@ -832,6 +833,9 @@ struct nfs_client *nfs4_set_ds_client(struct nfs_server *mds_srv, case XPRT_TRANSPORT_TCP: if (mds_clp->cl_nconnect > 1) { cl_init.nconnect = mds_clp->cl_nconnect; + if (ds_nconnect) + cl_init.nconnect = min(cl_init.nconnect, + ds_nconnect); cl_init.max_connect = NFS_MAX_TRANSPORTS; } } diff --git a/fs/nfs/pnfs.h b/fs/nfs/pnfs.h index 3a1c012ba18e42..5cde5db63d29fe 100644 --- a/fs/nfs/pnfs.h +++ b/fs/nfs/pnfs.h @@ -504,7 +504,8 @@ struct nfs4_pnfs_ds *nfs4_pnfs_ds_add(const struct net *net, void nfs4_pnfs_v3_ds_connect_unload(void); int nfs4_pnfs_ds_connect(struct nfs_server *mds_srv, struct nfs4_pnfs_ds *ds, struct nfs4_deviceid_node *devid, unsigned int timeo, - unsigned int retrans, u32 version, u32 minor_version, + unsigned int retrans, unsigned int nconnect, + u32 version, u32 minor_version, bool tightly_coupled); struct nfs4_pnfs_ds_addr *nfs4_decode_mp_ds_addr(struct net *net, struct xdr_stream *xdr, diff --git a/fs/nfs/pnfs_nfs.c b/fs/nfs/pnfs_nfs.c index c9bbcb765f4a84..7adb6f941cf26e 100644 --- a/fs/nfs/pnfs_nfs.c +++ b/fs/nfs/pnfs_nfs.c @@ -844,7 +844,8 @@ static struct nfs_client *(*get_v3_ds_connect)( int ds_addrlen, int ds_proto, unsigned int ds_timeo, - unsigned int ds_retrans); + unsigned int ds_retrans, + unsigned int ds_nconnect); static bool load_v3_ds_connect(void) { @@ -867,7 +868,8 @@ void nfs4_pnfs_v3_ds_connect_unload(void) static int _nfs4_pnfs_v3_ds_connect(struct nfs_server *mds_srv, struct nfs4_pnfs_ds *ds, unsigned int timeo, - unsigned int retrans) + unsigned int retrans, + unsigned int nconnect) { struct nfs_client *clp = ERR_PTR(-EIO); struct nfs_client *mds_clp = mds_srv->nfs_client; @@ -919,7 +921,7 @@ static int _nfs4_pnfs_v3_ds_connect(struct nfs_server *mds_srv, ds_proto = XPRT_TRANSPORT_TCP_TLS; clp = get_v3_ds_connect(mds_srv, &da->da_addr, da->da_addrlen, - ds_proto, timeo, retrans); + ds_proto, timeo, retrans, nconnect); if (IS_ERR(clp)) continue; clp->cl_rpcclient->cl_softerr = 0; @@ -942,6 +944,7 @@ static int _nfs4_pnfs_v4_ds_connect(struct nfs_server *mds_srv, struct nfs4_pnfs_ds *ds, unsigned int timeo, unsigned int retrans, + unsigned int nconnect, u32 minor_version, bool tightly_coupled) { @@ -1033,7 +1036,8 @@ static int _nfs4_pnfs_v4_ds_connect(struct nfs_server *mds_srv, clp = nfs4_set_ds_client(mds_srv, &da->da_addr, da->da_addrlen, ds_proto, - timeo, retrans, minor_version, + timeo, retrans, nconnect, + minor_version, tightly_coupled); if (IS_ERR(clp)) continue; @@ -1068,7 +1072,8 @@ static int _nfs4_pnfs_v4_ds_connect(struct nfs_server *mds_srv, */ int nfs4_pnfs_ds_connect(struct nfs_server *mds_srv, struct nfs4_pnfs_ds *ds, struct nfs4_deviceid_node *devid, unsigned int timeo, - unsigned int retrans, u32 version, u32 minor_version, + unsigned int retrans, unsigned int nconnect, + u32 version, u32 minor_version, bool tightly_coupled) { int err; @@ -1088,11 +1093,13 @@ int nfs4_pnfs_ds_connect(struct nfs_server *mds_srv, struct nfs4_pnfs_ds *ds, switch (version) { case 3: - err = _nfs4_pnfs_v3_ds_connect(mds_srv, ds, timeo, retrans); + err = _nfs4_pnfs_v3_ds_connect(mds_srv, ds, timeo, retrans, + nconnect); break; case 4: err = _nfs4_pnfs_v4_ds_connect(mds_srv, ds, timeo, retrans, - minor_version, tightly_coupled); + nconnect, minor_version, + tightly_coupled); break; default: dprintk("%s: unsupported DS version %d\n", __func__, version); From ea7f28f39bbf73f06acdd6c2f5bd5b6ba235eff8 Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Wed, 9 Sep 2026 13:11:47 -0400 Subject: [PATCH 0244/1352] NFSv4/flexfiles: report the intended opnum when DS connection setup fails When nfs4_ff_layout_prepare_ds() cannot establish a connection to a data server, it records a device error with opnum OP_ILLEGAL, since no operation ever made it to the wire. But servers can use the ff_ioerr4 opnum to distinguish failed read-class operations from write-class operations when deciding how to recover the affected mirror -- a mirror that may have missed writes needs to be brought back in sync, while one that merely failed to serve a read does not. OP_ILLEGAL gives the server nothing to act on. Every caller of nfs4_ff_layout_prepare_ds() knows which operation it was preparing to send, and the existing fail_return argument already divides the callers along the same boundary (false for READ, true for WRITE and COMMIT). Replace the boolean with the intended opnum, derive the layout return decision from it, and report it in the tracked device error instead of OP_ILLEGAL. Assisted-by: Claude:claude-fable-5 Signed-off-by: Benjamin Coddington Reviewed-by: Tigran Mkrtchyan Signed-off-by: Anna Schumaker --- fs/nfs/flexfilelayout/flexfilelayout.c | 13 ++++++++----- fs/nfs/flexfilelayout/flexfilelayout.h | 2 +- fs/nfs/flexfilelayout/flexfilelayoutdev.c | 14 ++++++++------ 3 files changed, 17 insertions(+), 12 deletions(-) diff --git a/fs/nfs/flexfilelayout/flexfilelayout.c b/fs/nfs/flexfilelayout/flexfilelayout.c index 2ea63fda0035b7..6fa6aa82d509da 100644 --- a/fs/nfs/flexfilelayout/flexfilelayout.c +++ b/fs/nfs/flexfilelayout/flexfilelayout.c @@ -883,7 +883,7 @@ ff_layout_choose_ds_for_read(struct pnfs_layout_segment *lseg, mirror_ds = ff_layout_get_mirror_ds(lseg->pls_layout, mirror, *dss_id); ds = nfs4_ff_layout_prepare_ds(lseg, mirror, mirror_ds, - *dss_id, false); + *dss_id, OP_READ); if (IS_ERR(ds)) { nfs4_ff_layout_put_deviceid(mirror_ds); ret = ERR_CAST(ds); @@ -1132,7 +1132,7 @@ ff_layout_pg_init_write(struct nfs_pageio_descriptor *pgio, mirror_ds = ff_layout_get_mirror_ds(pgio->pg_lseg->pls_layout, mirror, dss_id); ds = nfs4_ff_layout_prepare_ds(pgio->pg_lseg, mirror, - mirror_ds, dss_id, true); + mirror_ds, dss_id, OP_WRITE); if (IS_ERR(ds)) { nfs4_ff_layout_put_deviceid(mirror_ds); if (!ff_layout_no_fallback_to_mds(pgio->pg_lseg)) @@ -2188,7 +2188,8 @@ ff_layout_read_pagelist(struct nfs_pgio_header *hdr) mirror->dss_count, offset); mirror_ds = ff_layout_get_mirror_ds(lseg->pls_layout, mirror, dss_id); - ds = nfs4_ff_layout_prepare_ds(lseg, mirror, mirror_ds, dss_id, false); + ds = nfs4_ff_layout_prepare_ds(lseg, mirror, mirror_ds, dss_id, + OP_READ); if (IS_ERR(ds)) { ds_fatal_error = nfs_error_is_fatal(PTR_ERR(ds)); goto out_failed; @@ -2288,7 +2289,8 @@ ff_layout_write_pagelist(struct nfs_pgio_header *hdr, int sync) mirror->dss_count, offset); mirror_ds = ff_layout_get_mirror_ds(lseg->pls_layout, mirror, dss_id); - ds = nfs4_ff_layout_prepare_ds(lseg, mirror, mirror_ds, dss_id, true); + ds = nfs4_ff_layout_prepare_ds(lseg, mirror, mirror_ds, dss_id, + OP_WRITE); if (IS_ERR(ds)) { ds_fatal_error = nfs_error_is_fatal(PTR_ERR(ds)); goto out_failed; @@ -2398,7 +2400,8 @@ static int ff_layout_initiate_commit(struct nfs_commit_data *data, int how) mirror = FF_LAYOUT_COMP(lseg, idx); dss_id = calc_dss_id_from_commit(lseg, data->ds_commit_index); mirror_ds = ff_layout_get_mirror_ds(lseg->pls_layout, mirror, dss_id); - ds = nfs4_ff_layout_prepare_ds(lseg, mirror, mirror_ds, dss_id, true); + ds = nfs4_ff_layout_prepare_ds(lseg, mirror, mirror_ds, dss_id, + OP_COMMIT); if (IS_ERR(ds)) goto out_err; diff --git a/fs/nfs/flexfilelayout/flexfilelayout.h b/fs/nfs/flexfilelayout/flexfilelayout.h index 72b11034851a63..09c3cd6964fd63 100644 --- a/fs/nfs/flexfilelayout/flexfilelayout.h +++ b/fs/nfs/flexfilelayout/flexfilelayout.h @@ -256,7 +256,7 @@ nfs4_ff_layout_prepare_ds(struct pnfs_layout_segment *lseg, struct nfs4_ff_layout_mirror *mirror, struct nfs4_ff_layout_ds *mirror_ds, u32 dss_id, - bool fail_return); + enum nfs_opnum4 opnum); struct rpc_clnt * nfs4_ff_find_or_create_ds_client(const struct nfs4_ff_layout_ds *mirror_ds, diff --git a/fs/nfs/flexfilelayout/flexfilelayoutdev.c b/fs/nfs/flexfilelayout/flexfilelayoutdev.c index d52f485d0650e6..bc95628a811f21 100644 --- a/fs/nfs/flexfilelayout/flexfilelayoutdev.c +++ b/fs/nfs/flexfilelayout/flexfilelayoutdev.c @@ -379,7 +379,7 @@ ff_layout_get_mirror_ds(struct pnfs_layout_hdr *lo, * @mirror_ds: referenced device node for the stripe, from * ff_layout_get_mirror_ds() (may be an ERR_PTR) * @dss_id: DS stripe id to select stripe to use - * @fail_return: return layout on connect failure? + * @opnum: operation this connection is being prepared for * * Try to prepare a DS connection to accept an RPC call. This involves * selecting a mirror to use and connecting the client to it if it's not @@ -387,8 +387,10 @@ ff_layout_get_mirror_ds(struct pnfs_layout_hdr *lo, * * Since we only need a single functioning mirror to satisfy a read, we don't * want to return the layout if there is one. For writes though, any down - * mirror should result in a LAYOUTRETURN. @fail_return is how we distinguish - * between the two cases. + * mirror should result in a LAYOUTRETURN. @opnum is how we distinguish + * between the two cases. On failure, @opnum is also reported in the tracked + * device error so that the server can tell which class of I/O the client + * was unable to send to the mirror. * * Returns a pointer to a connected DS object on success or NULL on failure. */ @@ -397,7 +399,7 @@ nfs4_ff_layout_prepare_ds(struct pnfs_layout_segment *lseg, struct nfs4_ff_layout_mirror *mirror, struct nfs4_ff_layout_ds *mirror_ds, u32 dss_id, - bool fail_return) + enum nfs_opnum4 opnum) { struct nfs4_pnfs_ds *ds; struct inode *ino = lseg->pls_layout->plh_inode; @@ -448,9 +450,9 @@ nfs4_ff_layout_prepare_ds(struct pnfs_layout_segment *lseg, NULL : &mirror_ds->id_node, dss_id, lseg->pls_range.offset, lseg->pls_range.length, NFS4ERR_NXIO, - OP_ILLEGAL, GFP_NOIO); + opnum, GFP_NOIO); ff_layout_send_layouterror(lseg); - if (fail_return || !ff_layout_has_available_ds(lseg)) + if (opnum != OP_READ || !ff_layout_has_available_ds(lseg)) pnfs_error_mark_layout_for_return(ino, lseg); ds = ERR_PTR(status); out: From e2ebbf8f8c20ad65e53f81c77a815cc5bde6de17 Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Wed, 9 Sep 2026 13:11:48 -0400 Subject: [PATCH 0245/1352] pNFS: allow layout drivers to cancel I/O to a single device The ->cancel_io() layout operation cancels all in-flight I/O for a layout segment that is being returned. When the return is triggered by the failure of a single mirror instance, cancelling I/O to the healthy mirrors is both unnecessary and harmful: requests that were already transmitted cannot be un-sent, and will complete on the data servers after the layout has been returned. Give ->cancel_io() a device ID argument identifying the failed device, and thread it through pnfs_error_mark_layout_for_return() and pnfs_mark_matching_lsegs_return(). The flexfiles implementation compares it against the raw device ID from the layout, so that it can identify the failed mirror instance even when its device ID node was never instantiated. A NULL device ID preserves the existing cancel-everything behavior, and all callers pass NULL for now, so this patch makes no change in behavior. Assisted-by: Claude:claude-fable-5 Signed-off-by: Benjamin Coddington Reviewed-by: Tigran Mkrtchyan Signed-off-by: Anna Schumaker --- fs/nfs/blocklayout/blocklayout.c | 6 +++-- fs/nfs/callback_proc.c | 2 +- fs/nfs/filelayout/filelayout.c | 4 +-- fs/nfs/flexfilelayout/flexfilelayout.c | 16 +++++++----- fs/nfs/flexfilelayout/flexfilelayoutdev.c | 2 +- fs/nfs/pnfs.c | 32 ++++++++++++++--------- fs/nfs/pnfs.h | 14 ++++++---- 7 files changed, 46 insertions(+), 30 deletions(-) diff --git a/fs/nfs/blocklayout/blocklayout.c b/fs/nfs/blocklayout/blocklayout.c index d54a141a89b3bf..d86702e604f951 100644 --- a/fs/nfs/blocklayout/blocklayout.c +++ b/fs/nfs/blocklayout/blocklayout.c @@ -859,7 +859,8 @@ bl_pg_init_read(struct nfs_pageio_descriptor *pgio, struct nfs_page *req) if (pgio->pg_lseg && test_bit(NFS_LSEG_UNAVAILABLE, &pgio->pg_lseg->pls_flags)) { - pnfs_error_mark_layout_for_return(pgio->pg_inode, pgio->pg_lseg); + pnfs_error_mark_layout_for_return(pgio->pg_inode, pgio->pg_lseg, + NULL); pnfs_set_lo_fail(pgio->pg_lseg); nfs_pageio_reset_read_mds(pgio); } @@ -921,7 +922,8 @@ bl_pg_init_write(struct nfs_pageio_descriptor *pgio, struct nfs_page *req) if (pgio->pg_lseg && test_bit(NFS_LSEG_UNAVAILABLE, &pgio->pg_lseg->pls_flags)) { - pnfs_error_mark_layout_for_return(pgio->pg_inode, pgio->pg_lseg); + pnfs_error_mark_layout_for_return(pgio->pg_inode, pgio->pg_lseg, + NULL); pnfs_set_lo_fail(pgio->pg_lseg); nfs_pageio_reset_write_mds(pgio); } diff --git a/fs/nfs/callback_proc.c b/fs/nfs/callback_proc.c index c9b71dbae9ea65..9d836cb72e781d 100644 --- a/fs/nfs/callback_proc.c +++ b/fs/nfs/callback_proc.c @@ -292,7 +292,7 @@ static u32 initiate_file_draining(struct nfs_client *clp, switch (pnfs_mark_matching_lsegs_return(lo, &free_me_list, &args->cbl_range, be32_to_cpu(args->cbl_stateid.seqid), - args->cbl_layoutchanged)) { + args->cbl_layoutchanged, NULL)) { case 0: case -EBUSY: /* There are layout segments that need to be returned */ diff --git a/fs/nfs/filelayout/filelayout.c b/fs/nfs/filelayout/filelayout.c index 0d53277f972eb0..d1c08d529e02fd 100644 --- a/fs/nfs/filelayout/filelayout.c +++ b/fs/nfs/filelayout/filelayout.c @@ -186,7 +186,7 @@ static int filelayout_async_handle_error(struct rpc_task *task, dprintk("%s DS connection error %d\n", __func__, task->tk_status); nfs4_mark_deviceid_unavailable(devid); - pnfs_error_mark_layout_for_return(inode, lseg); + pnfs_error_mark_layout_for_return(inode, lseg, NULL); pnfs_set_lo_fail(lseg); rpc_wake_up(&tbl->slot_tbl_waitq); fallthrough; @@ -856,7 +856,7 @@ fl_pnfs_update_layout(struct inode *ino, status = filelayout_check_deviceid(lo, fl, gfp_flags); if (status) { - pnfs_error_mark_layout_for_return(ino, lseg); + pnfs_error_mark_layout_for_return(ino, lseg, NULL); pnfs_set_lo_fail(lseg); pnfs_put_lseg(lseg); lseg = NULL; diff --git a/fs/nfs/flexfilelayout/flexfilelayout.c b/fs/nfs/flexfilelayout/flexfilelayout.c index 6fa6aa82d509da..d27e0adb2709b0 100644 --- a/fs/nfs/flexfilelayout/flexfilelayout.c +++ b/fs/nfs/flexfilelayout/flexfilelayout.c @@ -1281,7 +1281,7 @@ static void ff_layout_resend_pnfs_read(struct nfs_pgio_header *hdr) mirror_ds = ff_layout_choose_any_ds_for_read(hdr->lseg, idx, &new_idx, hdr->args.offset, &dss_id); if (IS_ERR(mirror_ds)) { - pnfs_error_mark_layout_for_return(hdr->inode, hdr->lseg); + pnfs_error_mark_layout_for_return(hdr->inode, hdr->lseg, NULL); } else { nfs4_ff_layout_put_deviceid(mirror_ds); ff_layout_send_layouterror(hdr->lseg); @@ -1294,7 +1294,7 @@ static void ff_layout_reset_read(struct nfs_pgio_header *hdr) struct rpc_task *task = &hdr->task; pnfs_layoutcommit_inode(hdr->inode, false); - pnfs_error_mark_layout_for_return(hdr->inode, hdr->lseg); + pnfs_error_mark_layout_for_return(hdr->inode, hdr->lseg, NULL); if (!test_and_set_bit(NFS_IOHDR_REDO, &hdr->flags)) { dprintk("%s Reset task %5u for i/o through MDS " @@ -1594,7 +1594,7 @@ static void ff_layout_io_track_ds_error(struct pnfs_layout_segment *lseg, fallthrough; default: pnfs_error_mark_layout_for_return(lseg->pls_layout->plh_inode, - lseg); + lseg, NULL); } out: @@ -2256,7 +2256,7 @@ ff_layout_read_pagelist(struct nfs_pgio_header *hdr) * FF_FLAGS_NO_IO_THRU_MDS: force fresh LAYOUTGET, * never fall through to MDS I/O. */ - pnfs_error_mark_layout_for_return(hdr->inode, lseg); + pnfs_error_mark_layout_for_return(hdr->inode, lseg, NULL); return PNFS_TRY_AGAIN; } trace_pnfs_mds_fallback_read_pagelist(hdr->inode, @@ -2359,7 +2359,7 @@ ff_layout_write_pagelist(struct nfs_pgio_header *hdr, int sync) * FF_FLAGS_NO_IO_THRU_MDS: force fresh LAYOUTGET, * never fall through to MDS I/O. */ - pnfs_error_mark_layout_for_return(hdr->inode, lseg); + pnfs_error_mark_layout_for_return(hdr->inode, lseg, NULL); return PNFS_TRY_AGAIN; } trace_pnfs_mds_fallback_write_pagelist(hdr->inode, @@ -2490,7 +2490,8 @@ static bool ff_layout_match_io(const struct rpc_task *task, const void *data) return false; } -static void ff_layout_cancel_io(struct pnfs_layout_segment *lseg) +static void ff_layout_cancel_io(struct pnfs_layout_segment *lseg, + const struct nfs4_deviceid *devid) { struct nfs4_ff_layout_segment *flseg = FF_LAYOUT_LSEG(lseg); struct nfs4_ff_layout_mirror *mirror; @@ -2503,6 +2504,9 @@ static void ff_layout_cancel_io(struct pnfs_layout_segment *lseg) for (idx = 0; idx < flseg->mirror_array_cnt; idx++) { mirror = flseg->mirror_array[idx]; for (dss_id = 0; dss_id < mirror->dss_count; dss_id++) { + if (devid && memcmp(&mirror->dss[dss_id].devid, devid, + sizeof(*devid)) != 0) + continue; rcu_read_lock(); mirror_ds = rcu_dereference(mirror->dss[dss_id].mirror_ds); if (IS_ERR_OR_NULL(mirror_ds) || diff --git a/fs/nfs/flexfilelayout/flexfilelayoutdev.c b/fs/nfs/flexfilelayout/flexfilelayoutdev.c index bc95628a811f21..7bb0f2094e9df7 100644 --- a/fs/nfs/flexfilelayout/flexfilelayoutdev.c +++ b/fs/nfs/flexfilelayout/flexfilelayoutdev.c @@ -453,7 +453,7 @@ nfs4_ff_layout_prepare_ds(struct pnfs_layout_segment *lseg, opnum, GFP_NOIO); ff_layout_send_layouterror(lseg); if (opnum != OP_READ || !ff_layout_has_available_ds(lseg)) - pnfs_error_mark_layout_for_return(ino, lseg); + pnfs_error_mark_layout_for_return(ino, lseg, NULL); ds = ERR_PTR(status); out: return ds; diff --git a/fs/nfs/pnfs.c b/fs/nfs/pnfs.c index 79f671355e5b48..93a0852e1e8a66 100644 --- a/fs/nfs/pnfs.c +++ b/fs/nfs/pnfs.c @@ -433,7 +433,7 @@ bool nfs4_layout_refresh_old_stateid(nfs4_stateid *dst, } /* Try to update the seqid to the most recent */ err = pnfs_mark_matching_lsegs_return(lo, &head, &range, 0, - true); + true, NULL); if (err != -EBUSY) { dst->seqid = lo->plh_stateid.seqid; *dst_range = range; @@ -487,7 +487,8 @@ static int pnfs_mark_layout_stateid_return(struct pnfs_layout_hdr *lo, .length = NFS4_MAX_UINT64, }; - return pnfs_mark_matching_lsegs_return(lo, lseg_list, &range, seq, true); + return pnfs_mark_matching_lsegs_return(lo, lseg_list, &range, seq, true, + NULL); } static int @@ -525,7 +526,7 @@ pnfs_layout_io_set_failed(struct pnfs_layout_hdr *lo, u32 iomode) spin_lock(&inode->i_lock); pnfs_layout_set_fail_bit(lo, pnfs_iomode_to_fail_bit(iomode)); - pnfs_mark_matching_lsegs_return(lo, &head, &range, 0, true); + pnfs_mark_matching_lsegs_return(lo, &head, &range, 0, true, NULL); spin_unlock(&inode->i_lock); pnfs_free_lseg_list(&head); dprintk("%s Setting layout IOMODE_%s fail bit\n", __func__, @@ -740,7 +741,7 @@ pnfs_mark_matching_lsegs_invalid(struct pnfs_layout_hdr *lo, if (mark_lseg_invalid(lseg, tmp_list)) continue; remaining++; - pnfs_lseg_cancel_io(server, lseg); + pnfs_lseg_cancel_io(server, lseg, NULL); } dprintk("%s:Return %i\n", __func__, remaining); return remaining; @@ -1476,7 +1477,7 @@ _pnfs_return_layout(struct inode *ino) } valid_layout = pnfs_layout_is_valid(lo); pnfs_clear_layoutcommit(ino, &tmp_list); - pnfs_mark_matching_lsegs_return(lo, &tmp_list, &range, 0, true); + pnfs_mark_matching_lsegs_return(lo, &tmp_list, &range, 0, true, NULL); /* Don't send a LAYOUTRETURN if list was initially empty */ @@ -2658,7 +2659,8 @@ pnfs_layout_process(struct nfs4_layoutget *lgp) .iomode = IOMODE_ANY, .length = NFS4_MAX_UINT64, }; - pnfs_mark_matching_lsegs_return(lo, &free_me, &range, 0, true); + pnfs_mark_matching_lsegs_return(lo, &free_me, &range, 0, true, + NULL); goto out_forget; } else { /* We have a completely new layout */ @@ -2691,6 +2693,7 @@ pnfs_layout_process(struct nfs4_layoutget *lgp) * @return_range: describe layout segment ranges to be returned * @seq: stateid seqid to match * @cancel_io: signal io be cancelled + * @devid: only cancel io directed at this device (all devices if NULL) * * This function is mainly intended for use by layoutrecall. It attempts * to free the layout segment immediately, or else to mark it for return @@ -2705,7 +2708,8 @@ int pnfs_mark_matching_lsegs_return(struct pnfs_layout_hdr *lo, struct list_head *tmp_list, const struct pnfs_layout_range *return_range, - u32 seq, bool cancel_io) + u32 seq, bool cancel_io, + const struct nfs4_deviceid *devid) { struct pnfs_layout_segment *lseg, *next; struct nfs_server *server = NFS_SERVER(lo->plh_inode); @@ -2732,7 +2736,7 @@ pnfs_mark_matching_lsegs_return(struct pnfs_layout_hdr *lo, remaining++; set_bit(NFS_LSEG_LAYOUTRETURN, &lseg->pls_flags); if (cancel_io) - pnfs_lseg_cancel_io(server, lseg); + pnfs_lseg_cancel_io(server, lseg, devid); } if (remaining) { @@ -2750,7 +2754,8 @@ pnfs_mark_matching_lsegs_return(struct pnfs_layout_hdr *lo, static void pnfs_mark_layout_for_return(struct inode *inode, - const struct pnfs_layout_range *range) + const struct pnfs_layout_range *range, + const struct nfs4_deviceid *devid) { struct pnfs_layout_hdr *lo; bool return_now = false; @@ -2768,7 +2773,7 @@ pnfs_mark_layout_for_return(struct inode *inode, * for how it works. */ if (pnfs_mark_matching_lsegs_return(lo, &lo->plh_return_segs, range, 0, - true) != -EBUSY) { + true, devid) != -EBUSY) { const struct cred *cred; nfs4_stateid stateid; enum pnfs_iomode iomode; @@ -2785,7 +2790,8 @@ pnfs_mark_layout_for_return(struct inode *inode, } void pnfs_error_mark_layout_for_return(struct inode *inode, - struct pnfs_layout_segment *lseg) + struct pnfs_layout_segment *lseg, + const struct nfs4_deviceid *devid) { struct pnfs_layout_range range = { .iomode = lseg->pls_range.iomode, @@ -2793,7 +2799,7 @@ void pnfs_error_mark_layout_for_return(struct inode *inode, .length = NFS4_MAX_UINT64, }; - pnfs_mark_layout_for_return(inode, &range); + pnfs_mark_layout_for_return(inode, &range, devid); } EXPORT_SYMBOL_GPL(pnfs_error_mark_layout_for_return); @@ -2883,7 +2889,7 @@ static int pnfs_layout_return_unused_byserver(struct nfs_server *server, pnfs_get_layout_hdr(lo); pnfs_set_plh_return_info(lo, range->iomode, 0); if (pnfs_mark_matching_lsegs_return(lo, &lo->plh_return_segs, - range, 0, true) != 0 || + range, 0, true, NULL) != 0 || !pnfs_prepare_layoutreturn(lo, &stateid, &cred, &iomode)) { spin_unlock(&inode->i_lock); rcu_read_unlock(); diff --git a/fs/nfs/pnfs.h b/fs/nfs/pnfs.h index 5cde5db63d29fe..2774d4adf4c317 100644 --- a/fs/nfs/pnfs.h +++ b/fs/nfs/pnfs.h @@ -197,7 +197,8 @@ struct pnfs_layoutdriver_type { int (*prepare_layoutcommit) (struct nfs4_layoutcommit_args *args); int (*prepare_layoutstats) (struct nfs42_layoutstat_args *args); - void (*cancel_io)(struct pnfs_layout_segment *lseg); + void (*cancel_io)(struct pnfs_layout_segment *lseg, + const struct nfs4_deviceid *devid); }; struct pnfs_commit_ops { @@ -320,7 +321,8 @@ int pnfs_mark_matching_lsegs_invalid(struct pnfs_layout_hdr *lo, int pnfs_mark_matching_lsegs_return(struct pnfs_layout_hdr *lo, struct list_head *tmp_list, const struct pnfs_layout_range *recall_range, - u32 seq, bool cancel_io); + u32 seq, bool cancel_io, + const struct nfs4_deviceid *devid); int pnfs_mark_layout_stateid_invalid(struct pnfs_layout_hdr *lo, struct list_head *lseg_list); bool pnfs_roc(struct inode *ino, struct nfs4_layoutreturn_args *args, @@ -370,7 +372,8 @@ int pnfs_read_done_resend_to_mds(struct nfs_pgio_header *); int pnfs_write_done_resend_to_mds(struct nfs_pgio_header *); struct nfs4_threshold *pnfs_mdsthreshold_alloc(void); void pnfs_error_mark_layout_for_return(struct inode *inode, - struct pnfs_layout_segment *lseg); + struct pnfs_layout_segment *lseg, + const struct nfs4_deviceid *devid); void pnfs_layout_return_unused_byclid(struct nfs_client *clp, enum pnfs_iomode iomode); void pnfs_layout_reresolve_deviceid_byclid(struct nfs_client *clp, @@ -774,10 +777,11 @@ pnfs_lseg_request_intersecting(struct pnfs_layout_segment *lseg, struct nfs_page } static inline void pnfs_lseg_cancel_io(struct nfs_server *server, - struct pnfs_layout_segment *lseg) + struct pnfs_layout_segment *lseg, + const struct nfs4_deviceid *devid) { if (server->pnfs_curr_ld->cancel_io) - server->pnfs_curr_ld->cancel_io(lseg); + server->pnfs_curr_ld->cancel_io(lseg, devid); } extern unsigned int layoutstats_timer; From 792c80e8240fc05642fe5e35d988e25bd0eceba2 Mon Sep 17 00:00:00 2001 From: Benjamin Coddington Date: Wed, 9 Sep 2026 13:11:49 -0400 Subject: [PATCH 0246/1352] NFSv4/flexfiles: only cancel I/O to a failed mirror instance When an error causes the flexfiles driver to return a layout, ff_layout_cancel_io() kills every in-flight RPC for the layout segment, across all mirror instances. Cancelled requests that had already been transmitted to a healthy data server cannot be un-sent: they complete on the data server after the client has sent its LAYOUTRETURN, and the metadata server then observes writes to a file for which no write layout is outstanding. RFC 8881 Section 20.3.4 recommends that the client wait for the response from in-process or in-flight READ, WRITE, or COMMIT operations before returning the layout, and the machinery for that wait already exists: the LAYOUTRETURN is deferred until every request drops its layout segment reference, and requests that have not yet been transmitted exit at RPC prepare time once the segment has been invalidated. Cancellation is only needed to avoid waiting forever on a device that will never answer. Pass the failed instance's device ID when marking the layout for return, so that ff_layout_cancel_io() cancels only I/O directed at the device we have given up on. In-flight I/O to the remaining healthy instances drains normally -- typically within a round trip -- before the LAYOUTRETURN is sent. If a spared instance turns out to be unresponsive, its requests fail with their own device error, and the resulting layout return cancels its I/O in turn. Layout recalls with clora_changed set, bulk returns, and layout revocations continue to cancel I/O to every device, as do error paths where no single failed device can be identified. Assisted-by: Claude:claude-fable-5 Signed-off-by: Benjamin Coddington Reviewed-by: Tigran Mkrtchyan Signed-off-by: Anna Schumaker --- fs/nfs/flexfilelayout/flexfilelayout.c | 9 ++++++--- fs/nfs/flexfilelayout/flexfilelayoutdev.c | 3 ++- 2 files changed, 8 insertions(+), 4 deletions(-) diff --git a/fs/nfs/flexfilelayout/flexfilelayout.c b/fs/nfs/flexfilelayout/flexfilelayout.c index d27e0adb2709b0..94cc324b591f49 100644 --- a/fs/nfs/flexfilelayout/flexfilelayout.c +++ b/fs/nfs/flexfilelayout/flexfilelayout.c @@ -1594,7 +1594,8 @@ static void ff_layout_io_track_ds_error(struct pnfs_layout_segment *lseg, fallthrough; default: pnfs_error_mark_layout_for_return(lseg->pls_layout->plh_inode, - lseg, NULL); + lseg, + &mirror->dss[dss_id].devid); } out: @@ -2256,7 +2257,8 @@ ff_layout_read_pagelist(struct nfs_pgio_header *hdr) * FF_FLAGS_NO_IO_THRU_MDS: force fresh LAYOUTGET, * never fall through to MDS I/O. */ - pnfs_error_mark_layout_for_return(hdr->inode, lseg, NULL); + pnfs_error_mark_layout_for_return(hdr->inode, lseg, + &mirror->dss[dss_id].devid); return PNFS_TRY_AGAIN; } trace_pnfs_mds_fallback_read_pagelist(hdr->inode, @@ -2359,7 +2361,8 @@ ff_layout_write_pagelist(struct nfs_pgio_header *hdr, int sync) * FF_FLAGS_NO_IO_THRU_MDS: force fresh LAYOUTGET, * never fall through to MDS I/O. */ - pnfs_error_mark_layout_for_return(hdr->inode, lseg, NULL); + pnfs_error_mark_layout_for_return(hdr->inode, lseg, + &mirror->dss[dss_id].devid); return PNFS_TRY_AGAIN; } trace_pnfs_mds_fallback_write_pagelist(hdr->inode, diff --git a/fs/nfs/flexfilelayout/flexfilelayoutdev.c b/fs/nfs/flexfilelayout/flexfilelayoutdev.c index 7bb0f2094e9df7..6165c41fcf62ed 100644 --- a/fs/nfs/flexfilelayout/flexfilelayoutdev.c +++ b/fs/nfs/flexfilelayout/flexfilelayoutdev.c @@ -453,7 +453,8 @@ nfs4_ff_layout_prepare_ds(struct pnfs_layout_segment *lseg, opnum, GFP_NOIO); ff_layout_send_layouterror(lseg); if (opnum != OP_READ || !ff_layout_has_available_ds(lseg)) - pnfs_error_mark_layout_for_return(ino, lseg, NULL); + pnfs_error_mark_layout_for_return(ino, lseg, + &mirror->dss[dss_id].devid); ds = ERR_PTR(status); out: return ds; From 6ecf1b509f15b0e005dca548ef906d8842b22449 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?J=C3=A9r=C3=A9my=20Jean?= Date: Mon, 14 Sep 2026 20:42:38 +0000 Subject: [PATCH 0247/1352] NFS: filelayout: calculate dense stripe width in 64 bits MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit filelayout_get_dense_offset() calculates the stripe width from the stripe unit and count, where both values are provided by the server. Their product is stored on 32 bits, and can wrap to zero despite passing individual checks. For example, a 1 GiB stripe unit (0x40000000) and four stripes (4) cause the following div_u64() to raise a divide error during the first read through the dense layout. Update u32 to u64 and use div64_u64() instead of div_u64(). stripe_unit is a u32 and stripe_count is limited to 4096, so the product cannot overflow a u64. Fixes: cfe7f4120f8b ("NFSv4.1: filelayout i/o helpers") Assisted-by: Codex:gpt-5 Signed-off-by: Jérémy Jean Signed-off-by: Anna Schumaker --- fs/nfs/filelayout/filelayout.c | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/fs/nfs/filelayout/filelayout.c b/fs/nfs/filelayout/filelayout.c index d1c08d529e02fd..53808e42546a05 100644 --- a/fs/nfs/filelayout/filelayout.c +++ b/fs/nfs/filelayout/filelayout.c @@ -55,12 +55,13 @@ static loff_t filelayout_get_dense_offset(struct nfs4_filelayout_segment *flseg, loff_t offset) { - u32 stripe_width = flseg->stripe_unit * flseg->dsaddr->stripe_count; + u64 stripe_width = (u64)flseg->stripe_unit * + flseg->dsaddr->stripe_count; u64 stripe_no; u32 rem; offset -= flseg->pattern_offset; - stripe_no = div_u64(offset, stripe_width); + stripe_no = div64_u64(offset, stripe_width); div_u64_rem(offset, flseg->stripe_unit, &rem); return stripe_no * flseg->stripe_unit + rem; From 40564c8c7bb383c95739b5af62e90766b5d32759 Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Wed, 16 Sep 2026 11:29:32 +0300 Subject: [PATCH 0248/1352] drm/i915/lspcon: use u8 variable to hold 1-byte DPCD reads Passing a pointer to a u32 variable for a 1-byte reads is error prone, and only works because it's always initialized to 0 before the read, and the code's running on little-endian CPUs. Switch to u8 variables. In _lspcon_write_avi_infoframe_mca(), add a separate variable to hold the number of bytes written instead of reusing the same variable for two purposes. Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/66a74267a3da7619dda8dc2b1bb1bfcedc25df0d.1789547331.git.jani.nikula@intel.com Signed-off-by: Jani Nikula --- drivers/gpu/drm/i915/display/intel_lspcon.c | 14 ++++++++------ 1 file changed, 8 insertions(+), 6 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_lspcon.c b/drivers/gpu/drm/i915/display/intel_lspcon.c index 9ceabbc981a169..b306bf5f61309c 100644 --- a/drivers/gpu/drm/i915/display/intel_lspcon.c +++ b/drivers/gpu/drm/i915/display/intel_lspcon.c @@ -435,14 +435,14 @@ static bool _lspcon_write_avi_infoframe_parade(struct drm_dp_aux *aux, static bool _lspcon_write_avi_infoframe_mca(struct drm_dp_aux *aux, const u8 *buffer, ssize_t len) { - int ret; - u32 val = 0; + int ret, written = 0; u32 retry; u16 reg; const u8 *data = buffer; + u8 val; reg = LSPCON_MCA_AVI_IF_WRITE_OFFSET; - while (val < len) { + while (written < len) { /* DPCD write for AVI IF can fail on a slow FW day, so retry */ for (retry = 0; retry < 5; retry++) { ret = drm_dp_dpcd_write(aux, reg, (void *)data, 1); @@ -456,7 +456,9 @@ static bool _lspcon_write_avi_infoframe_mca(struct drm_dp_aux *aux, return false; } } - val++; reg++; data++; + written++; + reg++; + data++; } val = 0; @@ -612,8 +614,8 @@ void lspcon_set_infoframes(struct intel_encoder *encoder, static bool _lspcon_read_avi_infoframe_enabled_mca(struct drm_dp_aux *aux) { int ret; - u32 val = 0; u16 reg = LSPCON_MCA_AVI_IF_CTRL; + u8 val; ret = drm_dp_dpcd_read(aux, reg, &val, 1); if (ret < 0) { @@ -627,8 +629,8 @@ static bool _lspcon_read_avi_infoframe_enabled_mca(struct drm_dp_aux *aux) static bool _lspcon_read_avi_infoframe_enabled_parade(struct drm_dp_aux *aux) { int ret; - u32 val = 0; u16 reg = LSPCON_PARADE_AVI_IF_CTRL; + u8 val; ret = drm_dp_dpcd_read(aux, reg, &val, 1); if (ret < 0) { From a2d0701df0fcb4af26db7c576b7ee5f99e4e8fd4 Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Wed, 16 Sep 2026 11:29:33 +0300 Subject: [PATCH 0249/1352] drm/i915/lspcon: switch to drm_dp_dpcd_{read_byte, write_byte, write_data} Switch to the modern DPCD access functions that return negative error codes on errors, -EIO for incomplete access, and 0 on success. Use 1-byte accessors where sensible. Use local ret variables for the return value instead of checking inline. Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/a890746215c6c7a1e36a6bc880ae644dd06f1956.1789547331.git.jani.nikula@intel.com Signed-off-by: Jani Nikula --- drivers/gpu/drm/i915/display/intel_lspcon.c | 34 +++++++++------------ 1 file changed, 15 insertions(+), 19 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_lspcon.c b/drivers/gpu/drm/i915/display/intel_lspcon.c index b306bf5f61309c..07b5a9d010ff40 100644 --- a/drivers/gpu/drm/i915/display/intel_lspcon.c +++ b/drivers/gpu/drm/i915/display/intel_lspcon.c @@ -137,9 +137,7 @@ bool intel_lspcon_detect_hdr_capability(struct intel_digital_port *dig_port) u8 hdr_caps; int ret; - ret = drm_dp_dpcd_read(&intel_dp->aux, get_hdr_status_reg(lspcon), - &hdr_caps, 1); - + ret = drm_dp_dpcd_read_byte(&intel_dp->aux, get_hdr_status_reg(lspcon), &hdr_caps); if (ret < 0) { drm_dbg_kms(display->drm, "HDR capability detection failed\n"); lspcon->hdr_supported = false; @@ -244,10 +242,11 @@ static bool lspcon_wake_native_aux_ch(struct intel_lspcon *lspcon) { struct intel_dp *intel_dp = lspcon_to_intel_dp(lspcon); struct intel_display *display = to_intel_display(intel_dp); + int ret; u8 rev; - if (drm_dp_dpcd_readb(&lspcon_to_intel_dp(lspcon)->aux, DP_DPCD_REV, - &rev) != 1) { + ret = drm_dp_dpcd_read_byte(&lspcon_to_intel_dp(lspcon)->aux, DP_DPCD_REV, &rev); + if (ret < 0) { drm_dbg_kms(display->drm, "Native AUX CH down\n"); return false; } @@ -331,15 +330,14 @@ static bool lspcon_parade_fw_ready(struct drm_dp_aux *aux) { u8 avi_if_ctrl; u8 retry; - ssize_t ret; + int ret; /* Check if LSPCON FW is ready for data */ for (retry = 0; retry < 5; retry++) { if (retry) usleep_range(200, 300); - ret = drm_dp_dpcd_read(aux, LSPCON_PARADE_AVI_IF_CTRL, - &avi_if_ctrl, 1); + ret = drm_dp_dpcd_read_byte(aux, LSPCON_PARADE_AVI_IF_CTRL, &avi_if_ctrl); if (ret < 0) { drm_err(aux->drm_dev, "Failed to read AVI IF control\n"); return false; @@ -371,7 +369,7 @@ static bool _lspcon_parade_write_infoframe_blocks(struct drm_dp_aux *aux, reg = LSPCON_PARADE_AVI_IF_WRITE_OFFSET; data = avi_buf + block_count * 8; - ret = drm_dp_dpcd_write(aux, reg, data, 8); + ret = drm_dp_dpcd_write_data(aux, reg, data, 8); if (ret < 0) { drm_err(aux->drm_dev, "Failed to write AVI IF block %d\n", block_count); @@ -386,7 +384,7 @@ static bool _lspcon_parade_write_infoframe_blocks(struct drm_dp_aux *aux, */ reg = LSPCON_PARADE_AVI_IF_CTRL; avi_if_ctrl = LSPCON_PARADE_AVI_IF_KICKOFF | block_count; - ret = drm_dp_dpcd_write(aux, reg, &avi_if_ctrl, 1); + ret = drm_dp_dpcd_write_byte(aux, reg, avi_if_ctrl); if (ret < 0) { drm_err(aux->drm_dev, "Failed to update (0x%x), block %d\n", reg, block_count); @@ -445,8 +443,8 @@ static bool _lspcon_write_avi_infoframe_mca(struct drm_dp_aux *aux, while (written < len) { /* DPCD write for AVI IF can fail on a slow FW day, so retry */ for (retry = 0; retry < 5; retry++) { - ret = drm_dp_dpcd_write(aux, reg, (void *)data, 1); - if (ret == 1) { + ret = drm_dp_dpcd_write_byte(aux, reg, *data); + if (!ret) { break; } else if (retry < 4) { mdelay(50); @@ -461,9 +459,8 @@ static bool _lspcon_write_avi_infoframe_mca(struct drm_dp_aux *aux, data++; } - val = 0; reg = LSPCON_MCA_AVI_IF_CTRL; - ret = drm_dp_dpcd_read(aux, reg, &val, 1); + ret = drm_dp_dpcd_read_byte(aux, reg, &val); if (ret < 0) { drm_err(aux->drm_dev, "DPCD read failed, address 0x%x\n", reg); return false; @@ -473,14 +470,13 @@ static bool _lspcon_write_avi_infoframe_mca(struct drm_dp_aux *aux, val &= ~LSPCON_MCA_AVI_IF_HANDLED; val |= LSPCON_MCA_AVI_IF_KICKOFF; - ret = drm_dp_dpcd_write(aux, reg, &val, 1); + ret = drm_dp_dpcd_write_byte(aux, reg, val); if (ret < 0) { drm_err(aux->drm_dev, "DPCD read failed, address 0x%x\n", reg); return false; } - val = 0; - ret = drm_dp_dpcd_read(aux, reg, &val, 1); + ret = drm_dp_dpcd_read_byte(aux, reg, &val); if (ret < 0) { drm_err(aux->drm_dev, "DPCD read failed, address 0x%x\n", reg); return false; @@ -617,7 +613,7 @@ static bool _lspcon_read_avi_infoframe_enabled_mca(struct drm_dp_aux *aux) u16 reg = LSPCON_MCA_AVI_IF_CTRL; u8 val; - ret = drm_dp_dpcd_read(aux, reg, &val, 1); + ret = drm_dp_dpcd_read_byte(aux, reg, &val); if (ret < 0) { drm_err(aux->drm_dev, "DPCD read failed, address 0x%x\n", reg); return false; @@ -632,7 +628,7 @@ static bool _lspcon_read_avi_infoframe_enabled_parade(struct drm_dp_aux *aux) u16 reg = LSPCON_PARADE_AVI_IF_CTRL; u8 val; - ret = drm_dp_dpcd_read(aux, reg, &val, 1); + ret = drm_dp_dpcd_read_byte(aux, reg, &val); if (ret < 0) { drm_err(aux->drm_dev, "DPCD read failed, address 0x%x\n", reg); return false; From 8251875cd96257af98e2853788c58d7178d462a9 Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Wed, 16 Sep 2026 11:29:34 +0300 Subject: [PATCH 0250/1352] drm/i915/alpm: switch to drm_dp_dpcd_{read_byte, write_byte} Switch to the modern DPCD access functions that return negative error codes on errors, -EIO for incomplete access, and 0 on success. Rename an int r to ret while at it for clarity. Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/67868fd8a8c8e057ba82c4b142fabae2d0fa52d6.1789547331.git.jani.nikula@intel.com Signed-off-by: Jani Nikula --- drivers/gpu/drm/i915/display/intel_alpm.c | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_alpm.c b/drivers/gpu/drm/i915/display/intel_alpm.c index 10943539bc7c0b..9ef9d0e45258fd 100644 --- a/drivers/gpu/drm/i915/display/intel_alpm.c +++ b/drivers/gpu/drm/i915/display/intel_alpm.c @@ -628,7 +628,7 @@ void intel_alpm_enable_sink(struct intel_dp *intel_dp, intel_alpm_aux_less_wake_supported(intel_dp))) val |= DP_ALPM_MODE_AUX_LESS; - drm_dp_dpcd_writeb(&intel_dp->aux, DP_RECEIVER_ALPM_CONFIG, val); + drm_dp_dpcd_write_byte(&intel_dp->aux, DP_RECEIVER_ALPM_CONFIG, val); } void intel_alpm_lobf_enable(const struct intel_crtc_state *new_crtc_state) @@ -751,11 +751,11 @@ bool intel_alpm_get_error(struct intel_dp *intel_dp) { struct intel_display *display = to_intel_display(intel_dp); struct drm_dp_aux *aux = &intel_dp->aux; + int ret; u8 val; - int r; - r = drm_dp_dpcd_readb(aux, DP_RECEIVER_ALPM_STATUS, &val); - if (r != 1) { + ret = drm_dp_dpcd_read_byte(aux, DP_RECEIVER_ALPM_STATUS, &val); + if (ret < 0) { drm_err(display->drm, "Error reading ALPM status\n"); return true; } @@ -764,7 +764,7 @@ bool intel_alpm_get_error(struct intel_dp *intel_dp) drm_dbg_kms(display->drm, "ALPM lock timeout error\n"); /* Clearing error */ - drm_dp_dpcd_writeb(aux, DP_RECEIVER_ALPM_STATUS, val); + drm_dp_dpcd_write_byte(aux, DP_RECEIVER_ALPM_STATUS, val); return true; } From d613fc2cbe609fa55e1c14218f543ae82595c076 Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Wed, 16 Sep 2026 11:29:35 +0300 Subject: [PATCH 0251/1352] drm/i915/ddi: switch to drm_dp_dpcd_write_byte() Switch to the modern DPCD access functions that return negative error codes on errors, -EIO for incomplete access, and 0 on success. Use local ret variables for the return value instead of checking inline where sensible. This improves clarity. Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/f0652fabfd6143680decd826ec74cd5d1116195f.1789547331.git.jani.nikula@intel.com Signed-off-by: Jani Nikula --- drivers/gpu/drm/i915/display/intel_ddi.c | 16 ++++++++++------ 1 file changed, 10 insertions(+), 6 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_ddi.c b/drivers/gpu/drm/i915/display/intel_ddi.c index 3cdb06e81130b7..7efd4c4d499444 100644 --- a/drivers/gpu/drm/i915/display/intel_ddi.c +++ b/drivers/gpu/drm/i915/display/intel_ddi.c @@ -2332,12 +2332,14 @@ static void intel_dp_sink_set_msa_timing_par_ignore_state(struct intel_dp *intel bool enable) { struct intel_display *display = to_intel_display(intel_dp); + int ret; if (!crtc_state->vrr.enable) return; - if (drm_dp_dpcd_writeb(&intel_dp->aux, DP_DOWNSPREAD_CTRL, - enable ? DP_MSA_TIMING_PAR_IGNORE_EN : 0) <= 0) + ret = drm_dp_dpcd_write_byte(&intel_dp->aux, DP_DOWNSPREAD_CTRL, + enable ? DP_MSA_TIMING_PAR_IGNORE_EN : 0); + if (ret < 0) drm_dbg_kms(display->drm, "Failed to %s MSA_TIMING_PAR_IGNORE in the sink\n", str_enable_disable(enable)); @@ -2348,18 +2350,20 @@ static void intel_dp_sink_set_fec_ready(struct intel_dp *intel_dp, bool enable) { struct intel_display *display = to_intel_display(intel_dp); + int ret; if (!crtc_state->fec_enable) return; - if (drm_dp_dpcd_writeb(&intel_dp->aux, DP_FEC_CONFIGURATION, - enable ? DP_FEC_READY : 0) <= 0) + ret = drm_dp_dpcd_write_byte(&intel_dp->aux, DP_FEC_CONFIGURATION, + enable ? DP_FEC_READY : 0); + if (ret < 0) drm_dbg_kms(display->drm, "Failed to set FEC_READY to %s in the sink\n", str_enabled_disabled(enable)); if (enable && - drm_dp_dpcd_writeb(&intel_dp->aux, DP_FEC_STATUS, - DP_FEC_DECODE_EN_DETECTED | DP_FEC_DECODE_DIS_DETECTED) <= 0) + drm_dp_dpcd_write_byte(&intel_dp->aux, DP_FEC_STATUS, + DP_FEC_DECODE_EN_DETECTED | DP_FEC_DECODE_DIS_DETECTED) < 0) drm_dbg_kms(display->drm, "Failed to clear FEC detected flags\n"); } From 551dae8362eb18ef9233cef3ccc840502fb6e92c Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Wed, 16 Sep 2026 11:29:36 +0300 Subject: [PATCH 0252/1352] drm/i915/lspcon: switch to drm_dp_dpcd_{read_byte, read_data, write_byte, write_data}() Switch to the modern DPCD access functions that return negative error codes on errors, -EIO for incomplete access, and 0 on success. Use local ret variables for the return value instead of checking inline. Use %pe and ERR_PTR() for logging the errors. Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/f16cb57607634d411a42ca6cd80b10a713d7677f.1789547331.git.jani.nikula@intel.com Signed-off-by: Jani Nikula --- .../drm/i915/display/intel_dp_aux_backlight.c | 51 ++++++++++--------- 1 file changed, 27 insertions(+), 24 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_dp_aux_backlight.c b/drivers/gpu/drm/i915/display/intel_dp_aux_backlight.c index 266e042e00237c..f30da5f7e3d2ee 100644 --- a/drivers/gpu/drm/i915/display/intel_dp_aux_backlight.c +++ b/drivers/gpu/drm/i915/display/intel_dp_aux_backlight.c @@ -120,8 +120,8 @@ intel_dp_aux_supports_hdr_backlight(struct intel_connector *connector) intel_dp_wait_source_oui(intel_dp); - ret = drm_dp_dpcd_read(aux, INTEL_EDP_HDR_TCON_CAP0, tcon_cap, sizeof(tcon_cap)); - if (ret != sizeof(tcon_cap)) + ret = drm_dp_dpcd_read_data(aux, INTEL_EDP_HDR_TCON_CAP0, tcon_cap, sizeof(tcon_cap)); + if (ret < 0) return false; drm_dbg_kms(display->drm, @@ -178,8 +178,10 @@ intel_dp_aux_hdr_get_backlight(struct intel_connector *connector, enum pipe pipe struct intel_dp *intel_dp = enc_to_intel_dp(connector->encoder); u8 tmp; u8 buf[2] = {}; + int ret; - if (drm_dp_dpcd_readb(&intel_dp->aux, INTEL_EDP_HDR_GETSET_CTRL_PARAMS, &tmp) != 1) { + ret = drm_dp_dpcd_read_byte(&intel_dp->aux, INTEL_EDP_HDR_GETSET_CTRL_PARAMS, &tmp); + if (ret < 0) { drm_err(display->drm, "[CONNECTOR:%d:%s] Failed to read current backlight mode from DPCD\n", connector->base.base.id, connector->base.name); @@ -197,8 +199,9 @@ intel_dp_aux_hdr_get_backlight(struct intel_connector *connector, enum pipe pipe return panel->backlight.max; } - if (drm_dp_dpcd_read(&intel_dp->aux, INTEL_EDP_BRIGHTNESS_NITS_LSB, buf, - sizeof(buf)) != sizeof(buf)) { + ret = drm_dp_dpcd_read_data(&intel_dp->aux, INTEL_EDP_BRIGHTNESS_NITS_LSB, buf, + sizeof(buf)); + if (ret < 0) { drm_err(display->drm, "[CONNECTOR:%d:%s] Failed to read brightness from DPCD\n", connector->base.base.id, connector->base.name); @@ -215,12 +218,14 @@ intel_dp_aux_hdr_set_aux_backlight(const struct drm_connector_state *conn_state, struct drm_device *dev = connector->base.dev; struct intel_dp *intel_dp = enc_to_intel_dp(connector->encoder); u8 buf[4] = {}; + int ret; buf[0] = level & 0xFF; buf[1] = (level & 0xFF00) >> 8; - if (drm_dp_dpcd_write(&intel_dp->aux, INTEL_EDP_BRIGHTNESS_NITS_LSB, buf, - sizeof(buf)) != sizeof(buf)) + ret = drm_dp_dpcd_write_data(&intel_dp->aux, INTEL_EDP_BRIGHTNESS_NITS_LSB, buf, + sizeof(buf)); + if (ret < 0) drm_err(dev, "[CONNECTOR:%d:%s] Failed to write brightness level to DPCD\n", connector->base.base.id, connector->base.name); } @@ -258,13 +263,12 @@ intel_dp_aux_write_content_luminance(struct intel_connector *connector, buf[2] = hdr_metadata->hdmi_metadata_type1.max_fall & 0xFF; buf[3] = (hdr_metadata->hdmi_metadata_type1.max_fall & 0xFF00) >> 8; - ret = drm_dp_dpcd_write(&intel_dp->aux, - INTEL_EDP_HDR_CONTENT_LUMINANCE, - buf, sizeof(buf)); + ret = drm_dp_dpcd_write_data(&intel_dp->aux, + INTEL_EDP_HDR_CONTENT_LUMINANCE, + buf, sizeof(buf)); if (ret < 0) drm_dbg_kms(display->drm, - "Content Luminance DPCD reg write failed, err:-%d\n", - ret); + "Content Luminance DPCD reg write failed (%pe)\n", ERR_PTR(ret)); } static void @@ -313,11 +317,11 @@ intel_dp_aux_hdr_enable_backlight(const struct intel_crtc_state *crtc_state, intel_dp_wait_source_oui(intel_dp); - ret = drm_dp_dpcd_readb(&intel_dp->aux, INTEL_EDP_HDR_GETSET_CTRL_PARAMS, &old_ctrl); - if (ret != 1) { + ret = drm_dp_dpcd_read_byte(&intel_dp->aux, INTEL_EDP_HDR_GETSET_CTRL_PARAMS, &old_ctrl); + if (ret < 0) { drm_err(display->drm, - "[CONNECTOR:%d:%s] Failed to read current backlight control mode: %d\n", - connector->base.base.id, connector->base.name, ret); + "[CONNECTOR:%d:%s] Failed to read current backlight control mode (%pe)\n", + connector->base.base.id, connector->base.name, ERR_PTR(ret)); return; } @@ -338,7 +342,7 @@ intel_dp_aux_hdr_enable_backlight(const struct intel_crtc_state *crtc_state, intel_dp_aux_fill_hdr_tcon_params(conn_state, &ctrl); if (ctrl != old_ctrl && - drm_dp_dpcd_writeb(&intel_dp->aux, INTEL_EDP_HDR_GETSET_CTRL_PARAMS, ctrl) != 1) + drm_dp_dpcd_write_byte(&intel_dp->aux, INTEL_EDP_HDR_GETSET_CTRL_PARAMS, ctrl) < 0) drm_err(display->drm, "[CONNECTOR:%d:%s] Failed to configure DPCD brightness controls\n", connector->base.base.id, connector->base.name); @@ -392,13 +396,12 @@ intel_dp_aux_write_panel_luminance_override(struct intel_connector *connector) buf[2] = panel->backlight.max & 0xFF; buf[3] = (panel->backlight.max & 0xFF00) >> 8; - ret = drm_dp_dpcd_write(&intel_dp->aux, - INTEL_EDP_HDR_PANEL_LUMINANCE_OVERRIDE, - buf, sizeof(buf)); + ret = drm_dp_dpcd_write_data(&intel_dp->aux, + INTEL_EDP_HDR_PANEL_LUMINANCE_OVERRIDE, + buf, sizeof(buf)); if (ret < 0) drm_dbg_kms(display->drm, - "Panel Luminance DPCD reg write failed, err:-%d\n", - ret); + "Panel Luminance DPCD reg write failed (%pe)\n", ERR_PTR(ret)); } static int @@ -456,8 +459,8 @@ static u32 intel_dp_aux_vesa_get_backlight(struct intel_connector *connector, en int ret; if (panel->backlight.edp.vesa.luminance_control_support) { - ret = drm_dp_dpcd_read(&intel_dp->aux, DP_EDP_PANEL_TARGET_LUMINANCE_VALUE, buf, - sizeof(buf)); + ret = drm_dp_dpcd_read_data(&intel_dp->aux, DP_EDP_PANEL_TARGET_LUMINANCE_VALUE, buf, + sizeof(buf)); if (ret < 0) { drm_err(intel_dp->aux.drm_dev, "[CONNECTOR:%d:%s] Failed to read Luminance from DPCD\n", From b9290ffeb7096a3305891f2af8a4be96a517af59 Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Wed, 16 Sep 2026 11:29:37 +0300 Subject: [PATCH 0253/1352] drm/i915/dp: switch to drm_dp_dpcd_{read_byte, read_data, write_byte, write_data} Switch to the modern DPCD access functions that return negative error codes on errors, -EIO for incomplete access, and 0 on success. Use local ret variables for the return value instead of checking inline, where sensible. Rename local int err to ret for clarity and consistency. Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/ad0a642c726436b5d41f2315e1c431d65fe2ea24.1789547331.git.jani.nikula@intel.com Signed-off-by: Jani Nikula --- drivers/gpu/drm/i915/display/intel_dp.c | 142 +++++++++++++++--------- 1 file changed, 88 insertions(+), 54 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_dp.c b/drivers/gpu/drm/i915/display/intel_dp.c index 2480b8ce3bec71..ffddf4b3372852 100644 --- a/drivers/gpu/drm/i915/display/intel_dp.c +++ b/drivers/gpu/drm/i915/display/intel_dp.c @@ -3736,19 +3736,19 @@ static bool downstream_hpd_needs_d0(struct intel_dp *intel_dp) static int write_dsc_decompression_flag(struct drm_dp_aux *aux, u8 flag, bool set) { - int err; + int ret; u8 val; - err = drm_dp_dpcd_readb(aux, DP_DSC_ENABLE, &val); - if (err < 0) - return err; + ret = drm_dp_dpcd_read_byte(aux, DP_DSC_ENABLE, &val); + if (ret < 0) + return ret; if (set) val |= flag; else val &= ~flag; - return drm_dp_dpcd_writeb(aux, DP_DSC_ENABLE, val); + return drm_dp_dpcd_write_byte(aux, DP_DSC_ENABLE, val); } static void @@ -3912,6 +3912,7 @@ intel_dp_init_source_oui(struct intel_dp *intel_dp) struct intel_display *display = to_intel_display(intel_dp); u8 oui[] = { 0x00, 0xaa, 0x01 }; u8 buf[3] = {}; + int ret; if (READ_ONCE(intel_dp->oui_valid)) return; @@ -3922,7 +3923,8 @@ intel_dp_init_source_oui(struct intel_dp *intel_dp) * During driver init, we want to be careful and avoid changing the source OUI if it's * already set to what we want, so as to avoid clearing any state by accident */ - if (drm_dp_dpcd_read(&intel_dp->aux, DP_SOURCE_OUI, buf, sizeof(buf)) < 0) + ret = drm_dp_dpcd_read_data(&intel_dp->aux, DP_SOURCE_OUI, buf, sizeof(buf)); + if (ret < 0) drm_dbg_kms(display->drm, "Failed to read source OUI\n"); if (memcmp(oui, buf, sizeof(oui)) == 0) { @@ -3931,7 +3933,8 @@ intel_dp_init_source_oui(struct intel_dp *intel_dp) return; } - if (drm_dp_dpcd_write(&intel_dp->aux, DP_SOURCE_OUI, oui, sizeof(oui)) < 0) { + ret = drm_dp_dpcd_write_data(&intel_dp->aux, DP_SOURCE_OUI, oui, sizeof(oui)); + if (ret < 0) { drm_dbg_kms(display->drm, "Failed to write source OUI\n"); WRITE_ONCE(intel_dp->oui_valid, false); } @@ -3973,7 +3976,7 @@ void intel_dp_set_power(struct intel_dp *intel_dp, u8 mode) if (downstream_hpd_needs_d0(intel_dp)) return; - ret = drm_dp_dpcd_writeb(&intel_dp->aux, DP_SET_POWER, mode); + ret = drm_dp_dpcd_write_byte(&intel_dp->aux, DP_SET_POWER, mode); } else { struct intel_digital_port *dig_port = dp_to_dig_port(intel_dp); @@ -3987,17 +3990,17 @@ void intel_dp_set_power(struct intel_dp *intel_dp, u8 mode) * time to wake up. */ for (i = 0; i < 3; i++) { - ret = drm_dp_dpcd_writeb(&intel_dp->aux, DP_SET_POWER, mode); - if (ret == 1) + ret = drm_dp_dpcd_write_byte(&intel_dp->aux, DP_SET_POWER, mode); + if (!ret) break; msleep(1); } - if (ret == 1 && intel_lspcon_active(dig_port)) + if (!ret && intel_lspcon_active(dig_port)) intel_lspcon_wait_pcon_mode(dig_port); } - if (ret != 1) + if (ret < 0) drm_dbg_kms(display->drm, "[ENCODER:%d:%s] Set power to %s failed\n", encoder->base.base.id, encoder->base.name, @@ -4088,6 +4091,7 @@ bool intel_dp_initial_fastset_check(struct intel_encoder *encoder, static void intel_dp_get_pcon_dsc_cap(struct intel_dp *intel_dp) { struct intel_display *display = to_intel_display(intel_dp); + int ret; /* Clear the cached register set to avoid using stale values */ @@ -4096,9 +4100,10 @@ static void intel_dp_get_pcon_dsc_cap(struct intel_dp *intel_dp) if (!drm_dp_is_branch(intel_dp->dpcd)) return; - if (drm_dp_dpcd_read(&intel_dp->aux, DP_PCON_DSC_ENCODER, - intel_dp->pcon_dsc_dpcd, - sizeof(intel_dp->pcon_dsc_dpcd)) < 0) + ret = drm_dp_dpcd_read_data(&intel_dp->aux, DP_PCON_DSC_ENCODER, + intel_dp->pcon_dsc_dpcd, + sizeof(intel_dp->pcon_dsc_dpcd)); + if (ret < 0) drm_err(display->drm, "Failed to read DPCD register 0x%x\n", DP_PCON_DSC_ENCODER); @@ -4257,13 +4262,13 @@ int intel_dp_pcon_set_tmds_mode(struct intel_dp *intel_dp) /* Set PCON source control mode */ buf |= DP_PCON_ENABLE_SOURCE_CTL_MODE; - ret = drm_dp_dpcd_writeb(&intel_dp->aux, DP_PCON_HDMI_LINK_CONFIG_1, buf); + ret = drm_dp_dpcd_write_byte(&intel_dp->aux, DP_PCON_HDMI_LINK_CONFIG_1, buf); if (ret < 0) return ret; /* Set HDMI LINK ENABLE */ buf |= DP_PCON_ENABLE_HDMI_LINK; - ret = drm_dp_dpcd_writeb(&intel_dp->aux, DP_PCON_HDMI_LINK_CONFIG_1, buf); + ret = drm_dp_dpcd_write_byte(&intel_dp->aux, DP_PCON_HDMI_LINK_CONFIG_1, buf); if (ret < 0) return ret; @@ -4409,6 +4414,7 @@ void intel_dp_configure_protocol_converter(struct intel_dp *intel_dp, bool ycbcr444_to_420 = false; bool rgb_to_ycbcr = false; u8 tmp; + int ret; if (intel_dp->dpcd[DP_DPCD_REV] < 0x13) return; @@ -4418,8 +4424,9 @@ void intel_dp_configure_protocol_converter(struct intel_dp *intel_dp, tmp = intel_dp_has_hdmi_sink(intel_dp) ? DP_HDMI_DVI_OUTPUT_CONFIG : 0; - if (drm_dp_dpcd_writeb(&intel_dp->aux, - DP_PROTOCOL_CONVERTER_CONTROL_0, tmp) != 1) + ret = drm_dp_dpcd_write_byte(&intel_dp->aux, + DP_PROTOCOL_CONVERTER_CONTROL_0, tmp); + if (ret < 0) drm_dbg_kms(display->drm, "Failed to %s protocol converter HDMI mode\n", str_enable_disable(intel_dp_has_hdmi_sink(intel_dp))); @@ -4454,15 +4461,17 @@ void intel_dp_configure_protocol_converter(struct intel_dp *intel_dp, tmp = ycbcr444_to_420 ? DP_CONVERSION_TO_YCBCR420_ENABLE : 0; - if (drm_dp_dpcd_writeb(&intel_dp->aux, - DP_PROTOCOL_CONVERTER_CONTROL_1, tmp) != 1) + ret = drm_dp_dpcd_write_byte(&intel_dp->aux, + DP_PROTOCOL_CONVERTER_CONTROL_1, tmp); + if (ret < 0) drm_dbg_kms(display->drm, "Failed to %s protocol converter YCbCr 4:2:0 conversion mode\n", str_enable_disable(intel_dp->dfp.ycbcr_444_to_420)); tmp = rgb_to_ycbcr ? DP_CONVERSION_BT709_RGB_YCBCR_ENABLE : 0; - if (drm_dp_pcon_convert_rgb_to_ycbcr(&intel_dp->aux, tmp) < 0) + ret = drm_dp_pcon_convert_rgb_to_ycbcr(&intel_dp->aux, tmp); + if (ret < 0) drm_dbg_kms(display->drm, "Failed to %s protocol converter RGB->YCbCr conversion mode\n", str_enable_disable(tmp)); @@ -4472,8 +4481,8 @@ static u8 intel_dp_read_dprx_feature_enum(struct intel_dp *intel_dp) { u8 dprx = 0; - drm_dp_dpcd_read_data(&intel_dp->aux, DP_DPRX_FEATURE_ENUMERATION_LIST, - &dprx, sizeof(dprx)); + drm_dp_dpcd_read_byte(&intel_dp->aux, DP_DPRX_FEATURE_ENUMERATION_LIST, &dprx); + return dprx; } @@ -4492,9 +4501,8 @@ static int intel_dp_read_dsc_dpcd(struct drm_dp_aux *aux, { int ret; - ret = drm_dp_dpcd_read_data(aux, DP_DSC_SUPPORT, dsc_dpcd, - DP_DSC_RECEIVER_CAP_SIZE); - if (ret) { + ret = drm_dp_dpcd_read_data(aux, DP_DSC_SUPPORT, dsc_dpcd, DP_DSC_RECEIVER_CAP_SIZE); + if (ret < 0) { drm_dbg_kms(aux->drm_dev, "Could not read DSC DPCD register 0x%x Error: %pe\n", DP_DSC_SUPPORT, ERR_PTR(ret)); @@ -4510,7 +4518,7 @@ static int intel_dp_read_dsc_dpcd(struct drm_dp_aux *aux, static void init_dsc_overall_throughput_limits(struct intel_connector *connector, bool is_branch) { u8 branch_caps[DP_DSC_BRANCH_CAP_SIZE]; - int line_width; + int line_width, ret; connector->dp.dsc_branch_caps.overall_throughput.rgb_yuv444 = INT_MAX; connector->dp.dsc_branch_caps.overall_throughput.yuv422_420 = INT_MAX; @@ -4519,9 +4527,10 @@ static void init_dsc_overall_throughput_limits(struct intel_connector *connector if (!is_branch) return; - if (drm_dp_dpcd_read_data(connector->dp.dsc_decompression_aux, - DP_DSC_BRANCH_OVERALL_THROUGHPUT_0, branch_caps, - sizeof(branch_caps)) != 0) + ret = drm_dp_dpcd_read_data(connector->dp.dsc_decompression_aux, + DP_DSC_BRANCH_OVERALL_THROUGHPUT_0, branch_caps, + sizeof(branch_caps)); + if (ret < 0) return; connector->dp.dsc_branch_caps.overall_throughput.rgb_yuv444 = @@ -4539,6 +4548,7 @@ void intel_dp_get_dsc_sink_cap(u8 dpcd_rev, struct intel_connector *connector) { struct intel_display *display = to_intel_display(connector); + int ret; /* * Clear the cached register set to avoid using stale values @@ -4555,12 +4565,14 @@ void intel_dp_get_dsc_sink_cap(u8 dpcd_rev, if (dpcd_rev < DP_DPCD_REV_14) return; - if (intel_dp_read_dsc_dpcd(connector->dp.dsc_decompression_aux, - connector->dp.dsc_dpcd) < 0) + ret = intel_dp_read_dsc_dpcd(connector->dp.dsc_decompression_aux, + connector->dp.dsc_dpcd); + if (ret < 0) return; - if (drm_dp_dpcd_readb(connector->dp.dsc_decompression_aux, DP_FEC_CAPABILITY, - &connector->dp.fec_capability) < 0) { + ret = drm_dp_dpcd_read_byte(connector->dp.dsc_decompression_aux, DP_FEC_CAPABILITY, + &connector->dp.fec_capability); + if (ret < 0) { drm_dbg_kms(display->drm, "Could not read FEC DPCD register\n"); return; } @@ -4670,12 +4682,14 @@ static void intel_edp_mso_init(struct intel_dp *intel_dp) struct intel_display *display = to_intel_display(intel_dp); struct intel_connector *connector = intel_dp->attached_connector; struct drm_display_info *info = &connector->base.display_info; + int ret; u8 mso; if (intel_dp->edp_dpcd[0] < DP_EDP_14) return; - if (drm_dp_dpcd_readb(&intel_dp->aux, DP_EDP_MSO_LINK_CAPABILITIES, &mso) != 1) { + ret = drm_dp_dpcd_read_byte(&intel_dp->aux, DP_EDP_MSO_LINK_CAPABILITIES, &mso); + if (ret < 0) { drm_err(display->drm, "Failed to read MSO cap\n"); return; } @@ -4853,9 +4867,9 @@ intel_edp_init_dpcd(struct intel_dp *intel_dp, struct intel_connector *connector * method). The display control registers should read zero if they're * not supported anyway. */ - if (drm_dp_dpcd_read(&intel_dp->aux, DP_EDP_DPCD_REV, - intel_dp->edp_dpcd, sizeof(intel_dp->edp_dpcd)) == - sizeof(intel_dp->edp_dpcd)) { + ret = drm_dp_dpcd_read_data(&intel_dp->aux, DP_EDP_DPCD_REV, + intel_dp->edp_dpcd, sizeof(intel_dp->edp_dpcd)); + if (!ret) { drm_dbg_kms(display->drm, "eDP DPCD: %*ph\n", (int)sizeof(intel_dp->edp_dpcd), intel_dp->edp_dpcd); @@ -5084,6 +5098,7 @@ static bool intel_dp_get_sink_irq_esi(struct intel_dp *intel_dp, u8 *esi) { struct intel_display *display = to_intel_display(intel_dp); + int ret; /* * Display WA for HSD #13013007775: mtl/arl/lnl @@ -5092,24 +5107,30 @@ intel_dp_get_sink_irq_esi(struct intel_dp *intel_dp, u8 *esi) * inadvertently. */ if (IS_DISPLAY_VER(display, 14, 20) && !display->platform.battlemage) { - if (drm_dp_dpcd_read(&intel_dp->aux, DP_SINK_COUNT_ESI, esi, 3) != 3) + ret = drm_dp_dpcd_read_data(&intel_dp->aux, DP_SINK_COUNT_ESI, esi, 3); + if (ret < 0) return false; /* DP_SINK_COUNT_ESI + 3 == DP_LINK_SERVICE_IRQ_VECTOR_ESI0 */ - return drm_dp_dpcd_readb(&intel_dp->aux, DP_LINK_SERVICE_IRQ_VECTOR_ESI0, - &esi[3]) == 1; + ret = drm_dp_dpcd_read_byte(&intel_dp->aux, DP_LINK_SERVICE_IRQ_VECTOR_ESI0, + &esi[3]); + return ret == 0; } - return drm_dp_dpcd_read(&intel_dp->aux, DP_SINK_COUNT_ESI, esi, 4) == 4; + ret = drm_dp_dpcd_read_data(&intel_dp->aux, DP_SINK_COUNT_ESI, esi, 4); + + return ret == 0; } static bool intel_dp_ack_sink_irq_esi(struct intel_dp *intel_dp, u8 esi[4]) { int retry; + int ret; for (retry = 0; retry < 3; retry++) { - if (drm_dp_dpcd_write(&intel_dp->aux, DP_SINK_COUNT_ESI + 1, - &esi[1], 3) == 3) + ret = drm_dp_dpcd_write_data(&intel_dp->aux, DP_SINK_COUNT_ESI + 1, + &esi[1], 3); + if (!ret) return true; } @@ -5119,20 +5140,24 @@ static bool intel_dp_ack_sink_irq_esi(struct intel_dp *intel_dp, u8 esi[4]) /* Return %true if reading the ESI vector succeeded, %false otherwise. */ static bool intel_dp_get_sink_irq_esi_sst(struct intel_dp *intel_dp, u8 esi[4]) { + int ret; + memset(esi, 0, 4); /* * TODO: For DP_DPCD_REV >= 0x12 read * DP_SINK_COUNT_ESI and DP_DEVICE_SERVICE_IRQ_VECTOR_ESI0. */ - if (drm_dp_dpcd_read_data(&intel_dp->aux, DP_SINK_COUNT, esi, 2) != 0) + ret = drm_dp_dpcd_read_data(&intel_dp->aux, DP_SINK_COUNT, esi, 2); + if (ret < 0) return false; if (intel_dp->dpcd[DP_DPCD_REV] < DP_DPCD_REV_12) return true; /* TODO: Read DP_DEVICE_SERVICE_IRQ_VECTOR_ESI1 as well */ - if (drm_dp_dpcd_read_byte(&intel_dp->aux, DP_LINK_SERVICE_IRQ_VECTOR_ESI0, &esi[3]) != 0) + ret = drm_dp_dpcd_read_byte(&intel_dp->aux, DP_LINK_SERVICE_IRQ_VECTOR_ESI0, &esi[3]); + if (ret < 0) return false; return true; @@ -5141,18 +5166,22 @@ static bool intel_dp_get_sink_irq_esi_sst(struct intel_dp *intel_dp, u8 esi[4]) /* Return %true if acking the ESI vector IRQ events succeeded, %false otherwise. */ static bool intel_dp_ack_sink_irq_esi_sst(struct intel_dp *intel_dp, u8 esi[4]) { + int ret; + /* * TODO: For DP_DPCD_REV >= 0x12 write * DP_DEVICE_SERVICE_IRQ_VECTOR_ESI0 */ - if (drm_dp_dpcd_write_byte(&intel_dp->aux, DP_DEVICE_SERVICE_IRQ_VECTOR, esi[1]) != 0) + ret = drm_dp_dpcd_write_byte(&intel_dp->aux, DP_DEVICE_SERVICE_IRQ_VECTOR, esi[1]); + if (ret < 0) return false; if (intel_dp->dpcd[DP_DPCD_REV] < DP_DPCD_REV_12) return true; /* TODO: Read DP_DEVICE_SERVICE_IRQ_VECTOR_ESI1 as well */ - if (drm_dp_dpcd_write_byte(&intel_dp->aux, DP_LINK_SERVICE_IRQ_VECTOR_ESI0, esi[3]) != 0) + ret = drm_dp_dpcd_write_byte(&intel_dp->aux, DP_LINK_SERVICE_IRQ_VECTOR_ESI0, esi[3]); + if (ret < 0) return false; return true; @@ -5750,14 +5779,17 @@ intel_dp_handle_hdmi_link_status_change(struct intel_dp *intel_dp) { bool is_active; u8 buf = 0; + int ret; is_active = drm_dp_pcon_hdmi_link_active(&intel_dp->aux); if (intel_dp->frl.is_trained && !is_active) { - if (drm_dp_dpcd_readb(&intel_dp->aux, DP_PCON_HDMI_LINK_CONFIG_1, &buf) < 0) + ret = drm_dp_dpcd_read_byte(&intel_dp->aux, DP_PCON_HDMI_LINK_CONFIG_1, &buf); + if (ret < 0) return; buf &= ~DP_PCON_ENABLE_HDMI_LINK; - if (drm_dp_dpcd_writeb(&intel_dp->aux, DP_PCON_HDMI_LINK_CONFIG_1, buf) < 0) + ret = drm_dp_dpcd_write_byte(&intel_dp->aux, DP_PCON_HDMI_LINK_CONFIG_1, buf); + if (ret < 0) return; drm_dp_pcon_hdmi_frl_link_error_count(&intel_dp->aux, &intel_dp->attached_connector->base); @@ -6262,6 +6294,7 @@ static bool intel_dp_sink_supports_as_sdp_v2(struct intel_dp *intel_dp) { u8 rx_features; + int ret; /* * The DP spec does not explicitly provide the AS SDP v2 capability. @@ -6284,9 +6317,10 @@ intel_dp_sink_supports_as_sdp_v2(struct intel_dp *intel_dp) * support from Display ID. */ - if (drm_dp_dpcd_read_byte(&intel_dp->aux, - DP_DPRX_FEATURE_ENUMERATION_LIST_CONT_1, - &rx_features) == 1) { + ret = drm_dp_dpcd_read_byte(&intel_dp->aux, + DP_DPRX_FEATURE_ENUMERATION_LIST_CONT_1, + &rx_features); + if (!ret) { if (rx_features & DP_AS_SDP_FAVT_PAYLOAD_FIELDS_PARSING_SUPPORTED) return true; } From 41de7a1888fe1aff94411162dc31fb48b0234619 Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Mon, 21 Sep 2026 12:41:03 +0300 Subject: [PATCH 0254/1352] drm/i915/hdcp: switch to drm_dp_dpcd_{read_byte, read_data, write_data}() Switch to the modern DPCD access functions that return negative error codes on errors, -EIO for incomplete access, and 0 on success. This simplifies error handling all over the place. Convert the error logging to use %pe and ERR_PTR() while at it. Add comments to the read/write places that depend on the functions that return the number of bytes read/written for incomplete access. v2: return length from get_receiver_id_list_rx_info() (Sashiko) Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/20260921094103.701841-1-jani.nikula@intel.com Signed-off-by: Jani Nikula --- drivers/gpu/drm/i915/display/intel_dp_hdcp.c | 176 +++++++++---------- 1 file changed, 88 insertions(+), 88 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_dp_hdcp.c b/drivers/gpu/drm/i915/display/intel_dp_hdcp.c index 14ed0ea22dd30f..566a921d18a157 100644 --- a/drivers/gpu/drm/i915/display/intel_dp_hdcp.c +++ b/drivers/gpu/drm/i915/display/intel_dp_hdcp.c @@ -60,16 +60,15 @@ int intel_dp_hdcp_write_an_aksv(struct intel_digital_port *dig_port, { struct intel_display *display = to_intel_display(dig_port); u8 aksv[DRM_HDCP_KSV_LEN] = {}; - ssize_t dpcd_ret; + int ret; /* Output An first, that's easy */ - dpcd_ret = drm_dp_dpcd_write(&dig_port->dp.aux, DP_AUX_HDCP_AN, + ret = drm_dp_dpcd_write_data(&dig_port->dp.aux, DP_AUX_HDCP_AN, an, DRM_HDCP_AN_LEN); - if (dpcd_ret != DRM_HDCP_AN_LEN) { + if (ret < 0) { drm_dbg_kms(display->drm, - "Failed to write An over DP/AUX (%zd)\n", - dpcd_ret); - return dpcd_ret >= 0 ? -EIO : dpcd_ret; + "Failed to write An over DP/AUX (%pe)\n", ERR_PTR(ret)); + return ret; } /* @@ -79,14 +78,14 @@ int intel_dp_hdcp_write_an_aksv(struct intel_digital_port *dig_port, * the destination address which will tickle the hardware to output the * Aksv on our behalf after the header is sent. */ - dpcd_ret = drm_dp_dpcd_write(&dig_port->dp.aux, DP_AUX_HDCP_AKSV, + ret = drm_dp_dpcd_write_data(&dig_port->dp.aux, DP_AUX_HDCP_AKSV, aksv, DRM_HDCP_KSV_LEN); - if (dpcd_ret != DRM_HDCP_KSV_LEN) { + if (ret < 0) { drm_dbg_kms(display->drm, - "Failed to write Aksv over DP/AUX (%zd)\n", - dpcd_ret); - return dpcd_ret >= 0 ? -EIO : dpcd_ret; + "Failed to write Aksv over DP/AUX (%pe)\n", ERR_PTR(ret)); + return ret; } + return 0; } @@ -94,15 +93,16 @@ static int intel_dp_hdcp_read_bksv(struct intel_digital_port *dig_port, u8 *bksv) { struct intel_display *display = to_intel_display(dig_port); - ssize_t ret; + int ret; - ret = drm_dp_dpcd_read(&dig_port->dp.aux, DP_AUX_HDCP_BKSV, bksv, - DRM_HDCP_KSV_LEN); - if (ret != DRM_HDCP_KSV_LEN) { + ret = drm_dp_dpcd_read_data(&dig_port->dp.aux, DP_AUX_HDCP_BKSV, bksv, + DRM_HDCP_KSV_LEN); + if (ret < 0) { drm_dbg_kms(display->drm, - "Read Bksv from DP/AUX failed (%zd)\n", ret); - return ret >= 0 ? -EIO : ret; + "Read Bksv from DP/AUX failed (%pe)\n", ERR_PTR(ret)); + return ret; } + return 0; } @@ -110,20 +110,21 @@ static int intel_dp_hdcp_read_bstatus(struct intel_digital_port *dig_port, u8 *bstatus) { struct intel_display *display = to_intel_display(dig_port); - ssize_t ret; + int ret; /* * For some reason the HDMI and DP HDCP specs call this register * definition by different names. In the HDMI spec, it's called BSTATUS, * but in DP it's called BINFO. */ - ret = drm_dp_dpcd_read(&dig_port->dp.aux, DP_AUX_HDCP_BINFO, - bstatus, DRM_HDCP_BSTATUS_LEN); - if (ret != DRM_HDCP_BSTATUS_LEN) { + ret = drm_dp_dpcd_read_data(&dig_port->dp.aux, DP_AUX_HDCP_BINFO, + bstatus, DRM_HDCP_BSTATUS_LEN); + if (ret < 0) { drm_dbg_kms(display->drm, - "Read bstatus from DP/AUX failed (%zd)\n", ret); - return ret >= 0 ? -EIO : ret; + "Read bstatus from DP/AUX failed (%pe)\n", ERR_PTR(ret)); + return ret; } + return 0; } @@ -132,14 +133,13 @@ int intel_dp_hdcp_read_bcaps(struct drm_dp_aux *aux, struct intel_display *display, u8 *bcaps) { - ssize_t ret; + int ret; - ret = drm_dp_dpcd_read(aux, DP_AUX_HDCP_BCAPS, - bcaps, 1); - if (ret != 1) { + ret = drm_dp_dpcd_read_byte(aux, DP_AUX_HDCP_BCAPS, bcaps); + if (ret < 0) { drm_dbg_kms(display->drm, - "Read bcaps from DP/AUX failed (%zd)\n", ret); - return ret >= 0 ? -EIO : ret; + "Read bcaps from DP/AUX failed (%pe)\n", ERR_PTR(ret)); + return ret; } return 0; @@ -166,16 +166,16 @@ int intel_dp_hdcp_read_ri_prime(struct intel_digital_port *dig_port, u8 *ri_prime) { struct intel_display *display = to_intel_display(dig_port); - ssize_t ret; + int ret; - ret = drm_dp_dpcd_read(&dig_port->dp.aux, DP_AUX_HDCP_RI_PRIME, - ri_prime, DRM_HDCP_RI_LEN); - if (ret != DRM_HDCP_RI_LEN) { + ret = drm_dp_dpcd_read_data(&dig_port->dp.aux, DP_AUX_HDCP_RI_PRIME, + ri_prime, DRM_HDCP_RI_LEN); + if (ret < 0) { drm_dbg_kms(display->drm, - "Read Ri' from DP/AUX failed (%zd)\n", - ret); - return ret >= 0 ? -EIO : ret; + "Read Ri' from DP/AUX failed (%pe)\n", ERR_PTR(ret)); + return ret; } + return 0; } @@ -184,17 +184,18 @@ int intel_dp_hdcp_read_ksv_ready(struct intel_digital_port *dig_port, bool *ksv_ready) { struct intel_display *display = to_intel_display(dig_port); - ssize_t ret; u8 bstatus; + int ret; - ret = drm_dp_dpcd_read(&dig_port->dp.aux, DP_AUX_HDCP_BSTATUS, - &bstatus, 1); - if (ret != 1) { + ret = drm_dp_dpcd_read_byte(&dig_port->dp.aux, DP_AUX_HDCP_BSTATUS, + &bstatus); + if (ret < 0) { drm_dbg_kms(display->drm, - "Read bstatus from DP/AUX failed (%zd)\n", ret); - return ret >= 0 ? -EIO : ret; + "Read bstatus from DP/AUX failed (%pe)\n", ERR_PTR(ret)); + return ret; } *ksv_ready = bstatus & DP_BSTATUS_READY; + return 0; } @@ -203,23 +204,24 @@ int intel_dp_hdcp_read_ksv_fifo(struct intel_digital_port *dig_port, int num_downstream, u8 *ksv_fifo) { struct intel_display *display = to_intel_display(dig_port); - ssize_t ret; + int ret; int i; /* KSV list is read via 15 byte window (3 entries @ 5 bytes each) */ for (i = 0; i < num_downstream; i += 3) { size_t len = min(num_downstream - i, 3) * DRM_HDCP_KSV_LEN; - ret = drm_dp_dpcd_read(&dig_port->dp.aux, - DP_AUX_HDCP_KSV_FIFO, - ksv_fifo + i * DRM_HDCP_KSV_LEN, - len); - if (ret != len) { + ret = drm_dp_dpcd_read_data(&dig_port->dp.aux, + DP_AUX_HDCP_KSV_FIFO, + ksv_fifo + i * DRM_HDCP_KSV_LEN, + len); + if (ret < 0) { drm_dbg_kms(display->drm, - "Read ksv[%d] from DP/AUX failed (%zd)\n", - i, ret); - return ret >= 0 ? -EIO : ret; + "Read ksv[%d] from DP/AUX failed (%pe)\n", + i, ERR_PTR(ret)); + return ret; } } + return 0; } @@ -228,19 +230,19 @@ int intel_dp_hdcp_read_v_prime_part(struct intel_digital_port *dig_port, int i, u32 *part) { struct intel_display *display = to_intel_display(dig_port); - ssize_t ret; + int ret; if (i >= DRM_HDCP_V_PRIME_NUM_PARTS) return -EINVAL; - ret = drm_dp_dpcd_read(&dig_port->dp.aux, - DP_AUX_HDCP_V_PRIME(i), part, - DRM_HDCP_V_PRIME_PART_LEN); - if (ret != DRM_HDCP_V_PRIME_PART_LEN) { + ret = drm_dp_dpcd_read_data(&dig_port->dp.aux, DP_AUX_HDCP_V_PRIME(i), part, + DRM_HDCP_V_PRIME_PART_LEN); + if (ret < 0) { drm_dbg_kms(display->drm, - "Read v'[%d] from DP/AUX failed (%zd)\n", i, ret); - return ret >= 0 ? -EIO : ret; + "Read v'[%d] from DP/AUX failed (%pe)\n", i, ERR_PTR(ret)); + return ret; } + return 0; } @@ -258,14 +260,13 @@ bool intel_dp_hdcp_check_link(struct intel_digital_port *dig_port, struct intel_connector *connector) { struct intel_display *display = to_intel_display(dig_port); - ssize_t ret; u8 bstatus; + int ret; - ret = drm_dp_dpcd_read(&dig_port->dp.aux, DP_AUX_HDCP_BSTATUS, - &bstatus, 1); - if (ret != 1) { + ret = drm_dp_dpcd_read_byte(&dig_port->dp.aux, DP_AUX_HDCP_BSTATUS, &bstatus); + if (ret < 0) { drm_dbg_kms(display->drm, - "Read bstatus from DP/AUX failed (%zd)\n", ret); + "Read bstatus from DP/AUX failed (%pe)\n", ERR_PTR(ret)); return false; } @@ -346,15 +347,14 @@ intel_dp_hdcp2_read_rx_status(struct intel_connector *connector, struct intel_display *display = to_intel_display(connector); struct intel_digital_port *dig_port = intel_attached_dig_port(connector); struct drm_dp_aux *aux = &dig_port->dp.aux; - ssize_t ret; + int ret; - ret = drm_dp_dpcd_read(aux, - DP_HDCP_2_2_REG_RXSTATUS_OFFSET, rx_status, - HDCP_2_2_DP_RXSTATUS_LEN); - if (ret != HDCP_2_2_DP_RXSTATUS_LEN) { + ret = drm_dp_dpcd_read_data(aux, DP_HDCP_2_2_REG_RXSTATUS_OFFSET, rx_status, + HDCP_2_2_DP_RXSTATUS_LEN); + if (ret < 0) { drm_dbg_kms(display->drm, - "Read bstatus from DP/AUX failed (%zd)\n", ret); - return ret >= 0 ? -EIO : ret; + "Read bstatus from DP/AUX failed (%pe)\n", ERR_PTR(ret)); + return ret; } return 0; @@ -474,8 +474,8 @@ int intel_dp_hdcp2_write_msg(struct intel_connector *connector, len = bytes_to_write > DP_AUX_MAX_PAYLOAD_BYTES ? DP_AUX_MAX_PAYLOAD_BYTES : bytes_to_write; - ret = drm_dp_dpcd_write(aux, - offset, (void *)byte, len); + /* Note: This may return < len for partial writes. */ + ret = drm_dp_dpcd_write(aux, offset, byte, len); if (ret < 0) return ret; @@ -487,20 +487,20 @@ int intel_dp_hdcp2_write_msg(struct intel_connector *connector, return size; } +/* return number of bytes read on success */ static ssize_t get_receiver_id_list_rx_info(struct intel_connector *connector, u32 *dev_cnt, u8 *byte) { struct intel_digital_port *dig_port = intel_attached_dig_port(connector); struct drm_dp_aux *aux = &dig_port->dp.aux; - ssize_t ret; u8 *rx_info = byte; + int ret; - ret = drm_dp_dpcd_read(aux, - DP_HDCP_2_2_REG_RXINFO_OFFSET, - (void *)rx_info, HDCP_2_2_RXINFO_LEN); - if (ret != HDCP_2_2_RXINFO_LEN) - return ret >= 0 ? -EIO : ret; + ret = drm_dp_dpcd_read_data(aux, DP_HDCP_2_2_REG_RXINFO_OFFSET, + rx_info, HDCP_2_2_RXINFO_LEN); + if (ret < 0) + return ret; *dev_cnt = (HDCP_2_2_DEV_COUNT_HI(rx_info[0]) << 4 | HDCP_2_2_DEV_COUNT_LO(rx_info[1])); @@ -508,7 +508,7 @@ ssize_t get_receiver_id_list_rx_info(struct intel_connector *connector, if (*dev_cnt > HDCP_2_2_MAX_DEVICE_COUNT) *dev_cnt = HDCP_2_2_MAX_DEVICE_COUNT; - return ret; + return HDCP_2_2_RXINFO_LEN; } static @@ -566,11 +566,11 @@ int intel_dp_hdcp2_read_msg(struct intel_connector *connector, hdcp2_msg_data->msg_read_timeout); } - ret = drm_dp_dpcd_read(aux, offset, - (void *)byte, len); + /* Note: This may return < len for partial reads. */ + ret = drm_dp_dpcd_read(aux, offset, byte, len); if (ret < 0) { - drm_dbg_kms(display->drm, "msg_id %d, ret %zd\n", - msg_id, ret); + drm_dbg_kms(display->drm, "msg_id %d, ret %pe\n", + msg_id, ERR_PTR(ret)); return ret; } @@ -660,11 +660,11 @@ int _intel_dp_hdcp2_get_capability(struct drm_dp_aux *aux, * declare a monitor not capable of HDCP 2.2. */ for (i = 0; i < 3; i++) { - ret = drm_dp_dpcd_read(aux, - DP_HDCP_2_2_REG_RX_CAPS_OFFSET, - rx_caps, HDCP_2_2_RXCAPS_LEN); - if (ret != HDCP_2_2_RXCAPS_LEN) - return ret >= 0 ? -EIO : ret; + ret = drm_dp_dpcd_read_data(aux, + DP_HDCP_2_2_REG_RX_CAPS_OFFSET, + rx_caps, HDCP_2_2_RXCAPS_LEN); + if (ret < 0) + return ret; if (rx_caps[0] == HDCP_2_2_RX_CAPS_VERSION_VAL && HDCP_2_2_DP_HDCP_CAPABLE(rx_caps[2])) { From fc980a78c218a6f3f695827832dcd0cc5ff3ca37 Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Wed, 16 Sep 2026 11:29:39 +0300 Subject: [PATCH 0255/1352] drm/i915/dp: switch link training to drm_dp_dpcd_{read_byte, read_data, write_byte, write_data}() Switch to the modern DPCD access functions that return negative error codes on errors, -EIO for incomplete access, and 0 on success. Use local ret variables for the return value instead of checking inline, where sensible. Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/c88130676c7b8187a5e5954cdbd36c316950d89f.1789547331.git.jani.nikula@intel.com Signed-off-by: Jani Nikula --- .../drm/i915/display/intel_dp_link_training.c | 48 ++++++++++--------- 1 file changed, 26 insertions(+), 22 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_dp_link_training.c b/drivers/gpu/drm/i915/display/intel_dp_link_training.c index 9a692f4fdfee68..b6e8b13ee4dbd0 100644 --- a/drivers/gpu/drm/i915/display/intel_dp_link_training.c +++ b/drivers/gpu/drm/i915/display/intel_dp_link_training.c @@ -929,7 +929,7 @@ intel_dp_set_link_train(struct intel_dp *intel_dp, { int reg = intel_dp_training_pattern_set_reg(intel_dp, dp_phy); u8 buf[sizeof(intel_dp->train_set) + 1]; - int len; + int ret, len; intel_dp_program_link_training_pattern(intel_dp, crtc_state, dp_phy, dp_train_pat); @@ -939,7 +939,9 @@ intel_dp_set_link_train(struct intel_dp *intel_dp, memcpy(buf + 1, intel_dp->train_set, crtc_state->lane_count); len = crtc_state->lane_count + 1; - return drm_dp_dpcd_write(&intel_dp->aux, reg, buf, len) == len; + ret = drm_dp_dpcd_write_data(&intel_dp->aux, reg, buf, len); + + return !ret; } static char dp_training_pattern_name(u8 train_pat) @@ -1046,10 +1048,10 @@ intel_dp_update_link_train(struct intel_dp *intel_dp, intel_dp_set_signal_levels(intel_dp, crtc_state, dp_phy); - ret = drm_dp_dpcd_write(&intel_dp->aux, reg, - intel_dp->train_set, crtc_state->lane_count); + ret = drm_dp_dpcd_write_data(&intel_dp->aux, reg, + intel_dp->train_set, crtc_state->lane_count); - return ret == crtc_state->lane_count; + return !ret; } /* 128b/132b */ @@ -1115,7 +1117,7 @@ void intel_dp_link_training_set_mode(struct intel_dp *intel_dp, int link_rate, link_config[0] |= pr_with_as_sdp_enable ? DP_FIXED_VTOTAL_AS_SDP_EN_IN_PR_ACTIVE : 0; link_config[1] = drm_dp_is_uhbr_rate(link_rate) ? DP_SET_ANSI_128B132B : DP_SET_ANSI_8B10B; - drm_dp_dpcd_write(&intel_dp->aux, DP_DOWNSPREAD_CTRL, link_config, 2); + drm_dp_dpcd_write_data(&intel_dp->aux, DP_DOWNSPREAD_CTRL, link_config, 2); } static bool @@ -1163,8 +1165,8 @@ void intel_dp_link_training_set_bw(struct intel_dp *intel_dp, /* DP and eDP v1.3 and earlier link bw set method. */ u8 link_config[] = { link_bw, lane_count }; - drm_dp_dpcd_write(&intel_dp->aux, DP_LINK_BW_SET, link_config, - ARRAY_SIZE(link_config)); + drm_dp_dpcd_write_data(&intel_dp->aux, DP_LINK_BW_SET, link_config, + sizeof(link_config)); } else { /* * eDP v1.4 and later link rate set method. @@ -1175,8 +1177,8 @@ void intel_dp_link_training_set_bw(struct intel_dp *intel_dp, * eDP v1.5 sinks allow choosing either, and the last choice * shall be active. */ - drm_dp_dpcd_writeb(&intel_dp->aux, DP_LANE_COUNT_SET, lane_count); - drm_dp_dpcd_writeb(&intel_dp->aux, DP_LINK_RATE_SET, rate_select); + drm_dp_dpcd_write_byte(&intel_dp->aux, DP_LANE_COUNT_SET, lane_count); + drm_dp_dpcd_write_byte(&intel_dp->aux, DP_LINK_RATE_SET, rate_select); } } @@ -1286,8 +1288,8 @@ intel_dp_prepare_link_train(struct intel_dp *intel_dp, lt_dbg(intel_dp, DP_PHY_DPRX, "Reloading eDP link rates\n"); - drm_dp_dpcd_read(&intel_dp->aux, DP_SUPPORTED_LINK_RATES, - sink_rates, sizeof(sink_rates)); + drm_dp_dpcd_read_data(&intel_dp->aux, DP_SUPPORTED_LINK_RATES, + sink_rates, sizeof(sink_rates)); } if (link_bw) @@ -1612,7 +1614,7 @@ static void intel_dp_stop_post_lt_adj_req(struct intel_dp *intel_dp, if (crtc_state->enhanced_framing) lane_count |= DP_LANE_COUNT_ENHANCED_FRAME_EN; - drm_dp_dpcd_writeb(&intel_dp->aux, DP_LANE_COUNT_SET, lane_count); + drm_dp_dpcd_write_byte(&intel_dp->aux, DP_LANE_COUNT_SET, lane_count); } static bool intel_dp_disable_dpcd_training_pattern(struct intel_dp *intel_dp, @@ -1621,7 +1623,7 @@ static bool intel_dp_disable_dpcd_training_pattern(struct intel_dp *intel_dp, int reg = intel_dp_training_pattern_set_reg(intel_dp, dp_phy); u8 val = DP_TRAINING_PATTERN_DISABLE; - return drm_dp_dpcd_write(&intel_dp->aux, reg, &val, 1) == 1; + return drm_dp_dpcd_write_byte(&intel_dp->aux, reg, val) == 0; } static int @@ -1631,10 +1633,10 @@ intel_dp_128b132b_intra_hop(struct intel_dp *intel_dp, u8 sink_status; int ret; - ret = drm_dp_dpcd_readb(&intel_dp->aux, DP_SINK_STATUS, &sink_status); - if (ret != 1) { + ret = drm_dp_dpcd_read_byte(&intel_dp->aux, DP_SINK_STATUS, &sink_status); + if (ret < 0) { lt_dbg(intel_dp, DP_PHY_DPRX, "Failed to read sink status\n"); - return ret < 0 ? ret : -EIO; + return ret; } return sink_status & DP_INTRA_HOP_AUX_REPLY_INDICATION ? 1 : 0; @@ -2173,9 +2175,11 @@ intel_dp_128b132b_lane_cds(struct intel_dp *intel_dp, { u8 link_status[DP_LINK_STATUS_SIZE]; unsigned long deadline; + int ret; - if (drm_dp_dpcd_writeb(&intel_dp->aux, DP_TRAINING_PATTERN_SET, - DP_TRAINING_PATTERN_2_CDS) != 1) { + ret = drm_dp_dpcd_write_byte(&intel_dp->aux, DP_TRAINING_PATTERN_SET, + DP_TRAINING_PATTERN_2_CDS); + if (ret < 0) { lt_err(intel_dp, DP_PHY_DPRX, "Failed to start 128b/132b TPS2 CDS\n"); return false; } @@ -2363,9 +2367,9 @@ void intel_dp_128b132b_sdp_crc16(struct intel_dp *intel_dp, return; /* DP v2.0 SCR on SDP CRC16 for 128b/132b Link Layer */ - drm_dp_dpcd_writeb(&intel_dp->aux, - DP_SDP_ERROR_DETECTION_CONFIGURATION, - DP_SDP_CRC16_128B132B_EN); + drm_dp_dpcd_write_byte(&intel_dp->aux, + DP_SDP_ERROR_DETECTION_CONFIGURATION, + DP_SDP_CRC16_128B132B_EN); lt_dbg(intel_dp, DP_PHY_DPRX, "DP2.0 SDP CRC16 for 128b/132b enabled\n"); } From 791a8519ff8407fd17a9ef04cf97771b5a37435b Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Wed, 16 Sep 2026 11:29:40 +0300 Subject: [PATCH 0256/1352] drm/i915/dp-mst: switch to drm_dp_dpcd_read_byte() Switch to the modern DPCD access functions that return negative error codes on errors, -EIO for incomplete access, and 0 on success. Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/26dfbd4d99f86c19efbff89613728ab967ec4f67.1789547331.git.jani.nikula@intel.com Signed-off-by: Jani Nikula --- drivers/gpu/drm/i915/display/intel_dp_mst.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/gpu/drm/i915/display/intel_dp_mst.c b/drivers/gpu/drm/i915/display/intel_dp_mst.c index 0c362784afe4a2..7023619aaaba9e 100644 --- a/drivers/gpu/drm/i915/display/intel_dp_mst.c +++ b/drivers/gpu/drm/i915/display/intel_dp_mst.c @@ -2252,7 +2252,7 @@ bool intel_dp_mst_verify_dpcd_state(struct intel_dp *intel_dp) if (!intel_dp->is_mst) return true; - ret = drm_dp_dpcd_readb(intel_dp->mst.mgr.aux, DP_MSTM_CTRL, &val); + ret = drm_dp_dpcd_read_byte(intel_dp->mst.mgr.aux, DP_MSTM_CTRL, &val); /* Adjust the expected register value for SST + SideBand. */ if (ret < 0 || val != (DP_MST_EN | DP_UP_REQ_EN | DP_UPSTREAM_IS_SRC)) { From 3216d9cc254bb03c0bec1084d7bfccff2ecbcc7a Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Wed, 16 Sep 2026 11:29:41 +0300 Subject: [PATCH 0257/1352] drm/i915/dp-test: switch to drm_dp_dpcd_{read_byte, read_data, write_byte, write_data}() Switch to the modern DPCD access functions that return negative error codes on errors, -EIO for incomplete access, and 0 on success. This flags incomplete reads, and simplifies error handling all over the place. Rename local int status to ret for clarity and consistency. Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/ddf90aeb0dae3779d92381f550d95111398e7e02.1789547331.git.jani.nikula@intel.com Signed-off-by: Jani Nikula --- drivers/gpu/drm/i915/display/intel_dp_test.c | 57 ++++++++++---------- 1 file changed, 27 insertions(+), 30 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_dp_test.c b/drivers/gpu/drm/i915/display/intel_dp_test.c index 0551a1ce60d392..424183a6515592 100644 --- a/drivers/gpu/drm/i915/display/intel_dp_test.c +++ b/drivers/gpu/drm/i915/display/intel_dp_test.c @@ -147,25 +147,24 @@ bool intel_dp_test_compute_config(struct intel_connector *connector, static u8 intel_dp_autotest_link_training(struct intel_dp *intel_dp) { struct intel_display *display = to_intel_display(intel_dp); - int status = 0; + int ret; int test_link_rate; u8 test_lane_count, test_link_bw; /* (DP CTS 1.2) * 4.3.1.11 */ /* Read the TEST_LANE_COUNT and TEST_LINK_RTAE fields (DP CTS 3.1.4) */ - status = drm_dp_dpcd_readb(&intel_dp->aux, DP_TEST_LANE_COUNT, - &test_lane_count); - - if (status <= 0) { + ret = drm_dp_dpcd_read_byte(&intel_dp->aux, DP_TEST_LANE_COUNT, + &test_lane_count); + if (ret < 0) { drm_dbg_kms(display->drm, "Lane count read failed\n"); return DP_TEST_NAK; } test_lane_count &= DP_MAX_LANE_COUNT_MASK; - status = drm_dp_dpcd_readb(&intel_dp->aux, DP_TEST_LINK_RATE, - &test_link_bw); - if (status <= 0) { + ret = drm_dp_dpcd_read_byte(&intel_dp->aux, DP_TEST_LINK_RATE, + &test_link_bw); + if (ret < 0) { drm_dbg_kms(display->drm, "Link Rate read failed\n"); return DP_TEST_NAK; } @@ -188,35 +187,33 @@ static u8 intel_dp_autotest_video_pattern(struct intel_dp *intel_dp) u8 test_pattern; u8 test_misc; __be16 h_width, v_height; - int status = 0; + int ret; /* Read the TEST_PATTERN (DP CTS 3.1.5) */ - status = drm_dp_dpcd_readb(&intel_dp->aux, DP_TEST_PATTERN, - &test_pattern); - if (status <= 0) { + ret = drm_dp_dpcd_read_byte(&intel_dp->aux, DP_TEST_PATTERN, + &test_pattern); + if (ret < 0) { drm_dbg_kms(display->drm, "Test pattern read failed\n"); return DP_TEST_NAK; } if (test_pattern != DP_COLOR_RAMP) return DP_TEST_NAK; - status = drm_dp_dpcd_read(&intel_dp->aux, DP_TEST_H_WIDTH_HI, - &h_width, 2); - if (status <= 0) { + ret = drm_dp_dpcd_read_data(&intel_dp->aux, DP_TEST_H_WIDTH_HI, &h_width, 2); + if (ret < 0) { drm_dbg_kms(display->drm, "H Width read failed\n"); return DP_TEST_NAK; } - status = drm_dp_dpcd_read(&intel_dp->aux, DP_TEST_V_HEIGHT_HI, - &v_height, 2); - if (status <= 0) { + ret = drm_dp_dpcd_read_data(&intel_dp->aux, DP_TEST_V_HEIGHT_HI, + &v_height, 2); + if (ret < 0) { drm_dbg_kms(display->drm, "V Height read failed\n"); return DP_TEST_NAK; } - status = drm_dp_dpcd_readb(&intel_dp->aux, DP_TEST_MISC0, - &test_misc); - if (status <= 0) { + ret = drm_dp_dpcd_read_byte(&intel_dp->aux, DP_TEST_MISC0, &test_misc); + if (ret < 0) { drm_dbg_kms(display->drm, "TEST MISC read failed\n"); return DP_TEST_NAK; } @@ -274,8 +271,8 @@ static u8 intel_dp_autotest_edid(struct intel_dp *intel_dp) /* We have to write the checksum of the last block read */ block += block->extensions; - if (drm_dp_dpcd_writeb(&intel_dp->aux, DP_TEST_EDID_CHECKSUM, - block->checksum) <= 0) + if (drm_dp_dpcd_write_byte(&intel_dp->aux, DP_TEST_EDID_CHECKSUM, + block->checksum) < 0) drm_dbg_kms(display->drm, "Failed to write EDID checksum\n"); @@ -397,8 +394,8 @@ static void intel_dp_process_phy_request(struct intel_dp *intel_dp, intel_dp_phy_pattern_update(intel_dp, crtc_state); - drm_dp_dpcd_write(&intel_dp->aux, DP_TRAINING_LANE0_SET, - intel_dp->train_set, crtc_state->lane_count); + drm_dp_dpcd_write_data(&intel_dp->aux, DP_TRAINING_LANE0_SET, + intel_dp->train_set, crtc_state->lane_count); drm_dp_set_phy_test_pattern(&intel_dp->aux, data, intel_dp->dpcd[DP_DPCD_REV]); @@ -427,10 +424,10 @@ void intel_dp_test_request(struct intel_dp *intel_dp) struct intel_display *display = to_intel_display(intel_dp); u8 response = DP_TEST_NAK; u8 request = 0; - int status; + int ret; - status = drm_dp_dpcd_readb(&intel_dp->aux, DP_TEST_REQUEST, &request); - if (status <= 0) { + ret = drm_dp_dpcd_read_byte(&intel_dp->aux, DP_TEST_REQUEST, &request); + if (ret < 0) { drm_dbg_kms(display->drm, "Could not read test request from sink\n"); goto update_status; @@ -463,8 +460,8 @@ void intel_dp_test_request(struct intel_dp *intel_dp) intel_dp->compliance.test_type = request; update_status: - status = drm_dp_dpcd_writeb(&intel_dp->aux, DP_TEST_RESPONSE, response); - if (status <= 0) + ret = drm_dp_dpcd_write_byte(&intel_dp->aux, DP_TEST_RESPONSE, response); + if (ret < 0) drm_dbg_kms(display->drm, "Could not write test response to sink\n"); } From b6413a8f75e642bc4e75dcb3c25c7a8b5f451d33 Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Wed, 16 Sep 2026 11:29:42 +0300 Subject: [PATCH 0258/1352] drm/i915/psr: switch to drm_dp_dpcd_{read_byte, read_data, write_byte, write_data}() Switch to the modern DPCD access functions that return negative error codes on errors, -EIO for incomplete access, and 0 on success. This flags incomplete reads, and clarifies error handling all over the place. Use 1-byte accessors where sensible. Rename local int r to ret for clarity and consistency. Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/0a32cfdb64a37e586579a689bd8c3c4944596a31.1789547331.git.jani.nikula@intel.com Signed-off-by: Jani Nikula --- drivers/gpu/drm/i915/display/intel_psr.c | 65 ++++++++++++------------ 1 file changed, 33 insertions(+), 32 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_psr.c b/drivers/gpu/drm/i915/display/intel_psr.c index 872e253db1786b..f1d48b69a18fac 100644 --- a/drivers/gpu/drm/i915/display/intel_psr.c +++ b/drivers/gpu/drm/i915/display/intel_psr.c @@ -471,9 +471,10 @@ static u8 intel_dp_get_sink_sync_latency(struct intel_dp *intel_dp) { struct intel_display *display = to_intel_display(intel_dp); u8 val = 8; /* assume the worst if we can't read the value */ + int ret; - if (drm_dp_dpcd_readb(&intel_dp->aux, - DP_SYNCHRONIZATION_LATENCY_IN_SINK, &val) == 1) + ret = drm_dp_dpcd_read_byte(&intel_dp->aux, DP_SYNCHRONIZATION_LATENCY_IN_SINK, &val); + if (!ret) val &= DP_MAX_RESYNC_FRAME_COUNT_MASK; else drm_dbg_kms(display->drm, @@ -485,7 +486,7 @@ static void _psr_compute_su_granularity(struct intel_dp *intel_dp, struct intel_connector *connector) { struct intel_display *display = to_intel_display(intel_dp); - ssize_t r; + int ret; __le16 w; u8 y; @@ -500,19 +501,19 @@ static void _psr_compute_su_granularity(struct intel_dp *intel_dp, goto exit; } - r = drm_dp_dpcd_read(&intel_dp->aux, DP_PSR2_SU_X_GRANULARITY, &w, sizeof(w)); - if (r != sizeof(w)) + ret = drm_dp_dpcd_read_data(&intel_dp->aux, DP_PSR2_SU_X_GRANULARITY, &w, sizeof(w)); + if (ret < 0) drm_dbg_kms(display->drm, "Unable to read selective update x granularity\n"); /* * Spec says that if the value read is 0 the default granularity should * be used instead. */ - if (r != sizeof(w) || w == 0) + if (ret < 0 || w == 0) w = cpu_to_le16(4); - r = drm_dp_dpcd_read(&intel_dp->aux, DP_PSR2_SU_Y_GRANULARITY, &y, 1); - if (r != 1) { + ret = drm_dp_dpcd_read_byte(&intel_dp->aux, DP_PSR2_SU_Y_GRANULARITY, &y); + if (ret < 0) { drm_dbg_kms(display->drm, "Unable to read selective update y granularity\n"); y = 4; @@ -796,11 +797,11 @@ static void _panel_replay_enable_sink(struct intel_dp *intel_dp, panel_replay_config[1] |= DP_PANEL_REPLAY_SU_REGION_SCANLINE_CAPTURE; - drm_dp_dpcd_write(&intel_dp->aux, PANEL_REPLAY_CONFIG, - panel_replay_config, sizeof(panel_replay_config)); + drm_dp_dpcd_write_data(&intel_dp->aux, PANEL_REPLAY_CONFIG, + panel_replay_config, sizeof(panel_replay_config)); panel_replay_config_3 = intel_dp_as_sdp_transmission_time(); - drm_dp_dpcd_writeb(&intel_dp->aux, PANEL_REPLAY_CONFIG3, panel_replay_config_3); + drm_dp_dpcd_write_byte(&intel_dp->aux, PANEL_REPLAY_CONFIG3, panel_replay_config_3); } static void _psr_enable_sink(struct intel_dp *intel_dp, @@ -827,10 +828,10 @@ static void _psr_enable_sink(struct intel_dp *intel_dp, if (intel_dp->psr.entry_setup_frames > 0) val |= DP_PSR_FRAME_CAPTURE; - drm_dp_dpcd_writeb(&intel_dp->aux, DP_PSR_EN_CFG, val); + drm_dp_dpcd_write_byte(&intel_dp->aux, DP_PSR_EN_CFG, val); val |= DP_PSR_ENABLE; - drm_dp_dpcd_writeb(&intel_dp->aux, DP_PSR_EN_CFG, val); + drm_dp_dpcd_write_byte(&intel_dp->aux, DP_PSR_EN_CFG, val); } static void intel_psr_enable_sink(struct intel_dp *intel_dp, @@ -843,7 +844,7 @@ static void intel_psr_enable_sink(struct intel_dp *intel_dp, _psr_enable_sink(intel_dp, crtc_state); if (intel_dp_is_edp(intel_dp)) - drm_dp_dpcd_writeb(&intel_dp->aux, DP_SET_POWER, DP_SET_POWER_D0); + drm_dp_dpcd_write_byte(&intel_dp->aux, DP_SET_POWER, DP_SET_POWER_D0); } void intel_psr_panel_replay_enable_sink(struct intel_dp *intel_dp) @@ -854,8 +855,8 @@ void intel_psr_panel_replay_enable_sink(struct intel_dp *intel_dp) * ensure this bit is cleared/set accordingly. */ if (CAN_PANEL_REPLAY(intel_dp) && panel_replay_global_enabled(intel_dp)) - drm_dp_dpcd_writeb(&intel_dp->aux, PANEL_REPLAY_CONFIG, - DP_PANEL_REPLAY_ENABLE); + drm_dp_dpcd_write_byte(&intel_dp->aux, PANEL_REPLAY_CONFIG, + DP_PANEL_REPLAY_ENABLE); } static u32 intel_psr1_get_tp_time(struct intel_dp *intel_dp) @@ -2373,11 +2374,11 @@ static void intel_psr_disable_locked(struct intel_dp *intel_dp) /* Disable PSR on Sink */ if (!intel_dp->psr.panel_replay_enabled) { - drm_dp_dpcd_writeb(&intel_dp->aux, DP_PSR_EN_CFG, 0); + drm_dp_dpcd_write_byte(&intel_dp->aux, DP_PSR_EN_CFG, 0); if (intel_dp->psr.sel_update_enabled) - drm_dp_dpcd_writeb(&intel_dp->aux, - DP_RECEIVER_ALPM_CONFIG, 0); + drm_dp_dpcd_write_byte(&intel_dp->aux, + DP_RECEIVER_ALPM_CONFIG, 0); } /* Wa_16025596647 */ @@ -3517,7 +3518,7 @@ static void intel_psr_handle_irq(struct intel_dp *intel_dp) intel_psr_disable_locked(intel_dp); psr->sink_not_reliable = true; /* let's make sure that sink is awaken */ - drm_dp_dpcd_writeb(&intel_dp->aux, DP_SET_POWER, DP_SET_POWER_D0); + drm_dp_dpcd_write_byte(&intel_dp->aux, DP_SET_POWER, DP_SET_POWER_D0); } static void intel_psr_work(struct work_struct *work) @@ -3796,15 +3797,15 @@ static int psr_get_status_and_error_status(struct intel_dp *intel_dp, offset = intel_dp->psr.panel_replay_enabled ? DP_SINK_DEVICE_PR_AND_FRAME_LOCK_STATUS : DP_PSR_STATUS; - ret = drm_dp_dpcd_readb(aux, offset, status); - if (ret != 1) + ret = drm_dp_dpcd_read_byte(aux, offset, status); + if (ret < 0) return ret; offset = intel_dp->psr.panel_replay_enabled ? DP_PANEL_REPLAY_ERROR_STATUS : DP_PSR_ERROR_STATUS; - ret = drm_dp_dpcd_readb(aux, offset, error_status); - if (ret != 1) + ret = drm_dp_dpcd_read_byte(aux, offset, error_status); + if (ret < 0) return ret; *status = *status & DP_PSR_SINK_STATE_MASK; @@ -3829,11 +3830,11 @@ static void psr_capability_changed_check(struct intel_dp *intel_dp) { struct intel_display *display = to_intel_display(intel_dp); struct intel_psr *psr = &intel_dp->psr; + int ret; u8 val; - int r; - r = drm_dp_dpcd_readb(&intel_dp->aux, DP_PSR_ESI, &val); - if (r != 1) { + ret = drm_dp_dpcd_read_byte(&intel_dp->aux, DP_PSR_ESI, &val); + if (ret < 0) { drm_err(display->drm, "Error reading DP_PSR_ESI\n"); return; } @@ -3845,7 +3846,7 @@ static void psr_capability_changed_check(struct intel_dp *intel_dp) "Sink PSR capability changed, disabling PSR\n"); /* Clearing it */ - drm_dp_dpcd_writeb(&intel_dp->aux, DP_PSR_ESI, val); + drm_dp_dpcd_write_byte(&intel_dp->aux, DP_PSR_ESI, val); } } @@ -3913,10 +3914,10 @@ void intel_psr_short_pulse(struct intel_dp *intel_dp) "PSR_ERROR_STATUS unhandled errors %x\n", error_status & ~errors); /* clear status register */ - drm_dp_dpcd_writeb(&intel_dp->aux, - panel_replay_enabled ? - DP_PANEL_REPLAY_ERROR_STATUS : DP_PSR_ERROR_STATUS, - error_status); + drm_dp_dpcd_write_byte(&intel_dp->aux, + panel_replay_enabled ? + DP_PANEL_REPLAY_ERROR_STATUS : DP_PSR_ERROR_STATUS, + error_status); if (!psr->panel_replay_enabled) { psr_alpm_check(intel_dp); From bcaa4d1d69368b6034b1ceb226ffd0a203a5fbc7 Mon Sep 17 00:00:00 2001 From: Manuel Ebner Date: Wed, 2 Sep 2026 17:29:10 +0200 Subject: [PATCH 0259/1352] ipe: fix invalid sgid value in audit event documentation The audit event example in the IPE documentation contains 'sgid=)', which is not a valid Set Group ID value. Fix it to use 'sgid=0'. Fixes: ac6731870ed9 ("documentation: add IPE documentation") Signed-off-by: Manuel Ebner Reviewed-by: Randy Dunlap Signed-off-by: Fan Wu --- Documentation/admin-guide/LSM/ipe.rst | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/Documentation/admin-guide/LSM/ipe.rst b/Documentation/admin-guide/LSM/ipe.rst index a756d81585317c..bebf14fc241166 100644 --- a/Documentation/admin-guide/LSM/ipe.rst +++ b/Documentation/admin-guide/LSM/ipe.rst @@ -502,11 +502,11 @@ The following table lists the error codes that may appear in the errno field whi Event Examples:: type=1404 audit(1653425689.008:55): enforcing=0 old_enforcing=1 auid=4294967295 ses=4294967295 enabled=1 old-enabled=1 lsm=ipe res=1 - type=1300 audit(1653425689.008:55): arch=c000003e syscall=1 success=yes exit=2 a0=1 a1=55c1065e5c60 a2=2 a3=0 items=0 ppid=405 pid=441 auid=0 uid=0 gid=0 euid=0 suid=0 fsuid=0 egid=0 sgid=) + type=1300 audit(1653425689.008:55): arch=c000003e syscall=1 success=yes exit=2 a0=1 a1=55c1065e5c60 a2=2 a3=0 items=0 ppid=405 pid=441 auid=0 uid=0 gid=0 euid=0 suid=0 fsuid=0 egid=0 sgid=0 type=1327 audit(1653425689.008:55): proctitle="-bash" type=1404 audit(1653425689.008:55): enforcing=1 old_enforcing=0 auid=4294967295 ses=4294967295 enabled=1 old-enabled=1 lsm=ipe res=1 - type=1300 audit(1653425689.008:55): arch=c000003e syscall=1 success=yes exit=2 a0=1 a1=55c1065e5c60 a2=2 a3=0 items=0 ppid=405 pid=441 auid=0 uid=0 gid=0 euid=0 suid=0 fsuid=0 egid=0 sgid=) + type=1300 audit(1653425689.008:55): arch=c000003e syscall=1 success=yes exit=2 a0=1 a1=55c1065e5c60 a2=2 a3=0 items=0 ppid=405 pid=441 auid=0 uid=0 gid=0 euid=0 suid=0 fsuid=0 egid=0 sgid=0 type=1327 audit(1653425689.008:55): proctitle="-bash" This record will always be emitted in conjunction with a ``AUDITSYSCALL`` record for the ``write`` syscall. From 6620025ff272821e5c8612acbc83a0cb39886215 Mon Sep 17 00:00:00 2001 From: "Darrick J. Wong" Date: Tue, 22 Sep 2026 22:57:37 -0700 Subject: [PATCH 0260/1352] xfs: fix missing xfs_qm_adjust_dqlimits call in quotacheck repair LOLLM noticed that everywhere else in the kernel, a call to xfs_qm_adjust_dqlimits precedes every call to xfs_qm_adjust_dqtimers. In particular, mount-time quotacheck does this, but online quotacheck does not. Looking at xfs_qm_adjust_dqlimits, that function is in charge of conveying default limits to a dquot if that dquot's limits have been zeroed. That's quite possible in a repair, so we actually need to do that. However, there's a pre-existing pattern in the kernel -- for non-root dquots, first we call xfs_qm_adjust_dqlimits to set the dquot's limits to the defaults if they are zero, and then xfs_qm_adjust_dqtimers to start grace periods if the dquot's usage is above the softlimit. The grace period decision cannot be made correctly if we forget to import the default limits. Therefore, let's combine both into a single xfs_qm_adjust_dqenforcement helper that takes care of both pieces, which fixes quotacheck and makes it hard to repeat this mistake. Cc: stable@vger.kernel.org # v6.9 Fixes: 96ed2ae4a9b06b ("xfs: repair dquots based on live quotacheck results") Signed-off-by: Darrick J. Wong Assisted-by: LOLLM # finding obvious bugs Reviewed-by: Christoph Hellwig Signed-off-by: Carlos Maiolino --- fs/xfs/scrub/quota_repair.c | 5 +---- fs/xfs/scrub/quotacheck_repair.c | 3 +-- fs/xfs/xfs_dquot.c | 20 +++++++++++++++----- fs/xfs/xfs_dquot.h | 2 +- fs/xfs/xfs_qm.c | 6 +----- fs/xfs/xfs_trans_dquot.c | 6 +----- 6 files changed, 20 insertions(+), 22 deletions(-) diff --git a/fs/xfs/scrub/quota_repair.c b/fs/xfs/scrub/quota_repair.c index 59302e8afc7ef0..89f7ea4f92ef4a 100644 --- a/fs/xfs/scrub/quota_repair.c +++ b/fs/xfs/scrub/quota_repair.c @@ -248,10 +248,7 @@ xrep_quota_item( dq->q_flags |= XFS_DQFLAG_DIRTY; xfs_trans_dqjoin(sc->tp, dq); - if (dq->q_id) { - xfs_qm_adjust_dqlimits(dq); - xfs_qm_adjust_dqtimers(dq); - } + xfs_qm_adjust_dqenforcement(dq); xfs_trans_log_dquot(sc->tp, dq); return xfs_trans_roll(&sc->tp); diff --git a/fs/xfs/scrub/quotacheck_repair.c b/fs/xfs/scrub/quotacheck_repair.c index dbb522e1513b0b..48ee08df302a42 100644 --- a/fs/xfs/scrub/quotacheck_repair.c +++ b/fs/xfs/scrub/quotacheck_repair.c @@ -110,8 +110,7 @@ xqcheck_commit_dquot( /* Commit the dirty dquot to disk. */ dq->q_flags |= XFS_DQFLAG_DIRTY; - if (dq->q_id) - xfs_qm_adjust_dqtimers(dq); + xfs_qm_adjust_dqenforcement(dq); xfs_trans_log_dquot(xqc->sc->tp, dq); return xrep_trans_commit(xqc->sc); diff --git a/fs/xfs/xfs_dquot.c b/fs/xfs/xfs_dquot.c index e696ee36c2e8d2..6d22ead562da58 100644 --- a/fs/xfs/xfs_dquot.c +++ b/fs/xfs/xfs_dquot.c @@ -115,18 +115,15 @@ xfs_qm_dqdestroy( * We overwrite the dquot limits only if they are zero and this * is not the root dquot. */ -void +static void xfs_qm_adjust_dqlimits( struct xfs_dquot *dq) { struct xfs_mount *mp = dq->q_mount; struct xfs_quotainfo *q = mp->m_quotainfo; - struct xfs_def_quota *defq; + struct xfs_def_quota *defq = xfs_get_defquota(q, xfs_dquot_type(dq)); int prealloc = 0; - ASSERT(dq->q_id); - defq = xfs_get_defquota(q, xfs_dquot_type(dq)); - if (!dq->q_blk.softlimit) { dq->q_blk.softlimit = defq->blk.soft; prealloc = 1; @@ -223,6 +220,19 @@ xfs_qm_adjust_dqtimers( xfs_qm_adjust_res_timer(dq->q_mount, &dq->q_rtb, &defq->rtb); } +/* Adjust enforcement limits and timers after a change in usage. */ +void +xfs_qm_adjust_dqenforcement( + struct xfs_dquot *dq) +{ + if (dq->q_id == 0) + return; + + xfs_qm_adjust_dqlimits(dq); + xfs_qm_adjust_dqtimers(dq); + dq->q_flags |= XFS_DQFLAG_DIRTY; +} + /* * initialize a buffer full of dquots and log the whole thing */ diff --git a/fs/xfs/xfs_dquot.h b/fs/xfs/xfs_dquot.h index bbb824adca82ce..28c09704acc261 100644 --- a/fs/xfs/xfs_dquot.h +++ b/fs/xfs/xfs_dquot.h @@ -205,7 +205,7 @@ void xfs_qm_dqdestroy(struct xfs_dquot *dqp); int xfs_qm_dqflush(struct xfs_dquot *dqp, struct xfs_buf *bp); void xfs_qm_dqunpin_wait(struct xfs_dquot *dqp); void xfs_qm_adjust_dqtimers(struct xfs_dquot *d); -void xfs_qm_adjust_dqlimits(struct xfs_dquot *d); +void xfs_qm_adjust_dqenforcement(struct xfs_dquot *d); xfs_dqid_t xfs_qm_id_for_quotatype(struct xfs_inode *ip, xfs_dqtype_t type); int xfs_qm_dqget(struct xfs_mount *mp, xfs_dqid_t id, diff --git a/fs/xfs/xfs_qm.c b/fs/xfs/xfs_qm.c index 54d00d543b513a..008fed8624be2c 100644 --- a/fs/xfs/xfs_qm.c +++ b/fs/xfs/xfs_qm.c @@ -1294,11 +1294,7 @@ xfs_qm_quotacheck_dqadjust( * * There are no timers for the default values set in the root dquot. */ - if (dqp->q_id) { - xfs_qm_adjust_dqlimits(dqp); - xfs_qm_adjust_dqtimers(dqp); - } - + xfs_qm_adjust_dqenforcement(dqp); dqp->q_flags |= XFS_DQFLAG_DIRTY; out_unlock: mutex_unlock(&dqp->q_qlock); diff --git a/fs/xfs/xfs_trans_dquot.c b/fs/xfs/xfs_trans_dquot.c index 1606c614f205ae..93ec876792cc76 100644 --- a/fs/xfs/xfs_trans_dquot.c +++ b/fs/xfs/xfs_trans_dquot.c @@ -566,11 +566,7 @@ xfs_trans_apply_dquot_deltas( * Get any default limits in use. * Start/reset the timer(s) if needed. */ - if (dqp->q_id) { - xfs_qm_adjust_dqlimits(dqp); - xfs_qm_adjust_dqtimers(dqp); - } - + xfs_qm_adjust_dqenforcement(dqp); dqp->q_flags |= XFS_DQFLAG_DIRTY; /* * add this to the list of items to get logged From 2d72322bb2b75157935020c7118f307c6aa11fc7 Mon Sep 17 00:00:00 2001 From: "Darrick J. Wong" Date: Tue, 22 Sep 2026 22:57:53 -0700 Subject: [PATCH 0261/1352] xfs: online quotacheck must dirty dquot if enforcement adjustments needed LOLLM noticed that we have no way to force xchk_commit_dquot to call xfs_qm_adjust_dqenforcement if nothing else is wrong with the dquot. Therefore, add a new predicate to force the dirty flag if the dquot has zero limits and there are default limits; or if the grace period timer needs adjusting. Cc: stable@vger.kernel.org # v6.9 Fixes: 96ed2ae4a9b06b ("xfs: repair dquots based on live quotacheck results") Signed-off-by: Darrick J. Wong Assisted-by: LOLLM # finding obvious bugs Reviewed-by: Christoph Hellwig Signed-off-by: Carlos Maiolino --- fs/xfs/scrub/quotacheck_repair.c | 51 ++++++++++++++++++++++++++++++++ 1 file changed, 51 insertions(+) diff --git a/fs/xfs/scrub/quotacheck_repair.c b/fs/xfs/scrub/quotacheck_repair.c index 48ee08df302a42..b8334edf380c7f 100644 --- a/fs/xfs/scrub/quotacheck_repair.c +++ b/fs/xfs/scrub/quotacheck_repair.c @@ -39,6 +39,54 @@ * dquot is locked. */ +static bool +xqcheck_dqres_force_dirty( + const struct xfs_dquot_res *res, + const struct xfs_quota_limits *qlim) +{ + /* zero limits mean that we should set the default limits */ + if (res->softlimit == 0 && qlim->soft != 0) + return true; + if (res->hardlimit == 0 && qlim->hard != 0) + return true; + + /* do we need to adjust the timer setting? */ + if ((res->softlimit && res->count > res->softlimit) || + (res->hardlimit && res->count > res->hardlimit)) { + if (!res->timer) + return true; + } else { + if (res->timer) + return true; + } + + return false; +} + +/* Decide if we need to adjust the dquot limits or timers */ +static bool +xqcheck_dquot_force_dirty( + const struct xfs_dquot *dq) +{ + struct xfs_quotainfo *qi = dq->q_mount->m_quotainfo; + struct xfs_def_quota *defq; + + /* root dquot does not enforce limits */ + if (dq->q_id == 0) + return false; + + defq = xfs_get_defquota(qi, xfs_dquot_type(dq)); + + if (xqcheck_dqres_force_dirty(&dq->q_blk, &defq->blk)) + return true; + if (xqcheck_dqres_force_dirty(&dq->q_ino, &defq->ino)) + return true; + if (xqcheck_dqres_force_dirty(&dq->q_rtb, &defq->rtb)) + return true; + + return false; +} + /* Commit new counters to a dquot. */ static int xqcheck_commit_dquot( @@ -91,6 +139,9 @@ xqcheck_commit_dquot( dirty = true; } + if (!dirty && xqcheck_dquot_force_dirty(dq)) + dirty = true; + xcdq.flags |= (XQCHECK_DQUOT_REPAIR_SCANNED | XQCHECK_DQUOT_WRITTEN); error = xfarray_store(counts, dq->q_id, &xcdq); if (error == -EFBIG) { From 9d13120c34462d666b0efff0f98afbb7a96dd20e Mon Sep 17 00:00:00 2001 From: "Darrick J. Wong" Date: Tue, 22 Sep 2026 22:58:08 -0700 Subject: [PATCH 0262/1352] xfs: fix buffer overruns in xfs_ioc_attr_list When we converted the al_offset array in struct xfs_attrlist into a VLA, the size of the object shrank by 4 bytes. Unfortunately, the buffer size validation in the attrlist ioctl wasn't updated to notice this, so the al_offset[0] assignment blindly writes off the end of the buffer. LOLLM noticed the omitted check and complained. Cc: stable@vger.kernel.org # v6.5 Fixes: 371baf5c9750a2 ("xfs: convert flex-array declarations in struct xfs_attrlist*") Signed-off-by: Darrick J. Wong Assisted-by: LOLLM # finding obvious bugs Reviewed-by: Christoph Hellwig Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_handle.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/xfs/xfs_handle.c b/fs/xfs/xfs_handle.c index 0689cade8f74c2..fd9d4d8258fff2 100644 --- a/fs/xfs/xfs_handle.c +++ b/fs/xfs/xfs_handle.c @@ -409,7 +409,7 @@ xfs_ioc_attr_list( void *buffer; int error; - if (bufsize < sizeof(struct xfs_attrlist) || + if (bufsize < struct_size(alist, al_offset, 1) || bufsize > XFS_XATTR_LIST_MAX) return -EINVAL; From 37340d63ac1c80cd1ac349df55fe70f3dfd3b8d9 Mon Sep 17 00:00:00 2001 From: "Darrick J. Wong" Date: Tue, 22 Sep 2026 22:58:24 -0700 Subject: [PATCH 0263/1352] xfs: clean up after failed metafile relinking Once we start a metadir update to link in a file, we have to commit or cancel it, just like any other operation. LOLLM pointed out that I forgot that, so fix it. This fixes a bug in xfs_repair. However, in commit e80fbe1ad8eff7, we made xfs_metadir_cancel a static function within xfs_metadir.c, so we can't just add a xfs_metadir_cancel call to xfs_dqinode_metadir_link. Instead, create a new xfs_metadir_link_file helper in xfs_metadir.c that takes only the xfs_metadir_update object, and handles everything from start to finish. This enables us to make xfs_metadir_commit a static function too. Fixes: e80fbe1ad8eff7 ("xfs: use metadir for quota inodes") Signed-off-by: Darrick J. Wong Reviewed-by: Christoph Hellwig Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_dquot_buf.c | 13 +------------ fs/xfs/libxfs/xfs_metadir.c | 30 +++++++++++++++++++++++++++--- fs/xfs/libxfs/xfs_metadir.h | 5 +---- 3 files changed, 29 insertions(+), 19 deletions(-) diff --git a/fs/xfs/libxfs/xfs_dquot_buf.c b/fs/xfs/libxfs/xfs_dquot_buf.c index 77954d1d924cc3..6c8a9afdfb1c0f 100644 --- a/fs/xfs/libxfs/xfs_dquot_buf.c +++ b/fs/xfs/libxfs/xfs_dquot_buf.c @@ -454,19 +454,8 @@ xfs_dqinode_metadir_link( .path = xfs_dqinode_path(type), .ip = ip, }; - int error; - error = xfs_metadir_start_link(&upd); - if (error) - return error; - - error = xfs_metadir_link(&upd); - if (error) - return error; - - xfs_trans_log_inode(upd.tp, upd.ip, XFS_ILOG_CORE); - - return xfs_metadir_commit(&upd); + return xfs_metadir_link_file(&upd); } #endif /* __KERNEL__ */ diff --git a/fs/xfs/libxfs/xfs_metadir.c b/fs/xfs/libxfs/xfs_metadir.c index 7c6b086b73db61..1ff26e55b1864f 100644 --- a/fs/xfs/libxfs/xfs_metadir.c +++ b/fs/xfs/libxfs/xfs_metadir.c @@ -317,7 +317,7 @@ xfs_metadir_create( * Begin the process of linking a metadata file by allocating transactions * and locking whatever resources we're going to need. */ -int +static int xfs_metadir_start_link( struct xfs_metadir_update *upd) { @@ -364,7 +364,7 @@ xfs_metadir_start_link( * The path (up to the final component) must already exist, but the final * component must not already exist. */ -int +static int xfs_metadir_link( struct xfs_metadir_update *upd) { @@ -409,7 +409,7 @@ xfs_metadir_link( #endif /* ! __KERNEL__ */ /* Commit a metadir update and unlock/drop all resources. */ -int +static int xfs_metadir_commit( struct xfs_metadir_update *upd) { @@ -499,3 +499,27 @@ xfs_metadir_mkdir( return xfs_metadir_create_file(&upd, S_IFDIR, NULL, NULL, ipp); } + +#ifndef __KERNEL__ +/* Link a metadata file into a metadata directory. */ +int +xfs_metadir_link_file( + struct xfs_metadir_update *upd) +{ + int error; + + error = xfs_metadir_start_link(upd); + if (error) + return error; + + error = xfs_metadir_link(upd); + if (error) { + xfs_metadir_cancel(upd, error); + return error; + } + + xfs_trans_log_inode(upd->tp, upd->ip, XFS_ILOG_CORE); + + return xfs_metadir_commit(upd); +} +#endif /* ! __KERNEL__ */ diff --git a/fs/xfs/libxfs/xfs_metadir.h b/fs/xfs/libxfs/xfs_metadir.h index e434b9d1c93200..b64f9fc5ca7867 100644 --- a/fs/xfs/libxfs/xfs_metadir.h +++ b/fs/xfs/libxfs/xfs_metadir.h @@ -38,10 +38,7 @@ int xfs_metadir_create_file(struct xfs_metadir_update *upd, umode_t mode, xfs_metadir_createfn create, void *priv, struct xfs_inode **ipp); -int xfs_metadir_start_link(struct xfs_metadir_update *upd); -int xfs_metadir_link(struct xfs_metadir_update *upd); - -int xfs_metadir_commit(struct xfs_metadir_update *upd); +int xfs_metadir_link_file(struct xfs_metadir_update *upd); int xfs_metadir_mkdir(struct xfs_inode *dp, const char *path, struct xfs_inode **ipp); From 25ba4acda6bee103e781d24a944ad7b4794818b1 Mon Sep 17 00:00:00 2001 From: "Darrick J. Wong" Date: Tue, 22 Sep 2026 22:58:39 -0700 Subject: [PATCH 0264/1352] xfs: pass xfs_trans_resv object to reservation calculation helpers xfs_calc_namespace_reservations computes the directory tree related transaction reservations for a given xfs_trans_resv object. The helpers it relies on, however, read the live one from the xfs_mount even if we're doing this for minlogsize calculations. In practice this shouldn't be a big deal since the minlogsize and live reservation objects don't differ in a meaningful way, but LOLLM complained about the inconsistency so let's fix it anyway. Fixes: 7dba4a5fe1c5cd ("xfs: extend transaction reservations for parent attributes") Signed-off-by: Darrick J. Wong Assisted-by: LOLLM # finding obvious bugs Reviewed-by: Christoph Hellwig Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_trans_resv.c | 38 ++++++++++++++++++---------------- 1 file changed, 20 insertions(+), 18 deletions(-) diff --git a/fs/xfs/libxfs/xfs_trans_resv.c b/fs/xfs/libxfs/xfs_trans_resv.c index 3151e97ca8ff6d..1c20a7c27aa1bf 100644 --- a/fs/xfs/libxfs/xfs_trans_resv.c +++ b/fs/xfs/libxfs/xfs_trans_resv.c @@ -606,10 +606,10 @@ static inline unsigned int xfs_calc_pptr_replace_overhead(void) */ STATIC uint xfs_calc_rename_reservation( - struct xfs_mount *mp) + struct xfs_mount *mp, + struct xfs_trans_resv *resp) { unsigned int overhead = XFS_DQUOT_LOGRES; - struct xfs_trans_resv *resp = M_RES(mp); unsigned int t1, t2, t3 = 0; t1 = xfs_calc_inode_res(mp, 5) + @@ -715,10 +715,10 @@ xfs_link_log_count( */ STATIC uint xfs_calc_link_reservation( - struct xfs_mount *mp) + struct xfs_mount *mp, + struct xfs_trans_resv *resp) { unsigned int overhead = XFS_DQUOT_LOGRES; - struct xfs_trans_resv *resp = M_RES(mp); unsigned int t1, t2, t3 = 0; overhead += xfs_calc_iunlink_remove_reservation(mp); @@ -777,10 +777,10 @@ xfs_remove_log_count( */ STATIC uint xfs_calc_remove_reservation( - struct xfs_mount *mp) + struct xfs_mount *mp, + struct xfs_trans_resv *resp) { unsigned int overhead = XFS_DQUOT_LOGRES; - struct xfs_trans_resv *resp = M_RES(mp); unsigned int t1, t2, t3 = 0; overhead += xfs_calc_iunlink_add_reservation(mp); @@ -862,9 +862,9 @@ xfs_icreate_log_count( STATIC uint xfs_calc_icreate_reservation( - struct xfs_mount *mp) + struct xfs_mount *mp, + struct xfs_trans_resv *resp) { - struct xfs_trans_resv *resp = M_RES(mp); unsigned int overhead = XFS_DQUOT_LOGRES; unsigned int t1, t2, t3 = 0; @@ -911,9 +911,10 @@ xfs_mkdir_log_count( */ STATIC uint xfs_calc_mkdir_reservation( - struct xfs_mount *mp) + struct xfs_mount *mp, + struct xfs_trans_resv *resp) { - return xfs_calc_icreate_reservation(mp); + return xfs_calc_icreate_reservation(mp, resp); } static inline unsigned int @@ -940,9 +941,10 @@ xfs_symlink_log_count( */ STATIC uint xfs_calc_symlink_reservation( - struct xfs_mount *mp) + struct xfs_mount *mp, + struct xfs_trans_resv *resp) { - return xfs_calc_icreate_reservation(mp) + + return xfs_calc_icreate_reservation(mp, resp) + xfs_calc_buf_res(1, XFS_SYMLINK_MAXLEN); } @@ -1265,27 +1267,27 @@ xfs_calc_namespace_reservations( { ASSERT(resp->tr_attrsetm.tr_logres > 0); - resp->tr_rename.tr_logres = xfs_calc_rename_reservation(mp); + resp->tr_rename.tr_logres = xfs_calc_rename_reservation(mp, resp); resp->tr_rename.tr_logcount = xfs_rename_log_count(mp, resp); resp->tr_rename.tr_logflags |= XFS_TRANS_PERM_LOG_RES; - resp->tr_link.tr_logres = xfs_calc_link_reservation(mp); + resp->tr_link.tr_logres = xfs_calc_link_reservation(mp, resp); resp->tr_link.tr_logcount = xfs_link_log_count(mp, resp); resp->tr_link.tr_logflags |= XFS_TRANS_PERM_LOG_RES; - resp->tr_remove.tr_logres = xfs_calc_remove_reservation(mp); + resp->tr_remove.tr_logres = xfs_calc_remove_reservation(mp, resp); resp->tr_remove.tr_logcount = xfs_remove_log_count(mp, resp); resp->tr_remove.tr_logflags |= XFS_TRANS_PERM_LOG_RES; - resp->tr_symlink.tr_logres = xfs_calc_symlink_reservation(mp); + resp->tr_symlink.tr_logres = xfs_calc_symlink_reservation(mp, resp); resp->tr_symlink.tr_logcount = xfs_symlink_log_count(mp, resp); resp->tr_symlink.tr_logflags |= XFS_TRANS_PERM_LOG_RES; - resp->tr_create.tr_logres = xfs_calc_icreate_reservation(mp); + resp->tr_create.tr_logres = xfs_calc_icreate_reservation(mp, resp); resp->tr_create.tr_logcount = xfs_icreate_log_count(mp, resp); resp->tr_create.tr_logflags |= XFS_TRANS_PERM_LOG_RES; - resp->tr_mkdir.tr_logres = xfs_calc_mkdir_reservation(mp); + resp->tr_mkdir.tr_logres = xfs_calc_mkdir_reservation(mp, resp); resp->tr_mkdir.tr_logcount = xfs_mkdir_log_count(mp, resp); resp->tr_mkdir.tr_logflags |= XFS_TRANS_PERM_LOG_RES; } From a4d5e116f0397e7522fc52079333aacd90f42052 Mon Sep 17 00:00:00 2001 From: "Darrick J. Wong" Date: Tue, 22 Sep 2026 22:58:55 -0700 Subject: [PATCH 0265/1352] xfs: fix xfs_rename_space_res for non-pptr filesystems We don't need to reserve space for a parent pointer update for an existing target if parent pointers are disabled. Fix this regression (which LOLLM noticed) so that rename reservations go back to what they were before parent pointers. Cc: stable@vger.kernel.org # v6.10 Fixes: 5a8338c88284df ("xfs: Add parent pointers to rename") Signed-off-by: Darrick J. Wong Assisted-by: LOLLM # finding obvious bugs Reviewed-by: Christoph Hellwig Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_trans_space.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/fs/xfs/libxfs/xfs_trans_space.c b/fs/xfs/libxfs/xfs_trans_space.c index c4cd547033e584..7edd0d86f0bdb2 100644 --- a/fs/xfs/libxfs/xfs_trans_space.c +++ b/fs/xfs/libxfs/xfs_trans_space.c @@ -127,10 +127,10 @@ xfs_rename_space_res( if (has_whiteout) ret += xfs_parent_calc_space_res(mp, src_namelen); ret += 2 * xfs_parent_calc_space_res(mp, target_namelen); - } - if (target_exists) - ret += xfs_parent_calc_space_res(mp, target_namelen); + if (target_exists) + ret += xfs_parent_calc_space_res(mp, target_namelen); + } return ret; } From 469b4879f75273a94f972677150aefe4c639e130 Mon Sep 17 00:00:00 2001 From: "Darrick J. Wong" Date: Tue, 22 Sep 2026 22:59:11 -0700 Subject: [PATCH 0266/1352] xfs: fix ondisk symlink target validation in xrep_dinode_check_dfork LOLLM noticed that online repair of a broken symlink file could fail unnecessarily if a local-format symlink target isn't null terminated. The ondisk target isn't required to be null terminated, but repair enforces that anyway because it uses the validator for the incore symlink target. (The incore buffer is always null-terminated). Fix this by reverting the changes to xfs_symlink_shortform_verify and adding an ondisk-specific helper in inode_repair.c. Cc: stable@vger.kernel.org # v6.8 Fixes: e744cef2060559 ("xfs: zap broken inode forks") Signed-off-by: Darrick J. Wong Assisted-by: LOLLM # finding obvious bugs Reviewed-by: Christoph Hellwig Reviewed-by: Dave Chinner Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_inode_fork.c | 4 +--- fs/xfs/libxfs/xfs_symlink_remote.c | 8 ++++++-- fs/xfs/libxfs/xfs_symlink_remote.h | 2 +- fs/xfs/scrub/inode_repair.c | 26 +++++++++++++++++++++++++- 4 files changed, 33 insertions(+), 7 deletions(-) diff --git a/fs/xfs/libxfs/xfs_inode_fork.c b/fs/xfs/libxfs/xfs_inode_fork.c index 606a36526ce245..486fe7ab8b6110 100644 --- a/fs/xfs/libxfs/xfs_inode_fork.c +++ b/fs/xfs/libxfs/xfs_inode_fork.c @@ -683,9 +683,7 @@ xfs_ifork_verify_local_data( break; } case S_IFLNK: { - struct xfs_ifork *ifp = xfs_ifork_ptr(ip, XFS_DATA_FORK); - - fa = xfs_symlink_shortform_verify(ifp->if_data, ifp->if_bytes); + fa = xfs_symlink_shortform_verify(ip); break; } default: diff --git a/fs/xfs/libxfs/xfs_symlink_remote.c b/fs/xfs/libxfs/xfs_symlink_remote.c index b0dc3888bf1b40..0201a3d59b1a24 100644 --- a/fs/xfs/libxfs/xfs_symlink_remote.c +++ b/fs/xfs/libxfs/xfs_symlink_remote.c @@ -208,11 +208,15 @@ xfs_symlink_local_to_remote( */ xfs_failaddr_t xfs_symlink_shortform_verify( - void *sfp, - int64_t size) + struct xfs_inode *ip) { + struct xfs_ifork *ifp = xfs_ifork_ptr(ip, XFS_DATA_FORK); + char *sfp = (char *)ifp->if_data; + int size = ifp->if_bytes; char *endp = sfp + size; + ASSERT(ifp->if_format == XFS_DINODE_FMT_LOCAL); + /* * Zero length symlinks should never occur in memory as they are * never allowed to exist on disk. diff --git a/fs/xfs/libxfs/xfs_symlink_remote.h b/fs/xfs/libxfs/xfs_symlink_remote.h index c1672fe1f17bb2..3f6590602473d0 100644 --- a/fs/xfs/libxfs/xfs_symlink_remote.h +++ b/fs/xfs/libxfs/xfs_symlink_remote.h @@ -18,7 +18,7 @@ bool xfs_symlink_hdr_ok(xfs_ino_t ino, uint32_t offset, void xfs_symlink_local_to_remote(struct xfs_trans *tp, struct xfs_buf *bp, struct xfs_inode *ip, struct xfs_ifork *ifp, void *priv); -xfs_failaddr_t xfs_symlink_shortform_verify(void *sfp, int64_t size); +xfs_failaddr_t xfs_symlink_shortform_verify(struct xfs_inode *ip); int xfs_symlink_remote_read(struct xfs_inode *ip, char *link); int xfs_symlink_write_target(struct xfs_trans *tp, struct xfs_inode *ip, xfs_ino_t owner, const char *target_path, int pathlen, diff --git a/fs/xfs/scrub/inode_repair.c b/fs/xfs/scrub/inode_repair.c index b87c2214623383..fab4015f6a9506 100644 --- a/fs/xfs/scrub/inode_repair.c +++ b/fs/xfs/scrub/inode_repair.c @@ -1024,6 +1024,30 @@ xrep_dinode_bad_metabt_fork( return false; } +static xfs_failaddr_t +xrep_symlink_shortform_verify( + void *sfp, + int64_t size) +{ + /* + * Zero length symlinks should never occur in memory as they are + * never allowed to exist on disk. + */ + if (!size) + return __this_address; + + /* No negative sizes or overly long symlink targets. */ + if (size < 0 || size > XFS_SYMLINK_MAXLEN) + return __this_address; + + /* No NULLs in the target either. */ + if (memchr(sfp, 0, size)) + return __this_address; + + /* ondisk symlink target isn't null terminated, unlike incore */ + return NULL; +} + /* * Check the data fork for things that will fail the ifork verifiers or the * ifork formatters. @@ -1099,7 +1123,7 @@ xrep_dinode_check_dfork( return true; /* symlink structure must pass verification. */ if (S_ISLNK(mode) && - xfs_symlink_shortform_verify(dfork_ptr, data_size) != NULL) + xrep_symlink_shortform_verify(dfork_ptr, data_size) != NULL) return true; break; case XFS_DINODE_FMT_EXTENTS: From 713053a89e13433644f44c1423eaf260e27049a2 Mon Sep 17 00:00:00 2001 From: "Darrick J. Wong" Date: Tue, 22 Sep 2026 22:59:26 -0700 Subject: [PATCH 0267/1352] xfs: add missing healthmon trace strings LOLLM noticed that we don't have trace strings for all known healthmon types and domains. Fix that. Signed-off-by: Darrick J. Wong Assisted-by: LOLLM # finding obvious bugs Reviewed-by: Christoph Hellwig Signed-off-by: Carlos Maiolino --- fs/xfs/xfs_trace.h | 28 +++++++++++++++++++++++++--- 1 file changed, 25 insertions(+), 3 deletions(-) diff --git a/fs/xfs/xfs_trace.h b/fs/xfs/xfs_trace.h index 6aa379c2cf0cd1..0fc8927339b588 100644 --- a/fs/xfs/xfs_trace.h +++ b/fs/xfs/xfs_trace.h @@ -6030,32 +6030,54 @@ DEFINE_HEALTHMON_EVENT(xfs_healthmon_detach); DEFINE_HEALTHMON_EVENT(xfs_healthmon_report_unmount); #define XFS_HEALTHMON_TYPE_STRINGS \ + { XFS_HEALTHMON_RUNNING, "run" }, \ { XFS_HEALTHMON_LOST, "lost" }, \ { XFS_HEALTHMON_UNMOUNT, "unmount" }, \ + { XFS_HEALTHMON_SHUTDOWN, "shutdown" }, \ { XFS_HEALTHMON_SICK, "sick" }, \ { XFS_HEALTHMON_CORRUPT, "corrupt" }, \ { XFS_HEALTHMON_HEALTHY, "healthy" }, \ - { XFS_HEALTHMON_SHUTDOWN, "shutdown" } + { XFS_HEALTHMON_MEDIA_ERROR, "media" }, \ + { XFS_HEALTHMON_BUFREAD, "bufread" }, \ + { XFS_HEALTHMON_BUFWRITE, "bufwrite" }, \ + { XFS_HEALTHMON_DIOREAD, "dioread" }, \ + { XFS_HEALTHMON_DIOWRITE, "diowrite" }, \ + { XFS_HEALTHMON_DATALOST, "datalost" } #define XFS_HEALTHMON_DOMAIN_STRINGS \ { XFS_HEALTHMON_MOUNT, "mount" }, \ { XFS_HEALTHMON_FS, "fs" }, \ { XFS_HEALTHMON_AG, "ag" }, \ { XFS_HEALTHMON_INODE, "inode" }, \ - { XFS_HEALTHMON_RTGROUP, "rtgroup" } + { XFS_HEALTHMON_RTGROUP, "rtgroup" }, \ + { XFS_HEALTHMON_DATADEV, "datadev" }, \ + { XFS_HEALTHMON_RTDEV, "rtdev" }, \ + { XFS_HEALTHMON_LOGDEV, "logdev" }, \ + { XFS_HEALTHMON_FILERANGE, "filerange" } +TRACE_DEFINE_ENUM(XFS_HEALTHMON_RUNNING); TRACE_DEFINE_ENUM(XFS_HEALTHMON_LOST); -TRACE_DEFINE_ENUM(XFS_HEALTHMON_SHUTDOWN); TRACE_DEFINE_ENUM(XFS_HEALTHMON_UNMOUNT); +TRACE_DEFINE_ENUM(XFS_HEALTHMON_SHUTDOWN); TRACE_DEFINE_ENUM(XFS_HEALTHMON_SICK); TRACE_DEFINE_ENUM(XFS_HEALTHMON_CORRUPT); TRACE_DEFINE_ENUM(XFS_HEALTHMON_HEALTHY); +TRACE_DEFINE_ENUM(XFS_HEALTHMON_MEDIA_ERROR); +TRACE_DEFINE_ENUM(XFS_HEALTHMON_BUFREAD); +TRACE_DEFINE_ENUM(XFS_HEALTHMON_BUFWRITE); +TRACE_DEFINE_ENUM(XFS_HEALTHMON_DIOREAD); +TRACE_DEFINE_ENUM(XFS_HEALTHMON_DIOWRITE); +TRACE_DEFINE_ENUM(XFS_HEALTHMON_DATALOST); TRACE_DEFINE_ENUM(XFS_HEALTHMON_MOUNT); TRACE_DEFINE_ENUM(XFS_HEALTHMON_FS); TRACE_DEFINE_ENUM(XFS_HEALTHMON_AG); TRACE_DEFINE_ENUM(XFS_HEALTHMON_INODE); TRACE_DEFINE_ENUM(XFS_HEALTHMON_RTGROUP); +TRACE_DEFINE_ENUM(XFS_HEALTHMON_DATADEV); +TRACE_DEFINE_ENUM(XFS_HEALTHMON_RTDEV); +TRACE_DEFINE_ENUM(XFS_HEALTHMON_LOGDEV); +TRACE_DEFINE_ENUM(XFS_HEALTHMON_FILERANGE); DECLARE_EVENT_CLASS(xfs_healthmon_event_class, TP_PROTO(const struct xfs_healthmon *hm, From fdadc158cfbdde7ed357210194a1776f2b73bf51 Mon Sep 17 00:00:00 2001 From: "Darrick J. Wong" Date: Tue, 22 Sep 2026 22:59:42 -0700 Subject: [PATCH 0268/1352] xfarray: don't crash when sorting if array element crosses a folio LOLLM points out that if an array element crosses a folio boundary, xfile_get_folio returns a NULL folio pointer. If this happens, si->folio is also set to NULL, and calling folio_pos/folio_address will just crash the kernel. Teach this function to handle this condition by falling back to reading the array element into scratchpad memory. Cc: stable@vger.kernel.org # v6.6 Fixes: cf36f4f64c2d4e ("xfs: cache pages used for xfarray quicksort convergence") Signed-off-by: Darrick J. Wong Assisted-by: LOLLM # finding obvious bugs Reviewed-by: Christoph Hellwig Signed-off-by: Carlos Maiolino --- fs/xfs/scrub/xfarray.c | 69 ++++++++++++++++++++++++++++-------------- 1 file changed, 47 insertions(+), 22 deletions(-) diff --git a/fs/xfs/scrub/xfarray.c b/fs/xfs/scrub/xfarray.c index 2ce24bfe4c0fab..5ea3ce4bf8d846 100644 --- a/fs/xfs/scrub/xfarray.c +++ b/fs/xfs/scrub/xfarray.c @@ -794,6 +794,46 @@ xfarray_sort_scan_done( si->folio = NULL; } +static int +xfarray_sort_load_folio( + struct xfarray_sortinfo *si, + xfarray_idx_t idx, + loff_t idx_pos) +{ + struct folio *folio; + loff_t next_pos; + + folio = xfile_get_folio(si->array->xfile, idx_pos, si->array->obj_size, + XFILE_ALLOC); + if (IS_ERR(folio)) + return PTR_ERR(folio); + si->folio = folio; + + /* No folio? Get the caller to read into the scratchpad. */ + if (!si->folio) + return 0; + + si->first_folio_idx = xfarray_idx(si->array, + folio_pos(si->folio) + si->array->obj_size - 1); + + next_pos = folio_next_pos(si->folio); + si->last_folio_idx = xfarray_idx(si->array, next_pos - 1); + if (xfarray_pos(si->array, si->last_folio_idx + 1) > next_pos) + si->last_folio_idx--; + + /* + * If this folio still doesn't cover the desired element, it must cross + * a folio boundary. Get the caller to read into the scratchpad. + */ + if (idx < si->first_folio_idx || idx > si->last_folio_idx) { + xfarray_sort_scan_done(si); + return 0; + } + + trace_xfarray_sort_scan(si, idx); + return 0; +} + /* * Cache the folio backing the start of the given array element. If the array * element is contained entirely within the folio, return a pointer to the @@ -819,33 +859,18 @@ xfarray_sort_scan( (idx < si->first_folio_idx || idx > si->last_folio_idx)) xfarray_sort_scan_done(si); - /* Grab the first folio that backs this array element. */ + /* Grab the folio that backs this array element. */ if (!si->folio) { - struct folio *folio; - loff_t next_pos; - - folio = xfile_get_folio(si->array->xfile, idx_pos, - si->array->obj_size, XFILE_ALLOC); - if (IS_ERR(folio)) - return PTR_ERR(folio); - si->folio = folio; - - si->first_folio_idx = xfarray_idx(si->array, - folio_pos(si->folio) + si->array->obj_size - 1); - - next_pos = folio_next_pos(si->folio); - si->last_folio_idx = xfarray_idx(si->array, next_pos - 1); - if (xfarray_pos(si->array, si->last_folio_idx + 1) > next_pos) - si->last_folio_idx--; - - trace_xfarray_sort_scan(si, idx); + error = xfarray_sort_load_folio(si, idx, idx_pos); + if (error) + return error; } /* - * If this folio still doesn't cover the desired element, it must cross - * a folio boundary. Read into the scratchpad and we're done. + * If we don't have a folio mapping the entire array element, read into + * the scratchpad and we're done. */ - if (idx < si->first_folio_idx || idx > si->last_folio_idx) { + if (!si->folio) { void *temp = xfarray_scratch(si->array); error = xfile_load(si->array->xfile, temp, si->array->obj_size, From c49f46a0cbccc4666a1fba6b5ab0372439ccf012 Mon Sep 17 00:00:00 2001 From: "Darrick J. Wong" Date: Tue, 22 Sep 2026 22:59:57 -0700 Subject: [PATCH 0269/1352] xfarray: don't allow users to unset in the middle of an array Now that we've merged online repair and removed some clunky parts of the original online checking code, the only user of xfarray_unset is the free space btree repair code, and it only needs to be able to remove records from the end of the array. Let's remove all the code that handles "unset" array elements that are not at the end, because we can just reduce the array element count. Remove the "store anywhere" function because it was only ever used by the callers who used unset to remove elements in the middle of the array. Signed-off-by: Darrick J. Wong Reviewed-by: Christoph Hellwig Signed-off-by: Carlos Maiolino --- .../xfs/xfs-online-fsck-design.rst | 13 +-- fs/xfs/scrub/alloc_repair.c | 2 +- fs/xfs/scrub/xfarray.c | 108 ++---------------- fs/xfs/scrub/xfarray.h | 6 +- 4 files changed, 15 insertions(+), 114 deletions(-) diff --git a/Documentation/filesystems/xfs/xfs-online-fsck-design.rst b/Documentation/filesystems/xfs/xfs-online-fsck-design.rst index 3d9233f403dbb1..14767ce9fad43f 100644 --- a/Documentation/filesystems/xfs/xfs-online-fsck-design.rst +++ b/Documentation/filesystems/xfs/xfs-online-fsck-design.rst @@ -1973,8 +1973,7 @@ provide loading and storing of array elements at arbitrary array indices. Gaps are defined to be null records, and null records are defined to be a sequence of all zero bytes. Null records are detected by calling ``xfarray_element_is_null``. -They are created either by calling ``xfarray_unset`` to null out an existing -record or by never storing anything to an array index. +They are created by never storing anything to an array index. The second type of caller handles records that are not indexed by position and do not require multiple updates to a record. @@ -1991,9 +1990,7 @@ The typical use case here is constructing space extent reference counts from reverse mapping information. Records can be put in the bag in any order, they can be removed from the bag at any time, and uniqueness of records is left to callers. -The ``xfarray_store_anywhere`` function is used to insert a record in any -null record slot in the bag; and the ``xfarray_unset`` function removes a -record from the bag. +Note: Bags are now implemented with in-memory btrees for faster access. Iterating Array Elements ^^^^^^^^^^^^^^^^^^^^^^^^ @@ -2643,11 +2640,7 @@ generate refcount information from reverse mapping records. refcount record associating the block number range that we just walked to the size of the bag. -The bag-like structure in this case is a type 2 xfarray as discussed in the -:ref:`xfarray access patterns` section. -Reverse mappings are added to the bag using ``xfarray_store_anywhere`` and -removed via ``xfarray_unset``. -Bag members are examined through ``xfarray_iter`` loops. +The bag-like structure in this case is an in-memory btree. Case Study: Rebuilding File Fork Mapping Indices ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ diff --git a/fs/xfs/scrub/alloc_repair.c b/fs/xfs/scrub/alloc_repair.c index 2398e381959771..b6c5292ec80ea8 100644 --- a/fs/xfs/scrub/alloc_repair.c +++ b/fs/xfs/scrub/alloc_repair.c @@ -517,7 +517,7 @@ xrep_abt_reserve_space( * records (but doesn't break the sorting order), so we must * go around the loop once more to re-run _bload_init. */ - error = xfarray_unset(ra->free_records, record_nr); + error = xfarray_trim(ra->free_records, 1); if (error) break; ra->nr_real_records--; diff --git a/fs/xfs/scrub/xfarray.c b/fs/xfs/scrub/xfarray.c index 5ea3ce4bf8d846..685824efdb93f6 100644 --- a/fs/xfs/scrub/xfarray.c +++ b/fs/xfs/scrub/xfarray.c @@ -140,55 +140,20 @@ xfarray_load( xfarray_pos(array, idx)); } -/* Is this array element potentially unset? */ -static inline bool -xfarray_is_unset( - struct xfarray *array, - loff_t pos) -{ - void *temp = xfarray_scratch(array); - int error; - - if (array->unset_slots == 0) - return false; - - error = xfile_load(array->xfile, temp, array->obj_size, pos); - if (!error && xfarray_element_is_null(array, temp)) - return true; - - return false; -} - -/* - * Unset an array element. If @idx is the last element in the array, the - * array will be truncated. Otherwise, the entry will be zeroed. - */ +/* Remove the elements at the end of an array. */ int -xfarray_unset( - struct xfarray *array, - xfarray_idx_t idx) +xfarray_trim( + struct xfarray *array, + unsigned long long nr) { - void *temp = xfarray_scratch(array); - loff_t pos = xfarray_pos(array, idx); - int error; + loff_t new_eof; - if (idx >= array->nr) + if (nr > array->nr) return -ENODATA; - if (idx == array->nr - 1) { - array->nr--; - return 0; - } - - if (xfarray_is_unset(array, pos)) - return 0; - - memset(temp, 0, array->obj_size); - error = xfile_store(array->xfile, temp, array->obj_size, pos); - if (error) - return error; - - array->unset_slots++; + array->nr -= nr; + new_eof = xfarray_pos(array, array->nr); + xfile_discard(array->xfile, new_eof, MAX_LFS_FILESIZE - new_eof); return 0; } @@ -227,43 +192,6 @@ xfarray_element_is_null( return !memchr_inv(ptr, 0, array->obj_size); } -/* - * Store an element anywhere in the array that is unset. If there are no - * unset slots, append the element to the array. - */ -int -xfarray_store_anywhere( - struct xfarray *array, - const void *ptr) -{ - void *temp = xfarray_scratch(array); - loff_t endpos = xfarray_pos(array, array->nr); - loff_t pos; - int error; - - /* Find an unset slot to put it in. */ - for (pos = 0; - pos < endpos && array->unset_slots > 0; - pos += array->obj_size) { - error = xfile_load(array->xfile, temp, array->obj_size, - pos); - if (error || !xfarray_element_is_null(array, temp)) - continue; - - error = xfile_store(array->xfile, ptr, array->obj_size, - pos); - if (error) - return error; - - array->unset_slots--; - return 0; - } - - /* No unset slots found; attach it on the end. */ - array->unset_slots = 0; - return xfarray_append(array, ptr); -} - /* Return length of array. */ uint64_t xfarray_length( @@ -677,26 +605,10 @@ xfarray_qsort_pivot( /* Load the selected xfarray records into the pivot array. */ for (i = 0; i < XFARRAY_QSORT_PIVOT_NR; i++) { - xfarray_idx_t idx; - recp = xfarray_pivot_array_rec(parray, pivot_rec_sz, i); idxp = xfarray_pivot_array_idx(parray, pivot_rec_sz, i); - /* No unset records; load directly into the array. */ - if (likely(si->array->unset_slots == 0)) { - error = xfarray_sort_load(si, *idxp, recp); - if (error) - return error; - continue; - } - - /* - * Load non-null records into the scratchpad without changing - * the xfarray_idx_t in the pivot array. - */ - idx = *idxp; - xfarray_sort_bump_loads(si); - error = xfarray_load_next(si->array, &idx, recp); + error = xfarray_sort_load(si, *idxp, recp); if (error) return error; } diff --git a/fs/xfs/scrub/xfarray.h b/fs/xfs/scrub/xfarray.h index 5eeeeed13ae24a..d55225c7885b25 100644 --- a/fs/xfs/scrub/xfarray.h +++ b/fs/xfs/scrub/xfarray.h @@ -27,9 +27,6 @@ struct xfarray { /* Maximum possible array size. */ xfarray_idx_t max_nr; - /* Number of unset slots in the array below @nr. */ - uint64_t unset_slots; - /* Size of an array element. */ size_t obj_size; @@ -41,9 +38,8 @@ int xfarray_create(const char *descr, unsigned long long required_capacity, size_t obj_size, struct xfarray **arrayp); void xfarray_destroy(struct xfarray *array); int xfarray_load(struct xfarray *array, xfarray_idx_t idx, void *ptr); -int xfarray_unset(struct xfarray *array, xfarray_idx_t idx); +int xfarray_trim(struct xfarray *array, unsigned long long nr); int xfarray_store(struct xfarray *array, xfarray_idx_t idx, const void *ptr); -int xfarray_store_anywhere(struct xfarray *array, const void *ptr); bool xfarray_element_is_null(struct xfarray *array, const void *ptr); void xfarray_truncate(struct xfarray *array); unsigned long long xfarray_bytes(struct xfarray *array); From c7a7aa2c2bcfc40eb3608508733531f0ae57b41b Mon Sep 17 00:00:00 2001 From: "Darrick J. Wong" Date: Tue, 22 Sep 2026 23:00:13 -0700 Subject: [PATCH 0270/1352] xfs: don't allow sorting sparse arrays Now that we've reduced the functionality of xfarray_unset, let's add a new safeguard: no sorting of xfarrays with sparse holes in them. It's not clear what that even means, and nobody actually does this, so we're really just eliminating subtle logic bombs. Signed-off-by: Darrick J. Wong Reviewed-by: Christoph Hellwig Signed-off-by: Carlos Maiolino --- fs/xfs/scrub/xfarray.c | 14 ++++++++++++++ fs/xfs/scrub/xfarray.h | 3 +++ 2 files changed, 17 insertions(+) diff --git a/fs/xfs/scrub/xfarray.c b/fs/xfs/scrub/xfarray.c index 685824efdb93f6..7f85855588ed54 100644 --- a/fs/xfs/scrub/xfarray.c +++ b/fs/xfs/scrub/xfarray.c @@ -152,6 +152,9 @@ xfarray_trim( return -ENODATA; array->nr -= nr; + if (!array->nr) + array->possibly_sparse = false; + new_eof = xfarray_pos(array, array->nr); xfile_discard(array->xfile, new_eof, MAX_LFS_FILESIZE - new_eof); return 0; @@ -179,6 +182,8 @@ xfarray_store( if (ret) return ret; + if (idx > array->nr) + array->possibly_sparse = true; array->nr = max(array->nr, idx + 1); return 0; } @@ -853,6 +858,14 @@ xfarray_sort( return 0; if (array->nr >= QSORT_MAX_RECS) return -E2BIG; + if (array->possibly_sparse) { + /* + * What does it mean to sort an array with holes in it? + * Currently none of the users need this ability. + */ + ASSERT(!array->possibly_sparse); + return -EINVAL; + } error = xfarray_sortinfo_alloc(array, cmp_fn, flags, &si); if (error) @@ -1006,4 +1019,5 @@ xfarray_truncate( { xfile_discard(array->xfile, 0, MAX_LFS_FILESIZE); array->nr = 0; + array->possibly_sparse = false; } diff --git a/fs/xfs/scrub/xfarray.h b/fs/xfs/scrub/xfarray.h index d55225c7885b25..05ff65b09fcf41 100644 --- a/fs/xfs/scrub/xfarray.h +++ b/fs/xfs/scrub/xfarray.h @@ -32,6 +32,9 @@ struct xfarray { /* log2 of array element size, if possible. */ int obj_size_log; + + /* Might there be sparse holes in this array? */ + bool possibly_sparse; }; int xfarray_create(const char *descr, unsigned long long required_capacity, From afbe48d6f8e8b0b10a557b34f3f1c1299f9086d1 Mon Sep 17 00:00:00 2001 From: "Darrick J. Wong" Date: Tue, 22 Sep 2026 23:00:29 -0700 Subject: [PATCH 0271/1352] xfs: simply the free space btree repair code Now that the xfarray always knows how many valid records there are stored inside of it, get rid of the shadow variable. Signed-off-by: Darrick J. Wong Reviewed-by: Christoph Hellwig Signed-off-by: Carlos Maiolino --- fs/xfs/scrub/alloc_repair.c | 13 +++++-------- 1 file changed, 5 insertions(+), 8 deletions(-) diff --git a/fs/xfs/scrub/alloc_repair.c b/fs/xfs/scrub/alloc_repair.c index b6c5292ec80ea8..6e835240536181 100644 --- a/fs/xfs/scrub/alloc_repair.c +++ b/fs/xfs/scrub/alloc_repair.c @@ -108,9 +108,6 @@ struct xrep_abt { struct xfs_scrub *sc; - /* Number of non-null records in @free_records. */ - uint64_t nr_real_records; - /* get_records()'s position in the free space record array. */ xfarray_idx_t array_cur; @@ -403,7 +400,6 @@ xrep_abt_find_freespace( if (error) goto err_agfl; - ra->nr_real_records = xfarray_length(ra->free_records); err_agfl: xfs_trans_brelse(sc->tp, agfl_bp); err: @@ -446,15 +442,17 @@ xrep_abt_reserve_space( uint64_t required; unsigned int desired; unsigned int len; + const uint64_t nr_records = + xfarray_length(ra->free_records); /* Compute how many blocks we'll need. */ error = xfs_btree_bload_compute_geometry(cnt_cur, - &ra->new_cntbt.bload, ra->nr_real_records); + &ra->new_cntbt.bload, nr_records); if (error) break; error = xfs_btree_bload_compute_geometry(bno_cur, - &ra->new_bnobt.bload, ra->nr_real_records); + &ra->new_bnobt.bload, nr_records); if (error) break; @@ -470,7 +468,7 @@ xrep_abt_reserve_space( desired = required - allocated; /* We need space but there's none left; bye! */ - if (ra->nr_real_records == 0) { + if (nr_records == 0) { error = -ENOSPC; break; } @@ -520,7 +518,6 @@ xrep_abt_reserve_space( error = xfarray_trim(ra->free_records, 1); if (error) break; - ra->nr_real_records--; record_nr--; } while (1); From e22e1ac1f07e7f1e3f1ff923f0d5fa3fc4666460 Mon Sep 17 00:00:00 2001 From: "Darrick J. Wong" Date: Tue, 22 Sep 2026 23:00:44 -0700 Subject: [PATCH 0272/1352] xfarray: warn against sorting arrays with identical elements LOLLM complains about a potential underflow here if xfarray_qsort_push is called with lo==0. However, this isn't possible in most cases because filesystem metadata records cannot be identical. Signed-off-by: Darrick J. Wong Assisted-by: LOLLM # finding obvious bugs Reviewed-by: Christoph Hellwig Signed-off-by: Carlos Maiolino --- fs/xfs/scrub/xfarray.c | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/fs/xfs/scrub/xfarray.c b/fs/xfs/scrub/xfarray.c index 7f85855588ed54..c94f355607790e 100644 --- a/fs/xfs/scrub/xfarray.c +++ b/fs/xfs/scrub/xfarray.c @@ -682,6 +682,18 @@ xfarray_qsort_push( return -EFSCORRUPTED; } + /* + * Avoid the integer underflow below in (lo - 1). This shouldn't + * be possible because the pivot is the median of nine distinct + * filesystem metadata records, so at least four records will be less + * than the pivot, which means the pivot will not be in the low end of + * the range by the time we get here. + */ + if (lo == 0) { + ASSERT(lo != 0); + return -EFSCORRUPTED; + } + si->max_stack_used = max_t(uint8_t, si->max_stack_used, si->stack_depth + 2); From 21ff42031ece6a6623ce97a0f32fc59d97dbb76d Mon Sep 17 00:00:00 2001 From: Eric Sandeen Date: Tue, 22 Sep 2026 18:17:08 -0500 Subject: [PATCH 0273/1352] xfs: share the AG rmap btree with the rt rmap btree The AG rmap btree and the realtime rmap btree have several identical key and record ops. Share these to eliminate copied code. Signed-off-by: Eric Sandeen Reviewed-by: Christoph Hellwig Reviewed-by: Darrick J. Wong Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_rmap_btree.c | 16 +-- fs/xfs/libxfs/xfs_rmap_btree.h | 26 ++++ fs/xfs/libxfs/xfs_rtrmap_btree.c | 230 +++---------------------------- 3 files changed, 51 insertions(+), 221 deletions(-) diff --git a/fs/xfs/libxfs/xfs_rmap_btree.c b/fs/xfs/libxfs/xfs_rmap_btree.c index 10b3272238ebcb..5b283a5ddd130d 100644 --- a/fs/xfs/libxfs/xfs_rmap_btree.c +++ b/fs/xfs/libxfs/xfs_rmap_btree.c @@ -170,7 +170,7 @@ static inline __be64 ondisk_rec_offset_to_key(const union xfs_btree_rec *rec) return rec->rmap.rm_offset & ~cpu_to_be64(XFS_RMAP_OFF_UNWRITTEN); } -STATIC void +void xfs_rmapbt_init_key_from_rec( union xfs_btree_key *key, const union xfs_btree_rec *rec) @@ -187,7 +187,7 @@ xfs_rmapbt_init_key_from_rec( * the startblock for all records, and if the record is for a data/attr * fork mapping, we add blockcount-1 to the offset too. */ -STATIC void +void xfs_rmapbt_init_high_key_from_rec( union xfs_btree_key *key, const union xfs_btree_rec *rec) @@ -209,7 +209,7 @@ xfs_rmapbt_init_high_key_from_rec( key->rmap.rm_offset = cpu_to_be64(off); } -STATIC void +void xfs_rmapbt_init_rec_from_cur( struct xfs_btree_cur *cur, union xfs_btree_rec *rec) @@ -243,7 +243,7 @@ static inline uint64_t offset_keymask(uint64_t offset) return offset & ~XFS_RMAP_OFF_UNWRITTEN; } -STATIC int +int xfs_rmapbt_cmp_key_with_cur( struct xfs_btree_cur *cur, const union xfs_btree_key *key) @@ -257,7 +257,7 @@ xfs_rmapbt_cmp_key_with_cur( offset_keymask(xfs_rmap_irec_offset_pack(rec))); } -STATIC int +int xfs_rmapbt_cmp_two_keys( struct xfs_btree_cur *cur, const union xfs_btree_key *k1, @@ -390,7 +390,7 @@ const struct xfs_buf_ops xfs_rmapbt_buf_ops = { .verify_struct = xfs_rmapbt_verify, }; -STATIC int +int xfs_rmapbt_keys_inorder( struct xfs_btree_cur *cur, const union xfs_btree_key *k1, @@ -420,7 +420,7 @@ xfs_rmapbt_keys_inorder( return 0; } -STATIC int +int xfs_rmapbt_recs_inorder( struct xfs_btree_cur *cur, const union xfs_btree_rec *r1, @@ -450,7 +450,7 @@ xfs_rmapbt_recs_inorder( return 0; } -STATIC enum xbtree_key_contig +enum xbtree_key_contig xfs_rmapbt_keys_contiguous( struct xfs_btree_cur *cur, const union xfs_btree_key *key1, diff --git a/fs/xfs/libxfs/xfs_rmap_btree.h b/fs/xfs/libxfs/xfs_rmap_btree.h index 119b1567cd0ee8..7071dac745e1b8 100644 --- a/fs/xfs/libxfs/xfs_rmap_btree.h +++ b/fs/xfs/libxfs/xfs_rmap_btree.h @@ -11,6 +11,8 @@ struct xfs_btree_cur; struct xfs_mount; struct xbtree_afakeroot; struct xfbtree; +union xfs_btree_key; +union xfs_btree_rec; /* rmaps only exist on crc enabled filesystems */ #define XFS_RMAP_BLOCK_LEN XFS_BTREE_SBLOCK_CRC_LEN @@ -69,4 +71,28 @@ struct xfs_btree_cur *xfs_rmapbt_mem_cursor(struct xfs_perag *pag, int xfs_rmapbt_mem_init(struct xfs_mount *mp, struct xfbtree *xfbtree, struct xfs_buftarg *btp, xfs_agnumber_t agno); +/* + * Key and record btree ops. The rmap on-disk key/record format is identical + * for the AG rmap btree and the realtime rmap btree, so these are shared by + * both. + */ +void xfs_rmapbt_init_key_from_rec(union xfs_btree_key *key, + const union xfs_btree_rec *rec); +void xfs_rmapbt_init_high_key_from_rec(union xfs_btree_key *key, + const union xfs_btree_rec *rec); +void xfs_rmapbt_init_rec_from_cur(struct xfs_btree_cur *cur, + union xfs_btree_rec *rec); +int xfs_rmapbt_cmp_key_with_cur(struct xfs_btree_cur *cur, + const union xfs_btree_key *key); +int xfs_rmapbt_cmp_two_keys(struct xfs_btree_cur *cur, + const union xfs_btree_key *k1, const union xfs_btree_key *k2, + const union xfs_btree_key *mask); +int xfs_rmapbt_keys_inorder(struct xfs_btree_cur *cur, + const union xfs_btree_key *k1, const union xfs_btree_key *k2); +int xfs_rmapbt_recs_inorder(struct xfs_btree_cur *cur, + const union xfs_btree_rec *r1, const union xfs_btree_rec *r2); +enum xbtree_key_contig xfs_rmapbt_keys_contiguous(struct xfs_btree_cur *cur, + const union xfs_btree_key *key1, const union xfs_btree_key *key2, + const union xfs_btree_key *mask); + #endif /* __XFS_RMAP_BTREE_H__ */ diff --git a/fs/xfs/libxfs/xfs_rtrmap_btree.c b/fs/xfs/libxfs/xfs_rtrmap_btree.c index 5901d7efd3f676..2f00d0698ae17a 100644 --- a/fs/xfs/libxfs/xfs_rtrmap_btree.c +++ b/fs/xfs/libxfs/xfs_rtrmap_btree.c @@ -20,6 +20,7 @@ #include "xfs_btree_staging.h" #include "xfs_metafile.h" #include "xfs_rmap.h" +#include "xfs_rmap_btree.h" #include "xfs_rtrmap_btree.h" #include "xfs_trace.h" #include "xfs_cksum.h" @@ -113,60 +114,6 @@ xfs_rtrmapbt_get_dmaxrecs( return xfs_rtrmapbt_droot_maxrecs(cur->bc_ino.forksize, level == 0); } -/* - * Convert the ondisk record's offset field into the ondisk key's offset field. - * Fork and bmbt are significant parts of the rmap record key, but written - * status is merely a record attribute. - */ -static inline __be64 ondisk_rec_offset_to_key(const union xfs_btree_rec *rec) -{ - return rec->rmap.rm_offset & ~cpu_to_be64(XFS_RMAP_OFF_UNWRITTEN); -} - -STATIC void -xfs_rtrmapbt_init_key_from_rec( - union xfs_btree_key *key, - const union xfs_btree_rec *rec) -{ - key->rmap.rm_startblock = rec->rmap.rm_startblock; - key->rmap.rm_owner = rec->rmap.rm_owner; - key->rmap.rm_offset = ondisk_rec_offset_to_key(rec); -} - -STATIC void -xfs_rtrmapbt_init_high_key_from_rec( - union xfs_btree_key *key, - const union xfs_btree_rec *rec) -{ - uint64_t off; - int adj; - - adj = be32_to_cpu(rec->rmap.rm_blockcount) - 1; - - key->rmap.rm_startblock = rec->rmap.rm_startblock; - be32_add_cpu(&key->rmap.rm_startblock, adj); - key->rmap.rm_owner = rec->rmap.rm_owner; - key->rmap.rm_offset = ondisk_rec_offset_to_key(rec); - if (XFS_RMAP_NON_INODE_OWNER(be64_to_cpu(rec->rmap.rm_owner)) || - XFS_RMAP_IS_BMBT_BLOCK(be64_to_cpu(rec->rmap.rm_offset))) - return; - off = be64_to_cpu(key->rmap.rm_offset); - off = (XFS_RMAP_OFF(off) + adj) | (off & ~XFS_RMAP_OFF_MASK); - key->rmap.rm_offset = cpu_to_be64(off); -} - -STATIC void -xfs_rtrmapbt_init_rec_from_cur( - struct xfs_btree_cur *cur, - union xfs_btree_rec *rec) -{ - rec->rmap.rm_startblock = cpu_to_be32(cur->bc_rec.r.rm_startblock); - rec->rmap.rm_blockcount = cpu_to_be32(cur->bc_rec.r.rm_blockcount); - rec->rmap.rm_owner = cpu_to_be64(cur->bc_rec.r.rm_owner); - rec->rmap.rm_offset = cpu_to_be64( - xfs_rmap_irec_offset_pack(&cur->bc_rec.r)); -} - STATIC void xfs_rtrmapbt_init_ptr_from_cur( struct xfs_btree_cur *cur, @@ -175,69 +122,6 @@ xfs_rtrmapbt_init_ptr_from_cur( ptr->l = 0; } -/* - * Mask the appropriate parts of the ondisk key field for a key comparison. - * Fork and bmbt are significant parts of the rmap record key, but written - * status is merely a record attribute. - */ -static inline uint64_t offset_keymask(uint64_t offset) -{ - return offset & ~XFS_RMAP_OFF_UNWRITTEN; -} - -STATIC int -xfs_rtrmapbt_cmp_key_with_cur( - struct xfs_btree_cur *cur, - const union xfs_btree_key *key) -{ - struct xfs_rmap_irec *rec = &cur->bc_rec.r; - const struct xfs_rmap_key *kp = &key->rmap; - - return cmp_int(be32_to_cpu(kp->rm_startblock), rec->rm_startblock) ?: - cmp_int(be64_to_cpu(kp->rm_owner), rec->rm_owner) ?: - cmp_int(offset_keymask(be64_to_cpu(kp->rm_offset)), - offset_keymask(xfs_rmap_irec_offset_pack(rec))); -} - -STATIC int -xfs_rtrmapbt_cmp_two_keys( - struct xfs_btree_cur *cur, - const union xfs_btree_key *k1, - const union xfs_btree_key *k2, - const union xfs_btree_key *mask) -{ - const struct xfs_rmap_key *kp1 = &k1->rmap; - const struct xfs_rmap_key *kp2 = &k2->rmap; - int d; - - /* Doesn't make sense to mask off the physical space part */ - ASSERT(!mask || mask->rmap.rm_startblock); - - d = cmp_int(be32_to_cpu(kp1->rm_startblock), - be32_to_cpu(kp2->rm_startblock)); - if (d) - return d; - - if (!mask || mask->rmap.rm_owner) { - d = cmp_int(be64_to_cpu(kp1->rm_owner), - be64_to_cpu(kp2->rm_owner)); - if (d) - return d; - } - - if (!mask || mask->rmap.rm_offset) { - /* Doesn't make sense to allow offset but not owner */ - ASSERT(!mask || mask->rmap.rm_owner); - - d = cmp_int(offset_keymask(be64_to_cpu(kp1->rm_offset)), - offset_keymask(be64_to_cpu(kp2->rm_offset))); - if (d) - return d; - } - - return 0; -} - static xfs_failaddr_t xfs_rtrmapbt_verify( struct xfs_buf *bp) @@ -304,86 +188,6 @@ const struct xfs_buf_ops xfs_rtrmapbt_buf_ops = { .verify_struct = xfs_rtrmapbt_verify, }; -STATIC int -xfs_rtrmapbt_keys_inorder( - struct xfs_btree_cur *cur, - const union xfs_btree_key *k1, - const union xfs_btree_key *k2) -{ - uint32_t x; - uint32_t y; - uint64_t a; - uint64_t b; - - x = be32_to_cpu(k1->rmap.rm_startblock); - y = be32_to_cpu(k2->rmap.rm_startblock); - if (x < y) - return 1; - else if (x > y) - return 0; - a = be64_to_cpu(k1->rmap.rm_owner); - b = be64_to_cpu(k2->rmap.rm_owner); - if (a < b) - return 1; - else if (a > b) - return 0; - a = offset_keymask(be64_to_cpu(k1->rmap.rm_offset)); - b = offset_keymask(be64_to_cpu(k2->rmap.rm_offset)); - if (a <= b) - return 1; - return 0; -} - -STATIC int -xfs_rtrmapbt_recs_inorder( - struct xfs_btree_cur *cur, - const union xfs_btree_rec *r1, - const union xfs_btree_rec *r2) -{ - uint32_t x; - uint32_t y; - uint64_t a; - uint64_t b; - - x = be32_to_cpu(r1->rmap.rm_startblock); - y = be32_to_cpu(r2->rmap.rm_startblock); - if (x < y) - return 1; - else if (x > y) - return 0; - a = be64_to_cpu(r1->rmap.rm_owner); - b = be64_to_cpu(r2->rmap.rm_owner); - if (a < b) - return 1; - else if (a > b) - return 0; - a = offset_keymask(be64_to_cpu(r1->rmap.rm_offset)); - b = offset_keymask(be64_to_cpu(r2->rmap.rm_offset)); - if (a <= b) - return 1; - return 0; -} - -STATIC enum xbtree_key_contig -xfs_rtrmapbt_keys_contiguous( - struct xfs_btree_cur *cur, - const union xfs_btree_key *key1, - const union xfs_btree_key *key2, - const union xfs_btree_key *mask) -{ - ASSERT(!mask || mask->rmap.rm_startblock); - - /* - * We only support checking contiguity of the physical space component. - * If any callers ever need more specificity than that, they'll have to - * implement it here. - */ - ASSERT(!mask || (!mask->rmap.rm_owner && !mask->rmap.rm_offset)); - - return xbtree_key_contig(be32_to_cpu(key1->rmap.rm_startblock), - be32_to_cpu(key2->rmap.rm_startblock)); -} - static inline void xfs_rtrmapbt_move_ptrs( struct xfs_mount *mp, @@ -486,16 +290,16 @@ const struct xfs_btree_ops xfs_rtrmapbt_ops = { .get_minrecs = xfs_rtrmapbt_get_minrecs, .get_maxrecs = xfs_rtrmapbt_get_maxrecs, .get_dmaxrecs = xfs_rtrmapbt_get_dmaxrecs, - .init_key_from_rec = xfs_rtrmapbt_init_key_from_rec, - .init_high_key_from_rec = xfs_rtrmapbt_init_high_key_from_rec, - .init_rec_from_cur = xfs_rtrmapbt_init_rec_from_cur, + .init_key_from_rec = xfs_rmapbt_init_key_from_rec, + .init_high_key_from_rec = xfs_rmapbt_init_high_key_from_rec, + .init_rec_from_cur = xfs_rmapbt_init_rec_from_cur, .init_ptr_from_cur = xfs_rtrmapbt_init_ptr_from_cur, - .cmp_key_with_cur = xfs_rtrmapbt_cmp_key_with_cur, + .cmp_key_with_cur = xfs_rmapbt_cmp_key_with_cur, .buf_ops = &xfs_rtrmapbt_buf_ops, - .cmp_two_keys = xfs_rtrmapbt_cmp_two_keys, - .keys_inorder = xfs_rtrmapbt_keys_inorder, - .recs_inorder = xfs_rtrmapbt_recs_inorder, - .keys_contiguous = xfs_rtrmapbt_keys_contiguous, + .cmp_two_keys = xfs_rmapbt_cmp_two_keys, + .keys_inorder = xfs_rmapbt_keys_inorder, + .recs_inorder = xfs_rmapbt_recs_inorder, + .keys_contiguous = xfs_rmapbt_keys_contiguous, .broot_realloc = xfs_rtrmapbt_broot_realloc, }; @@ -595,16 +399,16 @@ const struct xfs_btree_ops xfs_rtrmapbt_mem_ops = { .free_block = xfbtree_free_block, .get_minrecs = xfbtree_get_minrecs, .get_maxrecs = xfbtree_get_maxrecs, - .init_key_from_rec = xfs_rtrmapbt_init_key_from_rec, - .init_high_key_from_rec = xfs_rtrmapbt_init_high_key_from_rec, - .init_rec_from_cur = xfs_rtrmapbt_init_rec_from_cur, + .init_key_from_rec = xfs_rmapbt_init_key_from_rec, + .init_high_key_from_rec = xfs_rmapbt_init_high_key_from_rec, + .init_rec_from_cur = xfs_rmapbt_init_rec_from_cur, .init_ptr_from_cur = xfbtree_init_ptr_from_cur, - .cmp_key_with_cur = xfs_rtrmapbt_cmp_key_with_cur, + .cmp_key_with_cur = xfs_rmapbt_cmp_key_with_cur, .buf_ops = &xfs_rtrmapbt_mem_buf_ops, - .cmp_two_keys = xfs_rtrmapbt_cmp_two_keys, - .keys_inorder = xfs_rtrmapbt_keys_inorder, - .recs_inorder = xfs_rtrmapbt_recs_inorder, - .keys_contiguous = xfs_rtrmapbt_keys_contiguous, + .cmp_two_keys = xfs_rmapbt_cmp_two_keys, + .keys_inorder = xfs_rmapbt_keys_inorder, + .recs_inorder = xfs_rmapbt_recs_inorder, + .keys_contiguous = xfs_rmapbt_keys_contiguous, }; /* Create a cursor for an in-memory btree. */ From e103e2c357261ea7bb6127b68478b661838fbf29 Mon Sep 17 00:00:00 2001 From: Eric Sandeen Date: Tue, 22 Sep 2026 18:17:09 -0500 Subject: [PATCH 0274/1352] xfs: share the AG refcount btree with the rt refcount btree The AG refcount btree and the realtime refcount btree have several identical key and record ops. Share these to eliminate copied code. Signed-off-by: Eric Sandeen Reviewed-by: Christoph Hellwig Reviewed-by: Darrick J. Wong Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_refcount_btree.c | 16 ++-- fs/xfs/libxfs/xfs_refcount_btree.h | 26 ++++++ fs/xfs/libxfs/xfs_rtrefcount_btree.c | 113 +++------------------------ 3 files changed, 43 insertions(+), 112 deletions(-) diff --git a/fs/xfs/libxfs/xfs_refcount_btree.c b/fs/xfs/libxfs/xfs_refcount_btree.c index 7e5f92c1ac56fd..2c6148a979942b 100644 --- a/fs/xfs/libxfs/xfs_refcount_btree.c +++ b/fs/xfs/libxfs/xfs_refcount_btree.c @@ -127,7 +127,7 @@ xfs_refcountbt_get_maxrecs( return cur->bc_mp->m_refc_mxr[level != 0]; } -STATIC void +void xfs_refcountbt_init_key_from_rec( union xfs_btree_key *key, const union xfs_btree_rec *rec) @@ -135,7 +135,7 @@ xfs_refcountbt_init_key_from_rec( key->refc.rc_startblock = rec->refc.rc_startblock; } -STATIC void +void xfs_refcountbt_init_high_key_from_rec( union xfs_btree_key *key, const union xfs_btree_rec *rec) @@ -147,7 +147,7 @@ xfs_refcountbt_init_high_key_from_rec( key->refc.rc_startblock = cpu_to_be32(x); } -STATIC void +void xfs_refcountbt_init_rec_from_cur( struct xfs_btree_cur *cur, union xfs_btree_rec *rec) @@ -174,7 +174,7 @@ xfs_refcountbt_init_ptr_from_cur( ptr->s = agf->agf_refcount_root; } -STATIC int +int xfs_refcountbt_cmp_key_with_cur( struct xfs_btree_cur *cur, const union xfs_btree_key *key) @@ -188,7 +188,7 @@ xfs_refcountbt_cmp_key_with_cur( return cmp_int(be32_to_cpu(kp->rc_startblock), start); } -STATIC int +int xfs_refcountbt_cmp_two_keys( struct xfs_btree_cur *cur, const union xfs_btree_key *k1, @@ -283,7 +283,7 @@ const struct xfs_buf_ops xfs_refcountbt_buf_ops = { .verify_struct = xfs_refcountbt_verify, }; -STATIC int +int xfs_refcountbt_keys_inorder( struct xfs_btree_cur *cur, const union xfs_btree_key *k1, @@ -293,7 +293,7 @@ xfs_refcountbt_keys_inorder( be32_to_cpu(k2->refc.rc_startblock); } -STATIC int +int xfs_refcountbt_recs_inorder( struct xfs_btree_cur *cur, const union xfs_btree_rec *r1, @@ -304,7 +304,7 @@ xfs_refcountbt_recs_inorder( be32_to_cpu(r2->refc.rc_startblock); } -STATIC enum xbtree_key_contig +enum xbtree_key_contig xfs_refcountbt_keys_contiguous( struct xfs_btree_cur *cur, const union xfs_btree_key *key1, diff --git a/fs/xfs/libxfs/xfs_refcount_btree.h b/fs/xfs/libxfs/xfs_refcount_btree.h index beb93bef6a8141..40eedfae6846ec 100644 --- a/fs/xfs/libxfs/xfs_refcount_btree.h +++ b/fs/xfs/libxfs/xfs_refcount_btree.h @@ -15,6 +15,8 @@ struct xfs_btree_cur; struct xfs_mount; struct xfs_perag; struct xbtree_afakeroot; +union xfs_btree_key; +union xfs_btree_rec; /* * Btree block header size @@ -69,4 +71,28 @@ unsigned int xfs_refcountbt_maxlevels_ondisk(void); int __init xfs_refcountbt_init_cur_cache(void); void xfs_refcountbt_destroy_cur_cache(void); +/* + * Key and record btree ops. The refcount on-disk key/record format is + * identical for the AG refcount btree and the realtime refcount btree, so + * these are shared by both. + */ +void xfs_refcountbt_init_key_from_rec(union xfs_btree_key *key, + const union xfs_btree_rec *rec); +void xfs_refcountbt_init_high_key_from_rec(union xfs_btree_key *key, + const union xfs_btree_rec *rec); +void xfs_refcountbt_init_rec_from_cur(struct xfs_btree_cur *cur, + union xfs_btree_rec *rec); +int xfs_refcountbt_cmp_key_with_cur(struct xfs_btree_cur *cur, + const union xfs_btree_key *key); +int xfs_refcountbt_cmp_two_keys(struct xfs_btree_cur *cur, + const union xfs_btree_key *k1, const union xfs_btree_key *k2, + const union xfs_btree_key *mask); +int xfs_refcountbt_keys_inorder(struct xfs_btree_cur *cur, + const union xfs_btree_key *k1, const union xfs_btree_key *k2); +int xfs_refcountbt_recs_inorder(struct xfs_btree_cur *cur, + const union xfs_btree_rec *r1, const union xfs_btree_rec *r2); +enum xbtree_key_contig xfs_refcountbt_keys_contiguous(struct xfs_btree_cur *cur, + const union xfs_btree_key *key1, const union xfs_btree_key *key2, + const union xfs_btree_key *mask); + #endif /* __XFS_REFCOUNT_BTREE_H__ */ diff --git a/fs/xfs/libxfs/xfs_rtrefcount_btree.c b/fs/xfs/libxfs/xfs_rtrefcount_btree.c index e91ff14577f9ad..697f5e62268573 100644 --- a/fs/xfs/libxfs/xfs_rtrefcount_btree.c +++ b/fs/xfs/libxfs/xfs_rtrefcount_btree.c @@ -20,6 +20,7 @@ #include "xfs_btree_staging.h" #include "xfs_rtrefcount_btree.h" #include "xfs_refcount.h" +#include "xfs_refcount_btree.h" #include "xfs_trace.h" #include "xfs_cksum.h" #include "xfs_error.h" @@ -113,41 +114,6 @@ xfs_rtrefcountbt_get_dmaxrecs( return xfs_rtrefcountbt_droot_maxrecs(cur->bc_ino.forksize, level == 0); } -STATIC void -xfs_rtrefcountbt_init_key_from_rec( - union xfs_btree_key *key, - const union xfs_btree_rec *rec) -{ - key->refc.rc_startblock = rec->refc.rc_startblock; -} - -STATIC void -xfs_rtrefcountbt_init_high_key_from_rec( - union xfs_btree_key *key, - const union xfs_btree_rec *rec) -{ - __u32 x; - - x = be32_to_cpu(rec->refc.rc_startblock); - x += be32_to_cpu(rec->refc.rc_blockcount) - 1; - key->refc.rc_startblock = cpu_to_be32(x); -} - -STATIC void -xfs_rtrefcountbt_init_rec_from_cur( - struct xfs_btree_cur *cur, - union xfs_btree_rec *rec) -{ - const struct xfs_refcount_irec *irec = &cur->bc_rec.rc; - uint32_t start; - - start = xfs_refcount_encode_startblock(irec->rc_startblock, - irec->rc_domain); - rec->refc.rc_startblock = cpu_to_be32(start); - rec->refc.rc_blockcount = cpu_to_be32(cur->bc_rec.rc.rc_blockcount); - rec->refc.rc_refcount = cpu_to_be32(cur->bc_rec.rc.rc_refcount); -} - STATIC void xfs_rtrefcountbt_init_ptr_from_cur( struct xfs_btree_cur *cur, @@ -156,33 +122,6 @@ xfs_rtrefcountbt_init_ptr_from_cur( ptr->l = 0; } -STATIC int -xfs_rtrefcountbt_cmp_key_with_cur( - struct xfs_btree_cur *cur, - const union xfs_btree_key *key) -{ - const struct xfs_refcount_key *kp = &key->refc; - const struct xfs_refcount_irec *irec = &cur->bc_rec.rc; - uint32_t start; - - start = xfs_refcount_encode_startblock(irec->rc_startblock, - irec->rc_domain); - return cmp_int(be32_to_cpu(kp->rc_startblock), start); -} - -STATIC int -xfs_rtrefcountbt_cmp_two_keys( - struct xfs_btree_cur *cur, - const union xfs_btree_key *k1, - const union xfs_btree_key *k2, - const union xfs_btree_key *mask) -{ - ASSERT(!mask || mask->refc.rc_startblock); - - return cmp_int(be32_to_cpu(k1->refc.rc_startblock), - be32_to_cpu(k2->refc.rc_startblock)); -} - static xfs_failaddr_t xfs_rtrefcountbt_verify( struct xfs_buf *bp) @@ -249,40 +188,6 @@ const struct xfs_buf_ops xfs_rtrefcountbt_buf_ops = { .verify_struct = xfs_rtrefcountbt_verify, }; -STATIC int -xfs_rtrefcountbt_keys_inorder( - struct xfs_btree_cur *cur, - const union xfs_btree_key *k1, - const union xfs_btree_key *k2) -{ - return be32_to_cpu(k1->refc.rc_startblock) < - be32_to_cpu(k2->refc.rc_startblock); -} - -STATIC int -xfs_rtrefcountbt_recs_inorder( - struct xfs_btree_cur *cur, - const union xfs_btree_rec *r1, - const union xfs_btree_rec *r2) -{ - return be32_to_cpu(r1->refc.rc_startblock) + - be32_to_cpu(r1->refc.rc_blockcount) <= - be32_to_cpu(r2->refc.rc_startblock); -} - -STATIC enum xbtree_key_contig -xfs_rtrefcountbt_keys_contiguous( - struct xfs_btree_cur *cur, - const union xfs_btree_key *key1, - const union xfs_btree_key *key2, - const union xfs_btree_key *mask) -{ - ASSERT(!mask || mask->refc.rc_startblock); - - return xbtree_key_contig(be32_to_cpu(key1->refc.rc_startblock), - be32_to_cpu(key2->refc.rc_startblock)); -} - static inline void xfs_rtrefcountbt_move_ptrs( struct xfs_mount *mp, @@ -383,16 +288,16 @@ const struct xfs_btree_ops xfs_rtrefcountbt_ops = { .get_minrecs = xfs_rtrefcountbt_get_minrecs, .get_maxrecs = xfs_rtrefcountbt_get_maxrecs, .get_dmaxrecs = xfs_rtrefcountbt_get_dmaxrecs, - .init_key_from_rec = xfs_rtrefcountbt_init_key_from_rec, - .init_high_key_from_rec = xfs_rtrefcountbt_init_high_key_from_rec, - .init_rec_from_cur = xfs_rtrefcountbt_init_rec_from_cur, + .init_key_from_rec = xfs_refcountbt_init_key_from_rec, + .init_high_key_from_rec = xfs_refcountbt_init_high_key_from_rec, + .init_rec_from_cur = xfs_refcountbt_init_rec_from_cur, .init_ptr_from_cur = xfs_rtrefcountbt_init_ptr_from_cur, - .cmp_key_with_cur = xfs_rtrefcountbt_cmp_key_with_cur, + .cmp_key_with_cur = xfs_refcountbt_cmp_key_with_cur, .buf_ops = &xfs_rtrefcountbt_buf_ops, - .cmp_two_keys = xfs_rtrefcountbt_cmp_two_keys, - .keys_inorder = xfs_rtrefcountbt_keys_inorder, - .recs_inorder = xfs_rtrefcountbt_recs_inorder, - .keys_contiguous = xfs_rtrefcountbt_keys_contiguous, + .cmp_two_keys = xfs_refcountbt_cmp_two_keys, + .keys_inorder = xfs_refcountbt_keys_inorder, + .recs_inorder = xfs_refcountbt_recs_inorder, + .keys_contiguous = xfs_refcountbt_keys_contiguous, .broot_realloc = xfs_rtrefcountbt_broot_realloc, }; From 4a5c7d1c38b283d66b19caf95930d53260f50411 Mon Sep 17 00:00:00 2001 From: Yun Zhou Date: Wed, 16 Sep 2026 08:43:35 +0800 Subject: [PATCH 0275/1352] xfs: move xfs_sync_sb_buf() out of libxfs into xfs_ioctl.c xfs_sync_sb_buf() dereferences mp->m_sb_bp and mp->m_rtsb_bp, which only exist in the kernel's struct xfs_mount, not in xfsprogs' libxfs. Keeping it in libxfs is a porting hazard. Its only caller, xfs_ioc_setlabel(), is kernel-only, so move it there and make it static. Fixes: c1351fb48eee ("xfs: don't hold buffer locks across sync transaction commit in xfs_sync_sb_buf") Suggested-by: Darrick J. Wong Suggested-by: Christoph Hellwig Signed-off-by: Yun Zhou Reviewed-by: Christoph Hellwig Signed-off-by: Carlos Maiolino --- fs/xfs/libxfs/xfs_sb.c | 40 ---------------------------------------- fs/xfs/libxfs/xfs_sb.h | 1 - fs/xfs/xfs_ioctl.c | 40 ++++++++++++++++++++++++++++++++++++++++ 3 files changed, 40 insertions(+), 41 deletions(-) diff --git a/fs/xfs/libxfs/xfs_sb.c b/fs/xfs/libxfs/xfs_sb.c index f0341adbb8790c..d2ef7b20a49ffa 100644 --- a/fs/xfs/libxfs/xfs_sb.c +++ b/fs/xfs/libxfs/xfs_sb.c @@ -1460,46 +1460,6 @@ xfs_update_secondary_sbs( return saved_error ? saved_error : error; } -/* - * Same behavior as xfs_sync_sb, except that it is always synchronous and it - * also writes the superblock buffer to disk sector 0 immediately. - */ -int -xfs_sync_sb_buf( - struct xfs_mount *mp, - bool update_rtsb) -{ - struct xfs_trans *tp; - int error; - - error = xfs_trans_alloc(mp, &M_RES(mp)->tr_sb, 0, 0, 0, &tp); - if (error) - return error; - - xfs_log_sb(tp); - if (update_rtsb) - xfs_log_rtsb(tp, xfs_trans_getsb(tp)); - xfs_trans_set_sync(tp); - error = xfs_trans_commit(tp); - if (error) - return error; - - /* Re-acquire and write the sb and rtsb to disk. */ - xfs_buf_lock(mp->m_sb_bp); - error = xfs_bwrite(mp->m_sb_bp); - xfs_buf_unlock(mp->m_sb_bp); - if (error) - return error; - - if (update_rtsb && mp->m_rtsb_bp) { - xfs_buf_lock(mp->m_rtsb_bp); - error = xfs_bwrite(mp->m_rtsb_bp); - xfs_buf_unlock(mp->m_rtsb_bp); - } - - return error; -} - void xfs_fs_geometry( struct xfs_mount *mp, diff --git a/fs/xfs/libxfs/xfs_sb.h b/fs/xfs/libxfs/xfs_sb.h index 34d0dd374e9b0b..77de65922213b1 100644 --- a/fs/xfs/libxfs/xfs_sb.h +++ b/fs/xfs/libxfs/xfs_sb.h @@ -15,7 +15,6 @@ struct xfs_perag; extern void xfs_log_sb(struct xfs_trans *tp); extern int xfs_sync_sb(struct xfs_mount *mp, bool wait); -extern int xfs_sync_sb_buf(struct xfs_mount *mp, bool update_rtsb); extern void xfs_sb_mount_common(struct xfs_mount *mp, struct xfs_sb *sbp); void xfs_sb_mount_rextsize(struct xfs_mount *mp, struct xfs_sb *sbp); void xfs_mount_sb_set_rextsize(struct xfs_mount *mp, diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c index 10673cf7f2d6fd..c0fc9b34f3933c 100644 --- a/fs/xfs/xfs_ioctl.c +++ b/fs/xfs/xfs_ioctl.c @@ -1035,6 +1035,46 @@ xfs_ioc_getlabel( return 0; } +/* + * Same behavior as xfs_sync_sb, except that it is always synchronous and it + * also writes the superblock buffer to disk sector 0 immediately. + */ +static int +xfs_sync_sb_buf( + struct xfs_mount *mp, + bool update_rtsb) +{ + struct xfs_trans *tp; + int error; + + error = xfs_trans_alloc(mp, &M_RES(mp)->tr_sb, 0, 0, 0, &tp); + if (error) + return error; + + xfs_log_sb(tp); + if (update_rtsb) + xfs_log_rtsb(tp, xfs_trans_getsb(tp)); + xfs_trans_set_sync(tp); + error = xfs_trans_commit(tp); + if (error) + return error; + + /* Re-acquire and write the sb and rtsb to disk. */ + xfs_buf_lock(mp->m_sb_bp); + error = xfs_bwrite(mp->m_sb_bp); + xfs_buf_unlock(mp->m_sb_bp); + if (error) + return error; + + if (update_rtsb && mp->m_rtsb_bp) { + xfs_buf_lock(mp->m_rtsb_bp); + error = xfs_bwrite(mp->m_rtsb_bp); + xfs_buf_unlock(mp->m_rtsb_bp); + } + + return error; +} + static int xfs_ioc_setlabel( struct file *filp, From bdacb4259953a6f47f2a6a9a3b70af91ae4669e0 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?J=C3=A9r=C3=A9my=20Jean?= Date: Mon, 17 Aug 2026 20:57:14 -0400 Subject: [PATCH 0276/1352] nfsd: preflight SEQUENCE replies before accepting a slot MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit nfsd4_sequence() narrows the reply buffer to the session's cached reply limit before it accepts the slot seqid. A client may negotiate ca_maxresponsesize_cached down to NFSD_MIN_HDR_SEQ_SZ, and nfsd4_alloc_slot() then gives every slot a zero-length sl_data[]. A COMPOUND tag can fill that narrowed buffer until it holds the SEQUENCE opcode but not the status word that follows. nfsd4_encode_operation() returns without running nfsd4_encode_sequence(), so cstate.data_offset stays zero. It leaves op->status at nfs_ok as well, so the COMPOUND is treated as having succeeded. nfsd4_store_cache_entry() declines to cache a lone SEQUENCE that returned an error. That test reads the status the operation reported, so it passes here. The copy starts at offset zero and takes the whole reply, RPC and COMPOUND headers included, into the zero-length sl_data[]. The COMPOUND tag is copied along with it, so the client picks most of the bytes written past the end of the slot: BUG: KASAN: slab-out-of-bounds in read_bytes_from_xdr_buf+0x1bc/0x390 Write of size 80 at addr ffff888003a549cd by task kunit_try_catch/24 __asan_memcpy+0x38/0x60 read_bytes_from_xdr_buf+0x1bc/0x390 nfsd4_sequence_done+0x5b0/0x810 nfs4svc_encode_compoundres+0x1bf/0x240 Check that the fixed-size SEQUENCE result, plus room for a following operation's error status, fits the negotiated limit before narrowing the buffer and consuming the slot seqid. The slot and its reply cache are left unchanged, as RFC 8881 Section 2.10.6.1.2 requires of an error returned from SEQUENCE. Fixes: 47ee52986472 ("nfsd4: adjust buflen to session channel limit") Cc: stable@vger.kernel.org Assisted-by: Codex:gpt-5 Signed-off-by: Jérémy Jean Tested-by: Mayank Jangid (OpenSec Intelligence) Link: https://patch.msgid.link/20260817-jean-v1-1-9e356596ab85@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfs4state.c | 19 ++++++++++++++++++- 1 file changed, 18 insertions(+), 1 deletion(-) diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c index 06e4192bc6938c..d33fb48e1f8963 100644 --- a/fs/nfsd/nfs4state.c +++ b/fs/nfsd/nfs4state.c @@ -5044,6 +5044,7 @@ __be32 nfsd4_sequence(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate, union nfsd4_op_u *u) { + struct nfsd4_compoundargs *args = rqstp->rq_argp; struct nfsd4_sequence *seq = &u->sequence; struct nfsd4_compoundres *resp = rqstp->rq_resp; struct xdr_stream *xdr = resp->xdr; @@ -5053,6 +5054,7 @@ nfsd4_sequence(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate, struct nfsd4_conn *conn; __be32 status; int buflen; + u32 maxlen, respsize; struct net *net = SVC_NET(rqstp); struct nfsd_net *nn = net_generic(net, nfsd_net_id); @@ -5130,7 +5132,22 @@ nfsd4_sequence(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate, session->se_fchannel.maxresp_sz; status = (seq->cachethis) ? nfserr_rep_too_big_to_cache : nfserr_rep_too_big; - if (xdr_restrict_buflen(xdr, buflen - rqstp->rq_auth_slack)) + if (buflen < rqstp->rq_auth_slack) + goto out_put_session; + maxlen = buflen - rqstp->rq_auth_slack; + + /* + * A SEQUENCE result too large for maxlen never reaches + * nfsd4_encode_sequence(), so cstate.data_offset stays zero and + * the reply cache overruns the slot. + */ + respsize = nfsd4_max_reply(rqstp, &args->ops[0]); + if (!nfsd4_last_compound_op(rqstp)) + respsize += COMPOUND_ERR_SLACK_SPACE; + if (xdr->buf->len + respsize > maxlen) + goto out_put_session; + + if (xdr_restrict_buflen(xdr, maxlen)) goto out_put_session; svc_reserve_auth(rqstp, buflen); From eb42e362082605b3ba15d53d9b72b1aaa25fbf74 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Mon, 17 Aug 2026 20:57:15 -0400 Subject: [PATCH 0277/1352] nfsd: set op->status when an operation's header cannot be encoded nfsd4_encode_operation() leaves op->status alone when the reply buffer has no room for the operation's opcode and status word. nfsd4_proc_compound() reads the unchanged nfs_ok as success and goes on to the next operation, so the reply counts an operation whose result was never encoded. Report the failure through nfsd4_check_resp_size(), which the rest of the function already uses. It returns NFS4ERR_REP_TOO_BIG, or NFS4ERR_REP_TOO_BIG_TO_CACHE on a session, and the COMPOUND ends at that operation. Two paths narrow the reply buffer: nfsd4_sequence(), which rejects a SEQUENCE result that does not fit, and nfsd4_encode_splice_read(), which can leave a single XDR word in the head page. Whether a COMPOUND reaches that boundary is unproven, so this is a guard rather than a fix. Link: https://patch.msgid.link/20260817-jean-v1-2-9e356596ab85@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfs4xdr.c | 13 +++++++++++-- 1 file changed, 11 insertions(+), 2 deletions(-) diff --git a/fs/nfsd/nfs4xdr.c b/fs/nfsd/nfs4xdr.c index 7d1b2d6f57f206..a154b02d82b3c4 100644 --- a/fs/nfsd/nfs4xdr.c +++ b/fs/nfsd/nfs4xdr.c @@ -6723,11 +6723,20 @@ nfsd4_encode_operation(struct nfsd4_compoundres *resp, struct nfsd4_op *op) unsigned int op_status_offset; nfsd4_enc encoder; - if (xdr_stream_encode_u32(xdr, op->opnum) != XDR_UNIT) + /* + * nfsd4_proc_compound() stops the COMPOUND early only + * when op->status is set, so a header that cannot be + * encoded has to report the failure here. + */ + if (xdr_stream_encode_u32(xdr, op->opnum) != XDR_UNIT) { + op->status = nfsd4_check_resp_size(resp, XDR_UNIT * 2); goto release; + } op_status_offset = xdr->buf->len; - if (!xdr_reserve_space(xdr, XDR_UNIT)) + if (!xdr_reserve_space(xdr, XDR_UNIT)) { + op->status = nfsd4_check_resp_size(resp, XDR_UNIT); goto release; + } if (op->opnum == OP_ILLEGAL) goto status; From f55ae50c51a9f63b70f97519aedfc7b0ee2b7291 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Mon, 17 Aug 2026 21:08:36 -0400 Subject: [PATCH 0278/1352] NFSD: Do not send CB_RECALL_ANY to NFSv4.0 clients deleg_reaper() sends CB_RECALL_ANY to every ACTIVE client holding delegations, but CB_RECALL_ANY is an NFSv4.1 operation. An NFSv4.0 client's callback service accepts only CB_GETATTR and CB_RECALL, so it replies OP_ILLEGAL. The decoder maps the unexpected opnum to -EIO, and nfsd4_cb_done() marks the client's callback channel down. Nothing brings the channel back. nfsd4_run_cb_work() sets NFSD4_CB_UP only for a minor version above zero, and the only nfsd4_probe_callback() call site an NFSv4.0 client reaches is nfsd4_setclientid_confirm(). One visit from the reaper therefore leaves the channel marked down until the client re-establishes its clientid. RENEW then returns NFS4ERR_CB_PATH_DOWN for as long as the client holds delegations. nfsd4_cb_channel_good() stops returning true, so the client is granted no further delegations. Skip clients at minor version zero. Fixes: 44df6f439a17 ("NFSD: add delegation reaper to react to low memory condition") Cc: stable@vger.kernel.org Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260817-recall-any-keep-count-v5-1-3b2cffce701e@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfs4state.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c index d33fb48e1f8963..3bcfcef417bba0 100644 --- a/fs/nfsd/nfs4state.c +++ b/fs/nfsd/nfs4state.c @@ -7962,6 +7962,8 @@ deleg_reaper(struct nfsd_net *nn) list_for_each_safe(pos, next, &nn->client_lru) { clp = list_entry(pos, struct nfs4_client, cl_lru); + if (clp->cl_minorversion == 0) + continue; if (clp->cl_state != NFSD4_ACTIVE) continue; if (list_empty(&clp->cl_delegations)) From 81f03503121a1c44d805477aaff396fb93f03bf2 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Mon, 17 Aug 2026 21:08:37 -0400 Subject: [PATCH 0279/1352] NFSD: Count the delegations held by each client struct nfs4_client records the delegations it holds on cl_delegations but keeps no count of them. deleg_reaper() walks nn->client_lru under nn->client_lock, but cl_delegations is serialized by nn->deleg_lock, which nests outside nn->client_lock. A caller there cannot take nn->deleg_lock to count the list. The cost tells against the walk as well: an O(n) count per client, on a pass that already visits every client. Add cl_deleg_count, maintained at the two sites that mutate cl_delegations. Both hold nn->deleg_lock, so the counter is already serialized against itself and needs no atomic of its own. The decrement sits below the delegation_hashed() test, next to the list_del_init it pairs with, so it runs only when the delegation really leaves the list. A reader that holds only nn->client_lock is not synchronized against either update site, so it can see a count that does not match the list. Such a reader marks the access with data_race() and may not depend on the value for correctness. No functional change. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260817-recall-any-keep-count-v5-2-3b2cffce701e@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfs4state.c | 2 ++ fs/nfsd/state.h | 2 ++ 2 files changed, 4 insertions(+) diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c index 3bcfcef417bba0..8da7635babbbfe 100644 --- a/fs/nfsd/nfs4state.c +++ b/fs/nfsd/nfs4state.c @@ -1527,6 +1527,7 @@ hash_delegation_locked(struct nfs4_delegation *dp, struct nfs4_file *fp) dp->dl_stid.sc_type = SC_TYPE_DELEG; list_add(&dp->dl_perfile, &fp->fi_delegations); list_add(&dp->dl_perclnt, &clp->cl_delegations); + clp->cl_deleg_count++; return 0; } @@ -1558,6 +1559,7 @@ unhash_delegation_locked(struct nfs4_delegation *dp, unsigned short statusmask) ++dp->dl_time; spin_lock(&fp->fi_lock); list_del_init(&dp->dl_perclnt); + dp->dl_stid.sc_client->cl_deleg_count--; list_del_init(&dp->dl_recall_lru); list_del_init(&dp->dl_perfile); spin_unlock(&fp->fi_lock); diff --git a/fs/nfsd/state.h b/fs/nfsd/state.h index c65b604e29f1be..cd9294f024bb98 100644 --- a/fs/nfsd/state.h +++ b/fs/nfsd/state.h @@ -633,6 +633,8 @@ struct nfs4_client { unsigned int cl_state; atomic_t cl_delegs_in_recall; + /* Length of cl_delegations, updated under nn->deleg_lock */ + unsigned int cl_deleg_count; struct nfsd4_cb_recall_any *cl_ra; time64_t cl_ra_time; From d8a16a7f365771c3b991833b61d463cd1ddc2a0b Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Mon, 17 Aug 2026 21:08:38 -0400 Subject: [PATCH 0280/1352] NFSD: Name directory delegations in the CB_RECALL_ANY type mask RFC 8881 Section 20.6.3 distinguishes an NFSv4.1 server implementation that shares one pool among all classes of recallable objects from one that keeps separate pools per class. NFSD falls in the former category. The CB_RECALL_ANY operation's craa_type_mask argument names the types of objects in the recallable resource pool, but NFSD's implementation does not name directory delegations, even though they are allocated through __alloc_init_deleg(), they are counted against the max_delegations budget, and the state shrinker reclaims them. Add RCA4_TYPE_MASK_DIR_DLG to craa_type_mask so clients that implement directory delegations consider them when choosing which delegations to return. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260817-recall-any-keep-count-v5-3-3b2cffce701e@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfs4state.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c index 8da7635babbbfe..f6da74ce16a920 100644 --- a/fs/nfsd/nfs4state.c +++ b/fs/nfsd/nfs4state.c @@ -7984,7 +7984,8 @@ deleg_reaper(struct nfsd_net *nn) clp->cl_ra_time = ktime_get_boottime_seconds(); clp->cl_ra->ra_keep = 0; clp->cl_ra->ra_bmval[0] = BIT(RCA4_TYPE_MASK_RDATA_DLG) | - BIT(RCA4_TYPE_MASK_WDATA_DLG); + BIT(RCA4_TYPE_MASK_WDATA_DLG) | + BIT(RCA4_TYPE_MASK_DIR_DLG); trace_nfsd_cb_recall_any(clp->cl_ra); nfsd4_run_cb(&clp->cl_ra->ra_cb); } From e4c6149709d6ecf9bead2a4309f87e53b7c6c09a Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Mon, 17 Aug 2026 21:08:39 -0400 Subject: [PATCH 0281/1352] NFSD: Send a meaningful CB_RECALL_ANY keep count deleg_reaper() sets craa_objects_to_keep to zero on every CB_RECALL_ANY. RFC 8881 Section 20.6.3 defines that field as the number of objects the client may keep, leaving the client to choose which of the excess to return, because the server cannot read lack of recent use as lack of usefulness. Zero asks for every delegation the client holds, including the ones backing files an application still has open. There is also no reason NFSD has to reclaim the entire delegation working set on the first sign of memory pressure. Derive the keep count from cl_deleg_count so that each callback asks for one delegation. Both the shrinker and the laundromat re-arm while their condition lasts, so a client with more to give is asked again on the next pass. The Linux client ignores craa_objects_to_keep and returns unused delegations selected from the type mask alone, so the count changes nothing for it. Fixes: 44df6f439a17 ("NFSD: add delegation reaper to react to low memory condition") Cc: stable@vger.kernel.org Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260817-recall-any-keep-count-v5-4-3b2cffce701e@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfs4state.c | 19 ++++++++++++++++--- 1 file changed, 16 insertions(+), 3 deletions(-) diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c index f6da74ce16a920..9d40b573a295af 100644 --- a/fs/nfsd/nfs4state.c +++ b/fs/nfsd/nfs4state.c @@ -7959,6 +7959,7 @@ deleg_reaper(struct nfsd_net *nn) { struct list_head *pos, *next; struct nfs4_client *clp; + unsigned int count; spin_lock(&nn->client_lock); list_for_each_safe(pos, next, &nn->client_lru) { @@ -7968,21 +7969,33 @@ deleg_reaper(struct nfsd_net *nn) continue; if (clp->cl_state != NFSD4_ACTIVE) continue; - if (list_empty(&clp->cl_delegations)) - continue; if (atomic_read(&clp->cl_delegs_in_recall)) continue; if (ktime_get_boottime_seconds() - clp->cl_ra_time < 5) continue; if (clp->cl_cb_state != NFSD4_CB_UP) continue; + /* + * This read races with hash_delegation_locked() and + * unhash_delegation_locked() on other CPUs. A stale + * count only skews the keep value; the next + * laundromat pass sees a more current one. + */ + count = data_race(READ_ONCE(clp->cl_deleg_count)); + if (!count) + continue; if (test_and_set_bit(NFSD4_CALLBACK_RUNNING, &clp->cl_ra->ra_cb.cb_flags)) continue; /* release in nfsd4_cb_recall_any_release */ kref_get(&clp->cl_nfsdfs.cl_ref); clp->cl_ra_time = ktime_get_boottime_seconds(); - clp->cl_ra->ra_keep = 0; + /* + * Ask for a single delegation. Recalling one before it + * is needed costs the client an OPEN when it next + * touches the file. + */ + clp->cl_ra->ra_keep = count - 1; clp->cl_ra->ra_bmval[0] = BIT(RCA4_TYPE_MASK_RDATA_DLG) | BIT(RCA4_TYPE_MASK_WDATA_DLG) | BIT(RCA4_TYPE_MASK_DIR_DLG); From 9ce2e3e3561dfcaf6ce777f24bf13cda04e98d12 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Mon, 17 Aug 2026 21:08:40 -0400 Subject: [PATCH 0282/1352] NFSD: Count delegations per network namespace The state shrinker is allocated per network namespace, but nfsd4_state_shrinker_count() reports num_delegations, which counts the delegations held by the whole host. Every namespace therefore reports every delegation on the server. Reclaim sees the population multiplied by the number of namespaces running NFSD. A namespace holding no delegations of its own still reports a nonzero count and queues its reaper, which then finds nothing to recall. Count the delegations in each namespace and report that instead. num_delegations stays for the admission check in __alloc_init_deleg() and the ceiling check in nfs4_laundromat(). Both compare against max_delegations, which is sized from host memory and so remains a host-wide limit. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260817-recall-any-keep-count-v5-5-3b2cffce701e@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/netns.h | 2 ++ fs/nfsd/nfs4state.c | 7 ++++++- 2 files changed, 8 insertions(+), 1 deletion(-) diff --git a/fs/nfsd/netns.h b/fs/nfsd/netns.h index 71eebfea020d50..bb62d19430bcf5 100644 --- a/fs/nfsd/netns.h +++ b/fs/nfsd/netns.h @@ -238,6 +238,8 @@ struct nfsd_net { int nfs4_max_clients; atomic_t nfsd_courtesy_clients; + /* per-namespace; num_delegations in nfs4state.c is host-wide */ + atomic_long_t nfsd_delegations; struct shrinker *nfsd_client_shrinker; struct work_struct nfsd_shrinker_work; diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c index 9d40b573a295af..ccf22ccb2839ad 100644 --- a/fs/nfsd/nfs4state.c +++ b/fs/nfsd/nfs4state.c @@ -1161,6 +1161,7 @@ static struct nfs4_ol_stateid * nfs4_alloc_open_stateid(struct nfs4_client *clp) */ static void nfs4_free_deleg(struct nfs4_stid *stid) { + struct nfsd_net *nn = net_generic(stid->sc_client->net, nfsd_net_id); struct nfs4_delegation *dp = delegstateid(stid); WARN_ON_ONCE(!list_empty(&stid->sc_cp_list)); @@ -1171,6 +1172,7 @@ static void nfs4_free_deleg(struct nfs4_stid *stid) nfsd41_cb_destroy_referring_call_list(&dp->dl_recall); kmem_cache_free(deleg_slab, stid); atomic_long_dec(&num_delegations); + atomic_long_dec(&nn->nfsd_delegations); } /* @@ -1255,6 +1257,7 @@ __alloc_init_deleg(struct nfs4_client *clp, struct nfs4_file *fp, struct nfs4_clnt_odstate *odstate, u32 dl_type, void (*sc_free)(struct nfs4_stid *)) { + struct nfsd_net *nn = net_generic(clp->net, nfsd_net_id); struct nfs4_delegation *dp; struct nfs4_stid *stid; long n; @@ -1263,6 +1266,7 @@ __alloc_init_deleg(struct nfs4_client *clp, struct nfs4_file *fp, return NULL; n = atomic_long_inc_return(&num_delegations); + atomic_long_inc(&nn->nfsd_delegations); if (n < 0 || n > max_delegations) goto out_dec; @@ -1295,6 +1299,7 @@ __alloc_init_deleg(struct nfs4_client *clp, struct nfs4_file *fp, return dp; out_dec: atomic_long_dec(&num_delegations); + atomic_long_dec(&nn->nfsd_delegations); return NULL; } @@ -5581,7 +5586,7 @@ nfsd4_state_shrinker_count(struct shrinker *shrink, struct shrink_control *sc) count = atomic_read(&nn->nfsd_courtesy_clients); if (!count) - count = atomic_long_read(&num_delegations); + count = atomic_long_read(&nn->nfsd_delegations); if (count) queue_work(laundry_wq, &nn->nfsd_shrinker_work); return (unsigned long)count; From d55a6bc1b02a7c0b8590d20dbc1139ff9c5b8915 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Mon, 17 Aug 2026 21:08:41 -0400 Subject: [PATCH 0283/1352] NFSD: Give delegations their own state shrinker Since commit 44df6f439a17 ("NFSD: add delegation reaper to react to low memory condition"), nfsd_client_shrinker has managed two unrelated populations of objects. One population is courtesy clients. Shrinking that population can be done synchronously and without risk of deadlock. The shrinker callback could return a precise count of the number of objects that were released. The other population is delegations. Shrinking that population requires sending a CB_RECALL_ANY; clients are not obligated to return any delegation. The shrinker callback is structurally unable to report progress. What's more, the single shrinker callback falls back to delegation reaping only when there are no courtesy clients left to reclaim. A single courtesy client is enough to keep a namespace's delegations out of the count it reports. To begin to resolve these issues, refactor the existing state shrinker into two: one for courtesy clients and one for reaping delegations. Each manages the size of its own population, and the shrinker names become namespace-specific. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260817-recall-any-keep-count-v5-6-3b2cffce701e@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/netns.h | 6 ++-- fs/nfsd/nfs4state.c | 77 +++++++++++++++++++++++++++++++++++---------- 2 files changed, 65 insertions(+), 18 deletions(-) diff --git a/fs/nfsd/netns.h b/fs/nfsd/netns.h index bb62d19430bcf5..ef01a1cf72acc8 100644 --- a/fs/nfsd/netns.h +++ b/fs/nfsd/netns.h @@ -240,8 +240,10 @@ struct nfsd_net { atomic_t nfsd_courtesy_clients; /* per-namespace; num_delegations in nfs4state.c is host-wide */ atomic_long_t nfsd_delegations; - struct shrinker *nfsd_client_shrinker; - struct work_struct nfsd_shrinker_work; + struct shrinker *nfsd_courtesy_shrinker; + struct shrinker *nfsd_deleg_shrinker; + struct work_struct nfsd_courtesy_work; + struct work_struct nfsd_deleg_work; /* last time an admin-revoke happened for NFSv4.0 */ time64_t nfs40_last_revoke; diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c index ccf22ccb2839ad..6b3296c33817d2 100644 --- a/fs/nfsd/nfs4state.c +++ b/fs/nfsd/nfs4state.c @@ -5579,16 +5579,27 @@ nfsd4_init_slabs(void) } static unsigned long -nfsd4_state_shrinker_count(struct shrinker *shrink, struct shrink_control *sc) +nfsd4_courtesy_shrinker_count(struct shrinker *shrink, + struct shrink_control *sc) { struct nfsd_net *nn = shrink->private_data; long count; count = atomic_read(&nn->nfsd_courtesy_clients); - if (!count) - count = atomic_long_read(&nn->nfsd_delegations); if (count) - queue_work(laundry_wq, &nn->nfsd_shrinker_work); + queue_work(laundry_wq, &nn->nfsd_courtesy_work); + return (unsigned long)count; +} + +static unsigned long +nfsd4_deleg_shrinker_count(struct shrinker *shrink, struct shrink_control *sc) +{ + struct nfsd_net *nn = shrink->private_data; + long count; + + count = atomic_long_read(&nn->nfsd_delegations); + if (count) + queue_work(laundry_wq, &nn->nfsd_deleg_work); return (unsigned long)count; } @@ -5598,6 +5609,25 @@ nfsd4_state_shrinker_scan(struct shrinker *shrink, struct shrink_control *sc) return SHRINK_STOP; } +static struct shrinker * +nfsd4_alloc_state_shrinker(struct nfsd_net *nn, const char *name, + unsigned long (*count)(struct shrinker *, + struct shrink_control *)) +{ + struct shrinker *shrink; + + shrink = shrinker_alloc(0, "%s:%s", name, nn->nfsd_name); + if (!shrink) + return NULL; + + shrink->count_objects = count; + shrink->scan_objects = nfsd4_state_shrinker_scan; + shrink->private_data = nn; + + shrinker_register(shrink); + return shrink; +} + void nfsd4_init_leases_net(struct nfsd_net *nn) { @@ -8011,12 +8041,20 @@ deleg_reaper(struct nfsd_net *nn) } static void -nfsd4_state_shrinker_worker(struct work_struct *work) +nfsd4_courtesy_shrinker_worker(struct work_struct *work) { struct nfsd_net *nn = container_of(work, struct nfsd_net, - nfsd_shrinker_work); + nfsd_courtesy_work); courtesy_client_reaper(nn); +} + +static void +nfsd4_deleg_shrinker_worker(struct work_struct *work) +{ + struct nfsd_net *nn = container_of(work, struct nfsd_net, + nfsd_deleg_work); + deleg_reaper(nn); } @@ -9979,21 +10017,26 @@ static int nfs4_state_create_net(struct net *net) INIT_DELAYED_WORK(&nn->laundromat_work, laundromat_main); /* Make sure this cannot run until client tracking is initialised */ disable_delayed_work(&nn->laundromat_work); - INIT_WORK(&nn->nfsd_shrinker_work, nfsd4_state_shrinker_worker); + INIT_WORK(&nn->nfsd_courtesy_work, nfsd4_courtesy_shrinker_worker); + INIT_WORK(&nn->nfsd_deleg_work, nfsd4_deleg_shrinker_worker); get_net(net); - nn->nfsd_client_shrinker = shrinker_alloc(0, "nfsd-client"); - if (!nn->nfsd_client_shrinker) + nn->nfsd_courtesy_shrinker = + nfsd4_alloc_state_shrinker(nn, "nfsd-courtesy", + nfsd4_courtesy_shrinker_count); + if (!nn->nfsd_courtesy_shrinker) goto err_shrinker; - nn->nfsd_client_shrinker->scan_objects = nfsd4_state_shrinker_scan; - nn->nfsd_client_shrinker->count_objects = nfsd4_state_shrinker_count; - nn->nfsd_client_shrinker->private_data = nn; - - shrinker_register(nn->nfsd_client_shrinker); + nn->nfsd_deleg_shrinker = + nfsd4_alloc_state_shrinker(nn, "nfsd-delegation", + nfsd4_deleg_shrinker_count); + if (!nn->nfsd_deleg_shrinker) + goto err_deleg_shrinker; return 0; +err_deleg_shrinker: + shrinker_free(nn->nfsd_courtesy_shrinker); err_shrinker: put_net(net); kfree(nn->sessionid_hashtbl); @@ -10094,8 +10137,10 @@ nfs4_state_shutdown_net(struct net *net) struct list_head *pos, *next, reaplist; struct nfsd_net *nn = net_generic(net, nfsd_net_id); - shrinker_free(nn->nfsd_client_shrinker); - cancel_work_sync(&nn->nfsd_shrinker_work); + shrinker_free(nn->nfsd_courtesy_shrinker); + shrinker_free(nn->nfsd_deleg_shrinker); + cancel_work_sync(&nn->nfsd_courtesy_work); + cancel_work_sync(&nn->nfsd_deleg_work); disable_delayed_work_sync(&nn->laundromat_work); locks_end_grace(&nn->nfsd4_manager); From 08bc0fa35bdc93e4b92a6555ccbec9ca95445f2a Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Mon, 17 Aug 2026 21:08:42 -0400 Subject: [PATCH 0284/1352] NFSD: Pace the state shrinker's scan requests Currently, neither of the scan callback functions records anything before returning SHRINK_STOP, so the size of each scan request is discarded. That size is the only real measure NFSD gets of reclaim pressure. Both count callbacks report their population whether or not the reaper is already queued to reclaim it, so reclaim asks again for work that is pending. Accumulate each courtesy scan request in nfsd_shrink_backlog and subtract the backlog from what that count callback reports. The worker retires the backlog once courtesy_client_reaper() has run. That reaper expires the clients synchronously, so the discount covers exactly the interval the work is pending. Delegations need a different bound. This is because deleg_reaper() only sends CB_RECALL_ANY and does not track how many delegations were actually returned by the targeted client. Report the delegations only once NFSD_RECALL_ANY_COOLDOWN_SECS have passed since the last sweep. deleg_reaper() skips any client it recalled from within that window, so an earlier scan request cannot produce another recall. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260817-recall-any-keep-count-v5-7-3b2cffce701e@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/netns.h | 6 +++ fs/nfsd/nfs4state.c | 92 ++++++++++++++++++++++++++++++++++++++------- 2 files changed, 85 insertions(+), 13 deletions(-) diff --git a/fs/nfsd/netns.h b/fs/nfsd/netns.h index ef01a1cf72acc8..23923cc4aa4712 100644 --- a/fs/nfsd/netns.h +++ b/fs/nfsd/netns.h @@ -245,6 +245,12 @@ struct nfsd_net { struct work_struct nfsd_courtesy_work; struct work_struct nfsd_deleg_work; + /* courtesy scan requests the reaper has not retired yet */ + atomic_long_t nfsd_shrink_backlog; + + /* when deleg_reaper() last swept the client list */ + time64_t nfsd_last_recall_any; + /* last time an admin-revoke happened for NFSv4.0 */ time64_t nfs40_last_revoke; diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c index 6b3296c33817d2..48ffbee3419c62 100644 --- a/fs/nfsd/nfs4state.c +++ b/fs/nfsd/nfs4state.c @@ -5578,41 +5578,88 @@ nfsd4_init_slabs(void) return -ENOMEM; } +#define NFSD_RECALL_ANY_COOLDOWN_SECS 5 + static unsigned long nfsd4_courtesy_shrinker_count(struct shrinker *shrink, struct shrink_control *sc) { struct nfsd_net *nn = shrink->private_data; - long count; + long backlog, count; count = atomic_read(&nn->nfsd_courtesy_clients); - if (count) - queue_work(laundry_wq, &nn->nfsd_courtesy_work); - return (unsigned long)count; + if (!count) + return 0; + + queue_work(laundry_wq, &nn->nfsd_courtesy_work); + + /* Work already queued is not available to reclaim again. */ + backlog = atomic_long_read(&nn->nfsd_shrink_backlog); + return count > backlog ? count - backlog : 0; } static unsigned long nfsd4_deleg_shrinker_count(struct shrinker *shrink, struct shrink_control *sc) { struct nfsd_net *nn = shrink->private_data; + time64_t elapsed; long count; count = atomic_long_read(&nn->nfsd_delegations); - if (count) - queue_work(laundry_wq, &nn->nfsd_deleg_work); - return (unsigned long)count; + if (!count) + return 0; + + /* + * Delegations the last sweep reached stay unreclaimable until + * deleg_reaper()'s cooldown expires. CB_RECALL_ANY leaves the + * choice of delegations to the client, so there is no return + * to wait on instead. + */ + elapsed = ktime_get_boottime_seconds() - + READ_ONCE(nn->nfsd_last_recall_any); + if (elapsed < NFSD_RECALL_ANY_COOLDOWN_SECS) + return 0; + + queue_work(laundry_wq, &nn->nfsd_deleg_work); + return count; } static unsigned long -nfsd4_state_shrinker_scan(struct shrinker *shrink, struct shrink_control *sc) +nfsd4_courtesy_shrinker_scan(struct shrinker *shrink, + struct shrink_control *sc) { + struct nfsd_net *nn = shrink->private_data; + + atomic_long_add(sc->nr_to_scan, &nn->nfsd_shrink_backlog); + queue_work(laundry_wq, &nn->nfsd_courtesy_work); + + /* + * The reaper runs from laundry_wq. Report no progress rather + * than claim memory that is not free yet. + */ + return SHRINK_STOP; +} + +static unsigned long +nfsd4_deleg_shrinker_scan(struct shrinker *shrink, struct shrink_control *sc) +{ + struct nfsd_net *nn = shrink->private_data; + + queue_work(laundry_wq, &nn->nfsd_deleg_work); + + /* + * The reaper sends CB_RECALL_ANY, so nothing is free when + * this returns. + */ return SHRINK_STOP; } static struct shrinker * nfsd4_alloc_state_shrinker(struct nfsd_net *nn, const char *name, unsigned long (*count)(struct shrinker *, - struct shrink_control *)) + struct shrink_control *), + unsigned long (*scan)(struct shrinker *, + struct shrink_control *)) { struct shrinker *shrink; @@ -5621,7 +5668,7 @@ nfsd4_alloc_state_shrinker(struct nfsd_net *nn, const char *name, return NULL; shrink->count_objects = count; - shrink->scan_objects = nfsd4_state_shrinker_scan; + shrink->scan_objects = scan; shrink->private_data = nn; shrinker_register(shrink); @@ -8006,7 +8053,8 @@ deleg_reaper(struct nfsd_net *nn) continue; if (atomic_read(&clp->cl_delegs_in_recall)) continue; - if (ktime_get_boottime_seconds() - clp->cl_ra_time < 5) + if (ktime_get_boottime_seconds() - clp->cl_ra_time < + NFSD_RECALL_ANY_COOLDOWN_SECS) continue; if (clp->cl_cb_state != NFSD4_CB_UP) continue; @@ -8038,6 +8086,12 @@ deleg_reaper(struct nfsd_net *nn) nfsd4_run_cb(&clp->cl_ra->ra_cb); } spin_unlock(&nn->client_lock); + + /* + * Stamp the sweep even when no recall went out. A sweep that + * found nothing eligible finds nothing on an immediate retry. + */ + WRITE_ONCE(nn->nfsd_last_recall_any, ktime_get_boottime_seconds()); } static void @@ -8045,8 +8099,16 @@ nfsd4_courtesy_shrinker_worker(struct work_struct *work) { struct nfsd_net *nn = container_of(work, struct nfsd_net, nfsd_courtesy_work); + long backlog; + /* + * Retire only the requests sampled here, so that requests + * arriving while the reaper runs are still discounted by + * nfsd4_courtesy_shrinker_count(). + */ + backlog = atomic_long_read(&nn->nfsd_shrink_backlog); courtesy_client_reaper(nn); + atomic_long_sub(backlog, &nn->nfsd_shrink_backlog); } static void @@ -10019,17 +10081,21 @@ static int nfs4_state_create_net(struct net *net) disable_delayed_work(&nn->laundromat_work); INIT_WORK(&nn->nfsd_courtesy_work, nfsd4_courtesy_shrinker_worker); INIT_WORK(&nn->nfsd_deleg_work, nfsd4_deleg_shrinker_worker); + atomic_long_set(&nn->nfsd_shrink_backlog, 0); + nn->nfsd_last_recall_any = 0; get_net(net); nn->nfsd_courtesy_shrinker = nfsd4_alloc_state_shrinker(nn, "nfsd-courtesy", - nfsd4_courtesy_shrinker_count); + nfsd4_courtesy_shrinker_count, + nfsd4_courtesy_shrinker_scan); if (!nn->nfsd_courtesy_shrinker) goto err_shrinker; nn->nfsd_deleg_shrinker = nfsd4_alloc_state_shrinker(nn, "nfsd-delegation", - nfsd4_deleg_shrinker_count); + nfsd4_deleg_shrinker_count, + nfsd4_deleg_shrinker_scan); if (!nn->nfsd_deleg_shrinker) goto err_deleg_shrinker; From 4ba83d9c0b48d7807c6e36260d6edcac660d5d29 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Mon, 17 Aug 2026 21:08:43 -0400 Subject: [PATCH 0285/1352] NFSD: Apportion CB_RECALL_ANY recalls among clients deleg_reaper() asks each eligible client to return one delegation whenever it runs, whether or not anything needs the memory. A delegation returned before it is needed costs the client an OPEN when it next touches the file. Nothing sizes the request either. The delegation scan callback discards nr_to_scan, which is reclaim's statement of how many objects it wants back. Record each delegation scan request in nfsd_deleg_backlog and pass the accumulated total to deleg_reaper(). Handing that total to every client would ask for it once per client, so scale it by each client's share of the delegations this sweep can reach. The count callback reports what is left after the outstanding requests, so concurrent reclaimers do not each ask for the same delegations. cl_ra_time keeps the next sweep from returning to the clients this one reached. Nothing is recalled until a scan arrives. nfs4_laundromat() is the exception. It has no scan request to pass, so it computes what must go for num_delegations to fall below max_delegations, and passes only this namespace's share. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260817-recall-any-keep-count-v5-8-3b2cffce701e@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/netns.h | 3 + fs/nfsd/nfs4state.c | 131 +++++++++++++++++++++++++++++++++++--------- 2 files changed, 109 insertions(+), 25 deletions(-) diff --git a/fs/nfsd/netns.h b/fs/nfsd/netns.h index 23923cc4aa4712..0ce7da20aba3c1 100644 --- a/fs/nfsd/netns.h +++ b/fs/nfsd/netns.h @@ -248,6 +248,9 @@ struct nfsd_net { /* courtesy scan requests the reaper has not retired yet */ atomic_long_t nfsd_shrink_backlog; + /* delegation scan requests the reaper has not retired yet */ + atomic_long_t nfsd_deleg_backlog; + /* when deleg_reaper() last swept the client list */ time64_t nfsd_last_recall_any; diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c index 48ffbee3419c62..88f6c0a0d8b3fe 100644 --- a/fs/nfsd/nfs4state.c +++ b/fs/nfsd/nfs4state.c @@ -93,7 +93,7 @@ static void nfs4_free_ol_stateid(struct nfs4_stid *stid); static void nfsd4_end_grace(struct nfsd_net *nn); static void _free_cpntf_state_locked(struct nfsd_net *nn, struct nfs4_cpntf_state *cps); static void nfsd4_file_hash_remove(struct nfs4_file *fi); -static void deleg_reaper(struct nfsd_net *nn); +static void deleg_reaper(struct nfsd_net *nn, unsigned long backlog); static void nfsd4_drop_revoked_stid(struct nfs4_stid *s) __releases(&s->sc_client->cl_lock); @@ -5603,7 +5603,7 @@ nfsd4_deleg_shrinker_count(struct shrinker *shrink, struct shrink_control *sc) { struct nfsd_net *nn = shrink->private_data; time64_t elapsed; - long count; + long backlog, count; count = atomic_long_read(&nn->nfsd_delegations); if (!count) @@ -5620,8 +5620,14 @@ nfsd4_deleg_shrinker_count(struct shrinker *shrink, struct shrink_control *sc) if (elapsed < NFSD_RECALL_ANY_COOLDOWN_SECS) return 0; - queue_work(laundry_wq, &nn->nfsd_deleg_work); - return count; + /* + * Unlike the courtesy shrinker, this one queues no work. + * Nothing is recalled until a scan request arrives. Subtract + * the requests already recorded, or concurrent reclaimers + * each see the whole namespace and stack a scan on top of it. + */ + backlog = atomic_long_read(&nn->nfsd_deleg_backlog); + return count > backlog ? count - backlog : 0; } static unsigned long @@ -5645,6 +5651,7 @@ nfsd4_deleg_shrinker_scan(struct shrinker *shrink, struct shrink_control *sc) { struct nfsd_net *nn = shrink->private_data; + atomic_long_add(sc->nr_to_scan, &nn->nfsd_deleg_backlog); queue_work(laundry_wq, &nn->nfsd_deleg_work); /* @@ -7894,6 +7901,7 @@ nfs4_laundromat(struct nfsd_net *nn) struct nfs4_cpntf_state *cps; struct nfs4_client *clp; copy_stateid_t *cps_t; + long held, host, n; int i; if (clients_still_reclaiming(nn)) { @@ -8007,8 +8015,22 @@ nfs4_laundromat(struct nfsd_net *nn) /* service the server-to-server copy delayed unmount list */ nfsd4_ssc_expire_umount(nn); #endif - if (atomic_long_read(&num_delegations) >= max_delegations) - deleg_reaper(nn); + /* + * set_max_delegations() computes a zero max_delegations on a + * server with very little memory. @host is a divisor below. + */ + host = atomic_long_read(&num_delegations); + if (host && host >= max_delegations) { + /* + * max_delegations bounds the host, but the laundromat + * runs once per network namespace. Requesting the whole + * overage in each would multiply the request, so take + * only this namespace's share. + */ + held = atomic_long_read(&nn->nfsd_delegations); + n = host - max_delegations + 1; + deleg_reaper(nn, DIV64_U64_ROUND_UP((u64)n * held, host)); + } out: return max_t(time64_t, lt.new_timeo, NFSD_LAUNDROMAT_MINTIMEOUT); } @@ -8036,27 +8058,58 @@ courtesy_client_reaper(struct nfsd_net *nn) nfs4_process_client_reaplist(&reaplist); } +/* The two passes in deleg_reaper() must agree on which clients are asked. */ +static bool +deleg_reaper_eligible(const struct nfs4_client *clp, time64_t now) +{ + if (clp->cl_minorversion == 0) + return false; + if (clp->cl_state != NFSD4_ACTIVE) + return false; + if (atomic_read(&clp->cl_delegs_in_recall)) + return false; + if (test_bit(NFSD4_CALLBACK_RUNNING, &clp->cl_ra->ra_cb.cb_flags)) + return false; + if (now - clp->cl_ra_time < NFSD_RECALL_ANY_COOLDOWN_SECS) + return false; + if (clp->cl_cb_state != NFSD4_CB_UP) + return false; + return true; +} + static void -deleg_reaper(struct nfsd_net *nn) +deleg_reaper(struct nfsd_net *nn, unsigned long backlog) { struct list_head *pos, *next; struct nfs4_client *clp; + unsigned long remaining, share, total; unsigned int count; + time64_t now; + + /* + * Recalling a delegation before it is needed costs the client + * an OPEN when it next touches the file. Leave + * nfsd_last_recall_any unstamped so the next sweep is not + * delayed. + */ + if (!backlog) + return; + now = ktime_get_boottime_seconds(); spin_lock(&nn->client_lock); - list_for_each_safe(pos, next, &nn->client_lru) { + + /* + * Only the clients this sweep asks contribute to the + * apportionment. Dividing the request among holders that are + * skipped under-serves it, and the shortfall goes nowhere: + * nfsd4_deleg_shrinker_worker() has already cleared + * nfsd_deleg_backlog. + */ + total = 0; + list_for_each(pos, &nn->client_lru) { clp = list_entry(pos, struct nfs4_client, cl_lru); - if (clp->cl_minorversion == 0) - continue; - if (clp->cl_state != NFSD4_ACTIVE) - continue; - if (atomic_read(&clp->cl_delegs_in_recall)) - continue; - if (ktime_get_boottime_seconds() - clp->cl_ra_time < - NFSD_RECALL_ANY_COOLDOWN_SECS) - continue; - if (clp->cl_cb_state != NFSD4_CB_UP) + if (!deleg_reaper_eligible(clp, now)) continue; /* * This read races with hash_delegation_locked() and @@ -8064,6 +8117,25 @@ deleg_reaper(struct nfsd_net *nn) * count only skews the keep value; the next * laundromat pass sees a more current one. */ + total += data_race(READ_ONCE(clp->cl_deleg_count)); + } + if (!total) + goto out; + + /* + * Reclaim asks in batches and is not bound by what the count + * callback reported, so the backlog can exceed what these + * clients hold. Cap it to keep each share within the client's + * own count. + */ + backlog = min(backlog, total); + remaining = backlog; + + list_for_each_safe(pos, next, &nn->client_lru) { + clp = list_entry(pos, struct nfs4_client, cl_lru); + + if (!deleg_reaper_eligible(clp, now)) + continue; count = data_race(READ_ONCE(clp->cl_deleg_count)); if (!count) continue; @@ -8072,26 +8144,34 @@ deleg_reaper(struct nfsd_net *nn) /* release in nfsd4_cb_recall_any_release */ kref_get(&clp->cl_nfsdfs.cl_ref); - clp->cl_ra_time = ktime_get_boottime_seconds(); + clp->cl_ra_time = now; /* - * Ask for a single delegation. Recalling one before it - * is needed costs the client an OPEN when it next - * touches the file. + * Rounding up guarantees every holder gives up at least + * one. The round-up can overshoot @backlog, so stop + * once the request is met. client_lru is ordered by + * last renewal, so the least active clients are asked + * first. */ - clp->cl_ra->ra_keep = count - 1; + share = DIV64_U64_ROUND_UP((u64)backlog * count, total); + share = min(share, remaining); + remaining -= share; + clp->cl_ra->ra_keep = count - share; clp->cl_ra->ra_bmval[0] = BIT(RCA4_TYPE_MASK_RDATA_DLG) | BIT(RCA4_TYPE_MASK_WDATA_DLG) | BIT(RCA4_TYPE_MASK_DIR_DLG); trace_nfsd_cb_recall_any(clp->cl_ra); nfsd4_run_cb(&clp->cl_ra->ra_cb); + if (!remaining) + break; } +out: spin_unlock(&nn->client_lock); /* * Stamp the sweep even when no recall went out. A sweep that * found nothing eligible finds nothing on an immediate retry. */ - WRITE_ONCE(nn->nfsd_last_recall_any, ktime_get_boottime_seconds()); + WRITE_ONCE(nn->nfsd_last_recall_any, now); } static void @@ -8117,7 +8197,7 @@ nfsd4_deleg_shrinker_worker(struct work_struct *work) struct nfsd_net *nn = container_of(work, struct nfsd_net, nfsd_deleg_work); - deleg_reaper(nn); + deleg_reaper(nn, atomic_long_xchg(&nn->nfsd_deleg_backlog, 0)); } static inline __be32 nfs4_check_fh(struct svc_fh *fhp, struct nfs4_stid *stp) @@ -10082,6 +10162,7 @@ static int nfs4_state_create_net(struct net *net) INIT_WORK(&nn->nfsd_courtesy_work, nfsd4_courtesy_shrinker_worker); INIT_WORK(&nn->nfsd_deleg_work, nfsd4_deleg_shrinker_worker); atomic_long_set(&nn->nfsd_shrink_backlog, 0); + atomic_long_set(&nn->nfsd_deleg_backlog, 0); nn->nfsd_last_recall_any = 0; get_net(net); From c7a057c5af414618f383c370aad53888c29aa126 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Tue, 18 Aug 2026 10:00:33 -0400 Subject: [PATCH 0286/1352] NFSD: Move the nfs3.h include out of nfsd.h fs/nfsd/nfsd.h is included throughout the server, yet it uses no NFSv3 protocol definition of its own. The include there served only to make those definitions reach the few source files that need them, by way of nfsd.h itself or the xdr.h chain that pulls it in. Give each consumer its own include and drop the one in nfsd.h, so the header no longer carries a dependency unrelated to its contents. nfsfh.c, nfsctl.c, nfs3xdr.c, nfs3proc.c, and nfs2acl.c reference NFS3_* definitions directly; add to each. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260818140035.12740-1-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfs2acl.c | 1 + fs/nfsd/nfs3proc.c | 1 + fs/nfsd/nfs3xdr.c | 1 + fs/nfsd/nfsctl.c | 1 + fs/nfsd/nfsd.h | 1 - fs/nfsd/nfsfh.c | 1 + 6 files changed, 5 insertions(+), 1 deletion(-) diff --git a/fs/nfsd/nfs2acl.c b/fs/nfsd/nfs2acl.c index 33610deda3b0a7..0a5c444fef99d8 100644 --- a/fs/nfsd/nfs2acl.c +++ b/fs/nfsd/nfs2acl.c @@ -10,6 +10,7 @@ /* FIXME: nfsacl.h is a broken header */ #include #include +#include #include "cache.h" #include "xdr3.h" #include "vfs.h" diff --git a/fs/nfsd/nfs3proc.c b/fs/nfsd/nfs3proc.c index 17bbe5d13f1833..60cd01b6a37d24 100644 --- a/fs/nfsd/nfs3proc.c +++ b/fs/nfsd/nfs3proc.c @@ -9,6 +9,7 @@ #include #include #include +#include #include "cache.h" #include "xdr3.h" diff --git a/fs/nfsd/nfs3xdr.c b/fs/nfsd/nfs3xdr.c index 090cea8e545dc6..a14e829e1c4192 100644 --- a/fs/nfsd/nfs3xdr.c +++ b/fs/nfsd/nfs3xdr.c @@ -8,6 +8,7 @@ */ #include +#include #include #include "xdr3.h" #include "auth.h" diff --git a/fs/nfsd/nfsctl.c b/fs/nfsd/nfsctl.c index efde963d909ad6..eeae33a19a963d 100644 --- a/fs/nfsd/nfsctl.c +++ b/fs/nfsd/nfsctl.c @@ -20,6 +20,7 @@ #include #include #include +#include #include "idmap.h" #include "nfsd.h" diff --git a/fs/nfsd/nfsd.h b/fs/nfsd/nfsd.h index 64315890eef589..a145294c59c87c 100644 --- a/fs/nfsd/nfsd.h +++ b/fs/nfsd/nfsd.h @@ -14,7 +14,6 @@ #include #include -#include #include #include diff --git a/fs/nfsd/nfsfh.c b/fs/nfsd/nfsfh.c index b1f3c22af52586..2bd6907f443f78 100644 --- a/fs/nfsd/nfsfh.c +++ b/fs/nfsd/nfsfh.c @@ -9,6 +9,7 @@ */ #include +#include #include #include From 044591fc4bbf3b399001209b8c4e3a28e3d6399e Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Tue, 18 Aug 2026 10:00:34 -0400 Subject: [PATCH 0287/1352] NFSD: Include where struct nfs_fh is used struct nfsd4_copy embeds a struct nfs_fh, and nlm_fopen() reads the size and data fields of one. Neither fs/nfsd/xdr4.h nor fs/nfsd/lockd.c includes the header that defines the type; both reach it by way of nfsd.h, which pulls in . Add the direct include to both files, so nfsd.h can later drop the it carries for no use of its own. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260818140035.12740-2-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/lockd.c | 1 + fs/nfsd/xdr4.h | 2 ++ 2 files changed, 3 insertions(+) diff --git a/fs/nfsd/lockd.c b/fs/nfsd/lockd.c index 5ec0f545606323..f24e45dc37a05b 100644 --- a/fs/nfsd/lockd.c +++ b/fs/nfsd/lockd.c @@ -9,6 +9,7 @@ #include #include +#include #include "nfsd.h" #include "nfserr.h" #include "vfs.h" diff --git a/fs/nfsd/xdr4.h b/fs/nfsd/xdr4.h index 7bbb375874ef80..b841bc462dac8d 100644 --- a/fs/nfsd/xdr4.h +++ b/fs/nfsd/xdr4.h @@ -37,6 +37,8 @@ #ifndef _LINUX_NFSD_XDR4_H #define _LINUX_NFSD_XDR4_H +#include + #include "state.h" #include "vfs.h" From 1f3b46a3b6dca2a6bd0b4f8554c993b3af474f49 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Tue, 18 Aug 2026 10:00:35 -0400 Subject: [PATCH 0288/1352] NFSD: Clean up header guards in fs/nfsd/xdr.h Make the header guards less ambiguous about their provenance. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260818140035.12740-3-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/xdr.h | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/fs/nfsd/xdr.h b/fs/nfsd/xdr.h index df540c940cefe4..52660b9f8dde43 100644 --- a/fs/nfsd/xdr.h +++ b/fs/nfsd/xdr.h @@ -1,8 +1,8 @@ /* SPDX-License-Identifier: GPL-2.0 */ /* XDR types for nfsd. This is mainly a typing exercise. */ -#ifndef LINUX_NFSD_H -#define LINUX_NFSD_H +#ifndef _LINUX_NFSD_XDR_H +#define _LINUX_NFSD_XDR_H #include #include "nfsd.h" @@ -175,4 +175,4 @@ bool svcxdr_encode_stat(struct xdr_stream *xdr, __be32 status); bool svcxdr_encode_fattr(struct svc_rqst *rqstp, struct xdr_stream *xdr, const struct svc_fh *fhp, const struct kstat *stat); -#endif /* LINUX_NFSD_H */ +#endif /* _LINUX_NFSD_XDR_H */ From e2002fed1a77b59dcade2facdd4e769f82e589cc Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Tue, 18 Aug 2026 14:41:51 -0400 Subject: [PATCH 0289/1352] NFSD: Resolve the recall-any mask names in the trace format RCA4_TYPE_MASK_* are enum constants, so the preprocessor cannot fold them into the print format that show_rca_mask() builds for the nfsd_cb_recall_any event. Nothing declares an eval map for them either, so trace_event_eval_update() has no substitution to apply at module load, and the event's format file ships the enumerator names verbatim. trace-cmd and perf cannot decode the bmval0 field. Declare the eval maps for the nine mask bits show_rca_mask() decodes. The format then carries the shift counts as integers, the same shape the SUNRPC trace points already emit from their BIT() flag decoders. Fixes: 638593be55c0 ("NFSD: add CB_RECALL_ANY tracepoints") Cc: stable@vger.kernel.org Reviewed-by: Jeff Layton Reviewed-by: Christoph Hellwig Link: https://patch.msgid.link/20260818184151.31180-1-cel@kernel.org Signed-off-by: Chuck Lever --- include/trace/misc/nfs.h | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/include/trace/misc/nfs.h b/include/trace/misc/nfs.h index b5fb77d7954b35..27781bd7a3f787 100644 --- a/include/trace/misc/nfs.h +++ b/include/trace/misc/nfs.h @@ -359,6 +359,16 @@ TRACE_DEFINE_ENUM(IOMODE_ANY); { IOMODE_RW, "RW" }, \ { IOMODE_ANY, "ANY" }) +TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_RDATA_DLG); +TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_WDATA_DLG); +TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_DIR_DLG); +TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_FILE_LAYOUT); +TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_BLK_LAYOUT); +TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_OBJ_LAYOUT_MIN); +TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_OBJ_LAYOUT_MAX); +TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_OTHER_LAYOUT_MIN); +TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_OTHER_LAYOUT_MAX); + #define show_rca_mask(x) \ __print_flags(x, "|", \ { BIT(RCA4_TYPE_MASK_RDATA_DLG), "RDATA_DLG" }, \ From 985f64f29bb8158dea3f5e1da74eba8a30a25262 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Fri, 21 Aug 2026 13:22:39 -0400 Subject: [PATCH 0290/1352] SUNRPC: Separate the TLS control-record receive from its policy svc_tcp_sock_recv_cmsg() receives the record at the head of the kTLS receive queue and decides what its content type means for the transport. The receive and the decision are one step, so a caller cannot learn a record's type without also acting on it. A later caller classifies the head of the queue with MSG_PEEK before it decides whether to consume the record. It needs the receive without the decision. Move the receive into svc_tcp_recv_cmsg(), which takes the recvmsg() flags and reports the octet count, the record type, and the message flags. The alert policy stays in svc_tcp_sock_recv_cmsg(), the helper's only caller in this patch. The zeroed control buffer replaces the msg_controllen check. An unfilled buffer reports record type zero, so a positive octet count with no record type now returns -EBADMSG instead of a count. The alert parse runs on a second msghdr over the alert buffer, so the copy and the parse share no iterator state. The rewind that commit bee47cb026e7 ("sunrpc: fix handling of server side tls alerts") added by hand is no longer needed. Link: https://patch.msgid.link/20260821-tls-read-sock-2-v1-1-7ffce164eb45@kernel.org Signed-off-by: Chuck Lever --- net/sunrpc/svcsock.c | 136 +++++++++++++++++++------------------------ 1 file changed, 59 insertions(+), 77 deletions(-) diff --git a/net/sunrpc/svcsock.c b/net/sunrpc/svcsock.c index 5a2d52284d7514..b402923c40f120 100644 --- a/net/sunrpc/svcsock.c +++ b/net/sunrpc/svcsock.c @@ -271,95 +271,77 @@ svc_tcp_sock_drain_record(struct socket *sock) } } -static int -svc_tcp_sock_process_cmsg(struct socket *sock, struct msghdr *msg, - struct cmsghdr *cmsg, int ret) +static int svc_tcp_recv_cmsg(struct socket *sock, int flags, + struct kvec *payload, u8 *type, + unsigned int *msg_flags) { - u8 content_type = tls_get_record_type(sock->sk, cmsg); - u8 level, description; + union { + struct cmsghdr cmsg; + u8 buf[CMSG_SPACE(sizeof(u8))]; + } u = {}; + struct msghdr msg = { + .msg_control = &u, + .msg_controllen = sizeof(u), + }; + int ret; - switch (content_type) { - case 0: - break; - case TLS_RECORD_TYPE_DATA: - /* TLS sets EOR at the end of each application data - * record, even though there might be more frames - * waiting to be decrypted. - */ - msg->msg_flags &= ~MSG_EOR; - break; - case TLS_RECORD_TYPE_ALERT: - tls_alert_recv(sock->sk, msg, &level, &description); - /* RFC 8446 Section 6: every alert but a closure alert is - * an error alert. - */ - switch (description) { - case TLS_ALERT_DESC_CLOSE_NOTIFY: - case TLS_ALERT_DESC_USER_CANCELED: - ret = -EAGAIN; - break; - default: - ret = -ENOTCONN; - } - break; - default: - /* discard this record type */ - ret = -EAGAIN; - } + iov_iter_kvec(&msg.msg_iter, ITER_DEST, payload, 1, payload->iov_len); + ret = sock_recvmsg(sock, &msg, flags); + if (ret < 0) + return ret; + *msg_flags = msg.msg_flags; + *type = tls_get_record_type(sock->sk, &u.cmsg); + if (!*type && ret) + return -EBADMSG; return ret; } static int -svc_tcp_sock_recv_cmsg(struct socket *sock, unsigned int *msg_flags) +svc_tcp_sock_recv_cmsg(struct socket *sock) { - union { - struct cmsghdr cmsg; - u8 buf[CMSG_SPACE(sizeof(u8))]; - } u; - u8 alert[2]; - struct kvec alert_kvec = { - .iov_base = alert, - .iov_len = sizeof(alert), - }; - struct msghdr msg = { - .msg_flags = *msg_flags, - .msg_control = &u, - .msg_controllen = sizeof(u), + u8 alert[2], type, level, description; + struct kvec recv_kvec = { + .iov_base = alert, + .iov_len = sizeof(alert), }; + unsigned int msg_flags; + struct msghdr msg = {}; int ret; - iov_iter_kvec(&msg.msg_iter, ITER_DEST, &alert_kvec, 1, - alert_kvec.iov_len); - ret = sock_recvmsg(sock, &msg, MSG_DONTWAIT); - /* put_cmsg() shrinks msg_controllen, so a short one means - * kTLS filled in u.cmsg. + ret = svc_tcp_recv_cmsg(sock, MSG_DONTWAIT, &recv_kvec, &type, + &msg_flags); + if (ret < 0 || !type) + return ret; + if (type != TLS_RECORD_TYPE_ALERT) { + /* An application data record carries RPC payload. + * Draining one breaks RPC fragment framing. + */ + if (type != TLS_RECORD_TYPE_DATA && !(msg_flags & MSG_EOR)) + svc_tcp_sock_drain_record(sock); + return -EAGAIN; + } + /* An Alert record carries exactly one two-octet message (RFC + * 8446 Section 5.1). recv_kvec caps the receive at two, so a + * longer record produces the same count. MSG_EOR appears only + * once kTLS has drained the whole record. */ - if (ret >= 0 && msg.msg_controllen < sizeof(u)) { - u8 content_type = tls_get_record_type(sock->sk, &u.cmsg); + if (ret != sizeof(alert) || !(msg_flags & MSG_EOR)) + return -EBADMSG; - /* Returning the count would credit the RPC stream with - * octets that never reached the caller's buffer. - */ - if (content_type != TLS_RECORD_TYPE_ALERT) { - /* An application data record carries RPC payload. - * Draining one breaks RPC fragment framing. - */ - if (content_type != TLS_RECORD_TYPE_DATA && - !(msg.msg_flags & MSG_EOR)) - svc_tcp_sock_drain_record(sock); - return -EAGAIN; - } - /* An Alert record carries exactly one two-octet message - * (RFC 8446 Section 5.1). alert_kvec caps the receive at two, - * so a longer record produces the same count. MSG_EOR appears - * only once kTLS has drained the whole record. - */ - if (ret != sizeof(alert) || !(msg.msg_flags & MSG_EOR)) - return -EBADMSG; - iov_iter_revert(&msg.msg_iter, ret); - ret = svc_tcp_sock_process_cmsg(sock, &msg, &u.cmsg, -EAGAIN); + iov_iter_kvec(&msg.msg_iter, ITER_DEST, &recv_kvec, 1, + recv_kvec.iov_len); + tls_alert_recv(sock->sk, &msg, &level, &description); + + /* RFC 8446 Section 6: every alert but a closure alert is + * an error alert. + */ + switch (description) { + case TLS_ALERT_DESC_CLOSE_NOTIFY: + case TLS_ALERT_DESC_USER_CANCELED: + return -EAGAIN; + default: + return -ENOTCONN; } - return ret; } static int @@ -372,7 +354,7 @@ svc_tcp_sock_recvmsg(struct svc_sock *svsk, struct msghdr *msg) if (msg->msg_flags & MSG_CTRUNC) { msg->msg_flags &= ~(MSG_CTRUNC | MSG_EOR); if (ret == 0 || ret == -EIO) { - ret = svc_tcp_sock_recv_cmsg(sock, &msg->msg_flags); + ret = svc_tcp_sock_recv_cmsg(sock); /* A control record delivers nothing to the caller, * and kTLS announces no data_ready for records it * already holds. Mark the transport ready so that From 78328b52fcbd11bcd09cba7a8f9383064386fd9a Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Fri, 21 Aug 2026 13:22:40 -0400 Subject: [PATCH 0291/1352] SUNRPC: Close the transport on an unhandled TLS record type svc_tcp_sock_recv_cmsg() drains a TLS record that is neither an alert nor application data, then returns -EAGAIN so the receive loop retries. It runs only on an established TLS session. A handshake record there carries a post-handshake message, and the server has no handler for one. Draining a KeyUpdate only delays the close. kTLS sets key_update_pending when it decrypts that record, and a later receive returns -EKEYEXPIRED. kTLS flags only a KeyUpdate, so any other post-handshake message disappears and the connection keeps running. Return -EPROTO for an unhandled record type. Any error but -EAGAIN closes the transport. An application data record keeps its -EAGAIN return. No NFS client is known to send a handshake record on an established connection. Link: https://patch.msgid.link/20260821-tls-read-sock-2-v1-2-7ffce164eb45@kernel.org Signed-off-by: Chuck Lever --- net/sunrpc/svcsock.c | 47 +++++++------------------------------------- 1 file changed, 7 insertions(+), 40 deletions(-) diff --git a/net/sunrpc/svcsock.c b/net/sunrpc/svcsock.c index b402923c40f120..ae1f3c474f8b96 100644 --- a/net/sunrpc/svcsock.c +++ b/net/sunrpc/svcsock.c @@ -238,39 +238,6 @@ static int svc_one_sock_name(struct svc_sock *svsk, char *buf, int remaining) return len; } -/* - * kTLS delivers a record only up to the caller's buffer and keeps - * the remainder on its receive list, where no further data_ready - * announces it. Consume the whole record. - */ -static void -svc_tcp_sock_drain_record(struct socket *sock) -{ - union { - struct cmsghdr cmsg; - u8 buf[CMSG_SPACE(sizeof(u8))]; - } u; - u8 discard[64]; - struct kvec discard_kvec = { - .iov_base = discard, - .iov_len = sizeof(discard), - }; - - for (;;) { - struct msghdr msg = { - .msg_control = &u, - .msg_controllen = sizeof(u), - }; - - iov_iter_kvec(&msg.msg_iter, ITER_DEST, &discard_kvec, 1, - discard_kvec.iov_len); - if (sock_recvmsg(sock, &msg, MSG_DONTWAIT) <= 0) - break; - if (msg.msg_flags & MSG_EOR) - break; - } -} - static int svc_tcp_recv_cmsg(struct socket *sock, int flags, struct kvec *payload, u8 *type, unsigned int *msg_flags) @@ -312,14 +279,14 @@ svc_tcp_sock_recv_cmsg(struct socket *sock) &msg_flags); if (ret < 0 || !type) return ret; - if (type != TLS_RECORD_TYPE_ALERT) { - /* An application data record carries RPC payload. - * Draining one breaks RPC fragment framing. - */ - if (type != TLS_RECORD_TYPE_DATA && !(msg_flags & MSG_EOR)) - svc_tcp_sock_drain_record(sock); + /* A data record reaches here only when kTLS queued an empty one + * ahead of the control record. Consuming it takes no payload, + * and the retry picks up the control record. + */ + if (type == TLS_RECORD_TYPE_DATA) return -EAGAIN; - } + if (type != TLS_RECORD_TYPE_ALERT) + return -EPROTO; /* An Alert record carries exactly one two-octet message (RFC * 8446 Section 5.1). recv_kvec caps the receive at two, so a * longer record produces the same count. MSG_EOR appears only From 43f94122a249c74ae8e8f7f5f2cae3f95a0c92ec Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Fri, 21 Aug 2026 13:22:41 -0400 Subject: [PATCH 0292/1352] SUNRPC: Flush a received record's pages once it is complete svc_tcp_read_msg() flushes the destination pages after every receive. A record that arrives in several pieces is therefore flushed once per piece. A page spanning two pieces is flushed twice. Nothing reads the message body before the record is complete, so no reader needs the intermediate flushes. Flush every page of the record in one pass from svc_tcp_recvfrom(), once the last fragment has arrived. Remove svc_flush_bvec() and its ARCH_IMPLEMENTS_FLUSH_DCACHE_PAGE guard. The guard avoided setting up a bvec iterator, and a plain walk of rq_pages compiles away on its own where flush_dcache_page() is an empty inline. Link: https://patch.msgid.link/20260821-tls-read-sock-2-v1-3-7ffce164eb45@kernel.org Signed-off-by: Chuck Lever --- net/sunrpc/svcsock.c | 36 +++++++++++++++--------------------- 1 file changed, 15 insertions(+), 21 deletions(-) diff --git a/net/sunrpc/svcsock.c b/net/sunrpc/svcsock.c index ae1f3c474f8b96..c7b94bd1898b38 100644 --- a/net/sunrpc/svcsock.c +++ b/net/sunrpc/svcsock.c @@ -334,25 +334,6 @@ svc_tcp_sock_recvmsg(struct svc_sock *svsk, struct msghdr *msg) return ret; } -#if ARCH_IMPLEMENTS_FLUSH_DCACHE_PAGE -static void svc_flush_bvec(const struct bio_vec *bvec, size_t size, size_t seek) -{ - struct bvec_iter bi = { - .bi_size = size + seek, - }; - struct bio_vec bv; - - bvec_iter_advance(bvec, &bi, seek & PAGE_MASK); - for_each_bvec(bv, bvec, bi, bi) - flush_dcache_page(bv.bv_page); -} -#else -static inline void svc_flush_bvec(const struct bio_vec *bvec, size_t size, - size_t seek) -{ -} -#endif - /* * Read from @rqstp's transport socket. The incoming message fills whole * pages in @rqstp's rq_pages array until the last page of the message @@ -380,8 +361,6 @@ static ssize_t svc_tcp_read_msg(struct svc_rqst *rqstp, size_t buflen, buflen -= seek; } len = svc_tcp_sock_recvmsg(svsk, &msg); - if (len > 0) - svc_flush_bvec(bvec, len, seek); /* If we read a full record, then assume there may be more * data to read (stream based sockets only!) @@ -1156,6 +1135,19 @@ static void svc_tcp_fragment_received(struct svc_sock *svsk) svsk->sk_marker = xdr_zero; } +/* + * Nothing reads the message body before the record is complete, so + * a single flush after the last fragment is enough. + */ +static void svc_tcp_flush_pages(struct svc_sock *svsk, + struct svc_rqst *rqstp) +{ + unsigned int pg, pages = DIV_ROUND_UP(svsk->sk_datalen, PAGE_SIZE); + + for (pg = 0; pg < pages; pg++) + flush_dcache_page(rqstp->rq_pages[pg]); +} + /** * svc_tcp_recvfrom - Receive data from a TCP socket * @rqstp: request structure into which to receive an RPC Call @@ -1202,6 +1194,8 @@ static int svc_tcp_recvfrom(struct svc_rqst *rqstp) if (svsk->sk_datalen < 8) goto err_nuts; + svc_tcp_flush_pages(svsk, rqstp); + rqstp->rq_arg.len = svsk->sk_datalen; rqstp->rq_arg.page_base = 0; if (rqstp->rq_arg.len <= rqstp->rq_arg.head[0].iov_len) { From 7975eeaca69a4cc12652ce2bf7b7c1b834297e5c Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Fri, 21 Aug 2026 13:22:42 -0400 Subject: [PATCH 0293/1352] SUNRPC: Receive RPC records with ->read_sock svc_tcp_recvfrom() reads an RPC record with two recvmsg() calls, one for the four-octet fragment marker and one for the body. Neither supplies a control-message buffer, so kTLS reports a TLS control record on either one by raising MSG_CTRUNC and delivering nothing. Both receives have to recognize that outcome and hand off to a recovery path that re-reads the record. Commit bee47cb026e7 ("sunrpc: fix handling of server side tls alerts") built that recovery path. It stays on both receives for as long as one recvmsg() serves data and control alike. Instead, parse the record stream in a ->read_sock actor. The data path then carries no control-message buffer at all. A control record at the head stops ->read_sock short, so svc_tcp_recv_ctrl_record() classifies the head of the receive queue with MSG_PEEK before consuming anything. skb_copy_bits() walks an skb from its head to reach the copy offset, so an actor that copies a page at a time rewalks a large GRO or kTLS skb once per page. Copy each callback's share of the record into rq_bvec with one skb_copy_datagram_iter() instead. A non-final fragment carries four octets of marker and may carry no payload. sk_datalen advances only by the payload, so neither a run of empty fragments nor a run of one-octet fragments trips the sv_max_mesg check before desc->count runs out. Only desc->count bounds such a run, four or five octets at a time. Cap the fragments per socket-lock hold. recvmsg() skips an urgent octet and clears the condition, so the old receive path advanced past it. ->read_sock consumes what precedes the urgent octet and then stops there for good. The data path calls no recvmsg() now, so the record stream cannot advance again. An RPC stream carries no urgent data, so close the connection. The read that first reaches the mark returns a positive count, so a zero-length read does not mark the stop. Key the close on an incomplete record. Link: https://patch.msgid.link/20260821-tls-read-sock-2-v1-4-7ffce164eb45@kernel.org Signed-off-by: Chuck Lever --- net/sunrpc/svcsock.c | 390 ++++++++++++++++++++++++++----------------- 1 file changed, 238 insertions(+), 152 deletions(-) diff --git a/net/sunrpc/svcsock.c b/net/sunrpc/svcsock.c index c7b94bd1898b38..fe307d8314c488 100644 --- a/net/sunrpc/svcsock.c +++ b/net/sunrpc/svcsock.c @@ -8,15 +8,6 @@ * evenly when servicing a single client. May need to modify the * svc_xprt_enqueue procedure... * - * TCP support is largely untested and may be a little slow. The problem - * is that we currently do two separate recvfrom's, one for the 4-byte - * record length, and the second for the actual record. This could possibly - * be improved by always reading a minimum size of around 100 bytes and - * tucking any superfluous bytes away in a temporary store. Still, that - * leaves write requests out in the rain. An alternative may be to peek at - * the first skb in the queue, and if it matches the next TCP sequence - * number, to extract the record marker. Yuck. - * * Copyright (C) 1995, 1996 Olaf Kirch */ @@ -263,10 +254,10 @@ static int svc_tcp_recv_cmsg(struct socket *sock, int flags, return ret; } -static int -svc_tcp_sock_recv_cmsg(struct socket *sock) +static int svc_tcp_recv_ctrl_record(struct svc_sock *svsk) { u8 alert[2], type, level, description; + struct socket *sock = svsk->sk_sock; struct kvec recv_kvec = { .iov_base = alert, .iov_len = sizeof(alert), @@ -275,18 +266,41 @@ svc_tcp_sock_recv_cmsg(struct socket *sock) struct msghdr msg = {}; int ret; - ret = svc_tcp_recv_cmsg(sock, MSG_DONTWAIT, &recv_kvec, &type, - &msg_flags); - if (ret < 0 || !type) - return ret; - /* A data record reaches here only when kTLS queued an empty one - * ahead of the control record. Consuming it takes no payload, - * and the retry picks up the control record. + if (!test_bit(XPT_TLS_SESSION, &svsk->sk_xprt.xpt_flags)) + return 0; + + /* A data record can become ready between ->read_sock returning + * and this probe. A plain receive would take two octets of it + * as RPC payload, so peek. */ - if (type == TLS_RECORD_TYPE_DATA) - return -EAGAIN; + ret = svc_tcp_recv_cmsg(sock, MSG_DONTWAIT | MSG_PEEK, + &recv_kvec, &type, &msg_flags); + if (ret == -EAGAIN || (!ret && !type)) + return 0; + if (ret < 0) + return ret; + if (type == TLS_RECORD_TYPE_DATA) { + /* The peek parks the decrypted record on ctx->rx_list, + * where it draws no further data_ready. Re-arm or the + * RPC hangs until the client times out. + */ + set_bit(XPT_DATA, &svsk->sk_xprt.xpt_flags); + return 0; + } if (type != TLS_RECORD_TYPE_ALERT) return -EPROTO; + + ret = svc_tcp_recv_cmsg(sock, MSG_DONTWAIT, &recv_kvec, &type, + &msg_flags); + /* The peek found a record at the head, so an -EAGAIN here is + * spurious. Propagating it strands the record with no later + * announcement, so return -EBADMSG, which closes the transport. + */ + if (ret == -EAGAIN) + return -EBADMSG; + if (ret < 0) + return ret; + /* An Alert record carries exactly one two-octet message (RFC * 8446 Section 5.1). recv_kvec caps the receive at two, so a * longer record produces the same count. MSG_EOR appears only @@ -300,77 +314,19 @@ svc_tcp_sock_recv_cmsg(struct socket *sock) tls_alert_recv(sock->sk, &msg, &level, &description); /* RFC 8446 Section 6: every alert but a closure alert is - * an error alert. + * an error alert. kTLS raises no data_ready for records it + * already holds, so re-arm for what sits behind the alert. */ switch (description) { case TLS_ALERT_DESC_CLOSE_NOTIFY: case TLS_ALERT_DESC_USER_CANCELED: + set_bit(XPT_DATA, &svsk->sk_xprt.xpt_flags); return -EAGAIN; default: return -ENOTCONN; } } -static int -svc_tcp_sock_recvmsg(struct svc_sock *svsk, struct msghdr *msg) -{ - int ret; - struct socket *sock = svsk->sk_sock; - - ret = sock_recvmsg(sock, msg, MSG_DONTWAIT); - if (msg->msg_flags & MSG_CTRUNC) { - msg->msg_flags &= ~(MSG_CTRUNC | MSG_EOR); - if (ret == 0 || ret == -EIO) { - ret = svc_tcp_sock_recv_cmsg(sock); - /* A control record delivers nothing to the caller, - * and kTLS announces no data_ready for records it - * already holds. Mark the transport ready so that - * the records behind this one can be received. - */ - if (ret == -EAGAIN) - set_bit(XPT_DATA, &svsk->sk_xprt.xpt_flags); - } - } - return ret; -} - -/* - * Read from @rqstp's transport socket. The incoming message fills whole - * pages in @rqstp's rq_pages array until the last page of the message - * has been received into a partial page. - */ -static ssize_t svc_tcp_read_msg(struct svc_rqst *rqstp, size_t buflen, - size_t seek) -{ - struct svc_sock *svsk = - container_of(rqstp->rq_xprt, struct svc_sock, sk_xprt); - struct bio_vec *bvec = rqstp->rq_bvec; - struct msghdr msg = { NULL }; - unsigned int i; - ssize_t len; - size_t t; - - clear_bit(XPT_DATA, &svsk->sk_xprt.xpt_flags); - - for (i = 0, t = 0; t < buflen; i++, t += PAGE_SIZE) - bvec_set_page(&bvec[i], rqstp->rq_pages[i], PAGE_SIZE, 0); - - iov_iter_bvec(&msg.msg_iter, ITER_DEST, bvec, i, buflen); - if (seek) { - iov_iter_advance(&msg.msg_iter, seek); - buflen -= seek; - } - len = svc_tcp_sock_recvmsg(svsk, &msg); - - /* If we read a full record, then assume there may be more - * data to read (stream based sockets only!) - */ - if (len == buflen) - set_bit(XPT_DATA, &svsk->sk_xprt.xpt_flags); - - return len; -} - /* * Set socket snd and rcv buffer lengths */ @@ -993,14 +949,14 @@ static struct svc_xprt *svc_tcp_accept(struct svc_xprt *xprt) return NULL; } -static size_t svc_tcp_restore_pages(struct svc_sock *svsk, - struct svc_rqst *rqstp) +static void svc_tcp_restore_pages(struct svc_sock *svsk, + struct svc_rqst *rqstp) { size_t len = svsk->sk_datalen; unsigned int i, npages; if (!len) - return 0; + return; npages = (len + PAGE_SIZE - 1) >> PAGE_SHIFT; for (i = 0; i < npages; i++) { if (rqstp->rq_pages[i] != NULL) @@ -1010,7 +966,6 @@ static size_t svc_tcp_restore_pages(struct svc_sock *svsk, svsk->sk_pages[i] = NULL; } rqstp->rq_arg.head[0].iov_base = page_address(rqstp->rq_pages[0]); - return len; } static void svc_tcp_save_pages(struct svc_sock *svsk, struct svc_rqst *rqstp) @@ -1049,50 +1004,6 @@ static void svc_tcp_clear_pages(struct svc_sock *svsk) svsk->sk_datalen = 0; } -/* - * Receive fragment record header into sk_marker. - */ -static ssize_t svc_tcp_read_marker(struct svc_sock *svsk, - struct svc_rqst *rqstp) -{ - ssize_t want, len; - - /* If we haven't gotten the record length yet, - * get the next four bytes. - */ - if (svsk->sk_tcplen < sizeof(rpc_fraghdr)) { - struct msghdr msg = { NULL }; - struct kvec iov; - - want = sizeof(rpc_fraghdr) - svsk->sk_tcplen; - iov.iov_base = ((char *)&svsk->sk_marker) + svsk->sk_tcplen; - iov.iov_len = want; - iov_iter_kvec(&msg.msg_iter, ITER_DEST, &iov, 1, want); - len = svc_tcp_sock_recvmsg(svsk, &msg); - if (len < 0) - return len; - svsk->sk_tcplen += len; - if (len < want) { - /* call again to read the remaining bytes */ - goto err_short; - } - trace_svcsock_marker(&svsk->sk_xprt, svsk->sk_marker); - if (svc_sock_reclen(svsk) + svsk->sk_datalen > - svsk->sk_xprt.xpt_server->sv_max_mesg) - goto err_too_large; - } - return svc_sock_reclen(svsk); - -err_too_large: - net_notice_ratelimited("svc: %s oversized RPC fragment (%u octets) from %pISpc\n", - svsk->sk_xprt.xpt_server->sv_name, - svc_sock_reclen(svsk), - (struct sockaddr *)&svsk->sk_xprt.xpt_remote); - svc_xprt_deferred_close(&svsk->sk_xprt); -err_short: - return -EAGAIN; -} - static int receive_cb_reply(struct svc_sock *svsk, struct svc_rqst *rqstp) { struct rpc_xprt *bc_xprt = svsk->sk_xprt.xpt_bc_xprt; @@ -1130,14 +1041,31 @@ static int receive_cb_reply(struct svc_sock *svsk, struct svc_rqst *rqstp) static void svc_tcp_fragment_received(struct svc_sock *svsk) { - /* If we have more data, signal svc_xprt_enqueue() to try again */ svsk->sk_tcplen = 0; svsk->sk_marker = xdr_zero; } /* - * Nothing reads the message body before the record is complete, so - * a single flush after the last fragment is enough. + * A non-final fragment carries four octets of marker and may carry + * no payload at all. sk_datalen advances only by the payload, so a + * run of tiny fragments exhausts ->read_sock's byte budget before + * the sv_max_mesg check trips, and a run of empty ones never trips + * it. Cap the fragments per socket-lock hold. The cap leaves the + * record incomplete, and svc_tcp_recvfrom() resumes it on the next + * call. + */ +#define SVC_TCP_MAX_FRAGS 256 + +struct svc_tcp_recv_ctx { + struct svc_rqst *rqstp; + unsigned int frags; + bool complete; +}; + +/* + * Nothing reads the message body before the message is complete, and + * partial receives refill the same pages. Flush once here, after the + * socket lock is released, rather than once per copy in the actor. */ static void svc_tcp_flush_pages(struct svc_sock *svsk, struct svc_rqst *rqstp) @@ -1148,15 +1076,127 @@ static void svc_tcp_flush_pages(struct svc_sock *svsk, flush_dcache_page(rqstp->rq_pages[pg]); } +/* + * Mapping the message's unfilled remainder would re-map untouched + * pages on every call, at a cost that grows with the message rather + * than with the octets copied. + */ +static void svc_tcp_recv_iter_init(struct svc_rqst *rqstp, + struct iov_iter *iter, size_t body_off, + size_t len) +{ + unsigned int first = body_off >> PAGE_SHIFT; + size_t seek = offset_in_page(body_off); + unsigned int i, pages = DIV_ROUND_UP(seek + len, PAGE_SIZE); + + for (i = 0; i < pages; i++) + bvec_set_page(&rqstp->rq_bvec[i], rqstp->rq_pages[first + i], + PAGE_SIZE, 0); + + iov_iter_bvec(iter, ITER_DEST, rqstp->rq_bvec, pages, seek + len); + iov_iter_advance(iter, seek); +} + +/* + * ->read_sock actor, called under the socket lock. sk_datalen is both + * the count of body octets received so far and their write offset into + * rq_pages. + */ +static int svc_tcp_recv_actor(read_descriptor_t *desc, struct sk_buff *skb, + unsigned int offset, size_t len) +{ + struct svc_tcp_recv_ctx *ctx = desc->arg.data; + struct svc_rqst *rqstp = ctx->rqstp; + struct svc_sock *svsk = + container_of(rqstp->rq_xprt, struct svc_sock, sk_xprt); + size_t reclen, received, want, take, n; + size_t consumed = 0; + + if (!desc->count) + return 0; + + len = min(len, desc->count); + + if (svsk->sk_tcplen < sizeof(rpc_fraghdr)) { + want = sizeof(rpc_fraghdr) - svsk->sk_tcplen; + n = min(want, len); + + if (skb_copy_bits(skb, offset, + (char *)&svsk->sk_marker + svsk->sk_tcplen, + n)) + goto fault; + svsk->sk_tcplen += n; + offset += n; + len -= n; + consumed += n; + desc->count -= n; + + if (svsk->sk_tcplen < sizeof(rpc_fraghdr)) + return consumed; + + trace_svcsock_marker(&svsk->sk_xprt, svsk->sk_marker); + if (svc_sock_reclen(svsk) + svsk->sk_datalen > + svsk->sk_xprt.xpt_server->sv_max_mesg) { + net_notice_ratelimited("svc: %s oversized RPC fragment (%u octets) from %pISpc\n", + svsk->sk_xprt.xpt_server->sv_name, + svc_sock_reclen(svsk), + (struct sockaddr *)&svsk->sk_xprt.xpt_remote); + desc->error = -EMSGSIZE; + desc->count = 0; + return consumed; + } + } + + reclen = svc_sock_reclen(svsk); + received = svsk->sk_tcplen - sizeof(rpc_fraghdr); + want = reclen - received; + take = min(want, len); + + if (take) { + struct iov_iter iter; + + svc_tcp_recv_iter_init(rqstp, &iter, svsk->sk_datalen, take); + if (skb_copy_datagram_iter(skb, offset, &iter, take)) + goto fault; + svsk->sk_datalen += take; + svsk->sk_tcplen += take; + consumed += take; + desc->count -= take; + } + + if (take == want) { + if (svc_sock_final_rec(svsk)) { + ctx->complete = true; + desc->count = 0; + } else { + svc_tcp_fragment_received(svsk); + if (++ctx->frags >= SVC_TCP_MAX_FRAGS) + desc->count = 0; + } + } + + return consumed; + +fault: + desc->error = -EFAULT; + desc->count = 0; + return consumed; +} + +static bool svc_tcp_at_urg_mark(struct sock *sk) +{ + const struct tcp_sock *tp = tcp_sk(sk); + + return tp->urg_data && tp->urg_seq == tp->copied_seq; +} + /** * svc_tcp_recvfrom - Receive data from a TCP socket * @rqstp: request structure into which to receive an RPC Call * * Called in a loop when XPT_DATA has been set. * - * Read the 4-byte stream record marker, then use the record length - * in that marker to set up exactly the resources needed to receive - * the next RPC message into @rqstp. + * Context: Process context. Takes and releases the socket lock. * * Returns: * On success, the number of bytes in a received RPC Call, or @@ -1171,26 +1211,64 @@ static int svc_tcp_recvfrom(struct svc_rqst *rqstp) struct svc_sock *svsk = container_of(rqstp->rq_xprt, struct svc_sock, sk_xprt); struct svc_serv *serv = svsk->sk_xprt.xpt_server; - size_t want, base; + struct svc_tcp_recv_ctx ctx = { + .rqstp = rqstp, + }; + read_descriptor_t desc = { + .arg.data = &ctx, + .count = serv->sv_max_mesg + sizeof(rpc_fraghdr), + }; + struct socket *sock = svsk->sk_sock; ssize_t len; __be32 *p; __be32 calldir; clear_bit(XPT_DATA, &svsk->sk_xprt.xpt_flags); - len = svc_tcp_read_marker(svsk, rqstp); - if (len < 0) - goto error; + svc_tcp_restore_pages(svsk, rqstp); - base = svc_tcp_restore_pages(svsk, rqstp); - want = len - (svsk->sk_tcplen - sizeof(rpc_fraghdr)); - len = svc_tcp_read_msg(rqstp, base + want, base); - if (len >= 0) { - trace_svcsock_tcp_recv(&svsk->sk_xprt, len); - svsk->sk_tcplen += len; - svsk->sk_datalen += len; + lock_sock(sock->sk); + len = sock->ops->read_sock(sock->sk, &desc, svc_tcp_recv_actor); + /* ->read_sock stops at urgent data and consumes none of it. + * Only recvmsg() clears the condition, and this path calls + * none, so every later read stops at the same octet. An RPC + * stream carries no urgent data, so close the connection. + * + * The read that first reaches the mark consumes the octets + * ahead of it, so the stop does not show up as a zero len. A + * record completed ahead of the mark is returned first. The + * XPT_DATA set below brings the next call back here with + * nothing left to consume. + */ + if (!ctx.complete && svc_tcp_at_urg_mark(sock->sk)) + desc.error = -EPROTO; + release_sock(sock->sk); + + /* ->read_sock returns the octets consumed before an actor + * failure, so a positive len can accompany desc.error. + */ + if (desc.error < 0) { + len = desc.error; + goto err_nuts; } - if (len != want || !svc_sock_final_rec(svsk)) + if (len >= 0) + trace_svcsock_tcp_recv(&svsk->sk_xprt, len); + + if (!ctx.complete) { + if (!desc.count) { + set_bit(XPT_DATA, &svsk->sk_xprt.xpt_flags); + goto err_incomplete; + } + /* A zero return leaves no record at the head to classify. + * -EINVAL means a control record sits there. Screen the + * other errors out first, because a probe calls + * sock_error(), whose xchg clears sk->sk_err as it reads. + */ + if (len <= 0 && len != -EINVAL) + goto err_incomplete; + + len = svc_tcp_recv_ctrl_record(svsk); goto err_incomplete; + } if (svsk->sk_datalen < 8) goto err_nuts; @@ -1211,6 +1289,12 @@ static int svc_tcp_recvfrom(struct svc_rqst *rqstp) else clear_bit(RQ_LOCAL, &rqstp->rq_flags); + /* Completing one message stops ->read_sock with whatever + * follows still queued, and no path from here re-arms XPT_DATA. + * The queued message would wait for unrelated traffic. + */ + set_bit(XPT_DATA, &svsk->sk_xprt.xpt_flags); + p = (__be32 *)rqstp->rq_arg.head[0].iov_base; calldir = p[1]; if (calldir) @@ -1235,19 +1319,21 @@ static int svc_tcp_recvfrom(struct svc_rqst *rqstp) svc_tcp_save_pages(svsk, rqstp); if (len < 0 && len != -EAGAIN) goto err_delete; - if (len == want) - svc_tcp_fragment_received(svsk); - else + if (svsk->sk_tcplen >= sizeof(rpc_fraghdr)) trace_svcsock_tcp_recv_short(&svsk->sk_xprt, svc_sock_reclen(svsk), svsk->sk_tcplen - sizeof(rpc_fraghdr)); + else + trace_svcsock_tcp_recv_eagain(&svsk->sk_xprt, 0); goto err_noclose; error: - if (len != -EAGAIN) - goto err_delete; trace_svcsock_tcp_recv_eagain(&svsk->sk_xprt, 0); goto err_noclose; err_nuts: + /* svc_tcp_save_pages() has not run, so svsk->sk_pages[] is + * empty. A non-zero sk_datalen makes the teardown-time + * svc_tcp_clear_pages() walk empty slots and WARN. + */ svsk->sk_datalen = 0; err_delete: trace_svcsock_tcp_recv_err(&svsk->sk_xprt, len); From b3f5c6ebc6b9d0098721a42beda88058f6d46354 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Fri, 21 Aug 2026 13:22:43 -0400 Subject: [PATCH 0294/1352] SUNRPC: Bypass sock_recvmsg() for the TLS control-record receive svc_tcp_recvfrom() parses the RPC record stream with ->read_sock, which calls neither security_socket_recvmsg() nor the sock:sock_recv_length tracepoint. svc_tcp_recv_cmsg() still goes through sock_recvmsg(), so an LSM mediates only the TLS control records on a server socket, and sock:sock_recv_length reports only those. Partial coverage is worse than none. It makes the RPC stream look mediated and observed when it is not. Until the record stream moved to ->read_sock, an LSM saw every octet NFSD read from a TCP socket. An SELinux policy that denies SOCKET__READ to NFSD blocked the receive. After this change no call on the server's TCP receive path consults an LSM, so that denial has no effect. Dispatch ->recvmsg directly so the whole receive path behaves one way. sock_recvmsg_nosec() reaches ->recvmsg through INDIRECT_CALL_INET(), so on a retpoline build the direct dispatch costs one indirect call per control record. Control records are rare on an established connection. Link: https://patch.msgid.link/20260821-tls-read-sock-2-v1-5-7ffce164eb45@kernel.org Signed-off-by: Chuck Lever --- net/sunrpc/svcsock.c | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/net/sunrpc/svcsock.c b/net/sunrpc/svcsock.c index fe307d8314c488..ef7ac080fcd3b2 100644 --- a/net/sunrpc/svcsock.c +++ b/net/sunrpc/svcsock.c @@ -229,10 +229,16 @@ static int svc_one_sock_name(struct svc_sock *svsk, char *buf, int remaining) return len; } +/* + * The ->read_sock data path invokes neither security_socket_recvmsg() + * nor the sock:sock_recv_length tracepoint. Dispatch ->recvmsg + * directly so the whole receive path behaves one way. + */ static int svc_tcp_recv_cmsg(struct socket *sock, int flags, struct kvec *payload, u8 *type, unsigned int *msg_flags) { + const struct proto_ops *ops = READ_ONCE(sock->ops); union { struct cmsghdr cmsg; u8 buf[CMSG_SPACE(sizeof(u8))]; @@ -244,7 +250,7 @@ static int svc_tcp_recv_cmsg(struct socket *sock, int flags, int ret; iov_iter_kvec(&msg.msg_iter, ITER_DEST, payload, 1, payload->iov_len); - ret = sock_recvmsg(sock, &msg, flags); + ret = ops->recvmsg(sock, &msg, msg_data_left(&msg), flags); if (ret < 0) return ret; *msg_flags = msg.msg_flags; From 068606f59baeba063cbab3e843bda743852405d5 Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Sat, 22 Aug 2026 10:10:33 +0200 Subject: [PATCH 0295/1352] NFSD: docs: Fix pNFS SCSI Kconfig symbol The pNFS SCSI layout server is controlled by NFSD_SCSILAYOUT. NFSD_SCSI has never existed. Fixes: f99d4fbdae67 ("nfsd: add SCSI layout support") Assisted-by: Codex:gpt-5.6-sol Signed-off-by: Karl Mehltretter Acked-by: Randy Dunlap Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260822081033.82098-1-kmehltretter@gmail.com Signed-off-by: Chuck Lever --- Documentation/admin-guide/nfs/pnfs-scsi-server.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Documentation/admin-guide/nfs/pnfs-scsi-server.rst b/Documentation/admin-guide/nfs/pnfs-scsi-server.rst index b202508d281d54..a3074450ac158f 100644 --- a/Documentation/admin-guide/nfs/pnfs-scsi-server.rst +++ b/Documentation/admin-guide/nfs/pnfs-scsi-server.rst @@ -16,7 +16,7 @@ addition to the MDS. As of now the file system needs to sit directly on the exported LUN, striping or concatenation of LUNs on the MDS and clients is not supported yet. -On a server built with CONFIG_NFSD_SCSI, the pNFS SCSI volume support is +On a server built with CONFIG_NFSD_SCSILAYOUT, the pNFS SCSI volume support is automatically enabled if the file system is exported using the "pnfs" option and the underlying SCSI device support persistent reservations. On the client make sure the kernel has the CONFIG_PNFS_BLOCK option From ff9def4aaf57adddcbab779e6c88629baeae2cfc Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sat, 22 Aug 2026 21:40:15 -0400 Subject: [PATCH 0296/1352] lockd: Fix use-after-free in nlmsvc_retry_blocked nlmsvc_retry_blocked() examines the block at the head of nlm_blocked under nlm_blocked_lock, then releases the lock before calling nlmsvc_grant_blocked() or retry_deferred_block(). The nlm_blocked list reference is all that keeps the block alive across that window. nlmsvc_grant_blocked() does take one of its own, but not until after the lock has been dropped. Unmounting the nfsd filesystem while a lock request is still blocked reaches nlmsvc_traverse_blocks(), which drops the list reference and frees the block along with the nlm_rqst hanging off it. BUG: KASAN: slab-use-after-free in nlm_async_call+0xd6/0x230 Read of size 8 at addr ffff88811b04c808 by task lockd/8377 nlm_async_call+0xd6/0x230 nlmsvc_retry_blocked+0x61c/0x800 lockd+0x144/0x1c0 Freed by task 8392: nlmsvc_release_block+0x231/0x290 nlmsvc_traverse_blocks+0x139/0x1b0 nlm_traverse_files+0x1aa/0xa00 nlmsvc_free_host_resources+0x12/0x60 nlm_shutdown_hosts_net+0x127/0x280 lockd_down+0xd5/0x1c0 Take a reference before releasing nlm_blocked_lock and drop it once the retry has run. Fixes: 0e4ac9d93515 ("lockd: handle fl_grant callbacks") Cc: stable@vger.kernel.org Reported-by: Shuangpeng Bai Closes: https://lore.kernel.org/linux-nfs/20260818235808.3458075-1-shuangpeng.kernel@gmail.com/ Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260822-lockd-retry-blocked-uaf-v3-1-761661eae60c@kernel.org Signed-off-by: Chuck Lever --- fs/lockd/svclock.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/fs/lockd/svclock.c b/fs/lockd/svclock.c index e628b5d355071f..8d83283d3e21c2 100644 --- a/fs/lockd/svclock.c +++ b/fs/lockd/svclock.c @@ -1023,6 +1023,7 @@ nlmsvc_retry_blocked(struct svc_rqst *rqstp) timeout = block->b_when - jiffies; break; } + kref_get(&block->b_count); spin_unlock(&nlm_blocked_lock); dprintk("nlmsvc_retry_blocked(%p, when=%ld)\n", @@ -1033,6 +1034,7 @@ nlmsvc_retry_blocked(struct svc_rqst *rqstp) retry_deferred_block(block); } else nlmsvc_grant_blocked(block); + nlmsvc_release_block(block); spin_lock(&nlm_blocked_lock); } spin_unlock(&nlm_blocked_lock); From c5d7d9a95187b9da4eb627a3fe4d45b1ec056c4d Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Sat, 22 Aug 2026 21:40:16 -0400 Subject: [PATCH 0297/1352] lockd: Serialize block retries against host teardown nlmsvc_grant_blocked() unlinks a block from nlm_blocked before it retries the lock, then re-inserts it. nlmsvc_traverse_blocks() skips a block that is not on nlm_blocked, so a teardown scan that runs during a retry passes it by and the retry puts it back. The surviving block pins its host. lockd warns that it could not shut down the host module, and the host outlives its network namespace. Hold the file's f_mutex across the retry, and extend the scan's hold across its unlink, so a scan and a retry of the same file can no longer interleave. Drop the mutex before releasing a block reference, since the last put takes f_mutex. A retry that waited out a scan re-checks under nlm_blocked_lock that its block is still queued and due. Reported-by: sashiko-bot Closes: https://sashiko.dev/#/patchset/20260819162247.2970703-1-cel@kernel.org?part=1 Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260822-lockd-retry-blocked-uaf-v3-2-761661eae60c@kernel.org Signed-off-by: Chuck Lever --- fs/lockd/svclock.c | 31 ++++++++++++++++++++++++++++--- 1 file changed, 28 insertions(+), 3 deletions(-) diff --git a/fs/lockd/svclock.c b/fs/lockd/svclock.c index 8d83283d3e21c2..495eacb3264f90 100644 --- a/fs/lockd/svclock.c +++ b/fs/lockd/svclock.c @@ -295,14 +295,17 @@ void nlmsvc_traverse_blocks(struct nlm_host *host, list_for_each_entry_safe(block, next, &file->f_blocks, b_flist) { if (!match(block->b_host, host)) continue; - /* Do not destroy blocks that are not on - * the global retry list - why? */ + /* + * nlmsvc_retry_blocked() holds f_mutex while the block + * is off nlm_blocked, so a block off the list here has + * been retired. + */ if (list_empty(&block->b_list)) continue; kref_get(&block->b_count); spin_unlock(&nlm_blocked_lock); - mutex_unlock(&file->f_mutex); nlmsvc_unlink_block(block); + mutex_unlock(&file->f_mutex); nlmsvc_release_block(block); goto restart; } @@ -1012,6 +1015,8 @@ nlmsvc_retry_blocked(struct svc_rqst *rqstp) { unsigned long timeout = MAX_SCHEDULE_TIMEOUT; struct nlm_block *block; + struct nlm_file *file; + bool due; spin_lock(&nlm_blocked_lock); while (!list_empty(&nlm_blocked) && !svc_thread_should_stop(rqstp)) { @@ -1026,6 +1031,25 @@ nlmsvc_retry_blocked(struct svc_rqst *rqstp) kref_get(&block->b_count); spin_unlock(&nlm_blocked_lock); + /* + * Hold f_mutex so nlmsvc_traverse_blocks() cannot scan + * the file while the retry has the block off nlm_blocked. + */ + file = block->b_file; + mutex_lock(&file->f_mutex); + spin_lock(&nlm_blocked_lock); + due = !list_empty(&block->b_list) && + block->b_when != NLM_NEVER && + !time_after(block->b_when, jiffies); + spin_unlock(&nlm_blocked_lock); + + if (!due) { + mutex_unlock(&file->f_mutex); + nlmsvc_release_block(block); + spin_lock(&nlm_blocked_lock); + continue; + } + dprintk("nlmsvc_retry_blocked(%p, when=%ld)\n", block, block->b_when); if (block->b_flags & B_QUEUED) { @@ -1034,6 +1058,7 @@ nlmsvc_retry_blocked(struct svc_rqst *rqstp) retry_deferred_block(block); } else nlmsvc_grant_blocked(block); + mutex_unlock(&file->f_mutex); nlmsvc_release_block(block); spin_lock(&nlm_blocked_lock); } From bae7f4cd79d87434656fedf7465c2e75186e40f1 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Mon, 24 Aug 2026 10:04:40 -0400 Subject: [PATCH 0298/1352] NFSD: Fix out-of-bounds read in the rpc_status dump nfsd_nl_rpc_status_get_dumpit() loads args->opcnt and args->ops separately, then walks ops[] before rechecking rq_status_counter. A COMPOUND that completes between the two loads runs nfsd4_release_compoundargs(), which zeroes opcnt and points ops back at the eight-entry inline array. A dump that already sampled an opcnt of 200 clamps it to the sixteen slots in rq_opnum, then indexes iops[0..15]. iops is the last member of struct nfsd4_compoundargs and rq_argp is allocated at exactly that size, so the walk runs off the end of the allocation. The trailing recheck discards the sampled data, but the read has already happened. NFSD_CMD_RPC_STATUS_GET carries no GENL_ADMIN_PERM, so an unprivileged local user can repeat the dump against a busy server until it lands in the window. Sample opcnt and ops into locals, then finish the counter recheck before dereferencing ops. An unchanged counter means both came from the same COMPOUND, where opcnt cannot exceed what ops holds. Fixes: bd9d6a3efa97 ("NFSD: add rpc_status netlink support") Cc: stable@vger.kernel.org Reported-by: Prabhakar Pujeri Closes: https://lore.kernel.org/linux-nfs/20260823113255.3417-1-prabhakar.pujeri@dell.com/ Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260824-nfsd-posix-acl-ownership-v1-1-090fffc608ec@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfsctl.c | 21 ++++++++++++++++++--- 1 file changed, 18 insertions(+), 3 deletions(-) diff --git a/fs/nfsd/nfsctl.c b/fs/nfsd/nfsctl.c index eeae33a19a963d..3a6945b3bd8b27 100644 --- a/fs/nfsd/nfsctl.c +++ b/fs/nfsd/nfsctl.c @@ -1588,14 +1588,29 @@ int nfsd_nl_rpc_status_get_dumpit(struct sk_buff *skb, rqstp->rq_proc == NFSPROC4_COMPOUND) { /* NFSv4 compound */ struct nfsd4_compoundargs *args; + struct nfsd4_op *ops; + u32 opcnt; int j; args = rqstp->rq_argp; - genl_rqstp.rq_opcnt = min_t(u32, args->opcnt, + opcnt = READ_ONCE(args->opcnt); + ops = READ_ONCE(args->ops); + + /* + * Finish the seqcount retry before + * dereferencing ops. An unchanged counter means + * opcnt and ops came from the same COMPOUND, + * where opcnt cannot exceed what ops holds. + */ + smp_rmb(); + if (READ_ONCE(rqstp->rq_status_counter) != + status_counter) + continue; + + genl_rqstp.rq_opcnt = min_t(u32, opcnt, ARRAY_SIZE(genl_rqstp.rq_opnum)); for (j = 0; j < genl_rqstp.rq_opcnt; j++) - genl_rqstp.rq_opnum[j] = - args->ops[j].opnum; + genl_rqstp.rq_opnum[j] = ops[j].opnum; } #endif /* CONFIG_NFSD_V4 */ From db4c1baf4f016d56aadb2fb5fbd77c7aed77eee7 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Mon, 24 Aug 2026 10:04:41 -0400 Subject: [PATCH 0299/1352] NFSD: Fix POSIX ACL leak in unexecuted NFSv4 COMPOUND operations nfsd4_decode_fattr4() allocates POSIX ACLs while decoding OP_OPEN, OP_CREATE, and OP_SETATTR, leaving the only reference to these ACLs in the operation's argument structure. Executing the operation hands that reference to struct nfsd_attrs, which drops it. However, if the operation is decoded but never executes, those ACLs are leaked. A client can repeat an aborting compound to force the server to leak memory. Give struct nfsd_attrs its own reference with posix_acl_dup() so nfsd_attrs_free() still balances the reference the operation took. Release the ACLs when the compound completes. Have the decoder record each ACL on the compound's temporary allocation chain, and give each chained item an optional release callback. The chain holds a reference for the life of the compound, so OP_OPEN no longer needs an op_release method. Fixes: 5fc51dfc2eb1 ("NFSD: Add support for XDR decoding POSIX draft ACLs") Cc: stable+noautosel@kernel.org # experimental, disabled by default Reported-by: Prabhakar Pujeri Closes: https://lore.kernel.org/linux-nfs/20260823113255.3417-1-prabhakar.pujeri@dell.com/ Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260824-nfsd-posix-acl-ownership-v1-2-090fffc608ec@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfs4proc.c | 30 ++++++------------------------ fs/nfsd/nfs4xdr.c | 41 +++++++++++++++++++++++++++-------------- fs/nfsd/xdr4.h | 1 + 3 files changed, 34 insertions(+), 38 deletions(-) diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c index 88385a161b4d04..bb74eef439388b 100644 --- a/fs/nfsd/nfs4proc.c +++ b/fs/nfsd/nfs4proc.c @@ -391,11 +391,8 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp, if (status) return status; } else { - /* The dpacl and pacl will get released by nfsd_attrs_free(). */ - attrs.na_dpacl = open->op_dpacl; - attrs.na_pacl = open->op_pacl; - open->op_dpacl = NULL; - open->op_pacl = NULL; + attrs.na_dpacl = posix_acl_dup(open->op_dpacl); + attrs.na_pacl = posix_acl_dup(open->op_pacl); } v_mtime = 0; @@ -795,13 +792,6 @@ static __be32 nfsd4_open_omfg(struct svc_rqst *rqstp, struct nfsd4_compound_stat return nfsd4_open(rqstp, cstate, &op->u); } -static void -nfsd4_open_release(union nfsd4_op_u *u) -{ - posix_acl_release(u->open.op_dpacl); - posix_acl_release(u->open.op_pacl); -} - /* * filehandle-manipulating ops. */ @@ -922,16 +912,13 @@ nfsd4_create(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate, struct nfsd_attrs attrs = { .na_iattr = &create->cr_iattr, .na_seclabel = &create->cr_label, - .na_dpacl = create->cr_dpacl, - .na_pacl = create->cr_pacl, + .na_dpacl = posix_acl_dup(create->cr_dpacl), + .na_pacl = posix_acl_dup(create->cr_pacl), }; struct svc_fh resfh; __be32 status; dev_t rdev; - create->cr_dpacl = NULL; - create->cr_pacl = NULL; - fh_init(&resfh, NFS4_FHSIZE); status = fh_verify(rqstp, &cstate->current_fh, S_IFDIR, NFSD_MAY_NOP); @@ -1341,8 +1328,8 @@ nfsd4_setattr(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate, struct nfsd_attrs attrs = { .na_iattr = &setattr->sa_iattr, .na_seclabel = &setattr->sa_label, - .na_pacl = setattr->sa_pacl, - .na_dpacl = setattr->sa_dpacl, + .na_pacl = posix_acl_dup(setattr->sa_pacl), + .na_dpacl = posix_acl_dup(setattr->sa_dpacl), }; bool save_no_wcc, deleg_attrs; struct nfs4_stid *st = NULL; @@ -1350,10 +1337,6 @@ nfsd4_setattr(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate, __be32 status = nfs_ok; int err; - /* Transfer ownership to attrs for cleanup via nfsd_attrs_free() */ - setattr->sa_pacl = NULL; - setattr->sa_dpacl = NULL; - deleg_attrs = setattr->sa_bmval[2] & (FATTR4_WORD2_TIME_DELEG_ACCESS | FATTR4_WORD2_TIME_DELEG_MODIFY); @@ -3943,7 +3926,6 @@ static const struct nfsd4_operation nfsd4_ops[] = { }, [OP_OPEN] = { .op_func = nfsd4_open, - .op_release = nfsd4_open_release, .op_flags = OP_HANDLES_WRONGSEC | OP_MODIFIES_SOMETHING, .op_name = "OP_OPEN", .op_rsize_bop = nfsd4_open_rsize, diff --git a/fs/nfsd/nfs4xdr.c b/fs/nfsd/nfs4xdr.c index a154b02d82b3c4..5bfbeb87394e6f 100644 --- a/fs/nfsd/nfs4xdr.c +++ b/fs/nfsd/nfs4xdr.c @@ -114,27 +114,27 @@ static int zero_clientid(clientid_t *clid) return (clid->cl_boot == 0) && (clid->cl_id == 0); } -/** - * svcxdr_tmpalloc - allocate memory to be freed after compound processing - * @argp: NFSv4 compound argument structure - * @len: length of buffer to allocate - * - * Allocates a buffer of size @len to be freed when processing the compound - * operation described in @argp finishes. - */ static void * -svcxdr_tmpalloc(struct nfsd4_compoundargs *argp, size_t len) +svcxdr_tmpalloc_release(struct nfsd4_compoundargs *argp, size_t len, + void (*release)(void *)) { struct svcxdr_tmpbuf *tb; tb = kmalloc_flex(*tb, buf, len); if (!tb) return NULL; + tb->release = release; tb->next = argp->to_free; argp->to_free = tb; return tb->buf; } +static void * +svcxdr_tmpalloc(struct nfsd4_compoundargs *argp, size_t len) +{ + return svcxdr_tmpalloc_release(argp, len, NULL); +} + /* * For xdr strings that need to be passed to other kernel api's * as null-terminated strings. @@ -442,10 +442,16 @@ nfsd4_decode_posixace4(struct nfsd4_compoundargs *argp, return status; } +static void svcxdr_release_pacl(void *p) +{ + posix_acl_release(*(struct posix_acl **)p); +} + static noinline __be32 nfsd4_decode_posixacl(struct nfsd4_compoundargs *argp, struct posix_acl **acl) { struct posix_acl_entry *ace; + struct posix_acl **slot; __be32 status; u32 count; @@ -485,6 +491,15 @@ nfsd4_decode_posixacl(struct nfsd4_compoundargs *argp, struct posix_acl **acl) if (count >= 3) sort_pacl_range(*acl, 0, count - 1); + slot = svcxdr_tmpalloc_release(argp, sizeof(*slot), + svcxdr_release_pacl); + if (!slot) { + posix_acl_release(*acl); + *acl = NULL; + return nfserr_jukebox; + } + *slot = *acl; + return nfs_ok; } @@ -677,7 +692,6 @@ nfsd4_decode_fattr4(struct nfsd4_compoundargs *argp, u32 *bmval, u32 bmlen, status = nfsd4_decode_posixacl(argp, &pacl); if (status) { - posix_acl_release(*dpaclp); *dpaclp = NULL; return status; } @@ -687,12 +701,8 @@ nfsd4_decode_fattr4(struct nfsd4_compoundargs *argp, u32 *bmval, u32 bmlen, /* request sanity: did attrlist4 contain the expected number of words? */ if (attrlist4_count != xdr_stream_pos(argp->xdr) - starting_pos) { -#ifdef CONFIG_NFSD_V4_POSIX_ACLS - posix_acl_release(*dpaclp); - posix_acl_release(*paclp); *dpaclp = NULL; *paclp = NULL; -#endif return nfserr_bad_xdr; } @@ -6846,7 +6856,10 @@ void nfsd4_release_compoundargs(struct svc_rqst *rqstp) } while (args->to_free) { struct svcxdr_tmpbuf *tb = args->to_free; + args->to_free = tb->next; + if (tb->release) + tb->release(tb->buf); kfree(tb); } } diff --git a/fs/nfsd/xdr4.h b/fs/nfsd/xdr4.h index b841bc462dac8d..de43a0da9668a0 100644 --- a/fs/nfsd/xdr4.h +++ b/fs/nfsd/xdr4.h @@ -800,6 +800,7 @@ bool nfsd4_cache_this_op(struct nfsd4_op *); */ struct svcxdr_tmpbuf { struct svcxdr_tmpbuf *next; + void (*release)(void *buf); char buf[]; }; From f04d17e710d156f229572a279eef48dcf842365f Mon Sep 17 00:00:00 2001 From: Ameer Hamza Date: Mon, 24 Aug 2026 22:36:41 +0500 Subject: [PATCH 0300/1352] nfsd: don't modify a session slot when replaying its cached reply nfsd4_sequence() claims a session slot by setting NFSD4_SLOT_INUSE under nn->client_lock. A reply served from the slot's reply cache does not claim it, and that distinction lived only in cstate->status, which nfsd4_sequence() set to nfserr_replay_cache. Commit cc028a10a48c ("NFSD: Hoist status code encoding into XDR encoder functions") moved nfsd4_proc_compound()'s cstate->status assignment below the out: label, and the replay path's goto out was the one path that relied on skipping it. The test in nfsd4_sequence_done() therefore no longer identifies a replay, and every replay now stores its reply and clears NFSD4_SLOT_INUSE as though it owned the slot. NFSD4_SLOT_INUSE is the interlock: check_slot_seqid() rejects every sequence id for a slot that is in use, before replay_matches_cache() runs. A replay never sets it, so two replays can be in flight on the same slot at once. One can read slot->sl_cred under nn->client_lock while the other's completion frees and rebuilds it from the XDR encoder without that lock. Two completions can also collide with each other and release the same group_info twice. free_svc_cred() leaves cr_uid, cr_gid and cr_flavor intact, so the reader passes every earlier test in same_creds() and dereferences a NULL cr_group_info. From a 6.12.91 production server: BUG: kernel NULL pointer dereference, address: 0000000000000004 CPU: 63 UID: 0 PID: 39015 Comm: nfsd RIP: 0010:same_creds+0x38/0xa0 [nfsd] RDX: 0000000000000000 Call Trace: nfsd4_sequence+0x6a8/0x910 [nfsd] nfsd4_proc_compound+0x345/0x670 [nfsd] nfsd_dispatch+0x100/0x220 [nfsd] svc_process_common+0x311/0x700 [sunrpc] svc_process+0x131/0x1c0 [sunrpc] svc_recv+0x7ef/0x9c0 [sunrpc] nfsd+0xa3/0x100 [nfsd] Kernel panic - not syncing: Fatal exception Since v6.14 a live session's slot table can shrink, and the unowned store becomes a use-after-free write. nfsd4_sequence() defers the shrink while a slot is in use, but it tests NFSD4_SLOT_INUSE, which a replay does not set, so free_session_slots() can kfree() the slot the replay still holds in cstate->slot. Record the claim in cstate->slot_owned when nfsd4_sequence() accepts a request and test that in nfsd4_sequence_done(), so only the request that claimed the slot updates its cached reply and clears NFSD4_SLOT_INUSE. svc_generic_init_request() zeroes the compound response before each request, so the flag starts clear. Fixes: cc028a10a48c ("NFSD: Hoist status code encoding into XDR encoder functions") Cc: stable@vger.kernel.org Assisted-by: Claude:claude-opus-5 Signed-off-by: Ameer Hamza Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260824173641.3274260-1-ameer.hamza@truenas.com Signed-off-by: Chuck Lever --- fs/nfsd/nfs4state.c | 7 ++++++- fs/nfsd/xdr4.h | 1 + 2 files changed, 7 insertions(+), 1 deletion(-) diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c index 88f6c0a0d8b3fe..db79eba61bbf93 100644 --- a/fs/nfsd/nfs4state.c +++ b/fs/nfsd/nfs4state.c @@ -5170,6 +5170,7 @@ nfsd4_sequence(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate, slot->sl_flags &= ~NFSD4_SLOT_CACHETHIS; cstate->slot = slot; + cstate->slot_owned = true; cstate->session = session; cstate->clp = clp; @@ -5244,7 +5245,11 @@ nfsd4_sequence_done(struct nfsd4_compoundres *resp) struct nfsd4_compound_state *cs = &resp->cstate; if (nfsd4_has_session(cs)) { - if (cs->status != nfserr_replay_cache) { + /* + * Only the request that claimed the slot may update its + * cached reply and clear NFSD4_SLOT_INUSE. + */ + if (cs->slot_owned) { nfsd4_store_cache_entry(resp); cs->slot->sl_flags &= ~NFSD4_SLOT_INUSE; } diff --git a/fs/nfsd/xdr4.h b/fs/nfsd/xdr4.h index de43a0da9668a0..ec198993429e8b 100644 --- a/fs/nfsd/xdr4.h +++ b/fs/nfsd/xdr4.h @@ -62,6 +62,7 @@ struct nfsd4_compound_state { struct nfsd4_slot *slot; int data_offset; bool spo_must_allowed; + bool slot_owned; size_t iovlen; u32 minorversion; __be32 status; From a3c2f6a39316948e3bd888bb5dd8279adef8f264 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Wed, 26 Aug 2026 15:44:43 -0400 Subject: [PATCH 0301/1352] NFS: Import NFS3ERR definitions enum nfs_stat in uapi/linux/nfs.h collects the on-the-wire status codes for every NFS version under version-agnostic NFSERR_* names. The in-kernel NFS client and server reference these names internally, which drags uapi/linux/nfs.h into many translation units that need nothing else from that header. xdrgen conversion will provide spec-based definitions of the NFS status enum constants. Replacing the existing internal references with version-specific names requires a version-specific spelling for each code that NFSD uses. linux/nfs4.h already supplies the NFSv4 codes, but NFSv3 has no counterpart header, and not every NFSv3 code has an NFS4ERR_* spelling: NFSERR_NOT_SYNC (10002) and NFSERR_REMOTE (71) have no NFSv4 equivalent at all, and NFSERR_JUKEBOX (10008) appears in NFSv4 as NFS4ERR_DELAY. Introduce the NFSv3 status codes as NFS3ERR_* in linux/nfs3.h so a subsequent patch can respell NFSD's status definitions without relying on uapi/linux/nfs.h. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260826194444.148243-1-cel@kernel.org Signed-off-by: Chuck Lever --- include/linux/nfs3.h | 35 +++++++++++++++++++++++++++++++++++ 1 file changed, 35 insertions(+) diff --git a/include/linux/nfs3.h b/include/linux/nfs3.h index 1d18da0860d52b..b6539a75edea07 100644 --- a/include/linux/nfs3.h +++ b/include/linux/nfs3.h @@ -7,6 +7,41 @@ #include +/* + * NFSv3 error status values. + * See RFC 1813 Section 2.5 + */ +enum { + NFS3ERR_PERM = 1, + NFS3ERR_NOENT = 2, + NFS3ERR_IO = 5, + NFS3ERR_NXIO = 6, + NFS3ERR_ACCES = 13, + NFS3ERR_EXIST = 17, + NFS3ERR_XDEV = 18, + NFS3ERR_NODEV = 19, + NFS3ERR_NOTDIR = 20, + NFS3ERR_ISDIR = 21, + NFS3ERR_INVAL = 22, + NFS3ERR_FBIG = 27, + NFS3ERR_NOSPC = 28, + NFS3ERR_ROFS = 30, + NFS3ERR_MLINK = 31, + NFS3ERR_NAMETOOLONG = 63, + NFS3ERR_NOTEMPTY = 66, + NFS3ERR_DQUOT = 69, + NFS3ERR_STALE = 70, + NFS3ERR_REMOTE = 71, + NFS3ERR_BADHANDLE = 10001, + NFS3ERR_NOT_SYNC = 10002, + NFS3ERR_BAD_COOKIE = 10003, + NFS3ERR_NOTSUPP = 10004, + NFS3ERR_TOOSMALL = 10005, + NFS3ERR_SERVERFAULT = 10006, + NFS3ERR_BADTYPE = 10007, + NFS3ERR_JUKEBOX = 10008, +}; + enum nfs3_stable_how { NFS_UNSTABLE = 0, NFS_DATA_SYNC = 1, From 6c936cf0e138ca282837d24b4daebe518bc122f3 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Wed, 26 Aug 2026 15:44:44 -0400 Subject: [PATCH 0302/1352] NFSD: Rework be32 nfserr definitions fs/nfsd/nfserr.h builds NFSD's internal __be32 nfserr_* values by wrapping the version-agnostic NFSERR_* status codes from uapi/linux/nfs.h in cpu_to_be32(), which it reaches through its include of linux/nfs.h. A subsequent patch replaces that include with the xdrgen-generated linux/sunrpc/xdrgen/nfs2.h. The generated header re-supplies only the eighteen NFSv2 status codes; the NFSv3 and NFSv4 codes that nfserr.h also uses (NFSERR_INVAL, NFSERR_JUKEBOX, NFSERR_RESOURCE, and the rest) would be left undeclared. Respell those definitions in terms of the version-specific NFS3ERR_* and NFS4ERR_* codes from linux/nfs3.h and linux/nfs4.h, and switch the one bare NFSERR_MOVED in nfs4xdr.c to NFS4ERR_MOVED. The wire values are unchanged; NFSD no longer depends on the version-agnostic NFSv3 and NFSv4 names. Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260826194444.148243-2-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfs4xdr.c | 2 +- fs/nfsd/nfserr.h | 129 +++++++++++++++++++++++----------------------- 2 files changed, 66 insertions(+), 65 deletions(-) diff --git a/fs/nfsd/nfs4xdr.c b/fs/nfsd/nfs4xdr.c index 5bfbeb87394e6f..00ddaac499c68b 100644 --- a/fs/nfsd/nfs4xdr.c +++ b/fs/nfsd/nfs4xdr.c @@ -3213,7 +3213,7 @@ static __be32 fattr_handle_absent_fs(u32 *bmval0, u32 *bmval1, u32 *bmval2, u32 *bmval1 & ~WORD1_ABSENT_FS_ATTRS) { if (*bmval0 & FATTR4_WORD0_RDATTR_ERROR || *bmval0 & FATTR4_WORD0_FS_LOCATIONS) - *rdattr_err = NFSERR_MOVED; + *rdattr_err = NFS4ERR_MOVED; else return nfserr_moved; } diff --git a/fs/nfsd/nfserr.h b/fs/nfsd/nfserr.h index 9b9df7aab220d7..d1c434b834b2c2 100644 --- a/fs/nfsd/nfserr.h +++ b/fs/nfsd/nfserr.h @@ -11,76 +11,77 @@ #define LINUX_NFSD_NFSERR_H #include +#include #include /* * These macros provide pre-xdr'ed values for faster operation. */ -#define nfs_ok cpu_to_be32(NFS_OK) -#define nfserr_perm cpu_to_be32(NFSERR_PERM) -#define nfserr_noent cpu_to_be32(NFSERR_NOENT) -#define nfserr_io cpu_to_be32(NFSERR_IO) -#define nfserr_nxio cpu_to_be32(NFSERR_NXIO) -#define nfserr_acces cpu_to_be32(NFSERR_ACCES) -#define nfserr_exist cpu_to_be32(NFSERR_EXIST) -#define nfserr_xdev cpu_to_be32(NFSERR_XDEV) -#define nfserr_nodev cpu_to_be32(NFSERR_NODEV) -#define nfserr_notdir cpu_to_be32(NFSERR_NOTDIR) -#define nfserr_isdir cpu_to_be32(NFSERR_ISDIR) -#define nfserr_inval cpu_to_be32(NFSERR_INVAL) -#define nfserr_fbig cpu_to_be32(NFSERR_FBIG) -#define nfserr_nospc cpu_to_be32(NFSERR_NOSPC) -#define nfserr_rofs cpu_to_be32(NFSERR_ROFS) -#define nfserr_mlink cpu_to_be32(NFSERR_MLINK) -#define nfserr_nametoolong cpu_to_be32(NFSERR_NAMETOOLONG) -#define nfserr_notempty cpu_to_be32(NFSERR_NOTEMPTY) -#define nfserr_dquot cpu_to_be32(NFSERR_DQUOT) -#define nfserr_stale cpu_to_be32(NFSERR_STALE) -#define nfserr_remote cpu_to_be32(NFSERR_REMOTE) -#define nfserr_wflush cpu_to_be32(NFSERR_WFLUSH) -#define nfserr_badhandle cpu_to_be32(NFSERR_BADHANDLE) -#define nfserr_notsync cpu_to_be32(NFSERR_NOT_SYNC) -#define nfserr_badcookie cpu_to_be32(NFSERR_BAD_COOKIE) -#define nfserr_notsupp cpu_to_be32(NFSERR_NOTSUPP) -#define nfserr_toosmall cpu_to_be32(NFSERR_TOOSMALL) -#define nfserr_serverfault cpu_to_be32(NFSERR_SERVERFAULT) -#define nfserr_badtype cpu_to_be32(NFSERR_BADTYPE) -#define nfserr_jukebox cpu_to_be32(NFSERR_JUKEBOX) -#define nfserr_denied cpu_to_be32(NFSERR_DENIED) -#define nfserr_deadlock cpu_to_be32(NFSERR_DEADLOCK) -#define nfserr_expired cpu_to_be32(NFSERR_EXPIRED) -#define nfserr_bad_cookie cpu_to_be32(NFSERR_BAD_COOKIE) -#define nfserr_same cpu_to_be32(NFSERR_SAME) -#define nfserr_clid_inuse cpu_to_be32(NFSERR_CLID_INUSE) -#define nfserr_stale_clientid cpu_to_be32(NFSERR_STALE_CLIENTID) -#define nfserr_resource cpu_to_be32(NFSERR_RESOURCE) -#define nfserr_moved cpu_to_be32(NFSERR_MOVED) -#define nfserr_nofilehandle cpu_to_be32(NFSERR_NOFILEHANDLE) -#define nfserr_minor_vers_mismatch cpu_to_be32(NFSERR_MINOR_VERS_MISMATCH) -#define nfserr_share_denied cpu_to_be32(NFSERR_SHARE_DENIED) -#define nfserr_stale_stateid cpu_to_be32(NFSERR_STALE_STATEID) -#define nfserr_old_stateid cpu_to_be32(NFSERR_OLD_STATEID) -#define nfserr_bad_stateid cpu_to_be32(NFSERR_BAD_STATEID) -#define nfserr_bad_seqid cpu_to_be32(NFSERR_BAD_SEQID) -#define nfserr_symlink cpu_to_be32(NFSERR_SYMLINK) -#define nfserr_not_same cpu_to_be32(NFSERR_NOT_SAME) -#define nfserr_lock_range cpu_to_be32(NFSERR_LOCK_RANGE) -#define nfserr_restorefh cpu_to_be32(NFSERR_RESTOREFH) -#define nfserr_attrnotsupp cpu_to_be32(NFSERR_ATTRNOTSUPP) -#define nfserr_bad_xdr cpu_to_be32(NFSERR_BAD_XDR) -#define nfserr_openmode cpu_to_be32(NFSERR_OPENMODE) -#define nfserr_badowner cpu_to_be32(NFSERR_BADOWNER) -#define nfserr_locks_held cpu_to_be32(NFSERR_LOCKS_HELD) -#define nfserr_op_illegal cpu_to_be32(NFSERR_OP_ILLEGAL) -#define nfserr_grace cpu_to_be32(NFSERR_GRACE) -#define nfserr_no_grace cpu_to_be32(NFSERR_NO_GRACE) -#define nfserr_reclaim_bad cpu_to_be32(NFSERR_RECLAIM_BAD) -#define nfserr_badname cpu_to_be32(NFSERR_BADNAME) -#define nfserr_admin_revoked cpu_to_be32(NFS4ERR_ADMIN_REVOKED) -#define nfserr_cb_path_down cpu_to_be32(NFSERR_CB_PATH_DOWN) -#define nfserr_locked cpu_to_be32(NFSERR_LOCKED) -#define nfserr_wrongsec cpu_to_be32(NFSERR_WRONGSEC) +#define nfs_ok cpu_to_be32(NFS_OK) +#define nfserr_perm cpu_to_be32(NFSERR_PERM) +#define nfserr_noent cpu_to_be32(NFSERR_NOENT) +#define nfserr_io cpu_to_be32(NFSERR_IO) +#define nfserr_nxio cpu_to_be32(NFSERR_NXIO) +#define nfserr_acces cpu_to_be32(NFSERR_ACCES) +#define nfserr_exist cpu_to_be32(NFSERR_EXIST) +#define nfserr_xdev cpu_to_be32(NFS3ERR_XDEV) +#define nfserr_nodev cpu_to_be32(NFSERR_NODEV) +#define nfserr_notdir cpu_to_be32(NFSERR_NOTDIR) +#define nfserr_isdir cpu_to_be32(NFSERR_ISDIR) +#define nfserr_inval cpu_to_be32(NFS3ERR_INVAL) +#define nfserr_fbig cpu_to_be32(NFSERR_FBIG) +#define nfserr_nospc cpu_to_be32(NFSERR_NOSPC) +#define nfserr_rofs cpu_to_be32(NFSERR_ROFS) +#define nfserr_mlink cpu_to_be32(NFS3ERR_MLINK) +#define nfserr_nametoolong cpu_to_be32(NFSERR_NAMETOOLONG) +#define nfserr_notempty cpu_to_be32(NFSERR_NOTEMPTY) +#define nfserr_dquot cpu_to_be32(NFSERR_DQUOT) +#define nfserr_stale cpu_to_be32(NFSERR_STALE) +#define nfserr_remote cpu_to_be32(NFS3ERR_REMOTE) +#define nfserr_wflush cpu_to_be32(NFSERR_WFLUSH) +#define nfserr_badhandle cpu_to_be32(NFS3ERR_BADHANDLE) +#define nfserr_notsync cpu_to_be32(NFS3ERR_NOT_SYNC) +#define nfserr_badcookie cpu_to_be32(NFS3ERR_BAD_COOKIE) +#define nfserr_notsupp cpu_to_be32(NFS3ERR_NOTSUPP) +#define nfserr_toosmall cpu_to_be32(NFS3ERR_TOOSMALL) +#define nfserr_serverfault cpu_to_be32(NFS3ERR_SERVERFAULT) +#define nfserr_badtype cpu_to_be32(NFS3ERR_BADTYPE) +#define nfserr_jukebox cpu_to_be32(NFS3ERR_JUKEBOX) #define nfserr_delay cpu_to_be32(NFS4ERR_DELAY) +#define nfserr_same cpu_to_be32(NFS4ERR_SAME) +#define nfserr_denied cpu_to_be32(NFS4ERR_DENIED) +#define nfserr_deadlock cpu_to_be32(NFS4ERR_DEADLOCK) +#define nfserr_expired cpu_to_be32(NFS4ERR_EXPIRED) +#define nfserr_bad_cookie cpu_to_be32(NFS4ERR_BAD_COOKIE) +#define nfserr_clid_inuse cpu_to_be32(NFS4ERR_CLID_INUSE) +#define nfserr_stale_clientid cpu_to_be32(NFS4ERR_STALE_CLIENTID) +#define nfserr_resource cpu_to_be32(NFS4ERR_RESOURCE) +#define nfserr_moved cpu_to_be32(NFS4ERR_MOVED) +#define nfserr_nofilehandle cpu_to_be32(NFS4ERR_NOFILEHANDLE) +#define nfserr_minor_vers_mismatch cpu_to_be32(NFS4ERR_MINOR_VERS_MISMATCH) +#define nfserr_share_denied cpu_to_be32(NFS4ERR_SHARE_DENIED) +#define nfserr_stale_stateid cpu_to_be32(NFS4ERR_STALE_STATEID) +#define nfserr_old_stateid cpu_to_be32(NFS4ERR_OLD_STATEID) +#define nfserr_bad_stateid cpu_to_be32(NFS4ERR_BAD_STATEID) +#define nfserr_bad_seqid cpu_to_be32(NFS4ERR_BAD_SEQID) +#define nfserr_symlink cpu_to_be32(NFS4ERR_SYMLINK) +#define nfserr_not_same cpu_to_be32(NFS4ERR_NOT_SAME) +#define nfserr_lock_range cpu_to_be32(NFS4ERR_LOCK_RANGE) +#define nfserr_restorefh cpu_to_be32(NFS4ERR_RESTOREFH) +#define nfserr_attrnotsupp cpu_to_be32(NFS4ERR_ATTRNOTSUPP) +#define nfserr_bad_xdr cpu_to_be32(NFS4ERR_BADXDR) +#define nfserr_openmode cpu_to_be32(NFS4ERR_OPENMODE) +#define nfserr_badowner cpu_to_be32(NFS4ERR_BADOWNER) +#define nfserr_locks_held cpu_to_be32(NFS4ERR_LOCKS_HELD) +#define nfserr_op_illegal cpu_to_be32(NFS4ERR_OP_ILLEGAL) +#define nfserr_grace cpu_to_be32(NFS4ERR_GRACE) +#define nfserr_no_grace cpu_to_be32(NFS4ERR_NO_GRACE) +#define nfserr_reclaim_bad cpu_to_be32(NFS4ERR_RECLAIM_BAD) +#define nfserr_badname cpu_to_be32(NFS4ERR_BADNAME) +#define nfserr_admin_revoked cpu_to_be32(NFS4ERR_ADMIN_REVOKED) +#define nfserr_cb_path_down cpu_to_be32(NFS4ERR_CB_PATH_DOWN) +#define nfserr_locked cpu_to_be32(NFS4ERR_LOCKED) +#define nfserr_wrongsec cpu_to_be32(NFS4ERR_WRONGSEC) #define nfserr_badiomode cpu_to_be32(NFS4ERR_BADIOMODE) #define nfserr_badlayout cpu_to_be32(NFS4ERR_BADLAYOUT) #define nfserr_bad_session_digest cpu_to_be32(NFS4ERR_BAD_SESSION_DIGEST) From a972c569eddb47f42f125058f38833a13e1d2a4a Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Thu, 27 Aug 2026 14:51:34 -0400 Subject: [PATCH 0303/1352] NFSD: Replace the use of include/trace/misc/nfs.h fs/nfsd/trace.h includes include/trace/misc/nfs.h for show_nfs4_seq4_status() and show_rca_mask(). That header pulls in linux/nfs.h, so every NFSD translation unit that reads trace.h also sees the NFS_OK, NFSERR_*, and file-type enumerators. I'm about to switch fs/nfsd/nfserr.h to an xdrgen-generated header, which defines enum nfsstat and enum ftype with those same names. Any translation unit that includes both headers then fails to build with enumerator redefinition errors. To address this, stop including trace/misc/nfs.h in fs/nfsd/trace.h. Link: https://patch.msgid.link/20260827185134.197322-1-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/trace.h | 45 +++++++++++++++++++++++++++++++++++++++- include/trace/misc/nfs.h | 22 -------------------- 2 files changed, 44 insertions(+), 23 deletions(-) diff --git a/fs/nfsd/trace.h b/fs/nfsd/trace.h index 7d7a1483109a00..2ae7f150a72ce9 100644 --- a/fs/nfsd/trace.h +++ b/fs/nfsd/trace.h @@ -13,7 +13,6 @@ #include #include #include -#include #include #include "export.h" @@ -790,6 +789,20 @@ TRACE_EVENT(nfsd_stateowner_replay, __entry->opnum, __entry->status) ); +#define show_nfs4_seq4_status(x) \ + __print_flags(x, "|", \ + { SEQ4_STATUS_CB_PATH_DOWN, "CB_PATH_DOWN" }, \ + { SEQ4_STATUS_CB_GSS_CONTEXTS_EXPIRING, "CB_GSS_CONTEXTS_EXPIRING" }, \ + { SEQ4_STATUS_CB_GSS_CONTEXTS_EXPIRED, "CB_GSS_CONTEXTS_EXPIRED" }, \ + { SEQ4_STATUS_EXPIRED_ALL_STATE_REVOKED, "EXPIRED_ALL_STATE_REVOKED" }, \ + { SEQ4_STATUS_EXPIRED_SOME_STATE_REVOKED, "EXPIRED_SOME_STATE_REVOKED" }, \ + { SEQ4_STATUS_ADMIN_STATE_REVOKED, "ADMIN_STATE_REVOKED" }, \ + { SEQ4_STATUS_RECALLABLE_STATE_REVOKED, "RECALLABLE_STATE_REVOKED" }, \ + { SEQ4_STATUS_LEASE_MOVED, "LEASE_MOVED" }, \ + { SEQ4_STATUS_RESTART_RECLAIM_NEEDED, "RESTART_RECLAIM_NEEDED" }, \ + { SEQ4_STATUS_CB_PATH_DOWN_SESSION, "CB_PATH_DOWN_SESSION" }, \ + { SEQ4_STATUS_BACKCHANNEL_FAULT, "BACKCHANNEL_FAULT" }) + TRACE_EVENT_CONDITION(nfsd_seq4_status, TP_PROTO( const struct svc_rqst *rqstp, @@ -1694,6 +1707,14 @@ TRACE_EVENT(nfsd_cb_setup_err, /* Not a real opcode, but there is no 0 operation. */ #define _CB_NULL 0 +TRACE_DEFINE_ENUM(OP_CB_GETATTR); +TRACE_DEFINE_ENUM(OP_CB_RECALL); +TRACE_DEFINE_ENUM(OP_CB_LAYOUTRECALL); +TRACE_DEFINE_ENUM(OP_CB_RECALL_ANY); +TRACE_DEFINE_ENUM(OP_CB_NOTIFY); +TRACE_DEFINE_ENUM(OP_CB_NOTIFY_LOCK); +TRACE_DEFINE_ENUM(OP_CB_OFFLOAD); + #define show_nfsd_cb_opcode(val) \ __print_symbolic(val, \ { _CB_NULL, "CB_NULL" }, \ @@ -1918,6 +1939,28 @@ TRACE_EVENT(nfsd_cb_offload, __entry->fh_hash, __entry->count, __entry->status) ); +TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_RDATA_DLG); +TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_WDATA_DLG); +TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_DIR_DLG); +TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_FILE_LAYOUT); +TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_BLK_LAYOUT); +TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_OBJ_LAYOUT_MIN); +TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_OBJ_LAYOUT_MAX); +TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_OTHER_LAYOUT_MIN); +TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_OTHER_LAYOUT_MAX); + +#define show_rca_mask(x) \ + __print_flags(x, "|", \ + { BIT(RCA4_TYPE_MASK_RDATA_DLG), "RDATA_DLG" }, \ + { BIT(RCA4_TYPE_MASK_WDATA_DLG), "WDATA_DLG" }, \ + { BIT(RCA4_TYPE_MASK_DIR_DLG), "DIR_DLG" }, \ + { BIT(RCA4_TYPE_MASK_FILE_LAYOUT), "FILE_LAYOUT" }, \ + { BIT(RCA4_TYPE_MASK_BLK_LAYOUT), "BLK_LAYOUT" }, \ + { BIT(RCA4_TYPE_MASK_OBJ_LAYOUT_MIN), "OBJ_LAYOUT_MIN" }, \ + { BIT(RCA4_TYPE_MASK_OBJ_LAYOUT_MAX), "OBJ_LAYOUT_MAX" }, \ + { BIT(RCA4_TYPE_MASK_OTHER_LAYOUT_MIN), "OTHER_LAYOUT_MIN" }, \ + { BIT(RCA4_TYPE_MASK_OTHER_LAYOUT_MAX), "OTHER_LAYOUT_MAX" }) + TRACE_EVENT(nfsd_cb_recall_any, TP_PROTO( const struct nfsd4_cb_recall_any *ra diff --git a/include/trace/misc/nfs.h b/include/trace/misc/nfs.h index 27781bd7a3f787..3146813fc4fe02 100644 --- a/include/trace/misc/nfs.h +++ b/include/trace/misc/nfs.h @@ -359,28 +359,6 @@ TRACE_DEFINE_ENUM(IOMODE_ANY); { IOMODE_RW, "RW" }, \ { IOMODE_ANY, "ANY" }) -TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_RDATA_DLG); -TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_WDATA_DLG); -TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_DIR_DLG); -TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_FILE_LAYOUT); -TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_BLK_LAYOUT); -TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_OBJ_LAYOUT_MIN); -TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_OBJ_LAYOUT_MAX); -TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_OTHER_LAYOUT_MIN); -TRACE_DEFINE_ENUM(RCA4_TYPE_MASK_OTHER_LAYOUT_MAX); - -#define show_rca_mask(x) \ - __print_flags(x, "|", \ - { BIT(RCA4_TYPE_MASK_RDATA_DLG), "RDATA_DLG" }, \ - { BIT(RCA4_TYPE_MASK_WDATA_DLG), "WDATA_DLG" }, \ - { BIT(RCA4_TYPE_MASK_DIR_DLG), "DIR_DLG" }, \ - { BIT(RCA4_TYPE_MASK_FILE_LAYOUT), "FILE_LAYOUT" }, \ - { BIT(RCA4_TYPE_MASK_BLK_LAYOUT), "BLK_LAYOUT" }, \ - { BIT(RCA4_TYPE_MASK_OBJ_LAYOUT_MIN), "OBJ_LAYOUT_MIN" }, \ - { BIT(RCA4_TYPE_MASK_OBJ_LAYOUT_MAX), "OBJ_LAYOUT_MAX" }, \ - { BIT(RCA4_TYPE_MASK_OTHER_LAYOUT_MIN), "OTHER_LAYOUT_MIN" }, \ - { BIT(RCA4_TYPE_MASK_OTHER_LAYOUT_MAX), "OTHER_LAYOUT_MAX" }) - #define show_nfs4_seq4_status(x) \ __print_flags(x, "|", \ { SEQ4_STATUS_CB_PATH_DOWN, "CB_PATH_DOWN" }, \ From e87ec9ce13397275fa91bf2003244c527453ef83 Mon Sep 17 00:00:00 2001 From: "Cen Zhang (Microsoft Security FORGE Labs)" Date: Fri, 28 Aug 2026 00:19:25 -0400 Subject: [PATCH 0304/1352] nfsd: hold cl_lock in client_has_openowners() client_has_openowners() walks clp->cl_openowners and reads so_stateids without clp->cl_lock. nfs4_put_stateowner() unhashes that openowner under cl_lock and then frees it, so a concurrent EXCHANGE_ID with mismatched creds can use-after-free the nfs4_openowner. BUG: KASAN: slab-use-after-free in client_has_state+0x10a/0x140 fs/nfsd/nfs4state.c:3718 client_has_openowners() nfsd4_exchange_id nfsd4_proc_compound nfsd_dispatch svc_process Take clp->cl_lock while client_has_openowners() walks the openowner list. Fixes: 4eaea1342507 ("nfsd: improve client_has_state to check for unused openowners") Cc: stable@vger.kernel.org Reported-by: Xiang Mei (Microsoft) Cc: AutonomousCodeSecurity@microsoft.com Signed-off-by: Cen Zhang (Microsoft Security FORGE Labs) Link: https://patch.msgid.link/20260828041925.36758-1-blbllhy@gmail.com Signed-off-by: Chuck Lever --- fs/nfsd/nfs4state.c | 12 +++++++++--- 1 file changed, 9 insertions(+), 3 deletions(-) diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c index db79eba61bbf93..1de6c6d757c3ed 100644 --- a/fs/nfsd/nfs4state.c +++ b/fs/nfsd/nfs4state.c @@ -4251,12 +4251,18 @@ nfsd4_set_ex_flags(struct nfs4_client *new, struct nfsd4_exchange_id *clid) static bool client_has_openowners(struct nfs4_client *clp) { struct nfs4_openowner *oo; + bool found = false; + spin_lock(&clp->cl_lock); list_for_each_entry(oo, &clp->cl_openowners, oo_perclient) { - if (!list_empty(&oo->oo_owner.so_stateids)) - return true; + if (!list_empty(&oo->oo_owner.so_stateids)) { + found = true; + break; + } } - return false; + spin_unlock(&clp->cl_lock); + + return found; } static bool client_has_state(struct nfs4_client *clp) From 2f1ee9d0649001c49d2703062ac99c8ca3fee00b Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Fri, 28 Aug 2026 09:50:34 -0400 Subject: [PATCH 0305/1352] svcrdma: Grant credits from the clamped sc_max_requests svc_rdma_accept() computes sc_fc_credits from sc_max_requests before the Receive Queue depth is checked against the device's max_qp_wr. When that check lowers sc_max_requests, the credit grant keeps the original value, so the server advertises more credits than it has Receives posted. A client that uses the full grant overruns the Receive Queue, and the connection is lost with an RNR error. Set sc_fc_credits after the clamp so the grant matches the number of Receives the server posts. Fixes: fc2e69db82c1 ("svcrdma: Clean up comment in svc_rdma_accept()") Cc: stable@vger.kernel.org Link: https://patch.msgid.link/20260828135036.796842-2-cel@kernel.org Signed-off-by: Chuck Lever --- net/sunrpc/xprtrdma/svc_rdma_transport.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/net/sunrpc/xprtrdma/svc_rdma_transport.c b/net/sunrpc/xprtrdma/svc_rdma_transport.c index 927269598ac24b..f949601b214471 100644 --- a/net/sunrpc/xprtrdma/svc_rdma_transport.c +++ b/net/sunrpc/xprtrdma/svc_rdma_transport.c @@ -471,7 +471,6 @@ static struct svc_xprt *svc_rdma_accept(struct svc_xprt *xprt) newxprt->sc_max_requests = svcrdma_max_requests; newxprt->sc_max_bc_requests = svcrdma_max_bc_requests; newxprt->sc_recv_batch = RPCRDMA_MAX_RECV_BATCH; - newxprt->sc_fc_credits = cpu_to_be32(newxprt->sc_max_requests); /* Qualify the transport's resource defaults with the * capabilities of this particular device. @@ -492,6 +491,8 @@ static struct svc_xprt *svc_rdma_accept(struct svc_xprt *xprt) newxprt->sc_max_bc_requests = 2; } + newxprt->sc_fc_credits = cpu_to_be32(newxprt->sc_max_requests); + /* Estimate the needed number of rdma_rw contexts. The maximum * Read and Write chunks have one segment each. Each request * can involve one Read chunk and either a Write chunk or Reply From c67bfd90b3a9f764ca8c1fd6cc3284c5f1394660 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Fri, 28 Aug 2026 09:50:35 -0400 Subject: [PATCH 0306/1352] svcrdma: Clear XPT_DATA when the last receive context is consumed svc_rdma_wc_receive() and svc_rdma_wc_read_done() set XPT_DATA after adding a completed context to sc_rq_dto_q or sc_read_complete_q. svc_rdma_recvfrom() dequeues one context and leaves XPT_DATA set, so the svc_xprt_received() that follows re-enqueues the transport and svc_xprt_enqueue() dispatches a second thread. That thread finds both queues empty and returns zero. Recheck the receive queues after each dequeue and clear XPT_DATA when the last context is taken, rather than only when a dequeue finds nothing. svc_xprt_received()'s kernel-doc no longer describes every transport, so relax its note about when XPT_DATA is cleared. Measured on one NFSv4.2 connection over 100GbE RoCE. A 4KB random read at queue depth 1 falls from 2.997 transport dequeues per RPC to 1.998, and from 96,629 to 82,398 server cycles per RPC. A 256KB random write falls from 3.270 dequeues to 2.004, and from 259,936 to 246,059 cycles. Each dispatch removed is worth about 10,000 cycles. The gain shrinks as the receive queues fill, since a leftover XPT_DATA then dispatches a thread that finds real work. An 8KB random write at queue depth 512 already runs at the two dequeues an RPC with a Read chunk requires, and shows no change. Throughput moves only where the server has no idle CPU to absorb the saving, so only the queue depth 1 read gains, by 1.8%. One dispatch per RPC remains. svc_rdma_send_ctxt_put() sets XPT_DATA to schedule a drain of sc_send_release_list, and svc_rdma_recvfrom() does not service that list. Link: https://patch.msgid.link/20260828135036.796842-3-cel@kernel.org Signed-off-by: Chuck Lever --- net/sunrpc/svc_xprt.c | 11 +++++++---- net/sunrpc/xprtrdma/svc_rdma_recvfrom.c | 18 ++++++++++++++---- 2 files changed, 21 insertions(+), 8 deletions(-) diff --git a/net/sunrpc/svc_xprt.c b/net/sunrpc/svc_xprt.c index 40040af588fb23..a60e972490ecf2 100644 --- a/net/sunrpc/svc_xprt.c +++ b/net/sunrpc/svc_xprt.c @@ -64,8 +64,10 @@ static LIST_HEAD(svc_xprt_class_list); * - Can be set or cleared at any time. * - After a set, svc_xprt_enqueue must be called to enqueue * the transport for processing. - * - After a clear, the transport must be read/accepted. - * If this succeeds, it must be set again. + * - After clearing XPT_CONN, the transport must be + * accepted. If this succeeds, the bit must be set again. + * - xpo_recvfrom decides when XPT_DATA is cleared; see + * svc_xprt_received. * XPT_CLOSE: * - Can set at any time. It is never cleared. * XPT_DEAD: @@ -218,8 +220,9 @@ EXPORT_SYMBOL_GPL(svc_xprt_init); * The caller must hold the XPT_BUSY bit and must * not thereafter touch transport data. * - * Note: XPT_DATA only gets cleared when a read-attempt finds no (or - * insufficient) data. + * Note: xpo_recvfrom decides when to clear XPT_DATA. A transport may + * leave the bit set until a read attempt finds no (or insufficient) + * data, or clear it as soon as it consumes the last queued receive. */ void svc_xprt_received(struct svc_xprt *xprt) { diff --git a/net/sunrpc/xprtrdma/svc_rdma_recvfrom.c b/net/sunrpc/xprtrdma/svc_rdma_recvfrom.c index fdfed1be97da9f..d029bcb7a5c0a0 100644 --- a/net/sunrpc/xprtrdma/svc_rdma_recvfrom.c +++ b/net/sunrpc/xprtrdma/svc_rdma_recvfrom.c @@ -925,8 +925,9 @@ static noinline void svc_rdma_read_complete(struct svc_rqst *rqstp, * %-ENOTCONN if posting failed (connection is lost), * %-EIO if rdma_rw initialization failed (DMA mapping, etc). * - * Called in a loop when XPT_DATA is set. XPT_DATA is cleared only - * when there are no remaining ctxt's to process. + * Called in a loop when XPT_DATA is set. XPT_DATA is cleared as + * soon as both receive queues are empty, so a consumed ctxt does + * not leave a stale bit behind. * * The next ctxt is removed from the "receive" lists. * @@ -960,6 +961,15 @@ int svc_rdma_recvfrom(struct svc_rqst *rqstp) ctxt = svc_rdma_next_recv_ctxt(&rdma_xprt->sc_read_complete_q); if (ctxt) { list_del(&ctxt->rc_list); + /* Producers add to these queues and set XPT_DATA under + * this lock, so the clear cannot race one. The clear can + * drop the XPT_DATA that svc_rdma_send_ctxt_put() sets + * for sc_send_release_list. svc_xprt_release() drains + * that list before this thread looks for more work. + */ + if (list_empty(&rdma_xprt->sc_read_complete_q) && + list_empty(&rdma_xprt->sc_rq_dto_q)) + clear_bit(XPT_DATA, &xprt->xpt_flags); spin_unlock(&rdma_xprt->sc_rq_dto_lock); svc_xprt_received(xprt); svc_rdma_read_complete(rqstp, ctxt); @@ -968,8 +978,8 @@ int svc_rdma_recvfrom(struct svc_rqst *rqstp) ctxt = svc_rdma_next_recv_ctxt(&rdma_xprt->sc_rq_dto_q); if (ctxt) list_del(&ctxt->rc_list); - else - /* No new incoming requests, terminate the loop */ + /* sc_read_complete_q was empty above, under this same lock. */ + if (list_empty(&rdma_xprt->sc_rq_dto_q)) clear_bit(XPT_DATA, &xprt->xpt_flags); spin_unlock(&rdma_xprt->sc_rq_dto_lock); From ad35b612ca9e117aa6a0b883ca15b3f245c587ea Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Fri, 28 Aug 2026 09:50:36 -0400 Subject: [PATCH 0307/1352] SUNRPC: Skip xpt_reserved accounting for non-UDP transports The xpt_reserved counter exists for UDP socket-buffer back-pressure. svc_udp_has_wspace() is the only has_wspace implementation that consults it, so on TCP and RDMA the counter is maintained and never read. svc_handle_xprt() adds to it once per RPC. svc_reserve() shrinks it again on each call from svc_process_common(), from svc_xprt_release(), and from each proc function that calls svc_reserve_auth(). Every shrinking call also runs svc_xprt_resource_released(), which issues an smp_mb() and can enqueue the transport. Add an xcl_flags field to svc_xprt_class and set SVC_XPRT_FLAG_WSPACE_RESERVE on the UDP class. Gate the xpt_reserved accounting on that flag. After the change, svc_reserve() no longer calls svc_xprt_resource_released() on TCP and RDMA. Two paths still cover that enqueue. svc_xprt_release() reaches the helper through svc_xprt_release_slot(), and svc_xprt_received() enqueues a transport whose XPT_DATA remains set. Link: https://patch.msgid.link/20260828135036.796842-4-cel@kernel.org Signed-off-by: Chuck Lever --- include/linux/sunrpc/svc_xprt.h | 5 ++++- net/sunrpc/svc_xprt.c | 25 ++++++++++++++++--------- net/sunrpc/svcsock.c | 1 + 3 files changed, 21 insertions(+), 10 deletions(-) diff --git a/include/linux/sunrpc/svc_xprt.h b/include/linux/sunrpc/svc_xprt.h index da2a2531e1106d..2af222f3ea2c2e 100644 --- a/include/linux/sunrpc/svc_xprt.h +++ b/include/linux/sunrpc/svc_xprt.h @@ -37,6 +37,9 @@ struct svc_xprt_class { struct list_head xcl_list; u32 xcl_max_payload; int xcl_ident; + u32 xcl_flags; +/* Set only on classes whose xpo_has_wspace() reads xpt_reserved */ +#define SVC_XPRT_FLAG_WSPACE_RESERVE BIT(0) }; /* @@ -59,7 +62,7 @@ struct svc_xprt { unsigned long xpt_flags; struct svc_serv *xpt_server; /* service for transport */ - atomic_t xpt_reserved; /* space on outq that is rsvd */ + atomic_t xpt_reserved; /* outq space rsvd, UDP only */ atomic_t xpt_nr_rqsts; /* Number of requests */ struct mutex xpt_mutex; /* to serialize sending data */ spinlock_t xpt_lock; /* protects sk_deferred diff --git a/net/sunrpc/svc_xprt.c b/net/sunrpc/svc_xprt.c index a60e972490ecf2..77e28dcc4d2ab3 100644 --- a/net/sunrpc/svc_xprt.c +++ b/net/sunrpc/svc_xprt.c @@ -478,11 +478,11 @@ static bool svc_xprt_ready(struct svc_xprt *xprt) /* * If another cpu has recently updated xpt_flags, - * sk_sock->flags, xpt_reserved, or xpt_nr_rqsts, we need to - * know about it; otherwise it's possible that both that cpu and - * this one could call svc_xprt_enqueue() without either - * svc_xprt_enqueue() recognizing that the conditions below - * are satisfied, and we could stall indefinitely: + * sk_sock->flags, xpt_reserved (UDP only), or xpt_nr_rqsts, + * we need to know about it; otherwise it's possible that both + * that cpu and this one could call svc_xprt_enqueue() without + * either svc_xprt_enqueue() recognizing that the conditions + * below are satisfied, and we could stall indefinitely: */ smp_rmb(); xpt_flags = READ_ONCE(xprt->xpt_flags); @@ -554,6 +554,10 @@ static struct svc_xprt *svc_xprt_dequeue(struct svc_pool *pool) * to make sure the reply fits. This function reduces that reserved * space to be the amount of space used already, plus @space. * + * The transport's reservation is tracked only on classes that set + * SVC_XPRT_FLAG_WSPACE_RESERVE. On the others, only @rqstp's + * reservation is updated. + * */ void svc_reserve(struct svc_rqst *rqstp, int space) { @@ -562,10 +566,12 @@ void svc_reserve(struct svc_rqst *rqstp, int space) space += rqstp->rq_res.head[0].iov_len; if (xprt && space < rqstp->rq_reserved) { - atomic_sub((rqstp->rq_reserved - space), - &xprt->xpt_reserved); + if (xprt->xpt_class->xcl_flags & SVC_XPRT_FLAG_WSPACE_RESERVE) { + atomic_sub((rqstp->rq_reserved - space), + &xprt->xpt_reserved); + svc_xprt_resource_released(xprt); + } rqstp->rq_reserved = space; - svc_xprt_resource_released(xprt); } } EXPORT_SYMBOL_GPL(svc_reserve); @@ -872,7 +878,8 @@ static void svc_handle_xprt(struct svc_rqst *rqstp, struct svc_xprt *xprt) else len = xprt->xpt_ops->xpo_recvfrom(rqstp); rqstp->rq_reserved = serv->sv_max_mesg; - atomic_add(rqstp->rq_reserved, &xprt->xpt_reserved); + if (xprt->xpt_class->xcl_flags & SVC_XPRT_FLAG_WSPACE_RESERVE) + atomic_add(rqstp->rq_reserved, &xprt->xpt_reserved); if (len <= 0) goto out; diff --git a/net/sunrpc/svcsock.c b/net/sunrpc/svcsock.c index ef7ac080fcd3b2..e5459d504b6a91 100644 --- a/net/sunrpc/svcsock.c +++ b/net/sunrpc/svcsock.c @@ -800,6 +800,7 @@ static struct svc_xprt_class svc_udp_class = { .xcl_ops = &svc_udp_ops, .xcl_max_payload = RPCSVC_MAXPAYLOAD_UDP, .xcl_ident = XPRT_TRANSPORT_UDP, + .xcl_flags = SVC_XPRT_FLAG_WSPACE_RESERVE, }; static void svc_udp_init(struct svc_sock *svsk, struct svc_serv *serv) From ac04dab23b5ff28fc7e41957824c5c439ae99887 Mon Sep 17 00:00:00 2001 From: Chuck Lever Date: Fri, 28 Aug 2026 19:18:41 -0400 Subject: [PATCH 0308/1352] NFSD: Return NFSERR_ISDIR for NFSv2 READ and WRITE on a non-regular file MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Commit 0ac903d1bfdc ("NFS: NFSERR_INVAL is not defined by NFSv2") changed the status NFSD returns for an NFSv2 READ or WRITE of a symlink or other non-regular file from NFSERR_INVAL to NFSERR_IO. RFC 1094 defines neither an INVAL nor a SYMLINK status. U-Boot's NFS client follows a symlink only when READ fails with NFSERR_ISDIR or NFSERR_INVAL. Since that commit it cannot load a boot image through a symlink exported by a Linux NFS server: the load completes with zero bytes transferred. Solaris returns NFSERR_ISDIR when the target of an NFSv2 READ or WRITE is not a regular file. Return the same status from NFSD. Other NFSv2 procedures continue to return NFSERR_IO. Fixes: 0ac903d1bfdc ("NFS: NFSERR_INVAL is not defined by NFSv2") Cc: stable@vger.kernel.org Reported-by: Jörg Sommer Closes: https://lore.kernel.org/linux-nfs/apE90ZRZB8IW_AiS@jo-so.de/ Reviewed-by: NeilBrown Reviewed-by: Jeff Layton Link: https://patch.msgid.link/20260828231841.282003-1-cel@kernel.org Signed-off-by: Chuck Lever --- fs/nfsd/nfsproc.c | 19 +++++++++++++++++-- 1 file changed, 17 insertions(+), 2 deletions(-) diff --git a/fs/nfsd/nfsproc.c b/fs/nfsd/nfsproc.c index 48541ef7644f4d..09d36083982504 100644 --- a/fs/nfsd/nfsproc.c +++ b/fs/nfsd/nfsproc.c @@ -40,6 +40,21 @@ static __be32 nfsd_map_status(__be32 status) return status; } +/* + * Because NFSv2 does not have an NFSERR_SYMLINK, Solaris returns + * NFSERR_ISDIR when the target of a READ or WRITE is any object + * that is not a regular file. + */ +static __be32 nfsd_map_io_status(__be32 status) +{ + switch (status) { + case nfserr_symlink: + case nfserr_wrong_type: + return nfserr_isdir; + } + return nfsd_map_status(status); +} + static __be32 nfsd_proc_null(struct svc_rqst *rqstp) { @@ -238,7 +253,7 @@ nfsd_proc_read(struct svc_rqst *rqstp) resp->status = fh_getattr(&resp->fh, &resp->stat); else if (resp->status == nfserr_jukebox) set_bit(RQ_DROPME, &rqstp->rq_flags); - resp->status = nfsd_map_status(resp->status); + resp->status = nfsd_map_io_status(resp->status); return rpc_success; } @@ -271,7 +286,7 @@ nfsd_proc_write(struct svc_rqst *rqstp) resp->status = fh_getattr(&resp->fh, &resp->stat); else if (resp->status == nfserr_jukebox) set_bit(RQ_DROPME, &rqstp->rq_flags); - resp->status = nfsd_map_status(resp->status); + resp->status = nfsd_map_io_status(resp->status); return rpc_success; } From 9663c15d93e39fbad397e759669d21f96777697e Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Tue, 22 Sep 2026 18:17:21 +0300 Subject: [PATCH 0309/1352] drm/xe/display: reduce includes in xe_display.c MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit xe_display.c has a bunch of extra leftover includes. Remove and simplify. Reviewed-by: Michał Grzelak Link: https://patch.msgid.link/20260922151721.1483978-1-jani.nikula@intel.com Signed-off-by: Jani Nikula --- drivers/gpu/drm/xe/display/xe_display.c | 10 +--------- 1 file changed, 1 insertion(+), 9 deletions(-) diff --git a/drivers/gpu/drm/xe/display/xe_display.c b/drivers/gpu/drm/xe/display/xe_display.c index 7b25c0814674c0..506f26e9c59a82 100644 --- a/drivers/gpu/drm/xe/display/xe_display.c +++ b/drivers/gpu/drm/xe/display/xe_display.c @@ -6,31 +6,23 @@ #include "xe_display.h" #include "regs/xe_irq_regs.h" -#include +#include -#include -#include #include #include -#include #include #include -#include -#include "intel_acpi.h" #include "intel_display.h" #include "intel_display_core.h" #include "intel_display_device.h" #include "intel_display_driver.h" #include "intel_display_irq.h" -#include "intel_display_types.h" #include "intel_dmc.h" #include "intel_dmc_wl.h" -#include "intel_dp.h" #include "intel_fbdev.h" #include "intel_hotplug.h" #include "intel_opregion.h" -#include "skl_watermark.h" #include "xe_device.h" #include "xe_display_bo.h" #include "xe_display_pcode.h" From 7c773ec6c87f1adddf8e9a175f034e54c30726f4 Mon Sep 17 00:00:00 2001 From: Lorenzo Bianconi Date: Wed, 9 Sep 2026 16:55:53 +0200 Subject: [PATCH 0310/1352] dt-bindings: PCI: toshiba,tc9563: Document embedded GPIO controller The TC9563 PCIe switch embeds a GPIO controller providing 37 GPIO lines, accessed through the switch's I2C interface. The controller is integrated in the switch rather than being a separate device, so expose it by adding the gpio-controller and #gpio-cells properties to the switch node itself, and extend the example to show the downstream ports' reset-gpios being driven from those lines. Signed-off-by: Lorenzo Bianconi Signed-off-by: Bjorn Helgaas Reviewed-by: Krzysztof Kozlowski Link: https://patch.msgid.link/20260909-pci-tc9563-aux-v5-1-c9b33f56c8d3@oss.qualcomm.com --- .../devicetree/bindings/pci/toshiba,tc9563.yaml | 14 +++++++++++++- 1 file changed, 13 insertions(+), 1 deletion(-) diff --git a/Documentation/devicetree/bindings/pci/toshiba,tc9563.yaml b/Documentation/devicetree/bindings/pci/toshiba,tc9563.yaml index f084830c6d0cb3..c4ebb99f50c37a 100644 --- a/Documentation/devicetree/bindings/pci/toshiba,tc9563.yaml +++ b/Documentation/devicetree/bindings/pci/toshiba,tc9563.yaml @@ -31,6 +31,11 @@ properties: description: GPIO controlling the RESX# pin. + gpio-controller: true + + '#gpio-cells': + const: 2 + vdd18-supply: true vdd09-supply: true @@ -128,7 +133,7 @@ examples: ranges; bus-range = <0x01 0xff>; - pcie@0,0 { + tc9563: pcie@0,0 { compatible = "pci1179,0623"; reg = <0x10000 0x0 0x0 0x0 0x0>; @@ -149,6 +154,9 @@ examples: resx-gpios = <&gpio 1 GPIO_ACTIVE_LOW>; + gpio-controller; + #gpio-cells = <2>; + pcie@1,0 { compatible = "pciclass,0604"; reg = <0x20800 0x0 0x0 0x0 0x0>; @@ -158,6 +166,8 @@ examples: ranges; bus-range = <0x03 0xff>; + reset-gpios = <&tc9563 2 GPIO_ACTIVE_LOW>; + toshiba,no-dfe-support; }; @@ -170,6 +180,8 @@ examples: ranges; bus-range = <0x04 0xff>; + reset-gpios = <&tc9563 3 GPIO_ACTIVE_LOW>; + toshiba,tx-amplitude-microvolt = <10>; }; From 3c06b0b0f3c963c954f775c836d2206a64da61b6 Mon Sep 17 00:00:00 2001 From: Alex Elder Date: Wed, 9 Sep 2026 16:55:54 +0200 Subject: [PATCH 0311/1352] gpio: tc9563: Add support for the embedded GPIO controller Add a driver for the GPIO controller embedded in the Toshiba TC9563 PCIe switch (and the Qualcomm QPS615). The device implements 37 GPIOs using two register banks: three registers control the first 32 GPIOs and three more control GPIOs 32-36. GPIOs 20 and 21 are reserved while GPIOs 22-24, 27-28, 31, and 34 are input-only. Register the driver as an auxiliary device driver. The TC9563 power controller creates the auxiliary device and provides a regmap that gives access to the GPIO registers, so use the gpio-regmap helpers to implement the GPIO chip. Signed-off-by: Alex Elder Co-developed-by: Daniel Thompson Signed-off-by: Daniel Thompson Co-developed-by: Lorenzo Bianconi Signed-off-by: Lorenzo Bianconi [bhelgaas: commit log per https://lore.kernel.org/arQ2mDjqHv4gmoOK@lore-desk] Signed-off-by: Bjorn Helgaas Reviewed-by: Linus Walleij Reviewed-by: Manivannan Sadhasivam Acked-by: Bartosz Golaszewski Link: https://patch.msgid.link/20260909-pci-tc9563-aux-v5-2-c9b33f56c8d3@oss.qualcomm.com --- drivers/gpio/Kconfig | 11 ++++ drivers/gpio/Makefile | 1 + drivers/gpio/gpio-tc9563.c | 99 +++++++++++++++++++++++++++++++++ include/linux/soc/qcom/tc9563.h | 16 ++++++ 4 files changed, 127 insertions(+) create mode 100644 drivers/gpio/gpio-tc9563.c create mode 100644 include/linux/soc/qcom/tc9563.h diff --git a/drivers/gpio/Kconfig b/drivers/gpio/Kconfig index a48586bb8edbae..2179eaffcc48b4 100644 --- a/drivers/gpio/Kconfig +++ b/drivers/gpio/Kconfig @@ -1830,6 +1830,17 @@ config GPIO_LTC4283 endmenu +config GPIO_TC9563 + tristate "Toshiba TC9563 GPIO support" + default m if ARCH_QCOM + select AUXILIARY_BUS + select GPIO_REGMAP + help + This enables support for the GPIO controller embedded in the Toshiba + TC9563 (and Qualcomm QPS615). This device connects to the host + via PCIe port, which is the upstream port on an internal PCIe + switch. + menu "PCI GPIO expanders" depends on PCI diff --git a/drivers/gpio/Makefile b/drivers/gpio/Makefile index dc9e6d643b5bca..792faa2668c993 100644 --- a/drivers/gpio/Makefile +++ b/drivers/gpio/Makefile @@ -182,6 +182,7 @@ obj-$(CONFIG_GPIO_SYSCON) += gpio-syscon.o obj-$(CONFIG_GPIO_TANGIER) += gpio-tangier.o obj-$(CONFIG_GPIO_TB10X) += gpio-tb10x.o obj-$(CONFIG_GPIO_TC3589X) += gpio-tc3589x.o +obj-$(CONFIG_GPIO_TC9563) += gpio-tc9563.o obj-$(CONFIG_GPIO_TEGRA186) += gpio-tegra186.o obj-$(CONFIG_GPIO_TEGRA) += gpio-tegra.o obj-$(CONFIG_GPIO_THUNDERX) += gpio-thunderx.o diff --git a/drivers/gpio/gpio-tc9563.c b/drivers/gpio/gpio-tc9563.c new file mode 100644 index 00000000000000..68c20c8cba7994 --- /dev/null +++ b/drivers/gpio/gpio-tc9563.c @@ -0,0 +1,99 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Copyright (C) 2026 by RISCstar Solutions Corporation. All rights reserved. + */ + +/* + * The Toshiba TC9563 implements a PCIe Gen 3 switch that connects an + * upstream x4 port to two downstream PCIe x2 ports. It incorporates + * an internal endpoint on a internal PCIe port that implements two + * Synopsys XGMAC Ethernet interfaces. + * + * 37 GPIOs are also implemented by an embedded GPIO controller. Three + * registers control the first 32 GPIOs (other than 20 and 21, which are + * reserved). Three other registers control GPIOs 32 through 36. GPIOs + * 22-24, 27-28, 31, and 34 are treated as "input only". + * + */ + +#include +#include +#include +#include +#include +#include + +/* + * There are two sets of registers, each representing (up to) 32 GPIOs with a + * stride of 4 bytes (IN1 is 4 bytes past IN0, EN1 is 4 bytes past EN0, etc.). + */ +#define TC9563_GPIO_COUNT 37 +#define TC9563_GPIO_PER_REG 32 +#define TC9563_GPIO_REG_STRIDE 4 + +static int tc9563_gpio_init_valid_mask(struct gpio_chip *gc, + unsigned long *valid_mask, + unsigned int ngpios) +{ + /* GPIOs 20 and 21 are reserved */ + bitmap_fill(valid_mask, ngpios); + bitmap_clear(valid_mask, 20, 2); + + return 0; +} + +static int tc9563_gpio_probe(struct auxiliary_device *adev, + const struct auxiliary_device_id *id) +{ + struct gpio_regmap_config config = { + .parent = &adev->dev, + .ngpio = TC9563_GPIO_COUNT, + .reg_stride = TC9563_GPIO_REG_STRIDE, + .ngpio_per_reg = TC9563_GPIO_PER_REG, + .reg_dat_base = GPIO_REGMAP_ADDR(TC9563_GPIO_IN0_OFFSET), + .reg_set_base = GPIO_REGMAP_ADDR(TC9563_GPIO_OUT0_OFFSET), + .reg_dir_in_base = GPIO_REGMAP_ADDR(TC9563_GPIO_EN0_OFFSET), + .init_valid_mask = tc9563_gpio_init_valid_mask, + }; + DECLARE_BITMAP(fixed_dir_mask, TC9563_GPIO_COUNT); + DECLARE_BITMAP(fixed_dir_out, TC9563_GPIO_COUNT); + + config.regmap = dev_get_platdata(&adev->dev); + if (!config.regmap) + return -EINVAL; + + /* + * Only some of our GPIOs are fixed direction: + * 22, 23, 24, 27, 28, 31, and 34 are input-only. + */ + bitmap_zero(fixed_dir_mask, TC9563_GPIO_COUNT); + bitmap_set(fixed_dir_mask, 22, 3); + bitmap_set(fixed_dir_mask, 27, 2); + set_bit(31, fixed_dir_mask); + set_bit(34, fixed_dir_mask); + config.fixed_direction_mask = fixed_dir_mask; + + bitmap_zero(fixed_dir_out, TC9563_GPIO_COUNT); + config.fixed_direction_output = fixed_dir_out; + + return PTR_ERR_OR_ZERO(devm_gpio_regmap_register(&adev->dev, &config)); +}; + +static const struct auxiliary_device_id tc9563_gpio_ids[] = { + { "pci_pwrctrl_tc9563." TC9563_GPIO_DEV_NAME }, + { /* sentinel */ } +}; +MODULE_DEVICE_TABLE(auxiliary, tc9563_gpio_ids); + +static struct auxiliary_driver tc9563_gpio_driver = { + .name = TC9563_GPIO_DEV_NAME, + .probe = tc9563_gpio_probe, + .id_table = tc9563_gpio_ids, +}; +module_auxiliary_driver(tc9563_gpio_driver); + +MODULE_AUTHOR("Alex Elder "); +MODULE_AUTHOR("Daniel Thompson "); +MODULE_AUTHOR("Lorenzo Bianconi "); +MODULE_DESCRIPTION("Toshiba TC9563 GPIO Driver"); +MODULE_LICENSE("GPL"); diff --git a/include/linux/soc/qcom/tc9563.h b/include/linux/soc/qcom/tc9563.h new file mode 100644 index 00000000000000..0dfd25747b9af7 --- /dev/null +++ b/include/linux/soc/qcom/tc9563.h @@ -0,0 +1,16 @@ +/* SPDX-License-Identifier: GPL-2.0-only */ +/* + * Copyright (c) 2026 Qualcomm Innovation Center, Inc. All rights reserved. + * Author: Lorenzo Bianconi + */ + +#ifndef __QCOM_TC9563_H +#define __QCOM_TC9563_H + +#define TC9563_GPIO_DEV_NAME "tc9563-gpio" + +#define TC9563_GPIO_IN0_OFFSET 0x801200 +#define TC9563_GPIO_EN0_OFFSET 0x801208 +#define TC9563_GPIO_OUT0_OFFSET 0x801210 + +#endif /* __QCOM_TC9563_H */ From f99bdd8b49fe5ea8f06d3f37c42e003f79205d4e Mon Sep 17 00:00:00 2001 From: Lorenzo Bianconi Date: Wed, 9 Sep 2026 16:55:55 +0200 Subject: [PATCH 0312/1352] PCI/pwrctrl: tc9563: Add GPIO auxiliary device support The TC9563 embeds a GPIO controller used for per-port reset signals. Create an auxiliary device for it so the gpio-tc9563 driver can register the GPIO chip and enable DT-based GPIO lookups. Pass the tc9563 regmap to the auxiliary device as its platform data. The pwrctrl driver does not wait for the GPIO chip to be probed. The per-port reset GPIO lookup, returning -EPROBE_DEFER until the chip is registered, is added in the next patch. Signed-off-by: Lorenzo Bianconi Signed-off-by: Bjorn Helgaas Reviewed-by: Bartosz Golaszewski Reviewed-by: Manivannan Sadhasivam Reviewed-by: Alex Elder Link: https://patch.msgid.link/20260909-pci-tc9563-aux-v5-3-c9b33f56c8d3@oss.qualcomm.com --- drivers/pci/pwrctrl/Kconfig | 1 + drivers/pci/pwrctrl/pci-pwrctrl-tc9563.c | 80 ++++++++++++++++++++++++ 2 files changed, 81 insertions(+) diff --git a/drivers/pci/pwrctrl/Kconfig b/drivers/pci/pwrctrl/Kconfig index 1952ab4f29b691..38aab596aa04b8 100644 --- a/drivers/pci/pwrctrl/Kconfig +++ b/drivers/pci/pwrctrl/Kconfig @@ -29,6 +29,7 @@ config PCI_PWRCTRL_TC9563 select PCI_PWRCTRL default m if ARCH_QCOM depends on I2C + depends on GPIO_TC9563 select REGMAP_I2C help Say Y here to enable the PCI Power Control driver of TC9563 PCIe diff --git a/drivers/pci/pwrctrl/pci-pwrctrl-tc9563.c b/drivers/pci/pwrctrl/pci-pwrctrl-tc9563.c index 59ad219c26c022..3fb862105fa588 100644 --- a/drivers/pci/pwrctrl/pci-pwrctrl-tc9563.c +++ b/drivers/pci/pwrctrl/pci-pwrctrl-tc9563.c @@ -4,12 +4,14 @@ */ #include +#include #include #include #include #include #include #include +#include #include #include #include @@ -18,6 +20,7 @@ #include #include #include +#include #include #include @@ -151,6 +154,8 @@ static const struct reg_sequence dsp2_pwroff_seq[] = { {TC9563_PORT_ACCESS_ENABLE, 0x8}, }; +static DEFINE_IDA(tc9563_pwrctrl_ida); + static int tc9563_pwrctrl_disable_port(struct tc9563_pwrctrl *tc9563, enum tc9563_pwrctrl_ports port) { @@ -393,6 +398,77 @@ static int tc9563_pwrctrl_parse_device_dt(struct device_node *node, return 0; } +static void tc9563_pwrctrl_adev_release(struct device *dev) +{ + struct auxiliary_device *adev = to_auxiliary_dev(dev); + + ida_free(&tc9563_pwrctrl_ida, adev->id); + of_node_put(adev->dev.of_node); + kfree(adev); +} + +static void tc9563_pwrctrl_adev_remove(void *data) +{ + struct auxiliary_device *adev = data; + + auxiliary_device_delete(adev); + auxiliary_device_uninit(adev); +} + +static int tc9563_pwrctrl_adev_add(struct device *dev, const char *name, + struct device_node *of_node, + void *priv_data) +{ + struct auxiliary_device *adev; + int id, ret; + + adev = kzalloc_obj(*adev); + if (!adev) + return -ENOMEM; + + id = ida_alloc(&tc9563_pwrctrl_ida, GFP_KERNEL); + if (id < 0) { + kfree(adev); + return id; + } + + adev->id = id; + adev->name = name; + adev->dev.parent = dev; + adev->dev.platform_data = priv_data; + adev->dev.release = tc9563_pwrctrl_adev_release; + adev->dev.of_node = of_node_get(of_node); + dev_set_of_node_reused(&adev->dev); + + ret = auxiliary_device_init(adev); + if (ret) { + ida_free(&tc9563_pwrctrl_ida, id); + of_node_put(adev->dev.of_node); + kfree(adev); + return ret; + } + + ret = auxiliary_device_add(adev); + if (ret) { + auxiliary_device_uninit(adev); + return ret; + } + + return devm_add_action_or_reset(dev, tc9563_pwrctrl_adev_remove, adev); +} + +static int tc9563_pwrctrl_add_gpio_adev(struct tc9563_pwrctrl *tc9563) +{ + struct device *dev = tc9563->pwrctrl.dev; + + if (!of_property_read_bool(dev->of_node, "gpio-controller") || + !of_property_present(dev->of_node, "#gpio-cells")) + return 0; + + return tc9563_pwrctrl_adev_add(dev, TC9563_GPIO_DEV_NAME, dev->of_node, + tc9563->regmap); +} + static int tc9563_pwrctrl_power_off(struct pci_pwrctrl *pwrctrl) { struct tc9563_pwrctrl *tc9563 = container_of(pwrctrl, @@ -596,6 +672,10 @@ static int tc9563_pwrctrl_probe(struct platform_device *pdev) tc9563->pwrctrl.power_on = tc9563_pwrctrl_power_on; tc9563->pwrctrl.power_off = tc9563_pwrctrl_power_off; + ret = tc9563_pwrctrl_add_gpio_adev(tc9563); + if (ret) + goto remove_i2c; + ret = devm_pci_pwrctrl_device_set_ready(dev, &tc9563->pwrctrl); if (ret) goto power_off; From c9ccc578d450dda93688f617046e43dc21ed0bcb Mon Sep 17 00:00:00 2001 From: Lorenzo Bianconi Date: Wed, 9 Sep 2026 16:55:56 +0200 Subject: [PATCH 0313/1352] PCI/pwrctrl: tc9563: Switch per-port reset to GPIO descriptor API Remove the local TC9563_GPIO_MASK and TC9563_GPIO_DEASSERT_BITS definitions, which are no longer used after switching to the GPIO descriptor API. Move TC9563_GPIO_CONFIG and TC9563_RESET_GPIO definitions in tc9563.h header file. Replace the direct regmap-based per-port reset logic in assert_deassert_reset() with gpiod_direction_output() calls, falling back to the legacy regmap approach only when no reset-gpios DT property is present for a given port. Add the reset GPIO pointer to struct tc9563_pwrctrl_cfg and introduce tc9563_pwrctrl_parse_reset_line() to look up reset-gpios from each PCI downstream port child node. The lookup is done lazily at the beginning of power_on(), returning -EPROBE_DEFER until the GPIO chip is registered. Signed-off-by: Lorenzo Bianconi Signed-off-by: Bjorn Helgaas Reviewed-by: Bartosz Golaszewski Reviewed-by: Manivannan Sadhasivam Reviewed-by: Alex Elder Link: https://patch.msgid.link/20260909-pci-tc9563-aux-v5-4-c9b33f56c8d3@oss.qualcomm.com --- drivers/pci/pwrctrl/pci-pwrctrl-tc9563.c | 85 +++++++++++++++++++----- include/linux/soc/qcom/tc9563.h | 3 + 2 files changed, 73 insertions(+), 15 deletions(-) diff --git a/drivers/pci/pwrctrl/pci-pwrctrl-tc9563.c b/drivers/pci/pwrctrl/pci-pwrctrl-tc9563.c index 3fb862105fa588..59bc0d77d3c444 100644 --- a/drivers/pci/pwrctrl/pci-pwrctrl-tc9563.c +++ b/drivers/pci/pwrctrl/pci-pwrctrl-tc9563.c @@ -26,9 +26,6 @@ #include "../pci.h" -#define TC9563_GPIO_CONFIG 0x801208 -#define TC9563_RESET_GPIO 0x801210 - #define TC9563_PORT_L0S_DELAY 0x82496c #define TC9563_PORT_L1_DELAY 0x824970 @@ -60,9 +57,6 @@ #define TC9563_POWER_CONTROL 0x82b09c #define TC9563_POWER_CONTROL_OVREN 0x82b2c8 -#define TC9563_GPIO_MASK 0xfffffff3 -#define TC9563_GPIO_DEASSERT_BITS 0xc /* Clear to deassert GPIO */ - #define TC9563_TX_MARGIN_MIN_UA 400000 /* @@ -88,6 +82,7 @@ struct tc9563_pwrctrl_cfg { u8 nfts[2]; /* GEN1 & GEN2 */ bool disable_dfe; bool disable_port; + struct gpio_desc *reset; }; #define TC9563_PWRCTL_MAX_SUPPLY 6 @@ -354,16 +349,40 @@ static int tc9563_pwrctrl_set_nfts(struct tc9563_pwrctrl *tc9563, static int tc9563_pwrctrl_assert_deassert_reset(struct tc9563_pwrctrl *tc9563, bool deassert) { - int ret, val; - - ret = regmap_write(tc9563->regmap, TC9563_GPIO_CONFIG, - TC9563_GPIO_MASK); - if (ret) - return ret; - - val = deassert ? TC9563_GPIO_DEASSERT_BITS : 0; + int i; + + for (i = 0; i < ARRAY_SIZE(tc9563->cfg); i++) { + int err; + + if (tc9563->cfg[i].reset) { + err = gpiod_direction_output(tc9563->cfg[i].reset, + !deassert); + if (err) + return err; + } else { + /* Fallback: legacy DTS without reset-gpios */ + switch (i) { + case TC9563_DSP1: + case TC9563_DSP2: + err = regmap_clear_bits(tc9563->regmap, + TC9563_GPIO_CONFIG, + BIT(i + 1)); + if (err) + return err; + + err = regmap_assign_bits(tc9563->regmap, + TC9563_RESET_GPIO, + BIT(i + 1), deassert); + if (err) + return err; + break; + default: + break; + } + } + } - return regmap_write(tc9563->regmap, TC9563_RESET_GPIO, val); + return 0; } static int tc9563_pwrctrl_parse_device_dt(struct device_node *node, @@ -398,6 +417,38 @@ static int tc9563_pwrctrl_parse_device_dt(struct device_node *node, return 0; } +static int tc9563_pwrctrl_parse_reset_line(struct tc9563_pwrctrl *tc9563) +{ + enum tc9563_pwrctrl_ports port = TC9563_USP; + struct device *dev = tc9563->pwrctrl.dev; + struct device_node *node = dev->of_node; + + for_each_child_of_node_scoped(node, child) { + struct tc9563_pwrctrl_cfg *cfg; + + if (++port >= TC9563_MAX) + break; + + cfg = &tc9563->cfg[port]; + if (cfg->reset) /* Already discovered */ + continue; + + cfg->reset = devm_fwnode_gpiod_get(dev, of_fwnode_handle(child), + "reset", GPIOD_ASIS, + NULL); + if (IS_ERR(cfg->reset)) { + int err = PTR_ERR(cfg->reset); + + cfg->reset = NULL; + if (err != -ENOENT) + return dev_err_probe(dev, err, + "failed to get reset\n"); + } + } + + return 0; +} + static void tc9563_pwrctrl_adev_release(struct device *dev) { struct auxiliary_device *adev = to_auxiliary_dev(dev); @@ -489,6 +540,10 @@ static int tc9563_pwrctrl_power_on(struct pci_pwrctrl *pwrctrl) struct tc9563_pwrctrl_cfg *cfg; int ret, i; + ret = tc9563_pwrctrl_parse_reset_line(tc9563); + if (ret) + return ret; + ret = regulator_bulk_enable(ARRAY_SIZE(tc9563->supplies), tc9563->supplies); if (ret < 0) diff --git a/include/linux/soc/qcom/tc9563.h b/include/linux/soc/qcom/tc9563.h index 0dfd25747b9af7..086f37a40d801d 100644 --- a/include/linux/soc/qcom/tc9563.h +++ b/include/linux/soc/qcom/tc9563.h @@ -13,4 +13,7 @@ #define TC9563_GPIO_EN0_OFFSET 0x801208 #define TC9563_GPIO_OUT0_OFFSET 0x801210 +#define TC9563_GPIO_CONFIG TC9563_GPIO_EN0_OFFSET +#define TC9563_RESET_GPIO TC9563_GPIO_OUT0_OFFSET + #endif /* __QCOM_TC9563_H */ From 708e96caa199c863b5601f30a0aeaad7fbc53b0f Mon Sep 17 00:00:00 2001 From: Lorenzo Bianconi Date: Wed, 9 Sep 2026 16:55:57 +0200 Subject: [PATCH 0314/1352] arm64: dts: qcom: qcs6490-rb3gen2: Enable TC9563 embedded GPIO controller The Toshiba TC9563 PCIe switch embeds a GPIO controller providing 37 GPIO lines. The controller is registered as an auxiliary device by the TC9563 power controller driver and accessed through the same i2c device. Describe the switch node itself as the embedded GPIO controller and use it to drive the PERST# reset lines of the two external downstream ports (pcie@1,0 and pcie@2,0), as expected by the TC9563 power controller after switching to the GPIO descriptor API for per-port resets. Signed-off-by: Lorenzo Bianconi Signed-off-by: Bjorn Helgaas Reviewed-by: Konrad Dybcio Reviewed-by: Bartosz Golaszewski Reviewed-by: Abel Vesa Reviewed-by: Alex Elder Reviewed-by: Manivannan Sadhasivam Reviewed-by: Alex Elder > --- Link: https://patch.msgid.link/20260909-pci-tc9563-aux-v5-5-c9b33f56c8d3@oss.qualcomm.com --- arch/arm64/boot/dts/qcom/qcs6490-rb3gen2.dts | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/arch/arm64/boot/dts/qcom/qcs6490-rb3gen2.dts b/arch/arm64/boot/dts/qcom/qcs6490-rb3gen2.dts index a13315bf0fb078..4da7342ca90d6e 100644 --- a/arch/arm64/boot/dts/qcom/qcs6490-rb3gen2.dts +++ b/arch/arm64/boot/dts/qcom/qcs6490-rb3gen2.dts @@ -867,7 +867,7 @@ &pcie1_port0 { reset-gpios = <&tlmm 2 GPIO_ACTIVE_LOW>; - pcie@0,0 { + tc9563: pcie@0,0 { compatible = "pci1179,0623"; reg = <0x10000 0x0 0x0 0x0 0x0>; #address-cells = <3>; @@ -891,6 +891,9 @@ pinctrl-0 = <&tc9563_resx_n>; pinctrl-names = "default"; + gpio-controller; + #gpio-cells = <2>; + pcie1_switch0_dsp1: pcie@1,0 { reg = <0x20800 0x0 0x0 0x0 0x0>; #address-cells = <3>; @@ -899,6 +902,7 @@ device_type = "pci"; ranges; bus-range = <0x3 0xff>; + reset-gpios = <&tc9563 2 GPIO_ACTIVE_LOW>; }; pcie@2,0 { @@ -909,6 +913,7 @@ device_type = "pci"; ranges; bus-range = <0x4 0xff>; + reset-gpios = <&tc9563 3 GPIO_ACTIVE_LOW>; /* Renesas μPD720201 PCIe USB3.0 Host Controller */ usb-controller@0,0 { From 9bafc322b7fca03717db78615b07bd401367b7c2 Mon Sep 17 00:00:00 2001 From: Scott Mayhew Date: Tue, 22 Sep 2026 14:24:36 -0400 Subject: [PATCH 0315/1352] nfs: split up block layout and SCSI layout support Add two new config options PNFS_BLOCK_LAYOUT and PNFS_SCSI_LAYOUT so that SCSI layouts can be enabled without requiring block layouts. Since block layouts are considered deprecated, PNFS_BLOCK_LAYOUT is off by default. The original PNFS_BLOCK config is now invisible and gets selected when either of PNFS_BLOCK_LAYOUT or PNFS_SCSI_LAYOUT are enabled. Also added a dependency on BLOCK to the Kconfig to fix undefined symbol warnings from the kernel test robot. Signed-off-by: Scott Mayhew Reviewed-by: Christoph Hellwig Signed-off-by: Anna Schumaker --- fs/nfs/Kconfig | 17 ++++- fs/nfs/blocklayout/Makefile | 3 +- fs/nfs/blocklayout/blocklayout.c | 113 ++++++++++++++++++++++--------- fs/nfs/blocklayout/dev.c | 32 ++++++++- 4 files changed, 129 insertions(+), 36 deletions(-) diff --git a/fs/nfs/Kconfig b/fs/nfs/Kconfig index 6bb30543eff00f..64c249f800a966 100644 --- a/fs/nfs/Kconfig +++ b/fs/nfs/Kconfig @@ -123,8 +123,23 @@ config PNFS_FILE_LAYOUT config PNFS_BLOCK tristate - depends on NFS_V4 && BLK_DEV_DM + +config PNFS_BLOCK_LAYOUT + bool "NFS client support for pNFS block layouts" + depends on NFS_V4 && BLOCK && BLK_DEV_DM + select PNFS_BLOCK + help + Enable support for the pNFS block-volume layout type (RFC 5663). + + If unsure, say N. + +config PNFS_SCSI_LAYOUT + bool "NFS client support for pNFS SCSI layouts" + depends on NFS_V4 && BLOCK default NFS_V4 + select PNFS_BLOCK + help + Enable suport for the pNFS SCSI layout type (RFC 8154). config PNFS_FLEXFILE_LAYOUT tristate diff --git a/fs/nfs/blocklayout/Makefile b/fs/nfs/blocklayout/Makefile index 7668a1bfb5fa57..3403cb7fe201c1 100644 --- a/fs/nfs/blocklayout/Makefile +++ b/fs/nfs/blocklayout/Makefile @@ -4,4 +4,5 @@ # obj-$(CONFIG_PNFS_BLOCK) += blocklayoutdriver.o -blocklayoutdriver-y += blocklayout.o dev.o extent_tree.o rpc_pipefs.o +blocklayoutdriver-y += blocklayout.o dev.o extent_tree.o +blocklayoutdriver-$(CONFIG_PNFS_BLOCK_LAYOUT) += rpc_pipefs.o diff --git a/fs/nfs/blocklayout/blocklayout.c b/fs/nfs/blocklayout/blocklayout.c index d86702e604f951..82873ff370ef68 100644 --- a/fs/nfs/blocklayout/blocklayout.c +++ b/fs/nfs/blocklayout/blocklayout.c @@ -470,18 +470,6 @@ static struct pnfs_layout_hdr *__bl_alloc_layout_hdr(struct inode *inode, return &bl->bl_layout; } -static struct pnfs_layout_hdr *bl_alloc_layout_hdr(struct inode *inode, - gfp_t gfp_flags) -{ - return __bl_alloc_layout_hdr(inode, gfp_flags, false); -} - -static struct pnfs_layout_hdr *sl_alloc_layout_hdr(struct inode *inode, - gfp_t gfp_flags) -{ - return __bl_alloc_layout_hdr(inode, gfp_flags, true); -} - static void bl_free_lseg(struct pnfs_layout_segment *lseg) { dprintk("%s enter\n", __func__); @@ -956,6 +944,13 @@ static const struct nfs_pageio_ops bl_pg_write_ops = { .pg_cleanup = pnfs_generic_pg_cleanup, }; +#ifdef CONFIG_PNFS_BLOCK_LAYOUT +static struct pnfs_layout_hdr *bl_alloc_layout_hdr(struct inode *inode, + gfp_t gfp_flags) +{ + return __bl_alloc_layout_hdr(inode, gfp_flags, false); +} + static struct pnfs_layoutdriver_type blocklayout_type = { .id = LAYOUT_BLOCK_VOLUME, .name = "LAYOUT_BLOCK_VOLUME", @@ -980,6 +975,48 @@ static struct pnfs_layoutdriver_type blocklayout_type = { .sync = pnfs_generic_sync, }; +static int __init pnfs_register_blocklayout(void) +{ + int ret; + + ret = bl_init_pipefs(); + if (ret) + return ret; + + ret = pnfs_register_layoutdriver(&blocklayout_type); + if (ret) { + bl_cleanup_pipefs(); + return ret; + } + + return 0; +} + +static void __exit pnfs_unregister_blocklayout(void) +{ + pnfs_unregister_layoutdriver(&blocklayout_type); + bl_cleanup_pipefs(); +} + +MODULE_ALIAS("nfs-layouttype4-3"); +#else +static int __init pnfs_register_blocklayout(void) +{ + return 0; +} + +static void __exit pnfs_unregister_blocklayout(void) +{ +} +#endif /* CONFIG_PNFS_BLOCK_LAYOUT */ + +#ifdef CONFIG_PNFS_SCSI_LAYOUT +static struct pnfs_layout_hdr *sl_alloc_layout_hdr(struct inode *inode, + gfp_t gfp_flags) +{ + return __bl_alloc_layout_hdr(inode, gfp_flags, true); +} + static struct pnfs_layoutdriver_type scsilayout_type = { .id = LAYOUT_SCSI, .name = "LAYOUT_SCSI", @@ -1004,6 +1041,28 @@ static struct pnfs_layoutdriver_type scsilayout_type = { .sync = pnfs_generic_sync, }; +static int __init pnfs_register_scsilayout(void) +{ + return pnfs_register_layoutdriver(&scsilayout_type); +} + +static void __exit pnfs_unregister_scsilayout(void) +{ + pnfs_unregister_layoutdriver(&scsilayout_type); +} + +MODULE_ALIAS("nfs-layouttype4-5"); +#else +static int __init pnfs_register_scsilayout(void) +{ + return 0; +} + +static void __exit pnfs_unregister_scsilayout(void) +{ +} +#endif /* CONFIG_PNFS_SCSI_LAYOUT */ + static int __init nfs4blocklayout_init(void) { @@ -1011,25 +1070,17 @@ static int __init nfs4blocklayout_init(void) dprintk("%s: NFSv4 Block Layout Driver Registering...\n", __func__); - ret = bl_init_pipefs(); + ret = pnfs_register_blocklayout(); if (ret) - goto out; + return ret; - ret = pnfs_register_layoutdriver(&blocklayout_type); - if (ret) - goto out_cleanup_pipe; + ret = pnfs_register_scsilayout(); + if (ret) { + pnfs_unregister_blocklayout(); + return ret; + } - ret = pnfs_register_layoutdriver(&scsilayout_type); - if (ret) - goto out_unregister_block; return 0; - -out_unregister_block: - pnfs_unregister_layoutdriver(&blocklayout_type); -out_cleanup_pipe: - bl_cleanup_pipefs(); -out: - return ret; } static void __exit nfs4blocklayout_exit(void) @@ -1037,13 +1088,9 @@ static void __exit nfs4blocklayout_exit(void) dprintk("%s: NFSv4 Block Layout Driver Unregistering...\n", __func__); - pnfs_unregister_layoutdriver(&scsilayout_type); - pnfs_unregister_layoutdriver(&blocklayout_type); - bl_cleanup_pipefs(); + pnfs_unregister_scsilayout(); + pnfs_unregister_blocklayout(); } -MODULE_ALIAS("nfs-layouttype4-3"); -MODULE_ALIAS("nfs-layouttype4-5"); - module_init(nfs4blocklayout_init); module_exit(nfs4blocklayout_exit); diff --git a/fs/nfs/blocklayout/dev.c b/fs/nfs/blocklayout/dev.c index c926b7e4382791..00462e6affb12f 100644 --- a/fs/nfs/blocklayout/dev.c +++ b/fs/nfs/blocklayout/dev.c @@ -15,6 +15,7 @@ #define NFSDBG_FACILITY NFSDBG_PNFS_LD +#ifdef CONFIG_PNFS_SCSI_LAYOUT static void bl_unregister_scsi(struct pnfs_block_dev *dev) { struct block_device *bdev = file_bdev(dev->bdev_file); @@ -45,6 +46,16 @@ static bool bl_register_scsi(struct pnfs_block_dev *dev) trace_bl_pr_key_reg(bdev, dev->pr_key); return true; } +#else +static void bl_unregister_scsi(struct pnfs_block_dev *dev) +{ +} + +static bool bl_register_scsi(struct pnfs_block_dev *dev) +{ + return false; +} +#endif /* CONFIG_PNFS_SCSI_LAYOUT */ static void bl_unregister_dev(struct pnfs_block_dev *dev) { @@ -292,7 +303,7 @@ static int bl_parse_deviceid(struct nfs_server *server, struct pnfs_block_dev *d, struct pnfs_block_volume *volumes, int idx, gfp_t gfp_mask); - +#ifdef CONFIG_PNFS_BLOCK_LAYOUT static int bl_parse_simple(struct nfs_server *server, struct pnfs_block_dev *d, struct pnfs_block_volume *volumes, int idx, gfp_t gfp_mask) @@ -320,7 +331,17 @@ bl_parse_simple(struct nfs_server *server, struct pnfs_block_dev *d, file_bdev(bdev_file)->bd_disk->disk_name); return 0; } +#else +static int +bl_parse_simple(struct nfs_server *server, struct pnfs_block_dev *d, + struct pnfs_block_volume *volumes, int idx, gfp_t gfp_mask) +{ + dprintk("unsupported volume type: %d\n", PNFS_BLOCK_VOLUME_SIMPLE); + return -EIO; +} +#endif /* CONFIG_PNFS_BLOCK_LAYOUT */ +#ifdef CONFIG_PNFS_SCSI_LAYOUT static bool bl_validate_designator(struct pnfs_block_volume *v) { @@ -449,6 +470,15 @@ bl_parse_scsi(struct nfs_server *server, struct pnfs_block_dev *d, d->bdev_file = NULL; return error; } +#else +static int +bl_parse_scsi(struct nfs_server *server, struct pnfs_block_dev *d, + struct pnfs_block_volume *volumes, int idx, gfp_t gfp_mask) +{ + dprintk("unsupported volume type: %d\n", PNFS_BLOCK_VOLUME_SCSI); + return -EIO; +} +#endif /* CONFIG_PNFS_SCSI_LAYOUT */ static int bl_parse_slice(struct nfs_server *server, struct pnfs_block_dev *d, From 17727f634433a4d1a7bc1764b90fa16eec860422 Mon Sep 17 00:00:00 2001 From: Suraj Kandpal Date: Tue, 22 Sep 2026 17:43:16 +0530 Subject: [PATCH 0316/1352] drm/i915/lspcon: log DPCD access errors Now that the DPCD accessors return a proper error code, include it in the error messages using %pe and ERR_PTR(), instead of just stating that the access failed. While at it, fix the AVI IF control write error message, which incorrectly reported a failed read. Signed-off-by: Suraj Kandpal Reviewed-by: Nemesa Garg Link: https://patch.msgid.link/20260922121316.1768918-1-suraj.kandpal@intel.com --- drivers/gpu/drm/i915/display/intel_lspcon.c | 34 +++++++++++++-------- 1 file changed, 21 insertions(+), 13 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_lspcon.c b/drivers/gpu/drm/i915/display/intel_lspcon.c index 07b5a9d010ff40..a61224523318c7 100644 --- a/drivers/gpu/drm/i915/display/intel_lspcon.c +++ b/drivers/gpu/drm/i915/display/intel_lspcon.c @@ -139,7 +139,8 @@ bool intel_lspcon_detect_hdr_capability(struct intel_digital_port *dig_port) ret = drm_dp_dpcd_read_byte(&intel_dp->aux, get_hdr_status_reg(lspcon), &hdr_caps); if (ret < 0) { - drm_dbg_kms(display->drm, "HDR capability detection failed\n"); + drm_dbg_kms(display->drm, "HDR capability detection failed (%pe)\n", + ERR_PTR(ret)); lspcon->hdr_supported = false; } else if (hdr_caps & 0x1) { drm_dbg_kms(display->drm, "LSPCON capable of HDR\n"); @@ -247,7 +248,7 @@ static bool lspcon_wake_native_aux_ch(struct intel_lspcon *lspcon) ret = drm_dp_dpcd_read_byte(&lspcon_to_intel_dp(lspcon)->aux, DP_DPCD_REV, &rev); if (ret < 0) { - drm_dbg_kms(display->drm, "Native AUX CH down\n"); + drm_dbg_kms(display->drm, "Native AUX CH down (%pe)\n", ERR_PTR(ret)); return false; } @@ -339,7 +340,8 @@ static bool lspcon_parade_fw_ready(struct drm_dp_aux *aux) ret = drm_dp_dpcd_read_byte(aux, LSPCON_PARADE_AVI_IF_CTRL, &avi_if_ctrl); if (ret < 0) { - drm_err(aux->drm_dev, "Failed to read AVI IF control\n"); + drm_err(aux->drm_dev, "Failed to read AVI IF control (%pe)\n", + ERR_PTR(ret)); return false; } @@ -371,8 +373,8 @@ static bool _lspcon_parade_write_infoframe_blocks(struct drm_dp_aux *aux, data = avi_buf + block_count * 8; ret = drm_dp_dpcd_write_data(aux, reg, data, 8); if (ret < 0) { - drm_err(aux->drm_dev, "Failed to write AVI IF block %d\n", - block_count); + drm_err(aux->drm_dev, "Failed to write AVI IF block %d (%pe)\n", + block_count, ERR_PTR(ret)); return false; } @@ -386,8 +388,8 @@ static bool _lspcon_parade_write_infoframe_blocks(struct drm_dp_aux *aux, avi_if_ctrl = LSPCON_PARADE_AVI_IF_KICKOFF | block_count; ret = drm_dp_dpcd_write_byte(aux, reg, avi_if_ctrl); if (ret < 0) { - drm_err(aux->drm_dev, "Failed to update (0x%x), block %d\n", - reg, block_count); + drm_err(aux->drm_dev, "Failed to update (0x%x), block %d (%pe)\n", + reg, block_count, ERR_PTR(ret)); return false; } @@ -450,7 +452,8 @@ static bool _lspcon_write_avi_infoframe_mca(struct drm_dp_aux *aux, mdelay(50); continue; } else { - drm_err(aux->drm_dev, "DPCD write failed at:0x%x\n", reg); + drm_err(aux->drm_dev, "DPCD write failed at:0x%x (%pe)\n", + reg, ERR_PTR(ret)); return false; } } @@ -462,7 +465,8 @@ static bool _lspcon_write_avi_infoframe_mca(struct drm_dp_aux *aux, reg = LSPCON_MCA_AVI_IF_CTRL; ret = drm_dp_dpcd_read_byte(aux, reg, &val); if (ret < 0) { - drm_err(aux->drm_dev, "DPCD read failed, address 0x%x\n", reg); + drm_err(aux->drm_dev, "DPCD read failed, address 0x%x (%pe)\n", + reg, ERR_PTR(ret)); return false; } @@ -472,13 +476,15 @@ static bool _lspcon_write_avi_infoframe_mca(struct drm_dp_aux *aux, ret = drm_dp_dpcd_write_byte(aux, reg, val); if (ret < 0) { - drm_err(aux->drm_dev, "DPCD read failed, address 0x%x\n", reg); + drm_err(aux->drm_dev, "DPCD write failed at:0x%x (%pe)\n", + reg, ERR_PTR(ret)); return false; } ret = drm_dp_dpcd_read_byte(aux, reg, &val); if (ret < 0) { - drm_err(aux->drm_dev, "DPCD read failed, address 0x%x\n", reg); + drm_err(aux->drm_dev, "DPCD read failed, address 0x%x (%pe)\n", + reg, ERR_PTR(ret)); return false; } @@ -615,7 +621,8 @@ static bool _lspcon_read_avi_infoframe_enabled_mca(struct drm_dp_aux *aux) ret = drm_dp_dpcd_read_byte(aux, reg, &val); if (ret < 0) { - drm_err(aux->drm_dev, "DPCD read failed, address 0x%x\n", reg); + drm_err(aux->drm_dev, "DPCD read failed, address 0x%x (%pe)\n", + reg, ERR_PTR(ret)); return false; } @@ -630,7 +637,8 @@ static bool _lspcon_read_avi_infoframe_enabled_parade(struct drm_dp_aux *aux) ret = drm_dp_dpcd_read_byte(aux, reg, &val); if (ret < 0) { - drm_err(aux->drm_dev, "DPCD read failed, address 0x%x\n", reg); + drm_err(aux->drm_dev, "DPCD read failed, address 0x%x (%pe)\n", + reg, ERR_PTR(ret)); return false; } From ebaa19c84f24f2fce51272d7f107ee704918f16b Mon Sep 17 00:00:00 2001 From: Suraj Kandpal Date: Mon, 21 Sep 2026 11:07:04 +0530 Subject: [PATCH 0317/1352] drm/i915/alpm: log DPCD access errors Now that the DPCD accessors return a proper error code, include it in the error messages using %pe and ERR_PTR(), instead of just stating that the access failed. Signed-off-by: Suraj Kandpal Reviewed-by: Nemesa Garg Link: https://patch.msgid.link/20260921053710.1586776-3-suraj.kandpal@intel.com --- drivers/gpu/drm/i915/display/intel_alpm.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/gpu/drm/i915/display/intel_alpm.c b/drivers/gpu/drm/i915/display/intel_alpm.c index 9ef9d0e45258fd..f54b29e193ee93 100644 --- a/drivers/gpu/drm/i915/display/intel_alpm.c +++ b/drivers/gpu/drm/i915/display/intel_alpm.c @@ -756,7 +756,7 @@ bool intel_alpm_get_error(struct intel_dp *intel_dp) ret = drm_dp_dpcd_read_byte(aux, DP_RECEIVER_ALPM_STATUS, &val); if (ret < 0) { - drm_err(display->drm, "Error reading ALPM status\n"); + drm_err(display->drm, "Error reading ALPM status (%pe)\n", ERR_PTR(ret)); return true; } From a27ef41465cb728931731e9f0bea568c198863f0 Mon Sep 17 00:00:00 2001 From: Suraj Kandpal Date: Mon, 21 Sep 2026 11:07:05 +0530 Subject: [PATCH 0318/1352] drm/i915/ddi: log DPCD access errors Now that the DPCD accessors return a proper error code, include it in the error messages using %pe and ERR_PTR(), instead of just stating that the access failed. Store the FEC_STATUS write result in ret as well, so that it can be logged the same way. Signed-off-by: Suraj Kandpal Reviewed-by: Nemesa Garg Link: https://patch.msgid.link/20260921053710.1586776-4-suraj.kandpal@intel.com --- drivers/gpu/drm/i915/display/intel_ddi.c | 20 ++++++++++++-------- 1 file changed, 12 insertions(+), 8 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_ddi.c b/drivers/gpu/drm/i915/display/intel_ddi.c index 7efd4c4d499444..a46ab5d3d70fc7 100644 --- a/drivers/gpu/drm/i915/display/intel_ddi.c +++ b/drivers/gpu/drm/i915/display/intel_ddi.c @@ -2341,8 +2341,8 @@ static void intel_dp_sink_set_msa_timing_par_ignore_state(struct intel_dp *intel enable ? DP_MSA_TIMING_PAR_IGNORE_EN : 0); if (ret < 0) drm_dbg_kms(display->drm, - "Failed to %s MSA_TIMING_PAR_IGNORE in the sink\n", - str_enable_disable(enable)); + "Failed to %s MSA_TIMING_PAR_IGNORE in the sink (%pe)\n", + str_enable_disable(enable), ERR_PTR(ret)); } static void intel_dp_sink_set_fec_ready(struct intel_dp *intel_dp, @@ -2358,13 +2358,17 @@ static void intel_dp_sink_set_fec_ready(struct intel_dp *intel_dp, ret = drm_dp_dpcd_write_byte(&intel_dp->aux, DP_FEC_CONFIGURATION, enable ? DP_FEC_READY : 0); if (ret < 0) - drm_dbg_kms(display->drm, "Failed to set FEC_READY to %s in the sink\n", - str_enabled_disabled(enable)); + drm_dbg_kms(display->drm, "Failed to set FEC_READY to %s in the sink (%pe)\n", + str_enabled_disabled(enable), ERR_PTR(ret)); - if (enable && - drm_dp_dpcd_write_byte(&intel_dp->aux, DP_FEC_STATUS, - DP_FEC_DECODE_EN_DETECTED | DP_FEC_DECODE_DIS_DETECTED) < 0) - drm_dbg_kms(display->drm, "Failed to clear FEC detected flags\n"); + if (enable) { + ret = drm_dp_dpcd_write_byte(&intel_dp->aux, DP_FEC_STATUS, + DP_FEC_DECODE_EN_DETECTED | + DP_FEC_DECODE_DIS_DETECTED); + if (ret < 0) + drm_dbg_kms(display->drm, "Failed to clear FEC detected flags (%pe)\n", + ERR_PTR(ret)); + } } static int wait_for_fec_detected(struct drm_dp_aux *aux, bool enabled) From 07d52ff1f649dea1298925021450ac000c3f52fc Mon Sep 17 00:00:00 2001 From: Suraj Kandpal Date: Mon, 21 Sep 2026 11:07:06 +0530 Subject: [PATCH 0319/1352] drm/i915/backlight: log DPCD access errors Now that the DPCD accessors return a proper error code, include it in the error messages using %pe and ERR_PTR(), instead of just stating that the access failed. Store the brightness control write result in ret as well, so that it can be logged the same way. Signed-off-by: Suraj Kandpal Reviewed-by: Nemesa Garg Link: https://patch.msgid.link/20260921053710.1586776-5-suraj.kandpal@intel.com --- .../drm/i915/display/intel_dp_aux_backlight.c | 29 ++++++++++--------- 1 file changed, 16 insertions(+), 13 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_dp_aux_backlight.c b/drivers/gpu/drm/i915/display/intel_dp_aux_backlight.c index f30da5f7e3d2ee..36aebc32971b77 100644 --- a/drivers/gpu/drm/i915/display/intel_dp_aux_backlight.c +++ b/drivers/gpu/drm/i915/display/intel_dp_aux_backlight.c @@ -183,8 +183,8 @@ intel_dp_aux_hdr_get_backlight(struct intel_connector *connector, enum pipe pipe ret = drm_dp_dpcd_read_byte(&intel_dp->aux, INTEL_EDP_HDR_GETSET_CTRL_PARAMS, &tmp); if (ret < 0) { drm_err(display->drm, - "[CONNECTOR:%d:%s] Failed to read current backlight mode from DPCD\n", - connector->base.base.id, connector->base.name); + "[CONNECTOR:%d:%s] Failed to read current backlight mode from DPCD (%pe)\n", + connector->base.base.id, connector->base.name, ERR_PTR(ret)); return 0; } @@ -203,8 +203,8 @@ intel_dp_aux_hdr_get_backlight(struct intel_connector *connector, enum pipe pipe sizeof(buf)); if (ret < 0) { drm_err(display->drm, - "[CONNECTOR:%d:%s] Failed to read brightness from DPCD\n", - connector->base.base.id, connector->base.name); + "[CONNECTOR:%d:%s] Failed to read brightness from DPCD (%pe)\n", + connector->base.base.id, connector->base.name, ERR_PTR(ret)); return 0; } @@ -226,8 +226,8 @@ intel_dp_aux_hdr_set_aux_backlight(const struct drm_connector_state *conn_state, ret = drm_dp_dpcd_write_data(&intel_dp->aux, INTEL_EDP_BRIGHTNESS_NITS_LSB, buf, sizeof(buf)); if (ret < 0) - drm_err(dev, "[CONNECTOR:%d:%s] Failed to write brightness level to DPCD\n", - connector->base.base.id, connector->base.name); + drm_err(dev, "[CONNECTOR:%d:%s] Failed to write brightness level to DPCD (%pe)\n", + connector->base.base.id, connector->base.name, ERR_PTR(ret)); } static void @@ -341,11 +341,14 @@ intel_dp_aux_hdr_enable_backlight(const struct intel_crtc_state *crtc_state, intel_dp_aux_fill_hdr_tcon_params(conn_state, &ctrl); - if (ctrl != old_ctrl && - drm_dp_dpcd_write_byte(&intel_dp->aux, INTEL_EDP_HDR_GETSET_CTRL_PARAMS, ctrl) < 0) - drm_err(display->drm, - "[CONNECTOR:%d:%s] Failed to configure DPCD brightness controls\n", - connector->base.base.id, connector->base.name); + if (ctrl != old_ctrl) { + ret = drm_dp_dpcd_write_byte(&intel_dp->aux, + INTEL_EDP_HDR_GETSET_CTRL_PARAMS, ctrl); + if (ret < 0) + drm_err(display->drm, + "[CONNECTOR:%d:%s] Failed to configure DPCD brightness controls (%pe)\n", + connector->base.base.id, connector->base.name, ERR_PTR(ret)); + } if (intel_dp_in_hdr_mode(conn_state)) { hdr_metadata = conn_state->hdr_output_metadata->data; @@ -463,8 +466,8 @@ static u32 intel_dp_aux_vesa_get_backlight(struct intel_connector *connector, en sizeof(buf)); if (ret < 0) { drm_err(intel_dp->aux.drm_dev, - "[CONNECTOR:%d:%s] Failed to read Luminance from DPCD\n", - connector->base.base.id, connector->base.name); + "[CONNECTOR:%d:%s] Failed to read Luminance from DPCD (%pe)\n", + connector->base.base.id, connector->base.name, ERR_PTR(ret)); return 0; } From 6591a9bf2586f3351fafd0d3a412fc0d480a9f81 Mon Sep 17 00:00:00 2001 From: Suraj Kandpal Date: Mon, 21 Sep 2026 11:07:07 +0530 Subject: [PATCH 0320/1352] drm/i915/dp: log DPCD access errors Now that the DPCD accessors return a proper error code, include it in the error messages using %pe and ERR_PTR(), instead of just stating that the access failed. Signed-off-by: Suraj Kandpal Reviewed-by: Nemesa Garg Link: https://patch.msgid.link/20260921053710.1586776-6-suraj.kandpal@intel.com --- drivers/gpu/drm/i915/display/intel_dp.c | 31 ++++++++++++++----------- 1 file changed, 17 insertions(+), 14 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_dp.c b/drivers/gpu/drm/i915/display/intel_dp.c index ffddf4b3372852..d8c0d9381471d1 100644 --- a/drivers/gpu/drm/i915/display/intel_dp.c +++ b/drivers/gpu/drm/i915/display/intel_dp.c @@ -3925,7 +3925,7 @@ intel_dp_init_source_oui(struct intel_dp *intel_dp) */ ret = drm_dp_dpcd_read_data(&intel_dp->aux, DP_SOURCE_OUI, buf, sizeof(buf)); if (ret < 0) - drm_dbg_kms(display->drm, "Failed to read source OUI\n"); + drm_dbg_kms(display->drm, "Failed to read source OUI (%pe)\n", ERR_PTR(ret)); if (memcmp(oui, buf, sizeof(oui)) == 0) { /* Assume the OUI was written now. */ @@ -3935,7 +3935,7 @@ intel_dp_init_source_oui(struct intel_dp *intel_dp) ret = drm_dp_dpcd_write_data(&intel_dp->aux, DP_SOURCE_OUI, oui, sizeof(oui)); if (ret < 0) { - drm_dbg_kms(display->drm, "Failed to write source OUI\n"); + drm_dbg_kms(display->drm, "Failed to write source OUI (%pe)\n", ERR_PTR(ret)); WRITE_ONCE(intel_dp->oui_valid, false); } @@ -4002,9 +4002,9 @@ void intel_dp_set_power(struct intel_dp *intel_dp, u8 mode) if (ret < 0) drm_dbg_kms(display->drm, - "[ENCODER:%d:%s] Set power to %s failed\n", + "[ENCODER:%d:%s] Set power to %s failed (%pe)\n", encoder->base.base.id, encoder->base.name, - mode == DP_SET_POWER_D0 ? "D0" : "D3"); + mode == DP_SET_POWER_D0 ? "D0" : "D3", ERR_PTR(ret)); } static bool @@ -4104,8 +4104,8 @@ static void intel_dp_get_pcon_dsc_cap(struct intel_dp *intel_dp) intel_dp->pcon_dsc_dpcd, sizeof(intel_dp->pcon_dsc_dpcd)); if (ret < 0) - drm_err(display->drm, "Failed to read DPCD register 0x%x\n", - DP_PCON_DSC_ENCODER); + drm_err(display->drm, "Failed to read DPCD register 0x%x (%pe)\n", + DP_PCON_DSC_ENCODER, ERR_PTR(ret)); drm_dbg_kms(display->drm, "PCON ENCODER DSC DPCD: %*ph\n", (int)sizeof(intel_dp->pcon_dsc_dpcd), intel_dp->pcon_dsc_dpcd); @@ -4428,8 +4428,9 @@ void intel_dp_configure_protocol_converter(struct intel_dp *intel_dp, DP_PROTOCOL_CONVERTER_CONTROL_0, tmp); if (ret < 0) drm_dbg_kms(display->drm, - "Failed to %s protocol converter HDMI mode\n", - str_enable_disable(intel_dp_has_hdmi_sink(intel_dp))); + "Failed to %s protocol converter HDMI mode (%pe)\n", + str_enable_disable(intel_dp_has_hdmi_sink(intel_dp)), + ERR_PTR(ret)); if (crtc_state->sink_format == INTEL_OUTPUT_FORMAT_YCBCR420) { switch (crtc_state->output_format) { @@ -4465,16 +4466,17 @@ void intel_dp_configure_protocol_converter(struct intel_dp *intel_dp, DP_PROTOCOL_CONVERTER_CONTROL_1, tmp); if (ret < 0) drm_dbg_kms(display->drm, - "Failed to %s protocol converter YCbCr 4:2:0 conversion mode\n", - str_enable_disable(intel_dp->dfp.ycbcr_444_to_420)); + "Failed to %s protocol converter YCbCr 4:2:0 conversion mode (%pe)\n", + str_enable_disable(intel_dp->dfp.ycbcr_444_to_420), + ERR_PTR(ret)); tmp = rgb_to_ycbcr ? DP_CONVERSION_BT709_RGB_YCBCR_ENABLE : 0; ret = drm_dp_pcon_convert_rgb_to_ycbcr(&intel_dp->aux, tmp); if (ret < 0) drm_dbg_kms(display->drm, - "Failed to %s protocol converter RGB->YCbCr conversion mode\n", - str_enable_disable(tmp)); + "Failed to %s protocol converter RGB->YCbCr conversion mode (%pe)\n", + str_enable_disable(tmp), ERR_PTR(ret)); } static u8 intel_dp_read_dprx_feature_enum(struct intel_dp *intel_dp) @@ -4573,7 +4575,8 @@ void intel_dp_get_dsc_sink_cap(u8 dpcd_rev, ret = drm_dp_dpcd_read_byte(connector->dp.dsc_decompression_aux, DP_FEC_CAPABILITY, &connector->dp.fec_capability); if (ret < 0) { - drm_dbg_kms(display->drm, "Could not read FEC DPCD register\n"); + drm_dbg_kms(display->drm, "Could not read FEC DPCD register (%pe)\n", + ERR_PTR(ret)); return; } @@ -4690,7 +4693,7 @@ static void intel_edp_mso_init(struct intel_dp *intel_dp) ret = drm_dp_dpcd_read_byte(&intel_dp->aux, DP_EDP_MSO_LINK_CAPABILITIES, &mso); if (ret < 0) { - drm_err(display->drm, "Failed to read MSO cap\n"); + drm_err(display->drm, "Failed to read MSO cap (%pe)\n", ERR_PTR(ret)); return; } From 1d9d3bc0dad332361e1363df50b058634dcd232d Mon Sep 17 00:00:00 2001 From: Suraj Kandpal Date: Mon, 21 Sep 2026 11:07:08 +0530 Subject: [PATCH 0321/1352] drm/i915/linktraining: log link training DPCD access errors Now that the DPCD accessors return a proper error code, include it in the error messages using %pe and ERR_PTR(), instead of just stating that the access failed. Signed-off-by: Suraj Kandpal Reviewed-by: Nemesa Garg Link: https://patch.msgid.link/20260921053710.1586776-7-suraj.kandpal@intel.com --- drivers/gpu/drm/i915/display/intel_dp_link_training.c | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_dp_link_training.c b/drivers/gpu/drm/i915/display/intel_dp_link_training.c index b6e8b13ee4dbd0..ac1aff6edf0528 100644 --- a/drivers/gpu/drm/i915/display/intel_dp_link_training.c +++ b/drivers/gpu/drm/i915/display/intel_dp_link_training.c @@ -1635,7 +1635,8 @@ intel_dp_128b132b_intra_hop(struct intel_dp *intel_dp, ret = drm_dp_dpcd_read_byte(&intel_dp->aux, DP_SINK_STATUS, &sink_status); if (ret < 0) { - lt_dbg(intel_dp, DP_PHY_DPRX, "Failed to read sink status\n"); + lt_dbg(intel_dp, DP_PHY_DPRX, "Failed to read sink status (%pe)\n", + ERR_PTR(ret)); return ret; } @@ -2180,7 +2181,8 @@ intel_dp_128b132b_lane_cds(struct intel_dp *intel_dp, ret = drm_dp_dpcd_write_byte(&intel_dp->aux, DP_TRAINING_PATTERN_SET, DP_TRAINING_PATTERN_2_CDS); if (ret < 0) { - lt_err(intel_dp, DP_PHY_DPRX, "Failed to start 128b/132b TPS2 CDS\n"); + lt_err(intel_dp, DP_PHY_DPRX, "Failed to start 128b/132b TPS2 CDS (%pe)\n", + ERR_PTR(ret)); return false; } From 0cddb1e03381532727bddd3bc18818af98c4c7d1 Mon Sep 17 00:00:00 2001 From: Suraj Kandpal Date: Mon, 21 Sep 2026 11:07:09 +0530 Subject: [PATCH 0322/1352] drm/i915/dptest: log DPCD access errors Now that the DPCD accessors return a proper error code, include it in the error messages using %pe and ERR_PTR(), instead of just stating that the access failed. Store the EDID checksum write result in ret as well, so that it can be logged the same way. Signed-off-by: Suraj Kandpal Reviewed-by: Nemesa Garg Link: https://patch.msgid.link/20260921053710.1586776-8-suraj.kandpal@intel.com --- drivers/gpu/drm/i915/display/intel_dp_test.c | 24 +++++++++++--------- 1 file changed, 13 insertions(+), 11 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_dp_test.c b/drivers/gpu/drm/i915/display/intel_dp_test.c index 424183a6515592..17cf2727312478 100644 --- a/drivers/gpu/drm/i915/display/intel_dp_test.c +++ b/drivers/gpu/drm/i915/display/intel_dp_test.c @@ -157,7 +157,7 @@ static u8 intel_dp_autotest_link_training(struct intel_dp *intel_dp) ret = drm_dp_dpcd_read_byte(&intel_dp->aux, DP_TEST_LANE_COUNT, &test_lane_count); if (ret < 0) { - drm_dbg_kms(display->drm, "Lane count read failed\n"); + drm_dbg_kms(display->drm, "Lane count read failed (%pe)\n", ERR_PTR(ret)); return DP_TEST_NAK; } test_lane_count &= DP_MAX_LANE_COUNT_MASK; @@ -165,7 +165,7 @@ static u8 intel_dp_autotest_link_training(struct intel_dp *intel_dp) ret = drm_dp_dpcd_read_byte(&intel_dp->aux, DP_TEST_LINK_RATE, &test_link_bw); if (ret < 0) { - drm_dbg_kms(display->drm, "Link Rate read failed\n"); + drm_dbg_kms(display->drm, "Link Rate read failed (%pe)\n", ERR_PTR(ret)); return DP_TEST_NAK; } test_link_rate = drm_dp_bw_code_to_link_rate(test_link_bw); @@ -193,7 +193,7 @@ static u8 intel_dp_autotest_video_pattern(struct intel_dp *intel_dp) ret = drm_dp_dpcd_read_byte(&intel_dp->aux, DP_TEST_PATTERN, &test_pattern); if (ret < 0) { - drm_dbg_kms(display->drm, "Test pattern read failed\n"); + drm_dbg_kms(display->drm, "Test pattern read failed (%pe)\n", ERR_PTR(ret)); return DP_TEST_NAK; } if (test_pattern != DP_COLOR_RAMP) @@ -201,20 +201,20 @@ static u8 intel_dp_autotest_video_pattern(struct intel_dp *intel_dp) ret = drm_dp_dpcd_read_data(&intel_dp->aux, DP_TEST_H_WIDTH_HI, &h_width, 2); if (ret < 0) { - drm_dbg_kms(display->drm, "H Width read failed\n"); + drm_dbg_kms(display->drm, "H Width read failed (%pe)\n", ERR_PTR(ret)); return DP_TEST_NAK; } ret = drm_dp_dpcd_read_data(&intel_dp->aux, DP_TEST_V_HEIGHT_HI, &v_height, 2); if (ret < 0) { - drm_dbg_kms(display->drm, "V Height read failed\n"); + drm_dbg_kms(display->drm, "V Height read failed (%pe)\n", ERR_PTR(ret)); return DP_TEST_NAK; } ret = drm_dp_dpcd_read_byte(&intel_dp->aux, DP_TEST_MISC0, &test_misc); if (ret < 0) { - drm_dbg_kms(display->drm, "TEST MISC read failed\n"); + drm_dbg_kms(display->drm, "TEST MISC read failed (%pe)\n", ERR_PTR(ret)); return DP_TEST_NAK; } if ((test_misc & DP_TEST_COLOR_FORMAT_MASK) != DP_COLOR_FORMAT_RGB) @@ -247,6 +247,7 @@ static u8 intel_dp_autotest_edid(struct intel_dp *intel_dp) u8 test_result = DP_TEST_ACK; struct intel_connector *intel_connector = intel_dp->attached_connector; struct drm_connector *connector = &intel_connector->base; + int ret; if (!intel_connector->detect_edid || connector->edid_corrupt || intel_dp->aux.i2c_defer_count > 6) { @@ -271,10 +272,11 @@ static u8 intel_dp_autotest_edid(struct intel_dp *intel_dp) /* We have to write the checksum of the last block read */ block += block->extensions; - if (drm_dp_dpcd_write_byte(&intel_dp->aux, DP_TEST_EDID_CHECKSUM, - block->checksum) < 0) + ret = drm_dp_dpcd_write_byte(&intel_dp->aux, DP_TEST_EDID_CHECKSUM, + block->checksum); + if (ret < 0) drm_dbg_kms(display->drm, - "Failed to write EDID checksum\n"); + "Failed to write EDID checksum (%pe)\n", ERR_PTR(ret)); test_result = DP_TEST_ACK | DP_TEST_EDID_CHECKSUM_WRITE; intel_dp->compliance.test_data.edid = INTEL_DP_RESOLUTION_PREFERRED; @@ -429,7 +431,7 @@ void intel_dp_test_request(struct intel_dp *intel_dp) ret = drm_dp_dpcd_read_byte(&intel_dp->aux, DP_TEST_REQUEST, &request); if (ret < 0) { drm_dbg_kms(display->drm, - "Could not read test request from sink\n"); + "Could not read test request from sink (%pe)\n", ERR_PTR(ret)); goto update_status; } @@ -463,7 +465,7 @@ void intel_dp_test_request(struct intel_dp *intel_dp) ret = drm_dp_dpcd_write_byte(&intel_dp->aux, DP_TEST_RESPONSE, response); if (ret < 0) drm_dbg_kms(display->drm, - "Could not write test response to sink\n"); + "Could not write test response to sink (%pe)\n", ERR_PTR(ret)); } /* phy test */ From a24f49a328477ef8520ec1d803c99adece6c6fb2 Mon Sep 17 00:00:00 2001 From: Suraj Kandpal Date: Mon, 21 Sep 2026 11:07:10 +0530 Subject: [PATCH 0323/1352] drm/i915/psr: log DPCD access errors Now that the DPCD accessors return a proper error code, include it in the error messages using %pe and ERR_PTR(), instead of just stating that the access failed. Signed-off-by: Suraj Kandpal Reviewed-by: Nemesa Garg Link: https://patch.msgid.link/20260921053710.1586776-9-suraj.kandpal@intel.com --- drivers/gpu/drm/i915/display/intel_psr.c | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_psr.c b/drivers/gpu/drm/i915/display/intel_psr.c index f1d48b69a18fac..00d0146b260024 100644 --- a/drivers/gpu/drm/i915/display/intel_psr.c +++ b/drivers/gpu/drm/i915/display/intel_psr.c @@ -478,7 +478,8 @@ static u8 intel_dp_get_sink_sync_latency(struct intel_dp *intel_dp) val &= DP_MAX_RESYNC_FRAME_COUNT_MASK; else drm_dbg_kms(display->drm, - "Unable to get sink synchronization latency, assuming 8 frames\n"); + "Unable to get sink synchronization latency, assuming 8 frames (%pe)\n", + ERR_PTR(ret)); return val; } @@ -504,7 +505,8 @@ static void _psr_compute_su_granularity(struct intel_dp *intel_dp, ret = drm_dp_dpcd_read_data(&intel_dp->aux, DP_PSR2_SU_X_GRANULARITY, &w, sizeof(w)); if (ret < 0) drm_dbg_kms(display->drm, - "Unable to read selective update x granularity\n"); + "Unable to read selective update x granularity (%pe)\n", + ERR_PTR(ret)); /* * Spec says that if the value read is 0 the default granularity should * be used instead. @@ -515,7 +517,8 @@ static void _psr_compute_su_granularity(struct intel_dp *intel_dp, ret = drm_dp_dpcd_read_byte(&intel_dp->aux, DP_PSR2_SU_Y_GRANULARITY, &y); if (ret < 0) { drm_dbg_kms(display->drm, - "Unable to read selective update y granularity\n"); + "Unable to read selective update y granularity (%pe)\n", + ERR_PTR(ret)); y = 4; } if (y == 0) @@ -3835,7 +3838,7 @@ static void psr_capability_changed_check(struct intel_dp *intel_dp) ret = drm_dp_dpcd_read_byte(&intel_dp->aux, DP_PSR_ESI, &val); if (ret < 0) { - drm_err(display->drm, "Error reading DP_PSR_ESI\n"); + drm_err(display->drm, "Error reading DP_PSR_ESI (%pe)\n", ERR_PTR(ret)); return; } From 83978027e89b8351deeec94ea92c968b62c167c2 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Krzysztof=20Wilczy=C5=84ski?= Date: Tue, 11 Aug 2026 09:45:04 +0000 Subject: [PATCH 0324/1352] PCI/sysfs: Stop reporting _DSM failures as -EPERM MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Currently, dsm_get_label() returns the literal -1 on every failure path. The sysfs read path passes that value to userspace as -EPERM. Reading the "label" or "acpi_index" attribute on a platform where the Device Name _DSM returns a malformed result then fails with: $ cat /sys/bus/pci/devices/0000:00:08.3/label cat: /sys/bus/pci/devices/0000:00:08.3/label: Operation not permitted Nothing in that path performs a permission check, and the read fails the same way for a privileged reader. The error code points at a cause that does not exist, and tools such as lspci report it on every invocation. Thus, return -ENODEV when the device has no ACPI companion, and -EIO when the _DSM evaluation fails or returns an object that cannot be parsed. Other _DSM users in the tree report such failures the same way. The set of reads that succeed, and the bytes they return, stay the same. Only the error code of reads that already fail differs. Link: https://github.com/pciutils/pciutils/issues/175 Fixes: 6058989bad05 ("PCI: Export ACPI _DSM provided firmware instance number and string name to sysfs") Signed-off-by: Krzysztof Wilczyński --- drivers/pci/pci-label.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/drivers/pci/pci-label.c b/drivers/pci/pci-label.c index 0c644651964043..255e0ecffb096d 100644 --- a/drivers/pci/pci-label.c +++ b/drivers/pci/pci-label.c @@ -160,12 +160,12 @@ static int dsm_get_label(struct device *dev, char *buf, int len = 0; if (!handle) - return -1; + return -ENODEV; obj = acpi_evaluate_dsm(handle, &pci_acpi_dsm_guid, 0x2, DSM_PCI_DEVICE_NAME, NULL); if (!obj) - return -1; + return -EIO; tmp = obj->package.elements; if (obj->type == ACPI_TYPE_PACKAGE && obj->package.count == 2 && @@ -190,7 +190,7 @@ static int dsm_get_label(struct device *dev, char *buf, ACPI_FREE(obj); - return len > 0 ? len : -1; + return len > 0 ? len : -EIO; } static ssize_t label_show(struct device *dev, struct device_attribute *attr, From 1d356143b020e0a311e32745edbf06f7eed5f1b6 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Krzysztof=20Wilczy=C5=84ski?= Date: Tue, 11 Aug 2026 09:45:19 +0000 Subject: [PATCH 0325/1352] PCI/sysfs: Decouple acpi_index from the optional device name element MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Currently, dsm_get_label() validates both elements of the Device Name _DSM result in a single conditional. The _DSM returns an ACPI package of two elements, the instance number and the device name, where the instance number is mandatory and the name is optional. Firmware that implements no name must return a NULL string for it. Reads of "acpi_index" therefore fail whenever the name element is malformed. That attribute exports only the instance number, and the two elements do not depend on each other. Thus, validate each element only for the attribute that exports it. So "acpi_index" now depends on the instance number alone, and "label" reads fail with -EIO when the name element is neither a string nor a buffer. The package elements pointer is read only after the object type has been checked. On platforms with a valid instance number and a malformed name element the "acpi_index" attribute starts returning data. Because udev derives the onboard interface name from "acpi_index", an interface on such a platform may be renamed once, on the first boot after this change. That is the attribute assuming the value the firmware always provided. Link: https://github.com/pciutils/pciutils/issues/175 Signed-off-by: Krzysztof Wilczyński --- drivers/pci/pci-label.c | 53 +++++++++++++++++++++++++---------------- 1 file changed, 33 insertions(+), 20 deletions(-) diff --git a/drivers/pci/pci-label.c b/drivers/pci/pci-label.c index 255e0ecffb096d..5b08f50653a37d 100644 --- a/drivers/pci/pci-label.c +++ b/drivers/pci/pci-label.c @@ -157,7 +157,7 @@ static int dsm_get_label(struct device *dev, char *buf, { acpi_handle handle = ACPI_HANDLE(dev); union acpi_object *obj, *tmp; - int len = 0; + int len; if (!handle) return -ENODEV; @@ -167,30 +167,43 @@ static int dsm_get_label(struct device *dev, char *buf, if (!obj) return -EIO; + if (obj->type != ACPI_TYPE_PACKAGE || obj->package.count != 2) { + len = -EIO; + goto out; + } + tmp = obj->package.elements; - if (obj->type == ACPI_TYPE_PACKAGE && obj->package.count == 2 && - tmp[0].type == ACPI_TYPE_INTEGER && - (tmp[1].type == ACPI_TYPE_STRING || - tmp[1].type == ACPI_TYPE_BUFFER)) { - /* - * The second string element is optional even when - * this _DSM is implemented; when not implemented, - * this entry must return a null string. - */ - if (attr == ACPI_ATTR_INDEX_SHOW) { - len = sysfs_emit(buf, "%llu\n", tmp->integer.value); - } else if (attr == ACPI_ATTR_LABEL_SHOW) { - if (tmp[1].type == ACPI_TYPE_STRING) - len = sysfs_emit(buf, "%s\n", - tmp[1].string.pointer); - else if (tmp[1].type == ACPI_TYPE_BUFFER) - len = dsm_label_utf16s_to_utf8s(tmp + 1, buf); - } + if (tmp[0].type != ACPI_TYPE_INTEGER) { + len = -EIO; + goto out; } + if (attr == ACPI_ATTR_INDEX_SHOW) { + len = sysfs_emit(buf, "%llu\n", tmp[0].integer.value); + goto out; + } + + /* + * Per PCI Firmware r3.3, sec 4.6.7, the device name is optional + * even when this _DSM is implemented. When not implemented, this + * entry must return a NULL string. + */ + switch (tmp[1].type) { + case ACPI_TYPE_STRING: + len = sysfs_emit(buf, "%s\n", tmp[1].string.pointer); + break; + case ACPI_TYPE_BUFFER: + len = dsm_label_utf16s_to_utf8s(&tmp[1], buf); + break; + default: + len = -EIO; + break; + } + +out: ACPI_FREE(obj); - return len > 0 ? len : -EIO; + return len; } static ssize_t label_show(struct device *dev, struct device_attribute *attr, From cd181482782ba80b652814c8a8b6212387c82a40 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Krzysztof=20Wilczy=C5=84ski?= Date: Tue, 11 Aug 2026 09:45:34 +0000 Subject: [PATCH 0326/1352] PCI/sysfs: Handle a malformed _DSM result in acpi_attr_is_visible() MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Currently, the "label" and "acpi_index" attributes are created whenever the _DSM function 0 bitmap advertises the Device Name function, without evaluating that function. The documented ABI states that each attribute is created only if the firmware has given a name or an instance number to the PCI device. The bitmap says nothing about the object the function returns. Firmware that advertises the function but returns an object that cannot be parsed therefore produces attributes that exist and fail every read, as in the report below. Thus, evaluate the Device Name _DSM when deciding attribute visibility and create each attribute only when the element it exports has a type the read path accepts, mirroring the checks performed at read time. The SMBIOS attribute group already follows this pattern by performing the DMI lookup in its is_visible() callback. The read side checks remain in place since the two evaluations happen at different times. This restores the behaviour from before commit 2fc59fe2ecdc ("PCI / pci-label: treat PCI label with index 0 as valid label"), which switched visibility from evaluating the function to checking only the function 0 bitmap. The check is now made for each attribute, not once for both. Fixes: 2fc59fe2ecdc ("PCI / pci-label: treat PCI label with index 0 as valid label") Closes: https://github.com/pciutils/pciutils/issues/175 Signed-off-by: Krzysztof Wilczyński --- Documentation/ABI/testing/sysfs-bus-pci | 14 ++++++++--- drivers/pci/pci-label.c | 33 ++++++++++++++++++++++++- 2 files changed, 42 insertions(+), 5 deletions(-) diff --git a/Documentation/ABI/testing/sysfs-bus-pci b/Documentation/ABI/testing/sysfs-bus-pci index 55ea1db749a180..c4ff6b53823345 100644 --- a/Documentation/ABI/testing/sysfs-bus-pci +++ b/Documentation/ABI/testing/sysfs-bus-pci @@ -244,10 +244,16 @@ Contact: Narendra K , linux-bugs@dell.com Description: Reading this attribute will provide the firmware given name (SMBIOS type 41 string or ACPI _DSM string) of - the PCI device. The attribute will be created only - if the firmware has given a name to the PCI device. - ACPI _DSM string name will be given priority if the - system firmware provides SMBIOS type 41 string also. + the PCI device. The attribute will be created only if the + firmware naming mechanism is implemented for the device: + a SMBIOS type 41 record with a non-empty reference + designation, or a Device Name _DSM that returns the name + as a string or a buffer. The value read is empty when the + firmware implements the _DSM but gives the device no name, + which per PCI Firmware r3.3, sec 4.6.7 the firmware reports + as a NULL string. ACPI _DSM string name will be given + priority if the system firmware provides SMBIOS type 41 + string also. Users: Userspace applications interested in knowing the firmware assigned name of the PCI device. diff --git a/drivers/pci/pci-label.c b/drivers/pci/pci-label.c index 5b08f50653a37d..abb941530c5605 100644 --- a/drivers/pci/pci-label.c +++ b/drivers/pci/pci-label.c @@ -230,11 +230,42 @@ static umode_t acpi_attr_is_visible(struct kobject *kobj, struct attribute *a, int n) { struct device *dev = kobj_to_dev(kobj); + union acpi_object *obj, *tmp; + umode_t mode = 0; if (!device_has_acpi_name(dev)) return 0; - return a->mode; + /* + * The bitmap from _DSM function 0 only advertises function 7, + * and whether the returned object can be parsed is a separate + * question. Evaluate it and expose each attribute only if the + * element it exports has one of the types the read path + * accepts, mirroring the checks in dsm_get_label(). + */ + obj = acpi_evaluate_dsm(ACPI_HANDLE(dev), &pci_acpi_dsm_guid, 0x2, + DSM_PCI_DEVICE_NAME, NULL); + if (!obj) + return 0; + + if (obj->type != ACPI_TYPE_PACKAGE || obj->package.count != 2) + goto out; + + tmp = obj->package.elements; + if (tmp[0].type != ACPI_TYPE_INTEGER) + goto out; + + if (a == &dev_attr_acpi_index.attr) + mode = a->mode; + else if (a == &dev_attr_label.attr && + (tmp[1].type == ACPI_TYPE_STRING || + tmp[1].type == ACPI_TYPE_BUFFER)) + mode = a->mode; + +out: + ACPI_FREE(obj); + + return mode; } const struct attribute_group pci_dev_acpi_attr_group = { From 9f8a65b456fe54a01c80a754be8d566891d53a80 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Krzysztof=20Wilczy=C5=84ski?= Date: Wed, 12 Aug 2026 10:58:45 +0000 Subject: [PATCH 0327/1352] PCI/sysfs: Pass the device name length in UTF-16 code units MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Currently, dsm_label_utf16s_to_utf8s() passes the ACPI buffer length to utf16s_to_utf8s() unchanged. The Device Name _DSM may return the device name as a buffer holding a UTF-16 string, and the ACPI length counts bytes while the converter counts wchar_t elements. The converter therefore receives a count twice the number of code units the buffer holds, and that count is its only bound on the input. A NUL code unit ends the conversion early, so a name that carries one is converted correctly and the error stays hidden. ACPICA zeroes the entire result allocation, and the padding placed after the buffer supplies that NUL for most lengths. Lengths that leave no padding, or a single byte of it, do not, and the conversion then runs up to its own length in bytes past the end of the allocation. The bytes beyond the allocation are decoded into the "label" attribute, which is world readable. Thus, divide the buffer length by the size of wchar_t so that the converter receives a count of code units. The division truncates, so the converter is never told to read more bytes than the buffer holds, including for an odd length. Reaching the out of bounds read needs firmware that returns the name as a buffer and omits the terminator. Fixes: 6058989bad05 ("PCI: Export ACPI _DSM provided firmware instance number and string name to sysfs") Cc: stable@vger.kernel.org Signed-off-by: Krzysztof Wilczyński --- drivers/pci/pci-label.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/pci/pci-label.c b/drivers/pci/pci-label.c index abb941530c5605..08c461f1142d7c 100644 --- a/drivers/pci/pci-label.c +++ b/drivers/pci/pci-label.c @@ -144,7 +144,7 @@ static int dsm_label_utf16s_to_utf8s(union acpi_object *obj, char *buf) int len; len = utf16s_to_utf8s((const wchar_t *)obj->buffer.pointer, - obj->buffer.length, + obj->buffer.length / sizeof(wchar_t), UTF16_LITTLE_ENDIAN, buf, PAGE_SIZE - 1); buf[len++] = '\n'; From 4735883c0d4bc3dae51b92218f00b2faff7d49b8 Mon Sep 17 00:00:00 2001 From: Aleksa Paunovic Date: Thu, 24 Sep 2026 19:21:07 -0600 Subject: [PATCH 0328/1352] riscv: Add support for early boot errata application on MIPS chips MIPS errata implementation previously skipped early boot application entirely. Although the only currently existing MIPS erratum does not require this, amending this now should make any future addition easier to implement. This commit was based on the existing T-Head implementation. Suggested-by: Jesse Taube Link: https://lore.kernel.org/linux-riscv/CADRr4bcbD57hmR0XGgo8BjgNp4shEADOondZEtwscCJ9-nxXRQ@mail.gmail.com/ Signed-off-by: Aleksa Paunovic Link: https://patch.msgid.link/20260810-p8700-early-boot-v1-1-5c4aa08d500f@htecgroup.com Signed-off-by: Paul Walmsley --- arch/riscv/Kconfig.errata | 1 + arch/riscv/errata/mips/Makefile | 6 ++++++ arch/riscv/errata/mips/errata.c | 38 ++++++++++++++++++++------------- 3 files changed, 30 insertions(+), 15 deletions(-) diff --git a/arch/riscv/Kconfig.errata b/arch/riscv/Kconfig.errata index 38ade0ea4e9cb7..1762ce7ae921c8 100644 --- a/arch/riscv/Kconfig.errata +++ b/arch/riscv/Kconfig.errata @@ -24,6 +24,7 @@ config ERRATA_ANDES_CMO config ERRATA_MIPS bool "MIPS errata" depends on RISCV_ALTERNATIVE + select RISCV_ALTERNATIVE_EARLY help All MIPS errata Kconfig depend on this Kconfig. Disabling this Kconfig will disable all MIPS errata. Please say "Y" diff --git a/arch/riscv/errata/mips/Makefile b/arch/riscv/errata/mips/Makefile index 6278c389b801ee..137e700d9d3f8e 100644 --- a/arch/riscv/errata/mips/Makefile +++ b/arch/riscv/errata/mips/Makefile @@ -1,5 +1,11 @@ ifdef CONFIG_RISCV_ALTERNATIVE_EARLY CFLAGS_errata.o := -mcmodel=medany +ifdef CONFIG_FTRACE +CFLAGS_REMOVE_errata.o = $(CC_FLAGS_FTRACE) +endif +ifdef CONFIG_KASAN +KASAN_SANITIZE_errata.o := n +endif endif obj-y += errata.o diff --git a/arch/riscv/errata/mips/errata.c b/arch/riscv/errata/mips/errata.c index 2c3dc2259e93e9..ac9a12d0a30c9d 100644 --- a/arch/riscv/errata/mips/errata.c +++ b/arch/riscv/errata/mips/errata.c @@ -7,12 +7,13 @@ #include #include #include +#include #include #include #include #include -static inline bool errata_probe_pause(void) +static inline bool errata_probe_pause(unsigned int stage) { if (!IS_ENABLED(CONFIG_ERRATA_MIPS_P8700_PAUSE_OPCODE)) return false; @@ -20,14 +21,17 @@ static inline bool errata_probe_pause(void) if (!riscv_isa_vendor_extension_available(MIPS_VENDOR_ID, XMIPSEXECTL)) return false; + if (stage == RISCV_ALTERNATIVES_EARLY_BOOT) + return false; + return true; } -static u32 mips_errata_probe(void) +static u32 mips_errata_probe(unsigned int stage) { u32 cpu_req_errata = 0; - if (errata_probe_pause()) + if (errata_probe_pause(stage)) cpu_req_errata |= BIT(ERRATA_MIPS_P8700_PAUSE_OPCODE); return cpu_req_errata; @@ -38,30 +42,34 @@ void mips_errata_patch_func(struct alt_entry *begin, struct alt_entry *end, unsigned int stage) { struct alt_entry *alt; - u32 cpu_req_errata = mips_errata_probe(); + u32 cpu_req_errata = mips_errata_probe(stage); u32 tmp; + void *oldptr, *altptr; BUILD_BUG_ON(ERRATA_MIPS_NUMBER >= RISCV_VENDOR_EXT_ALTERNATIVES_BASE); - if (stage == RISCV_ALTERNATIVES_EARLY_BOOT) - return; - for (alt = begin; alt < end; alt++) { if (alt->vendor_id != MIPS_VENDOR_ID) continue; - if (alt->patch_id >= ERRATA_MIPS_NUMBER) { - WARN(1, "MIPS errata id:%d not in kernel errata list\n", - alt->patch_id); + if (alt->patch_id >= ERRATA_MIPS_NUMBER) continue; - } tmp = (1U << alt->patch_id); if (cpu_req_errata & tmp) { - mutex_lock(&text_mutex); - patch_text_nosync(ALT_OLD_PTR(alt), ALT_ALT_PTR(alt), - alt->alt_len); - mutex_unlock(&text_mutex); + oldptr = ALT_OLD_PTR(alt); + altptr = ALT_ALT_PTR(alt); + + if (stage == RISCV_ALTERNATIVES_EARLY_BOOT) { + memcpy(oldptr, altptr, alt->alt_len); + } else { + mutex_lock(&text_mutex); + patch_text_nosync(oldptr, altptr, alt->alt_len); + mutex_unlock(&text_mutex); + } } } + + if (stage == RISCV_ALTERNATIVES_EARLY_BOOT) + local_flush_icache_all(); } From afdf722ab983f09dc06b2c71645964a80a7f1b57 Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Sat, 12 Sep 2026 09:34:44 +0200 Subject: [PATCH 0329/1352] netfs: Document the remote size member and its accessors netfs_inode.remote_i_size was renamed to _remote_i_size and accessors were added to prevent torn accesses. The netfs library documentation still shows the old member name. Update the struct excerpt and member description. Direct readers to netfs_read_remote_i_size() and netfs_write_remote_i_size(), and state that writes require inode->i_lock. Fixes: 2c8f4742bb76 ("netfs: Fix potential for tearing in ->remote_i_size and ->zero_point") Signed-off-by: Karl Mehltretter Link: https://patch.msgid.link/20260912073444.51152-1-kmehltretter@gmail.com Signed-off-by: Christian Brauner (Amutable) --- Documentation/filesystems/netfs_library.rst | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/Documentation/filesystems/netfs_library.rst b/Documentation/filesystems/netfs_library.rst index ddd799df6ce363..0c9786ffe19240 100644 --- a/Documentation/filesystems/netfs_library.rst +++ b/Documentation/filesystems/netfs_library.rst @@ -195,7 +195,7 @@ structure is defined:: struct inode inode; const struct netfs_request_ops *ops; struct fscache_cookie * cache; - loff_t remote_i_size; + loff_t _remote_i_size; unsigned long flags; ... }; @@ -229,11 +229,14 @@ filesystem: Local caching cookie, or NULL if no caching is enabled. This field does not exist if fscache is disabled. - * ``remote_i_size`` + * ``_remote_i_size`` The size of the file on the server. This differs from inode->i_size if local modifications have been made but not yet written back. + Use netfs_read_remote_i_size() and netfs_write_remote_i_size() to access + this field. Hold inode->i_lock when writing it. + * ``flags`` A set of flags, some of which the filesystem might be interested in: From c9e86618da6990f54b52e2eca9ffb27b4871bec8 Mon Sep 17 00:00:00 2001 From: David Howells Date: Thu, 10 Sep 2026 23:02:36 +0100 Subject: [PATCH 0330/1352] cachefiles: Clean up cachefiles_do_prepare_read() Clean up cachefiles_do_prepare_read() by merging it into cachefiles_prepare_read() now that the ondemand mode stuff has been removed. Signed-off-by: David Howells Link: https://patch.msgid.link/20260910220242.2165023-2-dhowells@redhat.com Reviewed-by: Paulo Alcantara cc: Marc Dionne cc: Paulo Alcantara cc: netfs@lists.linux.dev cc: linux-fsdevel@vger.kernel.org Signed-off-by: Christian Brauner (Amutable) --- fs/cachefiles/io.c | 37 +++++++++++++++---------------------- 1 file changed, 15 insertions(+), 22 deletions(-) diff --git a/fs/cachefiles/io.c b/fs/cachefiles/io.c index e61e885784d695..2c1cd703f4bf91 100644 --- a/fs/cachefiles/io.c +++ b/fs/cachefiles/io.c @@ -375,19 +375,23 @@ static int cachefiles_write(struct netfs_cache_resources *cres, term_func, term_func_priv); } -static inline enum netfs_io_source -cachefiles_do_prepare_read(struct netfs_cache_resources *cres, - uoff_t start, size_t *_len, loff_t i_size, - unsigned long *_flags, ino_t netfs_ino) +/* + * Prepare a read operation, shortening it to a cached/uncached boundary as + * appropriate. + */ +static enum netfs_io_source +cachefiles_prepare_read(struct netfs_io_subrequest *subreq, uoff_t i_size) { enum cachefiles_prepare_read_trace why; + struct netfs_cache_resources *cres = &subreq->rreq->cache_resources; struct cachefiles_object *object = NULL; struct cachefiles_cache *cache; struct fscache_cookie *cookie = fscache_cres_cookie(cres); const struct cred *saved_cred; struct file *file = cachefiles_cres_file(cres); enum netfs_io_source ret = NETFS_DOWNLOAD_FROM_SERVER; - size_t len = *_len; + uoff_t start = subreq->start; + size_t len = subreq->len; loff_t off, to; ino_t ino = file ? file_inode(file)->i_ino : 0; @@ -400,7 +404,7 @@ cachefiles_do_prepare_read(struct netfs_cache_resources *cres, } if (test_bit(FSCACHE_COOKIE_NO_DATA_TO_READ, &cookie->flags)) { - __set_bit(NETFS_SREQ_COPY_TO_CACHE, _flags); + __set_bit(NETFS_SREQ_COPY_TO_CACHE, &subreq->flags); why = cachefiles_trace_read_no_data; goto out_no_object; } @@ -441,7 +445,7 @@ cachefiles_do_prepare_read(struct netfs_cache_resources *cres, if (off > start) { off = round_up(off, cache->bsize); len = off - start; - *_len = len; + subreq->len = len; why = cachefiles_trace_read_found_part; goto download_and_store; } @@ -462,7 +466,7 @@ cachefiles_do_prepare_read(struct netfs_cache_resources *cres, else to = round_down(to, cache->bsize); len = to - start; - *_len = len; + subreq->len = len; } why = cachefiles_trace_read_have_data; @@ -470,26 +474,15 @@ cachefiles_do_prepare_read(struct netfs_cache_resources *cres, goto out; download_and_store: - __set_bit(NETFS_SREQ_COPY_TO_CACHE, _flags); + __set_bit(NETFS_SREQ_COPY_TO_CACHE, &subreq->flags); out: cachefiles_end_secure(cache, saved_cred); out_no_object: - trace_cachefiles_prep_read(object, start, len, *_flags, ret, why, ino, netfs_ino); + trace_cachefiles_prep_read(object, start, len, subreq->flags, ret, why, + ino, subreq->rreq->inode->i_ino); return ret; } -/* - * Prepare a read operation, shortening it to a cached/uncached - * boundary as appropriate. - */ -static enum netfs_io_source cachefiles_prepare_read(struct netfs_io_subrequest *subreq, - uoff_t i_size) -{ - return cachefiles_do_prepare_read(&subreq->rreq->cache_resources, - subreq->start, &subreq->len, i_size, - &subreq->flags, subreq->rreq->inode->i_ino); -} - /* * Prepare for a write to occur. */ From 4f973d314d5997888a35949049e52e28ce88d261 Mon Sep 17 00:00:00 2001 From: David Howells Date: Thu, 10 Sep 2026 23:02:37 +0100 Subject: [PATCH 0331/1352] netfs, cachefiles: Add a couple of traces for write failure Add a couple of subrequest traces for write failure in the cache. Signed-off-by: David Howells Link: https://patch.msgid.link/20260910220242.2165023-3-dhowells@redhat.com Reviewed-by: Paulo Alcantara cc: Marc Dionne cc: Paulo Alcantara cc: netfs@lists.linux.dev cc: linux-fsdevel@vger.kernel.org Signed-off-by: Christian Brauner (Amutable) --- fs/cachefiles/io.c | 8 ++++++-- include/trace/events/netfs.h | 2 ++ 2 files changed, 8 insertions(+), 2 deletions(-) diff --git a/fs/cachefiles/io.c b/fs/cachefiles/io.c index 2c1cd703f4bf91..dac48fdf85d13c 100644 --- a/fs/cachefiles/io.c +++ b/fs/cachefiles/io.c @@ -605,10 +605,14 @@ static void cachefiles_prepare_write_subreq(struct netfs_io_subrequest *subreq) stream->sreq_max_segs = BIO_MAX_VECS; if (!cachefiles_cres_file(cres)) { - if (!fscache_wait_for_operation(cres, FSCACHE_WANT_WRITE)) + if (!fscache_wait_for_operation(cres, FSCACHE_WANT_WRITE)) { + trace_netfs_sreq(subreq, netfs_sreq_trace_cache_waitfail); return netfs_prepare_write_failed(subreq); - if (!cachefiles_cres_file(cres)) + } + if (!cachefiles_cres_file(cres)) { + trace_netfs_sreq(subreq, netfs_sreq_trace_cache_nofile); return netfs_prepare_write_failed(subreq); + } } } diff --git a/include/trace/events/netfs.h b/include/trace/events/netfs.h index 303309be253f2a..35826aea1c2ddf 100644 --- a/include/trace/events/netfs.h +++ b/include/trace/events/netfs.h @@ -93,8 +93,10 @@ EM(netfs_sreq_trace_abandoned, "ABNDN") \ EM(netfs_sreq_trace_add_donations, "+DON ") \ EM(netfs_sreq_trace_added, "ADD ") \ + EM(netfs_sreq_trace_cache_nofile, "CA-!F") \ EM(netfs_sreq_trace_cache_nowrite, "CA-NW") \ EM(netfs_sreq_trace_cache_prepare, "CA-PR") \ + EM(netfs_sreq_trace_cache_waitfail, "CA-!W") \ EM(netfs_sreq_trace_cache_write, "CA-WR") \ EM(netfs_sreq_trace_cancel, "CANCL") \ EM(netfs_sreq_trace_clear, "CLEAR") \ From dd79f8c90ea8e1579cc35c089cbed9a892d32908 Mon Sep 17 00:00:00 2001 From: David Howells Date: Thu, 10 Sep 2026 23:02:38 +0100 Subject: [PATCH 0332/1352] cachefiles: Add a tracepoint to log insufficient space errors Add a tracepoint to log insufficient space errors. Signed-off-by: David Howells Link: https://patch.msgid.link/20260910220242.2165023-4-dhowells@redhat.com Reviewed-by: Paulo Alcantara cc: Marc Dionne cc: Paulo Alcantara cc: netfs@lists.linux.dev cc: linux-fsdevel@vger.kernel.org Signed-off-by: Christian Brauner (Amutable) --- fs/cachefiles/io.c | 15 +++++++++++---- fs/cachefiles/namei.c | 10 ++++++++-- include/trace/events/cachefiles.h | 30 +++++++++++++++++++++++++++++- 3 files changed, 48 insertions(+), 7 deletions(-) diff --git a/fs/cachefiles/io.c b/fs/cachefiles/io.c index dac48fdf85d13c..4b3ceda17426cf 100644 --- a/fs/cachefiles/io.c +++ b/fs/cachefiles/io.c @@ -533,10 +533,14 @@ int __cachefiles_prepare_write(struct cachefiles_object *object, * space, we need to see if it's fully allocated. If it's not, we may * want to cull it. */ - if (cachefiles_has_space(cache, 0, *_len / PAGE_SIZE, - cachefiles_has_space_check) == 0) + ret = cachefiles_has_space(cache, 0, *_len / PAGE_SIZE, + cachefiles_has_space_check); + if (ret == 0) return 0; /* Enough space to simply overwrite the whole block */ + if (ret == -ENOBUFS) + trace_cachefiles_no_space(object, cachefiles_trace_write_nospace_2); + pos = cachefiles_inject_read_error(); if (pos == 0) pos = vfs_llseek(file, start, SEEK_HOLE); @@ -565,8 +569,11 @@ int __cachefiles_prepare_write(struct cachefiles_object *object, return ret; check_space: - return cachefiles_has_space(cache, 0, *_len / PAGE_SIZE, - cachefiles_has_space_for_write); + ret = cachefiles_has_space(cache, 0, *_len / PAGE_SIZE, + cachefiles_has_space_for_write); + if (ret == -ENOBUFS) + trace_cachefiles_no_space(object, cachefiles_trace_write_nospace); + return ret; } static int cachefiles_prepare_write(struct netfs_cache_resources *cres, diff --git a/fs/cachefiles/namei.c b/fs/cachefiles/namei.c index 88955249a1a64b..4780ce6dc8309b 100644 --- a/fs/cachefiles/namei.c +++ b/fs/cachefiles/namei.c @@ -117,8 +117,11 @@ struct dentry *cachefiles_get_directory(struct cachefiles_cache *cache, if (d_is_negative(subdir)) { ret = cachefiles_has_space(cache, 1, 0, cachefiles_has_space_for_create); - if (ret < 0) + if (ret < 0) { + if (ret == -ENOBUFS) + trace_cachefiles_no_space(NULL, cachefiles_trace_mkdir_nospace); goto mkdir_error; + } _debug("attempt mkdir"); @@ -487,8 +490,11 @@ static bool cachefiles_create_file(struct cachefiles_object *object) ret = cachefiles_has_space(object->volume->cache, 1, 0, cachefiles_has_space_for_create); - if (ret < 0) + if (ret < 0) { + if (ret == -ENOBUFS) + trace_cachefiles_no_space(object, cachefiles_trace_create_nospace); return false; + } file = cachefiles_create_tmpfile(object); if (IS_ERR(file)) diff --git a/include/trace/events/cachefiles.h b/include/trace/events/cachefiles.h index 1938d51a945940..cd865c265c2451 100644 --- a/include/trace/events/cachefiles.h +++ b/include/trace/events/cachefiles.h @@ -80,11 +80,13 @@ enum cachefiles_prepare_read_trace { }; enum cachefiles_error_trace { + cachefiles_trace_create_nospace, cachefiles_trace_fallocate_error, cachefiles_trace_getxattr_error, cachefiles_trace_link_error, cachefiles_trace_lookup_error, cachefiles_trace_mkdir_error, + cachefiles_trace_mkdir_nospace, cachefiles_trace_notify_change_error, cachefiles_trace_open_error, cachefiles_trace_read_error, @@ -97,6 +99,8 @@ enum cachefiles_error_trace { cachefiles_trace_trunc_error, cachefiles_trace_unlink_error, cachefiles_trace_write_error, + cachefiles_trace_write_nospace, + cachefiles_trace_write_nospace_2, }; #endif @@ -161,11 +165,13 @@ enum cachefiles_error_trace { E_(cachefiles_trace_read_seek_nxio, "seek-enxio") #define cachefiles_error_traces \ + EM(cachefiles_trace_create_nospace, "create-nospace") \ EM(cachefiles_trace_fallocate_error, "fallocate") \ EM(cachefiles_trace_getxattr_error, "getxattr") \ EM(cachefiles_trace_link_error, "link") \ EM(cachefiles_trace_lookup_error, "lookup") \ EM(cachefiles_trace_mkdir_error, "mkdir") \ + EM(cachefiles_trace_mkdir_nospace, "mkdir-nospace") \ EM(cachefiles_trace_notify_change_error, "notify_change") \ EM(cachefiles_trace_open_error, "open") \ EM(cachefiles_trace_read_error, "read") \ @@ -177,7 +183,9 @@ enum cachefiles_error_trace { EM(cachefiles_trace_tmpfile_error, "tmpfile") \ EM(cachefiles_trace_trunc_error, "trunc") \ EM(cachefiles_trace_unlink_error, "unlink") \ - E_(cachefiles_trace_write_error, "write") + EM(cachefiles_trace_write_error, "write") \ + EM(cachefiles_trace_write_nospace, "write-nospace") \ + E_(cachefiles_trace_write_nospace_2, "write-nospace-2") /* @@ -694,6 +702,26 @@ TRACE_EVENT(cachefiles_io_error, __entry->error) ); +TRACE_EVENT(cachefiles_no_space, + TP_PROTO(struct cachefiles_object *obj, enum cachefiles_error_trace trace), + + TP_ARGS(obj, trace), + + TP_STRUCT__entry( + __field(unsigned int, obj) + __field(enum cachefiles_error_trace, trace) + ), + + TP_fast_assign( + __entry->obj = obj ? obj->debug_id : 0; + __entry->trace = trace; + ), + + TP_printk("o=%08x %s", + __entry->obj, + __print_symbolic(__entry->trace, cachefiles_error_traces)) + ); + #endif /* _TRACE_CACHEFILES_H */ /* This part must be outside protection */ From c8d890e089a14ff0b67c2b290c9e14f07421faf2 Mon Sep 17 00:00:00 2001 From: David Howells Date: Thu, 10 Sep 2026 23:02:39 +0100 Subject: [PATCH 0333/1352] cachefiles: Don't rely on backing fs storage map for most use cases Cachefiles currently uses the backing filesystem's idea of what data is held in a backing file and queries this by means of SEEK_DATA and SEEK_HOLE. However, this means it does two seek operations on the backing file for each individual read call it wants to prepare (unless the first returns -ENXIO). Worse, the backing filesystem is at liberty to insert or remove blocks of zeros in order to optimise its layout which may cause false positives and false negatives. The problem is that keeping track of what is dirty is tricky (if storing info in xattrs, which may have limited capacity and must be read and written as one piece) and expensive (in terms of diskspace at least) and is basically duplicating what a filesystem does. However, the most common write case, in which the application does { open(O_TRUNC); write(); write(); ... write(); close(); } where each write follows directly on from the previous and leaves no gaps in the file is reasonably easy to detect and can be noted in the primary xattr as CACHEFILES_CONTENT_ALL, indicating we have everything up to the object size stored. In this specific case, given that it is known that there are no holes in the file, there's no need to call SEEK_DATA/HOLE or use any other mechanism to track the contents. That speeds things up enormously. Even when it is necessary to use SEEK_DATA/HOLE, it may not be necessary to call it for each cache read subrequest generated. Implement this by adding support for the CACHEFILES_CONTENT_ALL content type (which is defined, but currently unused), which requires a slight adjustment in how backing files are managed. Specifically, the driver needs to know how much of the tail block is data and whether storing more data will create a hole. To this end, the way that the size of a backing file is managed is changed. Currently, the backing file is expanded to strictly match the size of the network file, but this can be changed to carry more useful information. This makes two pieces of metadata available: xattr.object_size and the backing file's i_size. Apply the following schema: (a) i_size is always a multiple of the DIO block size. (b) i_size is only updated to the end of the highest write stored. This is used to work out if we are following on without leaving a hole. (c) xattr.object_size is the size of the network filesystem file cached in this backing file. (d) xattr.object_size must point after the start of the last block (unless both are 0). (e) If xattr.object_size is at or after the block at the current end of the backing file (ie. i_size), then we have all the contents of the block (if xattr.content == CACHEFILES_CONTENT_ALL). (f) If xattr.object_size is somewhere in the middle of the last block, then the data following it is invalid and must be ignored. (g) If data is added to the last block, then that block must be fetched, modified and rewritten (it must be a buffered write through the pagecache and not DIO). (h) Writes to cache are rounded out to blocks on both sides and the folios used as sources must contain data for any lower gap and must have been cleared for any upper gap, and so will rewrite any non-data area in the tail block. To implement this, the following changes are made: (1) cookie->object_size is no longer updated when writes are copied into the pagecache, but rather only updated when a write request completes. This prevents object size miscomparison when checking the xattr causing the backing file to be invalidated (opening and marking the backing file and modifying the pagecache run in parallel). (2) The cache's current idea of the amount of data that should be stored in the backing file is kept track of in object->object_size. Possibly this is redundant with cookie->object_size, but the latter gets updated in some addition circumstances. (3) The size of the backing file at the start of a request is now tracked in struct netfs_cache_resources so that the partial EOF block can be located and cleaned. (4) The cache block size is now used consistently rather than using CACHEFILES_DIO_BLOCK_SIZE (4096). (5) The backing file size is no longer adjusted when looking up an object. (6) When shortening a file, if the new size is not block aligned, the part beyond the new size is cleared. If the file is truncated to zero, the content_info gets reset to CACHEFILES_CONTENT_NO_DATA. (7) A new struct, fscache_occupancy, is instituted to track the region being read. Netfslib allocates it and fills in the start and end of the region to be read then calls the ->query_occupancy() method to find and fill in the extents. It also indicates whether a recorded extent contains data or just contains a region that's all zeros (FSCACHE_EXTENT_DATA or FSCACHE_EXTENT_ZERO). (8) The ->prepare_read() cache method is changed such that, if given, it just limits the amount that can be read from the cache in one go. It no longer indicates what source of read should be done; that information is now obtained from ->query_occupancy(). (9) A new cache method, ->collect_write(), is added that is called when a contiguous series of writes have completed and a discontiguity or the end of the request has been hit. It it supplied with the start and length of the write made to the backing file and can use this information to update the cache metadata. (10) cachefiles_query_occupancy() is altered to find the next two "extents" of data stored in the backing file by doing SEEK_DATA/HOLE between the bounds set - unless it is known that there are no holes, in which case a whole-file first extent can be set. (11) cachefiles_collect_write() is implemented to take the collated write completion information and use this to update the cache metadata, in particular working out whether there's now a hole in the backing file requiring future use of SEEK_DATA/HOLE instead of just assuming the data is all present. It also uses fallocate(FALLOC_FL_ZERO_RANGE) to clean the part of a partial block that extended beyond the old object size. It might be better to perform a synchronous DIO write for this purpose, but that would mandate an RMW cycle. Ideally, it should be all zeros anyway, but, unfortunately, shared-writable mmap can interfere. (12) cachefiles_begin_operation() is updated to note the current backing file size and the cache DIO size. (13) cachefiles_create_tmpfile() no longer expands the backing file when it creates it. (14) cachefiles_set_object_xattr() is changed to use object->object_size rather than cookie->object_size. (15) cachefiles_check_auxdata() is altered to actually store the content type and to also set object->object_size. The cachefiles_coherency tracepoint is also modified to display xattr.object_size. (16) netfs_read_to_pagecache() is reworked. The cache ->prepare_read() method is replaced with ->query_occupancy() as the arbiter of what region of the file is read from where, and that retrieves up to two occupied extents of the backing file at once. The cache ->prepare_read() method is now repurposed to be the same as the equivalent network filesystem method and allows the cache to limit the size of the read before the iterator is prepared. netfs_single_dispatch_read() is similarly modified. (17) netfs_update_i_size() and afs_update_i_size() no longer call fscache_update_cookie() to update cookie->object_size. (18) Write collection now collates contiguous sequences of writes to the cache and calls the cache ->collect_write() method. Signed-off-by: David Howells Link: https://patch.msgid.link/20260910220242.2165023-5-dhowells@redhat.com Reviewed-by: Paulo Alcantara cc: Matthew Wilcox cc: linux-cifs@vger.kernel.org cc: netfs@lists.linux.dev cc: linux-fsdevel@vger.kernel.org Signed-off-by: Christian Brauner (Amutable) --- fs/afs/file.c | 1 - fs/cachefiles/interface.c | 90 ++----- fs/cachefiles/internal.h | 13 +- fs/cachefiles/io.c | 435 +++++++++++++++++++----------- fs/cachefiles/namei.c | 19 +- fs/cachefiles/xattr.c | 23 +- fs/netfs/buffered_read.c | 188 ++++++++----- fs/netfs/buffered_write.c | 3 - fs/netfs/internal.h | 2 + fs/netfs/objects.c | 1 + fs/netfs/read_retry.c | 2 + fs/netfs/read_single.c | 40 +-- fs/netfs/write_collect.c | 132 +++++++-- fs/netfs/write_issue.c | 18 ++ fs/netfs/write_retry.c | 3 + include/linux/fscache.h | 17 ++ include/linux/netfs.h | 38 ++- include/trace/events/cachefiles.h | 21 +- include/trace/events/netfs.h | 9 +- 19 files changed, 667 insertions(+), 388 deletions(-) diff --git a/fs/afs/file.c b/fs/afs/file.c index b7f461b9d5f506..4c78d3441785dd 100644 --- a/fs/afs/file.c +++ b/fs/afs/file.c @@ -447,7 +447,6 @@ void afs_set_i_size(struct afs_vnode *vnode, uoff_t new_i_size) } spin_unlock(&inode->i_lock); write_sequnlock(&vnode->cb_lock); - fscache_update_cookie(afs_vnode_cache(vnode), NULL, &new_i_size); } static void afs_update_i_size(struct inode *inode, uoff_t new_i_size) diff --git a/fs/cachefiles/interface.c b/fs/cachefiles/interface.c index a160d5c3e74c0e..789ff6abe92624 100644 --- a/fs/cachefiles/interface.c +++ b/fs/cachefiles/interface.c @@ -99,73 +99,6 @@ void cachefiles_put_object(struct cachefiles_object *object, _leave(""); } -/* - * Adjust the size of a cache file if necessary to match the DIO size. We keep - * the EOF marker a multiple of DIO blocks so that we don't fall back to doing - * non-DIO for a partial block straddling the EOF, but we also have to be - * careful of someone expanding the file and accidentally accreting the - * padding. - */ -static int cachefiles_adjust_size(struct cachefiles_object *object) -{ - struct iattr newattrs; - struct file *file = object->file; - uint64_t ni_size; - uoff_t oi_size; - int ret; - - ni_size = object->cookie->object_size; - ni_size = round_up(ni_size, CACHEFILES_DIO_BLOCK_SIZE); - - _enter("{OBJ%x},[%llu]", - object->debug_id, (unsigned long long) ni_size); - - if (!file) - return -ENOBUFS; - - oi_size = i_size_read(file_inode(file)); - if (oi_size == ni_size) - return 0; - - inode_lock(file_inode(file)); - - /* if there's an extension to a partial page at the end of the backing - * file, we need to discard the partial page so that we pick up new - * data after it */ - if (oi_size & ~PAGE_MASK && ni_size > oi_size) { - _debug("discard tail %llx", oi_size); - newattrs.ia_valid = ATTR_SIZE; - newattrs.ia_size = oi_size & PAGE_MASK; - ret = cachefiles_inject_remove_error(); - if (ret == 0) - ret = notify_change(&nop_mnt_idmap, file->f_path.dentry, - &newattrs, NULL); - if (ret < 0) - goto truncate_failed; - } - - newattrs.ia_valid = ATTR_SIZE; - newattrs.ia_size = ni_size; - ret = cachefiles_inject_write_error(); - if (ret == 0) - ret = notify_change(&nop_mnt_idmap, file->f_path.dentry, - &newattrs, NULL); - -truncate_failed: - inode_unlock(file_inode(file)); - - if (ret < 0) - trace_cachefiles_io_error(NULL, file_inode(file), ret, - cachefiles_trace_notify_change_error); - if (ret == -EIO) { - cachefiles_io_error_obj(object, "Size set failed"); - ret = -ENOBUFS; - } - - _leave(" = %d", ret); - return ret; -} - /* * Attempt to look up the nominated node in this cache */ @@ -198,7 +131,6 @@ static bool cachefiles_lookup_cookie(struct fscache_cookie *cookie) spin_lock(&cache->object_list_lock); list_add(&object->cache_link, &cache->object_list); spin_unlock(&cache->object_list_lock); - cachefiles_adjust_size(object); cachefiles_end_secure(cache, saved_cred); _leave(" = t"); @@ -232,7 +164,7 @@ static bool cachefiles_shorten_object(struct cachefiles_object *object, uoff_t i_size, dio_size; int ret; - dio_size = round_up(new_size, CACHEFILES_DIO_BLOCK_SIZE); + dio_size = round_up(new_size, cache->bsize); i_size = i_size_read(inode); trace_cachefiles_trunc(object, inode, i_size, dio_size, @@ -264,6 +196,7 @@ static bool cachefiles_shorten_object(struct cachefiles_object *object, } } + object->object_size = new_size; return true; } @@ -278,22 +211,31 @@ static void cachefiles_resize_cookie(struct netfs_cache_resources *cres, struct fscache_cookie *cookie = object->cookie; const struct cred *saved_cred; struct file *file = cachefiles_cres_file(cres); - uoff_t old_size = cookie->object_size; + uoff_t i_size = i_size_read(file_inode(file)); - _enter("%llu->%llu", old_size, new_size); + _enter("%llu->%llu", object->object_size, new_size); - if (new_size < old_size) { + /* If the file is being shrunk, we need to downsize the backing file + * and clear the end of the final block. + */ + if (new_size < object->object_size) { + if (new_size >= i_size) + goto out; cachefiles_begin_secure(cache, &saved_cred); cachefiles_shorten_object(object, file, new_size); cachefiles_end_secure(cache, saved_cred); object->cookie->object_size = new_size; + if (new_size == 0) + object->content_info = CACHEFILES_CONTENT_NO_DATA; return; } /* The file is being expanded. We don't need to do anything - * particularly. cookie->initial_size doesn't change and so the point - * at which we have to download before doesn't change. + * particularly. The tail of the last block should have been cleared + * both when it is written and when it is shrunk. */ +out: + object->object_size = new_size; cookie->object_size = new_size; } diff --git a/fs/cachefiles/internal.h b/fs/cachefiles/internal.h index 60bd801ada04f3..b2605111fd5642 100644 --- a/fs/cachefiles/internal.h +++ b/fs/cachefiles/internal.h @@ -16,8 +16,6 @@ #include #include -#define CACHEFILES_DIO_BLOCK_SIZE 4096 - struct cachefiles_cache; struct cachefiles_object; @@ -51,12 +49,17 @@ struct cachefiles_object { struct list_head cache_link; /* Link in cache->*_list */ struct file *file; /* The file representing this object */ char *d_name; /* Backing file name */ + unsigned long flags; +#define CACHEFILES_OBJECT_USING_TMPFILE 0 /* Have an unlinked tmpfile */ + uoff_t object_size; /* Size of the object stored + * (independent of cookie->object_size for + * coherency reasons) + */ + atomic64_t read_limit; /* Point beyond which uncommitted writes */ int debug_id; spinlock_t lock; refcount_t ref; - enum cachefiles_content content_info:8; /* Info about content presence */ - unsigned long flags; -#define CACHEFILES_OBJECT_USING_TMPFILE 0 /* Have an unlinked tmpfile */ + enum cachefiles_content content_info; /* Info about content presence */ }; /* diff --git a/fs/cachefiles/io.c b/fs/cachefiles/io.c index 4b3ceda17426cf..4f547d97356ec3 100644 --- a/fs/cachefiles/io.c +++ b/fs/cachefiles/io.c @@ -32,6 +32,8 @@ struct cachefiles_kiocb { u64 b_writing; }; +#define IS_ERR_VALUE_LL(x) unlikely((x) >= (unsigned long long)-MAX_ERRNO) + static inline void cachefiles_put_kiocb(struct cachefiles_kiocb *ki) { if (refcount_dec_and_test(&ki->ki_refcnt)) { @@ -193,60 +195,81 @@ static int cachefiles_read(struct netfs_cache_resources *cres, } /* - * Query the occupancy of the cache in a region, returning where the next chunk - * of data starts and how long it is. + * Query the occupancy of the cache in a region, returning the extent of the + * next two chunks of cached data and the next hole. */ static int cachefiles_query_occupancy(struct netfs_cache_resources *cres, - uoff_t start, size_t len, size_t granularity, - uoff_t *_data_start, size_t *_data_len) + struct fscache_occupancy *occ) { struct cachefiles_object *object; + struct inode *inode; struct file *file; - loff_t off, off2; - - *_data_start = -1; - *_data_len = 0; + uoff_t read_limit; + loff_t ret; + int i; if (!fscache_wait_for_operation(cres, FSCACHE_WANT_READ)) return -ENOBUFS; object = cachefiles_cres_object(cres); file = cachefiles_cres_file(cres); - granularity = max_t(size_t, object->volume->cache->bsize, granularity); + inode = file_inode(file); + occ->granularity = object->volume->cache->bsize; + /* Read read_limit before content_info. */ + read_limit = atomic64_read_acquire(&object->read_limit); + + _enter("%pD,%llu,%llx-%llx/%llx", + file, inode->i_ino, occ->query_from, occ->query_to, read_limit); + + if (read_limit == 0) + goto done; + + switch (READ_ONCE(object->content_info)) { + case CACHEFILES_CONTENT_ALL: + case CACHEFILES_CONTENT_SINGLE: + if (read_limit > occ->query_from) { + occ->cached_from[0] = 0; + occ->cached_to[0] = read_limit; + occ->cached_type[0] = FSCACHE_EXTENT_DATA; + occ->query_from = ULLONG_MAX; + } + goto done; + default: + break; + } - _enter("%pD,%llu,%llx,%zx/%llx", - file, file_inode(file)->i_ino, start, len, - i_size_read(file_inode(file))); + for (i = 0; i < ARRAY_SIZE(occ->cached_from); i++) { + ret = cachefiles_inject_read_error(); + if (ret == 0) + ret = vfs_llseek(file, occ->query_from, SEEK_DATA); + if (IS_ERR_VALUE_LL(ret)) { + if (ret != -ENXIO) + return ret; + occ->query_from = ULLONG_MAX; + goto done; + } + occ->cached_type[i] = FSCACHE_EXTENT_DATA; + occ->cached_from[i] = ret; + occ->query_from = ret; + + ret = cachefiles_inject_read_error(); + if (ret == 0) + ret = vfs_llseek(file, occ->query_from, SEEK_HOLE); + if (IS_ERR_VALUE_LL(ret)) { + if (ret != -ENXIO) + return ret; + occ->query_from = ULLONG_MAX; + goto done; + } + occ->cached_to[i] = ret; + occ->query_from = ret; + if (occ->query_from >= occ->query_to) + break; + } - off = cachefiles_inject_read_error(); - if (off == 0) - off = vfs_llseek(file, start, SEEK_DATA); - if (off == -ENXIO) - return -ENODATA; /* Beyond EOF */ - if (off < 0 && off >= (loff_t)-MAX_ERRNO) - return -ENOBUFS; /* Error. */ - if (round_up(off, granularity) >= start + len) - return -ENODATA; /* No data in range */ - - off2 = cachefiles_inject_read_error(); - if (off2 == 0) - off2 = vfs_llseek(file, off, SEEK_HOLE); - if (off2 == -ENXIO) - return -ENODATA; /* Beyond EOF */ - if (off2 < 0 && off2 >= (loff_t)-MAX_ERRNO) - return -ENOBUFS; /* Error. */ - - /* Round away partial blocks */ - off = round_up(off, granularity); - off2 = round_down(off2, granularity); - if (off2 <= off) - return -ENODATA; - - *_data_start = off; - if (off2 > start + len) - *_data_len = len; - else - *_data_len = off2 - off; +done: + _debug("query[0] %llx-%llx", occ->cached_from[0], occ->cached_to[0]); + _debug("query[1] %llx-%llx", occ->cached_from[1], occ->cached_to[1]); return 0; } @@ -375,114 +398,6 @@ static int cachefiles_write(struct netfs_cache_resources *cres, term_func, term_func_priv); } -/* - * Prepare a read operation, shortening it to a cached/uncached boundary as - * appropriate. - */ -static enum netfs_io_source -cachefiles_prepare_read(struct netfs_io_subrequest *subreq, uoff_t i_size) -{ - enum cachefiles_prepare_read_trace why; - struct netfs_cache_resources *cres = &subreq->rreq->cache_resources; - struct cachefiles_object *object = NULL; - struct cachefiles_cache *cache; - struct fscache_cookie *cookie = fscache_cres_cookie(cres); - const struct cred *saved_cred; - struct file *file = cachefiles_cres_file(cres); - enum netfs_io_source ret = NETFS_DOWNLOAD_FROM_SERVER; - uoff_t start = subreq->start; - size_t len = subreq->len; - loff_t off, to; - ino_t ino = file ? file_inode(file)->i_ino : 0; - - _enter("%zx @%llx/%llx", len, start, i_size); - - if (start >= i_size) { - ret = NETFS_FILL_WITH_ZEROES; - why = cachefiles_trace_read_after_eof; - goto out_no_object; - } - - if (test_bit(FSCACHE_COOKIE_NO_DATA_TO_READ, &cookie->flags)) { - __set_bit(NETFS_SREQ_COPY_TO_CACHE, &subreq->flags); - why = cachefiles_trace_read_no_data; - goto out_no_object; - } - - /* The object and the file may be being created in the background. */ - if (!file) { - why = cachefiles_trace_read_no_file; - if (!fscache_wait_for_operation(cres, FSCACHE_WANT_READ)) - goto out_no_object; - file = cachefiles_cres_file(cres); - if (!file) - goto out_no_object; - ino = file_inode(file)->i_ino; - } - - object = cachefiles_cres_object(cres); - cache = object->volume->cache; - cachefiles_begin_secure(cache, &saved_cred); - off = cachefiles_inject_read_error(); - if (off == 0) - off = vfs_llseek(file, start, SEEK_DATA); - if (off < 0 && off >= (loff_t)-MAX_ERRNO) { - if (off == (loff_t)-ENXIO) { - why = cachefiles_trace_read_seek_nxio; - goto download_and_store; - } - trace_cachefiles_io_error(object, file_inode(file), off, - cachefiles_trace_seek_error); - why = cachefiles_trace_read_seek_error; - goto out; - } - - if (off >= start + len) { - why = cachefiles_trace_read_found_hole; - goto download_and_store; - } - - if (off > start) { - off = round_up(off, cache->bsize); - len = off - start; - subreq->len = len; - why = cachefiles_trace_read_found_part; - goto download_and_store; - } - - to = cachefiles_inject_read_error(); - if (to == 0) - to = vfs_llseek(file, start, SEEK_HOLE); - if (to < 0 && to >= (loff_t)-MAX_ERRNO) { - trace_cachefiles_io_error(object, file_inode(file), to, - cachefiles_trace_seek_error); - why = cachefiles_trace_read_seek_error; - goto out; - } - - if (to < start + len) { - if (start + len >= i_size) - to = round_up(to, cache->bsize); - else - to = round_down(to, cache->bsize); - len = to - start; - subreq->len = len; - } - - why = cachefiles_trace_read_have_data; - ret = NETFS_READ_FROM_CACHE; - goto out; - -download_and_store: - __set_bit(NETFS_SREQ_COPY_TO_CACHE, &subreq->flags); -out: - cachefiles_end_secure(cache, saved_cred); -out_no_object: - trace_cachefiles_prep_read(object, start, len, subreq->flags, ret, why, - ino, subreq->rreq->inode->i_ino); - return ret; -} - /* * Prepare for a write to occur. */ @@ -497,7 +412,7 @@ int __cachefiles_prepare_write(struct cachefiles_object *object, int ret; /* Round to DIO size */ - start = round_down(*_start, PAGE_SIZE); + start = round_down(*_start, cache->bsize); if (start != *_start || *_len > upper_len) { /* Probably asked to cache a streaming write written into the * pagecache when the cookie was temporarily out of service to @@ -507,7 +422,7 @@ int __cachefiles_prepare_write(struct cachefiles_object *object, return -ENOBUFS; } - *_len = round_up(len, PAGE_SIZE); + *_len = round_up(len, cache->bsize); /* We need to work out whether there's sufficient disk space to perform * the write - but we can skip that check if we have space already @@ -533,7 +448,7 @@ int __cachefiles_prepare_write(struct cachefiles_object *object, * space, we need to see if it's fully allocated. If it's not, we may * want to cull it. */ - ret = cachefiles_has_space(cache, 0, *_len / PAGE_SIZE, + ret = cachefiles_has_space(cache, 0, *_len / cache->bsize, cachefiles_has_space_check); if (ret == 0) return 0; /* Enough space to simply overwrite the whole block */ @@ -569,7 +484,7 @@ int __cachefiles_prepare_write(struct cachefiles_object *object, return ret; check_space: - ret = cachefiles_has_space(cache, 0, *_len / PAGE_SIZE, + ret = cachefiles_has_space(cache, 0, *_len / cache->bsize, cachefiles_has_space_for_write); if (ret == -ENOBUFS) trace_cachefiles_no_space(object, cachefiles_trace_write_nospace); @@ -639,9 +554,9 @@ static void cachefiles_issue_write(struct netfs_io_subrequest *subreq) wreq->debug_id, subreq->debug_index, start, start + len - 1); /* We need to start on the cache granularity boundary */ - off = start & (CACHEFILES_DIO_BLOCK_SIZE - 1); + off = start & (cache->bsize - 1); if (off) { - pre = CACHEFILES_DIO_BLOCK_SIZE - off; + pre = cache->bsize - off; if (pre >= len) { fscache_count_dio_misfit(); netfs_write_subrequest_terminated(subreq, len); @@ -655,8 +570,8 @@ static void cachefiles_issue_write(struct netfs_io_subrequest *subreq) /* We also need to end on the cache granularity boundary */ if (start + len == wreq->i_size) { - size_t part = len % CACHEFILES_DIO_BLOCK_SIZE; - size_t need = CACHEFILES_DIO_BLOCK_SIZE - part; + size_t part = len & (cache->bsize - 1); + size_t need = cache->bsize - part; if (part && stream->submit_extendable_to >= need) { len += need; @@ -665,7 +580,7 @@ static void cachefiles_issue_write(struct netfs_io_subrequest *subreq) } } - post = len & (CACHEFILES_DIO_BLOCK_SIZE - 1); + post = len & (cache->bsize - 1); if (post) { len -= post; if (len == 0) { @@ -692,6 +607,198 @@ static void cachefiles_issue_write(struct netfs_io_subrequest *subreq) netfs_write_subrequest_terminated, subreq); } +/* + * Collect the result of buffered writeback to the cache. This includes + * copying a read to the cache. Netfslib collates the results, which might + * occur out of order, and delivers them to the cache so that it can update its + * content record. + * + * block_type is one of: + * - NETFS_CACHE_COLLECT_WRITE_DATA for a contiguous block of data + * - NETFS_CACHE_COLLECT_WRITE_GAP if a discontiguity was skipped + * - NETFS_CACHE_COLLECT_WRITE_CANCEL for a hole due to a failed/cancelled write + * + * The writes we made are all rounded out at both sides to the nearest DIO + * block boundary, so if the final block contains the EOF in the middle of it + * (rather than at the end), padding will have been written to the file. The + * backing file's filesize will have been updated if the write extended the + * file; the filesize may still change due to outstanding subreqs. + * + * The metadata in the cache file xattr records the size of the object we have + * stored, but the cache file EOF only goes up to where we've cached data to + * and, furthermore, is rounded up to the nearest DIO block boundary. + * + * Concurrent updates should be protected against by the caller. Netfslib + * holds NETFS_ICTX_WB_LOCK as a lock on writeback requests. DIO writes + * invalidate the cookie and caching is kept disabled until all users have + * unused the cookie. + */ +static void cachefiles_collect_write(struct netfs_io_request *wreq, + uoff_t start, size_t len, + enum netfs_cache_collect block_type) +{ + struct netfs_cache_resources *cres = &wreq->cache_resources; + struct cachefiles_object *object = cachefiles_cres_object(cres); + struct cachefiles_cache *cache = object->volume->cache; + struct inode *inode; + struct file *file = cachefiles_cres_file(cres); + uoff_t read_limit; + uoff_t old_size = cres->cache_i_size; + uoff_t new_size; + uoff_t data_to = object->object_size; + uoff_t end = start + len; + int ret; + + if (!file) + return; + + inode = file_inode(file); + new_size = i_size_read(inode); + + _enter("%llx,%zx,%x", start, len, cache->bsize); + + if (WARN_ON(old_size & (cache->bsize - 1)) || + WARN_ON(new_size & (cache->bsize - 1)) || + WARN_ON(start & (cache->bsize - 1)) || + WARN_ON(len & (cache->bsize - 1))) { + trace_cachefiles_io_error(object, inode, -EIO, + cachefiles_trace_alignment_error); + cachefiles_remove_object_xattr(cache, object, file->f_path.dentry); + return; + } + + /* If this is recording a gap, due to discontiguous writes or lack of + * cache space, then a hole may have been introduced into the backing + * file. Treat it as a zero-length data block. + */ + if (block_type == NETFS_CACHE_COLLECT_WRITE_GAP || + block_type == NETFS_CACHE_COLLECT_WRITE_CANCEL) { + start = end; + len = 0; + } + + /* Zeroth case: Single monolithic files are handled specially. + */ + if (wreq->origin == NETFS_WRITEBACK_SINGLE) { + if (block_type == NETFS_CACHE_COLLECT_WRITE_GAP || + block_type == NETFS_CACHE_COLLECT_WRITE_CANCEL) { + trace_cachefiles_trunc(object, inode, data_to, 0, + cachefiles_trunc_zap); + ret = cachefiles_inject_remove_error(); + if (ret == 0) + ret = vfs_truncate(&file->f_path, 0); + if (ret < 0) { + trace_cachefiles_io_error(object, inode, ret, + cachefiles_trace_trunc_error); + cachefiles_io_error_obj(object, "truncate failed %d", ret); + cachefiles_remove_object_xattr(cache, object, file->f_path.dentry); + return; + } + + object->content_info = CACHEFILES_CONTENT_NO_DATA; + read_limit = 0; + } else { + object->content_info = CACHEFILES_CONTENT_SINGLE; + read_limit = len; + } + goto update_sizes_2; + } + + /* First case: The backing file was empty. */ + if (old_size == 0) { + if (start == 0) + object->content_info = CACHEFILES_CONTENT_ALL; + else + object->content_info = CACHEFILES_CONTENT_BACKFS_MAP; + goto update_sizes; + } + + /* Second case: The backing file is entirely within the old object size + * and thus there can be no partial tail block to deal with in the + * cache file. + */ + if (old_size <= data_to) { + if (start > old_size) + goto discontiguous; + goto update_sizes; + } + + /* Third case: The write happened entirely within the bounds of the + * current cache file's size. + */ + if (end <= old_size) + goto update_sizes; + + /* Fourth case: The write overwrote the partial tail block and extended + * the file. We only need to update the object size because netfslib + * rounds out/pads cache writes to whole disk blocks. + */ + if (start < old_size) + goto update_sizes; + + /* Fifth case: The write started from the end of the whole tail block + * and extended the file. Just extend our notion of the filesize. + */ + if (start == old_size && old_size == data_to) + goto update_sizes; + + /* Sixth case: The write continued on from the partial tail block and + * extended the file. Need to clear the gap. + */ + if (start == old_size && old_size > data_to) + goto clear_gap; + +discontiguous: + /* Seventh case: The write was beyond the EOF on the cache file, so now + * there's a hole in the file and we can no longer say in the metadata + * that we can assume we have it all. We may also need to clear the + * end of the partial tail block. + */ + /* TODO: For the moment, we will have to use SEEK_HOLE/SEEK_DATA. */ + if (object->content_info != CACHEFILES_CONTENT_BACKFS_MAP) { + object->content_info = CACHEFILES_CONTENT_BACKFS_MAP; + trace_cachefiles_coherency(object, inode->i_ino, data_to, NULL, + CACHEFILES_CONTENT_BACKFS_MAP, + cachefiles_coherency_discontiguous); + } + +clear_gap: + /* We need to clear any partial padding that got jumped over. It + * *should* be all zeros, but shared-writable mmap exists... + */ + if (old_size > data_to) { + trace_cachefiles_trunc(object, inode, data_to, old_size, + cachefiles_trunc_clear_padding); + ret = cachefiles_inject_write_error(); + if (ret == 0) + ret = vfs_fallocate(file, FALLOC_FL_ZERO_RANGE, + data_to, old_size - data_to); + if (ret < 0) { + trace_cachefiles_io_error(object, inode, ret, + cachefiles_trace_fallocate_error); + cachefiles_io_error_obj(object, "fallocate zero pad failed %d", ret); + cachefiles_remove_object_xattr(cache, object, file->f_path.dentry); + return; + } + } + +update_sizes: + read_limit = umax(old_size, end); +update_sizes_2: + cres->cache_i_size = read_limit; + + /* We need to be careful setting the object_size: we may have written + * more to the cache than to the server (due to cache DIO rounding) and + * the i_size set on the netfs inode may include unwritten data that + * the server doesn't know about yet. + */ + object->object_size = umin(read_limit, wreq->i_size); + + /* Raise the limit at which reads can access the file. */ + /* Update read_limit after content_info */ + atomic64_set_release(&object->read_limit, read_limit); +} + /* * Clean up an operation. */ @@ -709,10 +816,10 @@ static const struct netfs_cache_ops cachefiles_netfs_cache_ops = { .read = cachefiles_read, .write = cachefiles_write, .issue_write = cachefiles_issue_write, - .prepare_read = cachefiles_prepare_read, .prepare_write = cachefiles_prepare_write, .prepare_write_subreq = cachefiles_prepare_write_subreq, .query_occupancy = cachefiles_query_occupancy, + .collect_write = cachefiles_collect_write, }; /* @@ -722,14 +829,20 @@ bool cachefiles_begin_operation(struct netfs_cache_resources *cres, enum fscache_want_state want_state) { struct cachefiles_object *object = cachefiles_cres_object(cres); + struct file *file; + + cres->dio_size = object->volume->cache->bsize; if (!cachefiles_cres_file(cres)) { cres->ops = &cachefiles_netfs_cache_ops; cres->object_id = object->debug_id; if (object->file) { spin_lock(&object->lock); - if (!cres->cache_priv2 && object->file) - cres->cache_priv2 = get_file(object->file); + file = object->file; + if (!cres->cache_priv2 && file) { + cres->cache_priv2 = get_file(file); + cres->cache_i_size = i_size_read(file_inode(file)); + } spin_unlock(&object->lock); } } diff --git a/fs/cachefiles/namei.c b/fs/cachefiles/namei.c index 4780ce6dc8309b..ca093840e5774d 100644 --- a/fs/cachefiles/namei.c +++ b/fs/cachefiles/namei.c @@ -417,7 +417,6 @@ struct file *cachefiles_create_tmpfile(struct cachefiles_object *object) struct dentry *fan = volume->fanout[(u8)object->cookie->key_hash]; struct file *file; const struct path parentpath = { .mnt = cache->mnt, .dentry = fan }; - uint64_t ni_size; long ret; @@ -445,23 +444,6 @@ struct file *cachefiles_create_tmpfile(struct cachefiles_object *object) if (!cachefiles_mark_inode_in_use(object, file_inode(file))) WARN_ON(1); - ni_size = object->cookie->object_size; - ni_size = round_up(ni_size, CACHEFILES_DIO_BLOCK_SIZE); - - if (ni_size > 0) { - trace_cachefiles_trunc(object, file_inode(file), 0, ni_size, - cachefiles_trunc_expand_tmpfile); - ret = cachefiles_inject_write_error(); - if (ret == 0) - ret = vfs_truncate(&file->f_path, ni_size); - if (ret < 0) { - trace_cachefiles_vfs_error( - object, file_inode(file), ret, - cachefiles_trace_trunc_error); - goto err_unuse; - } - } - ret = -EINVAL; if (unlikely(!file->f_op->read_iter) || unlikely(!file->f_op->write_iter)) { @@ -470,6 +452,7 @@ struct file *cachefiles_create_tmpfile(struct cachefiles_object *object) } out: cachefiles_end_secure(cache, saved_cred); + object->content_info = CACHEFILES_CONTENT_ALL; return file; err_unuse: diff --git a/fs/cachefiles/xattr.c b/fs/cachefiles/xattr.c index c70bf67e52b01a..551a3b0069c211 100644 --- a/fs/cachefiles/xattr.c +++ b/fs/cachefiles/xattr.c @@ -43,6 +43,7 @@ int cachefiles_set_object_xattr(struct cachefiles_object *object) struct dentry *dentry; struct file *file = object->file; unsigned int len = object->cookie->aux_len; + uoff_t object_size = object->cookie->object_size; int ret; if (!file) @@ -55,7 +56,7 @@ int cachefiles_set_object_xattr(struct cachefiles_object *object) if (!buf) return -ENOMEM; - buf->object_size = cpu_to_be64(object->cookie->object_size); + buf->object_size = cpu_to_be64(object_size); buf->zero_point = 0; buf->type = CACHEFILES_COOKIE_TYPE_DATA; buf->content = object->content_info; @@ -79,7 +80,7 @@ int cachefiles_set_object_xattr(struct cachefiles_object *object) trace_cachefiles_vfs_error(object, file_inode(file), ret, cachefiles_trace_setxattr_error); trace_cachefiles_coherency(object, file_inode(file)->i_ino, - buf->data, buf->content, + object_size, buf->data, buf->content, cachefiles_coherency_set_fail); if (ret != -ENOMEM) cachefiles_io_error_obj( @@ -87,7 +88,7 @@ int cachefiles_set_object_xattr(struct cachefiles_object *object) "Failed to set xattr with error %d", ret); } else { trace_cachefiles_coherency(object, file_inode(file)->i_ino, - buf->data, buf->content, + object_size, buf->data, buf->content, cachefiles_coherency_set_ok); } @@ -103,10 +104,12 @@ int cachefiles_check_auxdata(struct cachefiles_object *object, struct file *file { struct cachefiles_xattr *buf; struct dentry *dentry = file->f_path.dentry; + struct inode *inode = file_inode(file); unsigned int len = object->cookie->aux_len, tlen; const void *p = fscache_get_aux(object->cookie); enum cachefiles_coherency_trace why; ssize_t xlen; + uoff_t obj_size; int ret = -ESTALE; tlen = sizeof(struct cachefiles_xattr) + len; @@ -121,34 +124,39 @@ int cachefiles_check_auxdata(struct cachefiles_object *object, struct file *file if (xlen != tlen) { if (xlen < 0) { ret = xlen; - trace_cachefiles_vfs_error(object, file_inode(file), xlen, + trace_cachefiles_vfs_error(object, inode, xlen, cachefiles_trace_getxattr_error); } if (xlen == -EIO) cachefiles_io_error_obj( object, "Failed to read aux with error %zd", xlen); + obj_size = 0; why = cachefiles_coherency_check_xattr; goto out; } + obj_size = be64_to_cpu(buf->object_size); if (buf->type != CACHEFILES_COOKIE_TYPE_DATA) { why = cachefiles_coherency_check_type; } else if (memcmp(buf->data, p, len) != 0) { why = cachefiles_coherency_check_aux; - } else if (be64_to_cpu(buf->object_size) != object->cookie->object_size) { + } else if (obj_size != object->cookie->object_size) { why = cachefiles_coherency_check_objsize; } else if (buf->content == CACHEFILES_CONTENT_DIRTY) { // TODO: Begin conflict resolution pr_warn("Dirty object in cache\n"); why = cachefiles_coherency_check_dirty; } else { + object->content_info = buf->content; + object->object_size = obj_size; + atomic64_set(&object->read_limit, i_size_read(inode)); why = cachefiles_coherency_check_ok; ret = 0; } out: - trace_cachefiles_coherency(object, file_inode(file)->i_ino, + trace_cachefiles_coherency(object, inode->i_ino, obj_size, buf->data, buf->content, why); kfree(buf); return ret; @@ -163,6 +171,9 @@ int cachefiles_remove_object_xattr(struct cachefiles_cache *cache, { int ret; + trace_cachefiles_coherency(object, d_inode(dentry)->i_ino, 0, NULL, 0, + cachefiles_coherency_remove); + ret = cachefiles_inject_remove_error(); if (ret == 0) { ret = mnt_want_write(cache->mnt); diff --git a/fs/netfs/buffered_read.c b/fs/netfs/buffered_read.c index 68496e1a171f53..cc1b00f301d24b 100644 --- a/fs/netfs/buffered_read.c +++ b/fs/netfs/buffered_read.c @@ -137,21 +137,6 @@ static ssize_t netfs_prepare_read_iterator(struct netfs_io_subrequest *subreq) return subreq->len; } -static enum netfs_io_source netfs_cache_prepare_read(struct netfs_io_request *rreq, - struct netfs_io_subrequest *subreq, - uoff_t i_size) -{ - struct netfs_cache_resources *cres = &rreq->cache_resources; - enum netfs_io_source source; - - if (!cres->ops) - return NETFS_DOWNLOAD_FROM_SERVER; - source = cres->ops->prepare_read(subreq, i_size); - trace_netfs_sreq(subreq, netfs_sreq_trace_prepare); - return source; - -} - /* * Issue a read against the cache. * - Eats the caller's ref on subreq. @@ -166,6 +151,19 @@ static void netfs_read_cache_to_pagecache(struct netfs_io_request *rreq, netfs_cache_read_terminated, subreq); } +int netfs_read_query_cache(struct netfs_io_request *rreq, struct fscache_occupancy *occ) +{ + struct netfs_cache_resources *cres = &rreq->cache_resources; + + occ->granularity = PAGE_SIZE; + if (occ->query_from >= occ->query_to) + return 0; + if (!cres->ops) + return 0; + occ->query_from = round_up(occ->query_from, occ->granularity); + return cres->ops->query_occupancy(cres, occ); +} + void netfs_queue_read(struct netfs_io_request *rreq, struct netfs_io_subrequest *subreq) { @@ -268,6 +266,15 @@ static void netfs_mark_copy_to_cache(struct netfs_io_request *rreq, */ static void netfs_read_to_pagecache(struct netfs_io_request *rreq) { + struct fscache_occupancy _occ = { + .query_from = rreq->start, + .query_to = rreq->start + rreq->len, + .cached_from[0] = 0, + .cached_to[0] = 0, + .cached_from[1] = ULLONG_MAX, + .cached_to[1] = ULLONG_MAX, + }; + struct fscache_occupancy *occ = &_occ; struct folio_queue *fq = rreq->buffer.tail; unsigned int offset = 0; ssize_t size = rreq->len; @@ -275,11 +282,97 @@ static void netfs_read_to_pagecache(struct netfs_io_request *rreq) int ret = 0, slot = 0; do { + int (*prepare_read)(struct netfs_io_subrequest *subreq) = NULL; struct netfs_io_subrequest *subreq; enum netfs_io_source source; ssize_t slice; + uoff_t hole_to, cache_to; + size_t len = size; + bool copy = false; + + /* If we don't have any, find out the next couple of data + * extents from the cache, containing of following the + * specified start offset. Holes have to be fetched from the + * server; data regions from the cache. + */ + hole_to = occ->cached_from[0]; + cache_to = occ->cached_to[0]; + if (start >= cache_to) { + /* Extent exhausted; shuffle down. */ + int i; + + for (i = 0; i < ARRAY_SIZE(occ->cached_from) - 1; i++) { + occ->cached_from[i] = occ->cached_from[i + 1]; + occ->cached_to[i] = occ->cached_to[i + 1]; + occ->cached_type[i] = occ->cached_type[i + 1]; + } + occ->cached_from[i] = ULLONG_MAX; + occ->cached_to[i] = ULLONG_MAX; + + if (occ->cached_from[0] != ULLONG_MAX) + continue; + + /* Get new extents */ + ret = netfs_read_query_cache(rreq, occ); + if (ret < 0) + break; + continue; + } + + uoff_t zero_point = netfs_read_zero_point(rreq->inode); + uoff_t zlimit = umin(zero_point, rreq->i_size); + + _debug("rsub %llx %llx-%llx", start, hole_to, cache_to); + + if (start >= hole_to && start < cache_to) { + /* Overlap with a cached region, where the cache may + * record a block of zeroes. + */ + _debug("cached s=%llx c=%llx l=%zx", start, cache_to, size); + len = umin(cache_to - start, size); + len = round_up(len, occ->granularity); + if (occ->cached_type[0] == FSCACHE_EXTENT_ZERO) { + source = NETFS_FILL_WITH_ZEROES; + netfs_stat(&netfs_n_rh_zero); + } else { + source = NETFS_READ_FROM_CACHE; + prepare_read = rreq->cache_resources.ops->prepare_read; + } + } else if (start >= zlimit && size > 0) { + /* If this range lies beyond the zero-point, that part + * can just be cleared locally. + */ + _debug("zero %llx-%llx", start, start + size); + len = size; + source = NETFS_FILL_WITH_ZEROES; + if (rreq->cache_resources.ops) + copy = true; + netfs_stat(&netfs_n_rh_zero); + } else { + /* Read a cache hole from the server. If any part of + * this range lies beyond the zero-point or the EOF, + * that part can just be cleared locally. + */ + uoff_t limit = min3(zlimit, start + size, hole_to); + + _debug("limit %llx %llx", rreq->i_size, zero_point); + _debug("download %llx-%llx", start, start + size); + len = umin(limit - start, ULONG_MAX); + source = NETFS_DOWNLOAD_FROM_SERVER; + prepare_read = rreq->netfs_ops->prepare_read; + if (rreq->cache_resources.ops) + copy = true; + netfs_stat(&netfs_n_rh_download); + } - subreq = netfs_alloc_subrequest(rreq, NETFS_SOURCE_UNKNOWN); + if (len == 0) { + pr_err("ZERO-LEN READ: R=%08x l=%zx/%zx s=%llx z=%llx i=%llx", + rreq->debug_id, len, size, + start, zero_point, rreq->i_size); + break; + } + + subreq = netfs_alloc_subrequest(rreq, source); if (!subreq) { ret = -ENOMEM; break; @@ -287,66 +380,23 @@ static void netfs_read_to_pagecache(struct netfs_io_request *rreq) subreq->start = start; subreq->len = size; + if (copy) + __set_bit(NETFS_SREQ_COPY_TO_CACHE, &subreq->flags); netfs_queue_read(rreq, subreq); - source = netfs_cache_prepare_read(rreq, subreq, rreq->i_size); - subreq->source = source; - if (source == NETFS_DOWNLOAD_FROM_SERVER) { - uoff_t zero_point = netfs_read_zero_point(rreq->inode); - uoff_t zp = umin(zero_point, rreq->i_size); - size_t len = subreq->len; - - if (unlikely(rreq->origin == NETFS_READ_SINGLE)) - zp = rreq->i_size; - if (subreq->start >= zp) { - subreq->source = source = NETFS_FILL_WITH_ZEROES; - goto fill_with_zeroes; - } + rreq->io_streams[0].sreq_max_len = MAX_RW_COUNT; + rreq->io_streams[0].sreq_max_segs = INT_MAX; - if (len > zp - subreq->start) - len = zp - subreq->start; - if (len == 0) { - pr_err("ZERO-LEN READ: R=%08x[%x] l=%zx/%zx s=%llx z=%llx i=%llx", - rreq->debug_id, subreq->debug_index, - subreq->len, size, - subreq->start, zero_point, rreq->i_size); + if (prepare_read) { + ret = prepare_read(subreq); + if (ret < 0) { netfs_cancel_read(subreq, ret); break; } - subreq->len = len; - - netfs_stat(&netfs_n_rh_download); - if (rreq->netfs_ops->prepare_read) { - ret = rreq->netfs_ops->prepare_read(subreq); - if (ret < 0) { - netfs_cancel_read(subreq, ret); - break; - } - trace_netfs_sreq(subreq, netfs_sreq_trace_prepare); - } - goto issue; - } - - fill_with_zeroes: - if (source == NETFS_FILL_WITH_ZEROES) { - subreq->source = NETFS_FILL_WITH_ZEROES; - trace_netfs_sreq(subreq, netfs_sreq_trace_submit); - netfs_stat(&netfs_n_rh_zero); - goto issue; + trace_netfs_sreq(subreq, netfs_sreq_trace_prepare); } - if (source == NETFS_READ_FROM_CACHE) { - trace_netfs_sreq(subreq, netfs_sreq_trace_submit); - goto issue; - } - - pr_err("Unexpected read source %u\n", source); - WARN_ON_ONCE(1); - netfs_cancel_read(subreq, ret); - break; - - issue: slice = netfs_prepare_read_iterator(subreq); if (slice < 0) { ret = slice; @@ -360,11 +410,11 @@ static void netfs_read_to_pagecache(struct netfs_io_request *rreq) if (fq) { /* See if the cache indicated this should be cached. */ - bool copy = test_bit(NETFS_SREQ_COPY_TO_CACHE, &subreq->flags); - + copy = test_bit(NETFS_SREQ_COPY_TO_CACHE, &subreq->flags); netfs_mark_copy_to_cache(rreq, &fq, &slot, &offset, slice, copy); } + trace_netfs_sreq(subreq, netfs_sreq_trace_submit); netfs_issue_read(rreq, subreq); netfs_maybe_bulk_drop_ra_refs(rreq); diff --git a/fs/netfs/buffered_write.c b/fs/netfs/buffered_write.c index ead22980075fea..49b47252f67502 100644 --- a/fs/netfs/buffered_write.c +++ b/fs/netfs/buffered_write.c @@ -54,9 +54,6 @@ void netfs_update_i_size(struct netfs_inode *ctx, struct inode *inode, i_size = i_size_read(inode); if (end > i_size) { i_size_write(inode, end); -#if IS_ENABLED(CONFIG_FSCACHE) - fscache_update_cookie(ctx->cache, NULL, &end); -#endif gap = SECTOR_SIZE - (i_size & (SECTOR_SIZE - 1)); if (copied > gap) { diff --git a/fs/netfs/internal.h b/fs/netfs/internal.h index 4891e6e3c5db68..b8591abc90a985 100644 --- a/fs/netfs/internal.h +++ b/fs/netfs/internal.h @@ -23,6 +23,8 @@ /* * buffered_read.c */ +int netfs_read_query_cache(struct netfs_io_request *rreq, + struct fscache_occupancy *occ); void netfs_queue_read(struct netfs_io_request *rreq, struct netfs_io_subrequest *subreq); void netfs_cache_read_terminated(void *priv, ssize_t transferred_or_error); diff --git a/fs/netfs/objects.c b/fs/netfs/objects.c index 9c6ea718692a7d..cb28917e6e2046 100644 --- a/fs/netfs/objects.c +++ b/fs/netfs/objects.c @@ -44,6 +44,7 @@ struct netfs_io_request *netfs_alloc_request(struct address_space *mapping, rreq->gfp = gfp; rreq->start = start; rreq->collected_to = start; + rreq->cache_coll_to = start; rreq->cleaned_to = start; rreq->len = len; rreq->progress_at = 0; diff --git a/fs/netfs/read_retry.c b/fs/netfs/read_retry.c index 396a05432a7e92..5bd8dee5a83467 100644 --- a/fs/netfs/read_retry.c +++ b/fs/netfs/read_retry.c @@ -271,6 +271,7 @@ void netfs_retry_reads(struct netfs_io_request *rreq) struct netfs_io_stream *stream = &rreq->io_streams[0]; netfs_stat(&netfs_n_rh_retry_read_req); + trace_netfs_rreq(rreq, netfs_rreq_trace_retry_begin); /* Wait for all outstanding I/O to quiesce before performing retries as * we may need to renegotiate the I/O sizes. @@ -281,6 +282,7 @@ void netfs_retry_reads(struct netfs_io_request *rreq) trace_netfs_rreq(rreq, netfs_rreq_trace_resubmit); netfs_retry_read_subrequests(rreq); + trace_netfs_rreq(rreq, netfs_rreq_trace_retry_end); } /* diff --git a/fs/netfs/read_single.c b/fs/netfs/read_single.c index 81295c055cedbc..b248e34bd0c86d 100644 --- a/fs/netfs/read_single.c +++ b/fs/netfs/read_single.c @@ -58,20 +58,6 @@ static int netfs_single_begin_cache_read(struct netfs_io_request *rreq, struct n return fscache_begin_read_operation(&rreq->cache_resources, netfs_i_cookie(ctx)); } -static void netfs_single_cache_prepare_read(struct netfs_io_request *rreq, - struct netfs_io_subrequest *subreq) -{ - struct netfs_cache_resources *cres = &rreq->cache_resources; - - if (!cres->ops) { - subreq->source = NETFS_DOWNLOAD_FROM_SERVER; - return; - } - subreq->source = cres->ops->prepare_read(subreq, rreq->i_size); - trace_netfs_sreq(subreq, netfs_sreq_trace_prepare); - -} - static void netfs_single_read_cache(struct netfs_io_request *rreq, struct netfs_io_subrequest *subreq) { @@ -89,10 +75,27 @@ static void netfs_single_read_cache(struct netfs_io_request *rreq, */ static int netfs_single_dispatch_read(struct netfs_io_request *rreq) { + struct fscache_occupancy occ = { + .query_from = 0, + .query_to = rreq->len, + .cached_from[0] = ULLONG_MAX, + .cached_to[0] = ULLONG_MAX, + .cached_from[1] = ULLONG_MAX, + .cached_to[1] = ULLONG_MAX, + }; struct netfs_io_subrequest *subreq; + enum netfs_io_source source = NETFS_DOWNLOAD_FROM_SERVER; int ret = 0; - subreq = netfs_alloc_subrequest(rreq, NETFS_SOURCE_UNKNOWN); + /* Try to use the cache if the cache content matches the size of the + * remote file. + */ + netfs_read_query_cache(rreq, &occ); + if (occ.cached_from[0] == 0 && + occ.cached_to[0] >= rreq->len) + source = NETFS_READ_FROM_CACHE; + + subreq = netfs_alloc_subrequest(rreq, source); if (!subreq) return -ENOMEM; @@ -102,7 +105,6 @@ static int netfs_single_dispatch_read(struct netfs_io_request *rreq) netfs_queue_read(rreq, subreq); - netfs_single_cache_prepare_read(rreq, subreq); switch (subreq->source) { case NETFS_DOWNLOAD_FROM_SERVER: netfs_stat(&netfs_n_rh_download); @@ -117,6 +119,12 @@ static int netfs_single_dispatch_read(struct netfs_io_request *rreq) rreq->submitted += subreq->len; break; case NETFS_READ_FROM_CACHE: + if (rreq->cache_resources.ops->prepare_read) { + ret = rreq->cache_resources.ops->prepare_read(subreq); + if (ret < 0) + goto cancel; + } + netfs_all_subreqs_queued(rreq); trace_netfs_sreq(subreq, netfs_sreq_trace_submit); netfs_single_read_cache(rreq, subreq); diff --git a/fs/netfs/write_collect.c b/fs/netfs/write_collect.c index 6d99d4a6f7808c..6e8ea534230df0 100644 --- a/fs/netfs/write_collect.c +++ b/fs/netfs/write_collect.c @@ -188,6 +188,26 @@ static void netfs_writeback_unlock_folios(struct netfs_io_request *wreq, wreq->buffer.first_tail_slot = slot; } +/* + * Collect cache results. + */ +static void netfs_cache_collect(struct netfs_io_request *wreq, + struct netfs_io_stream *stream, + enum netfs_cache_collect block_type) +{ + struct netfs_cache_resources *cres = &wreq->cache_resources; + + if (stream->source != NETFS_WRITE_TO_CACHE || + wreq->cache_coll_to >= stream->collected_to) + return; + + if (cres->ops && cres->ops->collect_write) + cres->ops->collect_write(wreq, wreq->cache_coll_to, + stream->collected_to - wreq->cache_coll_to, + block_type); + wreq->cache_coll_to = stream->collected_to; +} + /* * Collect and assess the results of various write subrequests. We may need to * retry some of the results - or even do an RMW cycle for content crypto. @@ -235,13 +255,19 @@ static void netfs_collect_write_results(struct netfs_io_request *wreq) /* Read first subreq pointer before IN_PROGRESS flag. */ while (front) { + enum netfs_cache_collect cache_collect; + trace_netfs_collect_sreq(wreq, front); //_debug("sreq [%x] %llx %zx/%zx", // front->debug_index, front->start, front->transferred, front->len); if (stream->collected_to < front->start) { trace_netfs_collect_gap(wreq, stream, issued_to, 'F'); + if (stream->cache_collect != NETFS_CACHE_COLLECT_WRITE_GAP) + netfs_cache_collect(wreq, stream, stream->cache_collect); stream->collected_to = front->start; + netfs_cache_collect(wreq, stream, NETFS_CACHE_COLLECT_WRITE_GAP); + stream->cache_collect = NETFS_CACHE_COLLECT_WRITE_GAP; } /* Stall if the front is still undergoing I/O. */ @@ -249,7 +275,6 @@ static void netfs_collect_write_results(struct netfs_io_request *wreq) notes |= HIT_PENDING; break; } - smp_rmb(); /* Read counters after I-P flag. */ if (stream->failed) { stream->collected_to = front->start + front->len; @@ -262,15 +287,44 @@ static void netfs_collect_write_results(struct netfs_io_request *wreq) stream->transferred_valid = true; notes |= MADE_PROGRESS; } - if (test_bit(NETFS_SREQ_FAILED, &front->flags)) { - stream->failed = true; - stream->error = front->error; - if (stream->source == NETFS_UPLOAD_TO_SERVER) - mapping_set_error(wreq->mapping, front->error); - notes |= NEED_REASSESS | SAW_FAILURE; + + /* Handle failed or cancelled subreqs. Failure of + * cache writes are handled differently to upload + * failures. Cache writes aren't fatal, provided we're + * not doing disconnected operation, and so we can kind + * of treat them as if they had succeeded - except that + * we need to log any holes they cause. + */ + switch (stream->source) { + case NETFS_UPLOAD_TO_SERVER: + if (test_bit(NETFS_SREQ_FAILED, &front->flags)) { + if (!stream->failed) { + stream->failed = true; + stream->error = front->error; + mapping_set_error(wreq->mapping, front->error); + break; + } + notes |= NEED_REASSESS | SAW_FAILURE; + } + break; + + case NETFS_WRITE_TO_CACHE: + cache_collect = test_bit(NETFS_SREQ_CANCELLED, &front->flags) ? + NETFS_CACHE_COLLECT_WRITE_CANCEL : + NETFS_CACHE_COLLECT_WRITE_DATA; + if (cache_collect != stream->cache_collect && + stream->cache_collect != NETFS_CACHE_COLLECT_WRITE_GAP) { + trace_netfs_rreq(wreq, netfs_rreq_trace_cache_fail_collect); + netfs_cache_collect(wreq, stream, stream->cache_collect); + } + stream->cache_collect = cache_collect; + break; + + default: + WARN_ON(1); break; } - if (front->transferred < front->len) { + if (test_bit(NETFS_SREQ_NEED_RETRY, &front->flags)) { stream->need_retry = true; notes |= NEED_RETRY | MADE_PROGRESS; break; @@ -359,6 +413,7 @@ static void netfs_collect_write_results(struct netfs_io_request *wreq) */ bool netfs_write_collection(struct netfs_io_request *wreq) { + struct netfs_io_stream *cstream = &wreq->io_streams[1]; struct netfs_inode *ictx = netfs_inode(wreq->inode); size_t transferred; bool transferred_valid = false; @@ -393,13 +448,19 @@ bool netfs_write_collection(struct netfs_io_request *wreq) wreq->transferred = transferred; trace_netfs_rreq(wreq, netfs_rreq_trace_write_done); - if (wreq->io_streams[1].active && - wreq->io_streams[1].failed && - ictx->ops->invalidate_cache) { - /* Cache write failure doesn't prevent writeback completion - * unless we're in disconnected mode. - */ - ictx->ops->invalidate_cache(wreq); + if (cstream->active) { + if (test_bit(NETFS_RREQ_CACHE_ERROR, &wreq->flags)) { + if (ictx->ops->invalidate_cache) { + /* Cache write failure doesn't prevent + * writeback completion unless we're in + * disconnected mode. + */ + trace_netfs_rreq(wreq, netfs_rreq_trace_inval_cache); + ictx->ops->invalidate_cache(wreq); + } + } else if (!cstream->failed) { + netfs_cache_collect(wreq, cstream, cstream->cache_collect); + } } _debug("finished"); @@ -483,24 +544,51 @@ void netfs_write_subrequest_terminated(void *_op, ssize_t transferred_or_error) if (IS_ERR_VALUE(transferred_or_error)) { subreq->error = transferred_or_error; - /* if need retry is set, error should not matter */ - if (!test_bit(NETFS_SREQ_NEED_RETRY, &subreq->flags)) { - set_bit(NETFS_SREQ_FAILED, &subreq->flags); - trace_netfs_failure(wreq, subreq, transferred_or_error, netfs_fail_write); - } switch (subreq->source) { case NETFS_WRITE_TO_CACHE: + /* We don't mark a cache-write subreq as failed. + * Instead we tell the issuer to produce dummy subreqs + * instead and make a note if we need to invalidate the + * cache at the end. We also don't pause the loop that + * grabs pages and launches upload subreqs. + * + * Note that we need to distinguish between -ENOBUFS + * (no space available in the cache) and other errors. + * In the former case, we can keep the data we have, + * though we might have to change the way the on-disk + * data is tracked. + */ netfs_stat(&netfs_n_wh_write_failed); + if (test_bit(NETFS_SREQ_NEED_RETRY, &subreq->flags)) + break; + + trace_netfs_failure(wreq, subreq, transferred_or_error, netfs_fail_write); + __set_bit(NETFS_SREQ_CANCELLED, &subreq->flags); + set_bit(NETFS_RREQ_CACHE_STOP, &wreq->flags); + if (transferred_or_error == -ENOBUFS) + trace_netfs_rreq(wreq, netfs_rreq_trace_cache_no_space); + else if (!test_and_set_bit(NETFS_RREQ_CACHE_ERROR, &wreq->flags)) + trace_netfs_rreq(wreq, netfs_rreq_trace_cache_failed); + subreq->transferred = subreq->len; break; + case NETFS_UPLOAD_TO_SERVER: + /* If need_retry is set, error should not matter */ + if (!test_bit(NETFS_SREQ_NEED_RETRY, &subreq->flags)) { + set_bit(NETFS_SREQ_FAILED, &subreq->flags); + trace_netfs_failure(wreq, subreq, transferred_or_error, + netfs_fail_upload); + } + + set_bit(NETFS_RREQ_PAUSE, &wreq->flags); + trace_netfs_rreq(wreq, netfs_rreq_trace_set_pause); netfs_stat(&netfs_n_wh_upload_failed); break; + default: break; } - trace_netfs_rreq(wreq, netfs_rreq_trace_set_pause); - set_bit(NETFS_RREQ_PAUSE, &wreq->flags); } else { if (WARN(transferred_or_error > subreq->len - subreq->transferred, "Subreq excess write: R=%x[%x] %zd > %zu - %zu", diff --git a/fs/netfs/write_issue.c b/fs/netfs/write_issue.c index e7cb496f63246a..3989b4ec0c4b43 100644 --- a/fs/netfs/write_issue.c +++ b/fs/netfs/write_issue.c @@ -111,6 +111,8 @@ struct netfs_io_request *netfs_create_write_req(struct address_space *mapping, goto nomem; wreq->cleaned_to = wreq->start; + if (wreq->cache_resources.dio_size > 1) + wreq->cache_coll_to = round_down(wreq->start, wreq->cache_resources.dio_size); wreq->io_streams[0].stream_nr = 0; wreq->io_streams[0].source = NETFS_UPLOAD_TO_SERVER; @@ -231,6 +233,21 @@ static void netfs_do_issue_write(struct netfs_io_stream *stream, _enter("R=%x[%x],%zx", wreq->debug_id, subreq->debug_index, subreq->len); + if (stream->source == NETFS_WRITE_TO_CACHE && + unlikely(test_bit(NETFS_RREQ_CACHE_STOP, &wreq->flags))) { + size_t dio_size = wreq->cache_resources.dio_size; + size_t len, disp; + + disp = subreq->start & (dio_size - 1); + len = round_up(subreq->len + disp, dio_size); + + subreq->start -= disp; + subreq->len = len; + + __set_bit(NETFS_SREQ_CANCELLED, &subreq->flags); + return netfs_write_subrequest_terminated(subreq, subreq->len); + } + if (test_bit(NETFS_SREQ_FAILED, &subreq->flags)) return netfs_write_subrequest_terminated(subreq, subreq->error); @@ -264,6 +281,7 @@ void netfs_issue_write(struct netfs_io_request *wreq, if (!subreq) return; + stream->construct = NULL; subreq->io_iter.count = subreq->len; netfs_do_issue_write(stream, subreq); diff --git a/fs/netfs/write_retry.c b/fs/netfs/write_retry.c index d7d5348496fc8d..2f20577563e145 100644 --- a/fs/netfs/write_retry.c +++ b/fs/netfs/write_retry.c @@ -210,6 +210,7 @@ void netfs_retry_writes(struct netfs_io_request *wreq) int s; netfs_stat(&netfs_n_wh_retry_write_req); + trace_netfs_rreq(wreq, netfs_rreq_trace_retry_begin); /* Wait for all outstanding I/O to quiesce before performing retries as * we may need to renegotiate the I/O sizes. @@ -234,4 +235,6 @@ void netfs_retry_writes(struct netfs_io_request *wreq) netfs_retry_write_stream(wreq, stream); } } + + trace_netfs_rreq(wreq, netfs_rreq_trace_retry_end); } diff --git a/include/linux/fscache.h b/include/linux/fscache.h index e19fca38382b9b..f2d958bd1f480b 100644 --- a/include/linux/fscache.h +++ b/include/linux/fscache.h @@ -147,6 +147,23 @@ struct fscache_cookie { }; }; +enum fscache_extent_type { + FSCACHE_EXTENT_DATA, + FSCACHE_EXTENT_ZERO, +} __mode(byte); + +/* + * Cache occupancy information. + */ +struct fscache_occupancy { + unsigned long long query_from; /* Point to query from */ + unsigned long long query_to; /* Point to query to */ + unsigned long long cached_from[2]; /* Point at which cache extents start */ + unsigned long long cached_to[2]; /* Point at which cache extents end */ + unsigned int granularity; /* Granularity desired */ + enum fscache_extent_type cached_type[2]; /* Type of cache extent */ +}; + /* * slow-path functions for when there is actually caching available, and the * netfs does actually have a valid token diff --git a/include/linux/netfs.h b/include/linux/netfs.h index 71fdd6ef43a7d2..67e010b6994b31 100644 --- a/include/linux/netfs.h +++ b/include/linux/netfs.h @@ -22,6 +22,7 @@ enum netfs_sreq_ref_trace; typedef struct mempool mempool_t; +struct fscache_occupancy; struct folio_queue; /** @@ -125,6 +126,12 @@ static inline struct netfs_group *netfs_folio_group(struct folio *folio) return priv; } +enum netfs_cache_collect { + NETFS_CACHE_COLLECT_WRITE_GAP, /* Gap in collection, no state either way */ + NETFS_CACHE_COLLECT_WRITE_DATA, /* Currently collecting good writes */ + NETFS_CACHE_COLLECT_WRITE_CANCEL, /* Currently collecting cancelled writes */ +}; + /* * Stream of I/O subrequests going to a particular destination, such as the * server or the local cache. This is mainly intended for writing where we may @@ -152,6 +159,7 @@ struct netfs_io_stream { bool need_retry; /* T if this stream needs retrying */ bool failed; /* T if this stream failed */ bool transferred_valid; /* T is ->transferred is valid */ + enum netfs_cache_collect cache_collect; /* Current writeback cache collect state */ }; /* @@ -161,9 +169,11 @@ struct netfs_cache_resources { const struct netfs_cache_ops *ops; void *cache_priv; void *cache_priv2; + uoff_t cache_i_size; /* Initial size of cache file */ unsigned int cookie_id; /* Cache cookie debug ID */ unsigned int object_id; /* Cache object debug ID */ unsigned int inval_counter; /* object->inval_counter at begin_op */ + unsigned int dio_size; /* DIO block size */ }; /* @@ -197,6 +207,7 @@ struct netfs_io_subrequest { #define NETFS_SREQ_IN_PROGRESS 8 /* Unlocked when the subrequest completes */ #define NETFS_SREQ_NEED_RETRY 9 /* Set if the filesystem requests a retry */ #define NETFS_SREQ_FAILED 10 /* Set if the subreq failed unretryably */ +#define NETFS_SREQ_CANCELLED 11 /* Set if the subreq was cancelled by netfslib */ }; enum netfs_io_origin { @@ -252,6 +263,7 @@ struct netfs_io_request { uoff_t start; /* Start position */ atomic64_t issued_to; /* Write issuer folio cursor */ uoff_t collected_to; /* Point we've collected to */ + uoff_t cache_coll_to; /* Point the cache has collected to */ uoff_t cleaned_to; /* Position we've cleaned folios to */ uoff_t abandon_to; /* Position to abandon folios to */ const struct folio *no_unlock_folio; /* Don't unlock this folio after read */ @@ -273,11 +285,13 @@ struct netfs_io_request { #define NETFS_RREQ_FAILED 3 /* The request failed */ #define NETFS_RREQ_RETRYING 4 /* Set if we're in the retry path */ #define NETFS_RREQ_SHORT_TRANSFER 5 /* Set if we have a short transfer */ -#define NETFS_RREQ_OFFLOAD_COLLECTION 8 /* Offload collection to workqueue */ -#define NETFS_RREQ_NO_UNLOCK_FOLIO 9 /* Don't unlock no_unlock_folio on completion */ +#define NETFS_RREQ_CACHE_STOP 8 /* Set to stop caching (ENOBUFS or error) */ +#define NETFS_RREQ_CACHE_ERROR 9 /* Set if we got an error from the cache */ #define NETFS_RREQ_CANCEL_CACHING 10 /* Set to cancel caching */ -#define NETFS_RREQ_UPLOAD_TO_SERVER 11 /* Need to write to the server */ -#define NETFS_RREQ_USE_IO_ITER 12 /* Use ->io_iter rather than ->i_pages */ +#define NETFS_RREQ_OFFLOAD_COLLECTION 12 /* Offload collection to workqueue */ +#define NETFS_RREQ_NO_UNLOCK_FOLIO 13 /* Don't unlock no_unlock_folio on completion */ +#define NETFS_RREQ_UPLOAD_TO_SERVER 14 /* Need to write to the server */ +#define NETFS_RREQ_USE_IO_ITER 15 /* Use ->io_iter rather than ->i_pages */ #define NETFS_RREQ_NEED_PUT_RA_REFS 17 /* Need to put the folio refs RA gave us */ #ifdef CONFIG_NETFS_PGPRIV2 #define NETFS_RREQ_USE_PGPRIV2 31 /* [DEPRECATED] Use PG_private_2 to mark @@ -359,8 +373,7 @@ struct netfs_cache_ops { /* Prepare a read operation, shortening it to a cached/uncached * boundary as appropriate. */ - enum netfs_io_source (*prepare_read)(struct netfs_io_subrequest *subreq, - uoff_t i_size); + int (*prepare_read)(struct netfs_io_subrequest *subreq); /* Prepare a write subrequest, working out if we're allowed to do it * and finding out the maximum amount of data to gather before @@ -380,8 +393,17 @@ struct netfs_cache_ops { * next chunk of data starts and how long it is. */ int (*query_occupancy)(struct netfs_cache_resources *cres, - uoff_t start, size_t len, size_t granularity, - uoff_t *_data_start, size_t *_data_len); + struct fscache_occupancy *occ); + + /* Collect the result of buffered writeback to the cache. This + * includes copying a read to the cache. block_type is one of: + * - NETFS_CACHE_COLLECT_WRITE_DATA for a block of data + * - NETFS_CACHE_COLLECT_WRITE_GAP if a discontiguity was skipped + * - NETFS_CACHE_COLLECT_WRITE_CANCEL for a cancellation gap + */ + void (*collect_write)(struct netfs_io_request *wreq, + uoff_t start, size_t len, + enum netfs_cache_collect block_type); }; /* High-level read API. */ diff --git a/include/trace/events/cachefiles.h b/include/trace/events/cachefiles.h index cd865c265c2451..a19233e8ae78e6 100644 --- a/include/trace/events/cachefiles.h +++ b/include/trace/events/cachefiles.h @@ -52,6 +52,8 @@ enum cachefiles_coherency_trace { cachefiles_coherency_check_ok, cachefiles_coherency_check_type, cachefiles_coherency_check_xattr, + cachefiles_coherency_discontiguous, + cachefiles_coherency_remove, cachefiles_coherency_set_fail, cachefiles_coherency_set_ok, cachefiles_coherency_vol_check_cmp, @@ -63,9 +65,11 @@ enum cachefiles_coherency_trace { }; enum cachefiles_trunc_trace { + cachefiles_trunc_clear_padding, cachefiles_trunc_dio_adjust, cachefiles_trunc_expand_tmpfile, cachefiles_trunc_shrink, + cachefiles_trunc_zap, }; enum cachefiles_prepare_read_trace { @@ -80,6 +84,7 @@ enum cachefiles_prepare_read_trace { }; enum cachefiles_error_trace { + cachefiles_trace_alignment_error, cachefiles_trace_create_nospace, cachefiles_trace_fallocate_error, cachefiles_trace_getxattr_error, @@ -140,6 +145,8 @@ enum cachefiles_error_trace { EM(cachefiles_coherency_check_ok, "OK ") \ EM(cachefiles_coherency_check_type, "BAD type") \ EM(cachefiles_coherency_check_xattr, "BAD xatt") \ + EM(cachefiles_coherency_discontiguous, "--- gap ") \ + EM(cachefiles_coherency_remove, "REMOVE ") \ EM(cachefiles_coherency_set_fail, "SET fail") \ EM(cachefiles_coherency_set_ok, "SET ok ") \ EM(cachefiles_coherency_vol_check_cmp, "VOL BAD cmp ") \ @@ -150,9 +157,11 @@ enum cachefiles_error_trace { E_(cachefiles_coherency_vol_set_ok, "VOL SET ok ") #define cachefiles_trunc_traces \ + EM(cachefiles_trunc_clear_padding, "CLRPAD") \ EM(cachefiles_trunc_dio_adjust, "DIOADJ") \ EM(cachefiles_trunc_expand_tmpfile, "EXPTMP") \ - E_(cachefiles_trunc_shrink, "SHRINK") + EM(cachefiles_trunc_shrink, "SHRINK") \ + E_(cachefiles_trunc_zap, "ZAP ") #define cachefiles_prepare_read_traces \ EM(cachefiles_trace_read_after_eof, "after-eof ") \ @@ -165,6 +174,7 @@ enum cachefiles_error_trace { E_(cachefiles_trace_read_seek_nxio, "seek-enxio") #define cachefiles_error_traces \ + EM(cachefiles_trace_alignment_error, "align") \ EM(cachefiles_trace_create_nospace, "create-nospace") \ EM(cachefiles_trace_fallocate_error, "fallocate") \ EM(cachefiles_trace_getxattr_error, "getxattr") \ @@ -379,12 +389,12 @@ TRACE_EVENT(cachefiles_rename, TRACE_EVENT(cachefiles_coherency, TP_PROTO(struct cachefiles_object *obj, - ino_t ino, + ino_t ino, uoff_t obj_size, const void *disk_aux, enum cachefiles_content content, enum cachefiles_coherency_trace why), - TP_ARGS(obj, ino, disk_aux, content, why), + TP_ARGS(obj, ino, obj_size, disk_aux, content, why), /* Note that obj may be NULL */ TP_STRUCT__entry( @@ -392,6 +402,7 @@ TRACE_EVENT(cachefiles_coherency, __field(enum cachefiles_coherency_trace, why) __field(enum cachefiles_content, content) __field(u64, ino) + __field(u64, obj_size) __field(u64, aux) __field(u64, disk_aux) ), @@ -406,6 +417,7 @@ TRACE_EVENT(cachefiles_coherency, __entry->why = why; __entry->content = content; __entry->ino = ino; + __entry->obj_size = obj_size; __entry->aux = be64_to_cpup((__be64 *)obj->cookie->inline_aux); /* cachefiles_xattr::data is 2-byte aligned but not 8-byte aligned. */ @@ -420,10 +432,11 @@ TRACE_EVENT(cachefiles_coherency, } ), - TP_printk("o=%08x %s B=%llx c=%u aux=%llx dsk=%llx", + TP_printk("o=%08x %s B=%llx oz=%llx c=%u aux=%llx dsk=%llx", __entry->obj, __print_symbolic(__entry->why, cachefiles_coherency_traces), __entry->ino, + __entry->obj_size, __entry->content, __entry->aux, __entry->disk_aux) diff --git a/include/trace/events/netfs.h b/include/trace/events/netfs.h index 35826aea1c2ddf..bf1e1f185b05c4 100644 --- a/include/trace/events/netfs.h +++ b/include/trace/events/netfs.h @@ -49,6 +49,10 @@ #define netfs_rreq_traces \ EM(netfs_rreq_trace_all_queued, "ALL-Q ") \ EM(netfs_rreq_trace_assess, "ASSESS ") \ + EM(netfs_rreq_trace_cache_cancelled, "CA-CNCL") \ + EM(netfs_rreq_trace_cache_failed, "CA-FAIL") \ + EM(netfs_rreq_trace_cache_fail_collect, "CA-F-CO") \ + EM(netfs_rreq_trace_cache_no_space, "CA-NOSP") \ EM(netfs_rreq_trace_collect, "COLLECT") \ EM(netfs_rreq_trace_complete, "COMPLET") \ EM(netfs_rreq_trace_copy, "COPY ") \ @@ -57,11 +61,14 @@ EM(netfs_rreq_trace_end_copy_to_cache, "END-C2C") \ EM(netfs_rreq_trace_free, "FREE ") \ EM(netfs_rreq_trace_intr, "INTR ") \ + EM(netfs_rreq_trace_inval_cache, "INVL-CA") \ EM(netfs_rreq_trace_ki_complete, "KI-CMPL") \ EM(netfs_rreq_trace_ra_put_ref, "RA-PUT ") \ EM(netfs_rreq_trace_recollect, "RECLLCT") \ EM(netfs_rreq_trace_redirty, "REDIRTY") \ EM(netfs_rreq_trace_resubmit, "RESUBMT") \ + EM(netfs_rreq_trace_retry_begin, "RETRY-BEGIN") \ + EM(netfs_rreq_trace_retry_end, "RETRY-END") \ EM(netfs_rreq_trace_set_abandon, "S-ABNDN") \ EM(netfs_rreq_trace_set_pause, "PAUSE ") \ EM(netfs_rreq_trace_unlock, "UNLOCK ") \ @@ -135,12 +142,12 @@ #define netfs_failures \ EM(netfs_fail_check_write_begin, "check-write-begin") \ - EM(netfs_fail_copy_to_cache, "copy-to-cache") \ EM(netfs_fail_dio_read_short, "dio-read-short") \ EM(netfs_fail_dio_read_zero, "dio-read-zero") \ EM(netfs_fail_read, "read") \ EM(netfs_fail_short_read, "short-read") \ EM(netfs_fail_prepare_write, "prep-write") \ + EM(netfs_fail_upload, "upload") \ E_(netfs_fail_write, "write") #define netfs_rreq_ref_traces \ From f96cbb18159fe6c316da59d83f10eb6372aa96a3 Mon Sep 17 00:00:00 2001 From: David Howells Date: Thu, 10 Sep 2026 23:02:40 +0100 Subject: [PATCH 0334/1352] cachefiles: Preset the state xattr when creating a new file With a really small cache, cachefiles is likely to see a lot of writes hitting ENOSPC - and this can include setxattr that sets the state xattr on a cachefile - but we don't really want to successfully fill a cache file only to have to scrap it because we can't set the xattr. Instead, preset the xattr when we create the tmpfile we're going to use, and scrap the file at that point if we get ENOSPC. Only if setxattr succeeds do we allow data to be written to the file. Note that there is a potential performance loss in that writes to the cache have to be delayed until this is completed - but we do the tmpfile/setxattr in parallel, starting when the file is opened and only have to wait once writeback occurs. Signed-off-by: David Howells Link: https://patch.msgid.link/20260910220242.2165023-6-dhowells@redhat.com Reviewed-by: Paulo Alcantara cc: Marc Dionne cc: Paulo Alcantara cc: netfs@lists.linux.dev cc: linux-fsdevel@vger.kernel.org Signed-off-by: Christian Brauner (Amutable) --- fs/cachefiles/internal.h | 1 + fs/cachefiles/namei.c | 5 ++++ fs/cachefiles/xattr.c | 59 +++++++++++++++++++++++++++++++++++++++- 3 files changed, 64 insertions(+), 1 deletion(-) diff --git a/fs/cachefiles/internal.h b/fs/cachefiles/internal.h index b2605111fd5642..664be64ab5384b 100644 --- a/fs/cachefiles/internal.h +++ b/fs/cachefiles/internal.h @@ -283,6 +283,7 @@ void cachefiles_withdraw_volume(struct cachefiles_volume *volume); /* * xattr.c */ +int cachefiles_preset_object_xattr(struct cachefiles_object *object, struct file *file); extern int cachefiles_set_object_xattr(struct cachefiles_object *object); extern int cachefiles_check_auxdata(struct cachefiles_object *object, struct file *file); diff --git a/fs/cachefiles/namei.c b/fs/cachefiles/namei.c index ca093840e5774d..ef656a319ede2e 100644 --- a/fs/cachefiles/namei.c +++ b/fs/cachefiles/namei.c @@ -450,6 +450,11 @@ struct file *cachefiles_create_tmpfile(struct cachefiles_object *object) pr_notice("Cache does not support read_iter and write_iter\n"); goto err_unuse; } + + /* Preallocate space for the xattr. */ + ret = cachefiles_preset_object_xattr(object, file); + if (ret < 0) + goto err_unuse; out: cachefiles_end_secure(cache, saved_cred); object->content_info = CACHEFILES_CONTENT_ALL; diff --git a/fs/cachefiles/xattr.c b/fs/cachefiles/xattr.c index 551a3b0069c211..8ebb713482e3e7 100644 --- a/fs/cachefiles/xattr.c +++ b/fs/cachefiles/xattr.c @@ -34,6 +34,57 @@ struct cachefiles_vol_xattr { __u8 data[]; /* netfs volume coherency data */ } __packed; +/* + * Preset the state xattr on a cache file to allocate space for it. + */ +int cachefiles_preset_object_xattr(struct cachefiles_object *object, struct file *file) +{ + struct cachefiles_xattr *buf; + struct dentry *dentry = file->f_path.dentry; + unsigned int len = object->cookie->aux_len; + int ret; + + buf = kzalloc(sizeof(struct cachefiles_xattr) + min(len, sizeof(__be64)), GFP_KERNEL); + if (!buf) + return -ENOMEM; + + buf->type = CACHEFILES_COOKIE_TYPE_DATA; + buf->content = CACHEFILES_CONTENT_DIRTY; + + ret = cachefiles_inject_write_error(); + if (ret == 0) { + ret = mnt_want_write_file(file); + if (ret == 0) { + ret = vfs_setxattr(&nop_mnt_idmap, dentry, + cachefiles_xattr_cache, buf, + sizeof(struct cachefiles_xattr) + len, 0); + mnt_drop_write_file(file); + } + } + if (ret < 0) { + trace_cachefiles_vfs_error(object, file_inode(file), ret, + cachefiles_trace_setxattr_error); + trace_cachefiles_coherency(object, file_inode(file)->i_ino, + object->object_size, + buf->data, buf->content, + cachefiles_coherency_set_fail); + switch (ret) { + case -ENOMEM: + case -ENOSPC: + break; + default: + cachefiles_io_error_obj( + object, + "Failed to set xattr with error %d", ret); + break; + } + } + + kfree(buf); + _leave(" = %d", ret); + return ret; +} + /* * set the state xattr on a cache file */ @@ -82,10 +133,16 @@ int cachefiles_set_object_xattr(struct cachefiles_object *object) trace_cachefiles_coherency(object, file_inode(file)->i_ino, object_size, buf->data, buf->content, cachefiles_coherency_set_fail); - if (ret != -ENOMEM) + switch (ret) { + case -ENOMEM: + break; + case -ENOSPC: + default: cachefiles_io_error_obj( object, "Failed to set xattr with error %d", ret); + break; + } } else { trace_cachefiles_coherency(object, file_inode(file)->i_ino, object_size, buf->data, buf->content, From 28612ec5303410ec009d6c617446243a727aa2ce Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Wed, 26 Aug 2026 18:06:40 +0200 Subject: [PATCH 0335/1352] powerpc: remove coredump support A task that dumps open spufs context adds a bunch of extra elf notes describing the SPU state. It is the only reason do_coredump() unshares the file descriptor table. We could make this conditional on spufs but eh. Nothing can consume those notes anymore. gdb dropped Cell Broadband Engine debugging in 9.1 and binutils removed it in 2.34. That's about 6 years ago. So no program can actually read an SPU note out of a core file and probably never did in recent history. Note that the IBM Cell blades that shipped the Cell processor were removed in commit 05bf59fbeef3 ("powerpc/cell: Remove support for IBM Cell Blades"). The PlayStation 3 is the only platform left and nothing there produces or reads these notes. So remove it. Link: https://patch.msgid.link/20260826-work-spufs-coredump-v1-1-579e72a7ab66@kernel.org Acked-by: Arnd Bergmann Signed-off-by: Christian Brauner (Amutable) --- arch/powerpc/Kconfig | 1 - arch/powerpc/include/asm/elf.h | 6 - arch/powerpc/include/asm/spu.h | 3 - arch/powerpc/platforms/cell/Kconfig | 1 - arch/powerpc/platforms/cell/spu_syscalls.c | 20 -- arch/powerpc/platforms/cell/spufs/Makefile | 1 - arch/powerpc/platforms/cell/spufs/coredump.c | 183 ------------------- arch/powerpc/platforms/cell/spufs/file.c | 114 ------------ arch/powerpc/platforms/cell/spufs/spufs.h | 12 -- arch/powerpc/platforms/cell/spufs/syscalls.c | 4 - 10 files changed, 345 deletions(-) delete mode 100644 arch/powerpc/platforms/cell/spufs/coredump.c diff --git a/arch/powerpc/Kconfig b/arch/powerpc/Kconfig index 2580e27e432874..40c874fe2f53ac 100644 --- a/arch/powerpc/Kconfig +++ b/arch/powerpc/Kconfig @@ -160,7 +160,6 @@ config PPC select ARCH_HAS_UBSAN select ARCH_HAS_VDSO_ARCH_DATA select ARCH_HAVE_NMI_SAFE_CMPXCHG - select ARCH_HAVE_EXTRA_ELF_NOTES if SPU_BASE select ARCH_KEEP_MEMBLOCK select ARCH_MHP_MEMMAP_ON_MEMORY_ENABLE if PPC_RADIX_MMU select ARCH_MIGHT_HAVE_PC_PARPORT diff --git a/arch/powerpc/include/asm/elf.h b/arch/powerpc/include/asm/elf.h index bb4b94444d3e8a..5dc8c4923eb1df 100644 --- a/arch/powerpc/include/asm/elf.h +++ b/arch/powerpc/include/asm/elf.h @@ -123,12 +123,6 @@ extern int arch_setup_additional_pages(struct linux_binprm *bprm, (0x7ff >> (PAGE_SHIFT - 12)) : \ (0x3ffff >> (PAGE_SHIFT - 12))) -#ifdef CONFIG_SPU_BASE -/* Notes used in ET_CORE. Note name is "SPU//". */ -#define NT_SPU 1 - -#endif /* CONFIG_SPU_BASE */ - #ifdef CONFIG_PPC64 #define get_cache_geometry(level) \ diff --git a/arch/powerpc/include/asm/spu.h b/arch/powerpc/include/asm/spu.h index 96ad4510c89542..7152285b6268fd 100644 --- a/arch/powerpc/include/asm/spu.h +++ b/arch/powerpc/include/asm/spu.h @@ -210,15 +210,12 @@ extern long spu_sys_callback(struct spu_syscall_block *s); /* syscalls implemented in spufs */ struct file; -struct coredump_params; struct spufs_calls { long (*create_thread)(const char __user *name, unsigned int flags, umode_t mode, struct file *neighbor); long (*spu_run)(struct file *filp, __u32 __user *unpc, __u32 __user *ustatus); - int (*coredump_extra_notes_size)(void); - int (*coredump_extra_notes_write)(struct coredump_params *cprm); void (*notify_spus_active)(void); struct module *owner; }; diff --git a/arch/powerpc/platforms/cell/Kconfig b/arch/powerpc/platforms/cell/Kconfig index db65bfcd1e7498..6bd26815c33103 100644 --- a/arch/powerpc/platforms/cell/Kconfig +++ b/arch/powerpc/platforms/cell/Kconfig @@ -10,7 +10,6 @@ config SPU_FS tristate "SPU file system" default m depends on PPC_CELL - depends on COREDUMP select SPU_BASE help The SPU file system is used to access Synergistic Processing diff --git a/arch/powerpc/platforms/cell/spu_syscalls.c b/arch/powerpc/platforms/cell/spu_syscalls.c index 000894e07b027d..8be81207e886a1 100644 --- a/arch/powerpc/platforms/cell/spu_syscalls.c +++ b/arch/powerpc/platforms/cell/spu_syscalls.c @@ -88,26 +88,6 @@ SYSCALL_DEFINE3(spu_run,int, fd, __u32 __user *, unpc, __u32 __user *, ustatus) return calls->spu_run(fd_file(arg), unpc, ustatus); } -#ifdef CONFIG_COREDUMP -int elf_coredump_extra_notes_size(void) -{ - CLASS(spufs_calls, calls)(); - if (!calls) - return 0; - - return calls->coredump_extra_notes_size(); -} - -int elf_coredump_extra_notes_write(struct coredump_params *cprm) -{ - CLASS(spufs_calls, calls)(); - if (!calls) - return 0; - - return calls->coredump_extra_notes_write(cprm); -} -#endif - void notify_spus_active(void) { struct spufs_calls *calls; diff --git a/arch/powerpc/platforms/cell/spufs/Makefile b/arch/powerpc/platforms/cell/spufs/Makefile index 52e4c80ec8d031..60319d4ff25a6a 100644 --- a/arch/powerpc/platforms/cell/spufs/Makefile +++ b/arch/powerpc/platforms/cell/spufs/Makefile @@ -4,7 +4,6 @@ obj-$(CONFIG_SPU_FS) += spufs.o spufs-y += inode.o file.o context.o syscalls.o spufs-y += sched.o backing_ops.o hw_ops.o run.o gang.o spufs-y += switch.o fault.o lscsa_alloc.o -spufs-$(CONFIG_COREDUMP) += coredump.o # magic for the trace events CFLAGS_sched.o := -I$(src) diff --git a/arch/powerpc/platforms/cell/spufs/coredump.c b/arch/powerpc/platforms/cell/spufs/coredump.c deleted file mode 100644 index 301ee7d8b7df02..00000000000000 --- a/arch/powerpc/platforms/cell/spufs/coredump.c +++ /dev/null @@ -1,183 +0,0 @@ -// SPDX-License-Identifier: GPL-2.0-or-later -/* - * SPU core dump code - * - * (C) Copyright 2006 IBM Corp. - * - * Author: Dwayne Grant McConnell - */ - -#include -#include -#include -#include -#include -#include -#include -#include -#include - -#include - -#include "spufs.h" - -static int spufs_ctx_note_size(struct spu_context *ctx, int dfd) -{ - int i, sz, total = 0; - char *name; - char fullname[80]; - - for (i = 0; spufs_coredump_read[i].name != NULL; i++) { - name = spufs_coredump_read[i].name; - sz = spufs_coredump_read[i].size; - - sprintf(fullname, "SPU/%d/%s", dfd, name); - - total += sizeof(struct elf_note); - total += roundup(strlen(fullname) + 1, 4); - total += roundup(sz, 4); - } - - return total; -} - -static int match_context(const void *v, struct file *file, unsigned fd) -{ - struct spu_context *ctx; - if (file->f_op != &spufs_context_fops) - return 0; - ctx = SPUFS_I(file_inode(file))->i_ctx; - if (ctx->flags & SPU_CREATE_NOSCHED) - return 0; - return fd + 1; -} - -/* - * The additional architecture-specific notes for Cell are various - * context files in the spu context. - * - * This function iterates over all open file descriptors and sees - * if they are a directory in spufs. In that case we use spufs - * internal functionality to dump them without needing to actually - * open the files. - */ -/* - * descriptor table is not shared, so files can't change or go away. - */ -static struct spu_context *coredump_next_context(int *fd) -{ - struct spu_context *ctx = NULL; - struct file *file; - int n = iterate_fd(current->files, *fd, match_context, NULL); - if (!n) - return NULL; - *fd = n - 1; - - file = fget_raw(*fd); - if (file) { - ctx = SPUFS_I(file_inode(file))->i_ctx; - get_spu_context(ctx); - fput(file); - } - - return ctx; -} - -int spufs_coredump_extra_notes_size(void) -{ - struct spu_context *ctx; - int size = 0, rc, fd; - - fd = 0; - while ((ctx = coredump_next_context(&fd)) != NULL) { - rc = spu_acquire_saved(ctx); - if (rc) { - put_spu_context(ctx); - break; - } - - rc = spufs_ctx_note_size(ctx, fd); - spu_release_saved(ctx); - if (rc < 0) { - put_spu_context(ctx); - break; - } - - size += rc; - - /* start searching the next fd next time */ - fd++; - put_spu_context(ctx); - } - - return size; -} - -static int spufs_arch_write_note(struct spu_context *ctx, int i, - struct coredump_params *cprm, int dfd) -{ - size_t sz = spufs_coredump_read[i].size; - char fullname[80]; - struct elf_note en; - int ret; - - sprintf(fullname, "SPU/%d/%s", dfd, spufs_coredump_read[i].name); - en.n_namesz = strlen(fullname) + 1; - en.n_descsz = sz; - en.n_type = NT_SPU; - - if (!dump_emit(cprm, &en, sizeof(en))) - return -EIO; - if (!dump_emit(cprm, fullname, en.n_namesz)) - return -EIO; - if (!dump_align(cprm, 4)) - return -EIO; - - if (spufs_coredump_read[i].dump) { - ret = spufs_coredump_read[i].dump(ctx, cprm); - if (ret < 0) - return ret; - } else { - char buf[32]; - - ret = snprintf(buf, sizeof(buf), "0x%.16llx", - spufs_coredump_read[i].get(ctx)); - if (ret >= sizeof(buf)) - return sizeof(buf); - - /* count trailing the NULL: */ - if (!dump_emit(cprm, buf, ret + 1)) - return -EIO; - } - - dump_skip_to(cprm, roundup(cprm->pos - ret + sz, 4)); - return 0; -} - -int spufs_coredump_extra_notes_write(struct coredump_params *cprm) -{ - struct spu_context *ctx; - int fd, j, rc; - - fd = 0; - while ((ctx = coredump_next_context(&fd)) != NULL) { - rc = spu_acquire_saved(ctx); - if (rc) - return rc; - - for (j = 0; spufs_coredump_read[j].name != NULL; j++) { - rc = spufs_arch_write_note(ctx, j, cprm, fd); - if (rc) { - spu_release_saved(ctx); - return rc; - } - } - - spu_release_saved(ctx); - - /* start searching the next fd next time */ - fd++; - } - - return 0; -} diff --git a/arch/powerpc/platforms/cell/spufs/file.c b/arch/powerpc/platforms/cell/spufs/file.c index de7494748fecd6..98c47bafaf6781 100644 --- a/arch/powerpc/platforms/cell/spufs/file.c +++ b/arch/powerpc/platforms/cell/spufs/file.c @@ -9,7 +9,6 @@ #undef DEBUG -#include #include #include #include @@ -130,14 +129,6 @@ static ssize_t spufs_attr_write(struct file *file, const char __user *buf, return ret; } -static ssize_t spufs_dump_emit(struct coredump_params *cprm, void *buf, - size_t size) -{ - if (!dump_emit(cprm, buf, size)) - return -EIO; - return size; -} - #define DEFINE_SPUFS_SIMPLE_ATTRIBUTE(__fops, __get, __set, __fmt) \ static int __fops ## _open(struct inode *inode, struct file *file) \ { \ @@ -180,12 +171,6 @@ spufs_mem_release(struct inode *inode, struct file *file) return 0; } -static ssize_t -spufs_mem_dump(struct spu_context *ctx, struct coredump_params *cprm) -{ - return spufs_dump_emit(cprm, ctx->ops->get_ls(ctx), LS_SIZE); -} - static ssize_t spufs_mem_read(struct file *file, char __user *buffer, size_t size, loff_t *pos) @@ -466,13 +451,6 @@ spufs_regs_open(struct inode *inode, struct file *file) return 0; } -static ssize_t -spufs_regs_dump(struct spu_context *ctx, struct coredump_params *cprm) -{ - return spufs_dump_emit(cprm, ctx->csa.lscsa->gprs, - sizeof(ctx->csa.lscsa->gprs)); -} - static ssize_t spufs_regs_read(struct file *file, char __user *buffer, size_t size, loff_t *pos) @@ -523,13 +501,6 @@ static const struct file_operations spufs_regs_fops = { .llseek = generic_file_llseek, }; -static ssize_t -spufs_fpcr_dump(struct spu_context *ctx, struct coredump_params *cprm) -{ - return spufs_dump_emit(cprm, &ctx->csa.lscsa->fpcr, - sizeof(ctx->csa.lscsa->fpcr)); -} - static ssize_t spufs_fpcr_read(struct file *file, char __user * buffer, size_t size, loff_t * pos) @@ -953,15 +924,6 @@ spufs_signal1_release(struct inode *inode, struct file *file) return 0; } -static ssize_t spufs_signal1_dump(struct spu_context *ctx, - struct coredump_params *cprm) -{ - if (!ctx->csa.spu_chnlcnt_RW[3]) - return 0; - return spufs_dump_emit(cprm, &ctx->csa.spu_chnldata_RW[3], - sizeof(ctx->csa.spu_chnldata_RW[3])); -} - static ssize_t __spufs_signal1_read(struct spu_context *ctx, char __user *buf, size_t len) { @@ -1086,15 +1048,6 @@ spufs_signal2_release(struct inode *inode, struct file *file) return 0; } -static ssize_t spufs_signal2_dump(struct spu_context *ctx, - struct coredump_params *cprm) -{ - if (!ctx->csa.spu_chnlcnt_RW[4]) - return 0; - return spufs_dump_emit(cprm, &ctx->csa.spu_chnldata_RW[4], - sizeof(ctx->csa.spu_chnldata_RW[4])); -} - static ssize_t __spufs_signal2_read(struct spu_context *ctx, char __user *buf, size_t len) { @@ -1924,15 +1877,6 @@ static const struct file_operations spufs_caps_fops = { .release = single_release, }; -static ssize_t spufs_mbox_info_dump(struct spu_context *ctx, - struct coredump_params *cprm) -{ - if (!(ctx->csa.prob.mb_stat_R & 0x0000ff)) - return 0; - return spufs_dump_emit(cprm, &ctx->csa.prob.pu_mb_R, - sizeof(ctx->csa.prob.pu_mb_R)); -} - static ssize_t spufs_mbox_info_read(struct file *file, char __user *buf, size_t len, loff_t *pos) { @@ -1962,15 +1906,6 @@ static const struct file_operations spufs_mbox_info_fops = { .llseek = generic_file_llseek, }; -static ssize_t spufs_ibox_info_dump(struct spu_context *ctx, - struct coredump_params *cprm) -{ - if (!(ctx->csa.prob.mb_stat_R & 0xff0000)) - return 0; - return spufs_dump_emit(cprm, &ctx->csa.priv2.puint_mb_R, - sizeof(ctx->csa.priv2.puint_mb_R)); -} - static ssize_t spufs_ibox_info_read(struct file *file, char __user *buf, size_t len, loff_t *pos) { @@ -2005,13 +1940,6 @@ static size_t spufs_wbox_info_cnt(struct spu_context *ctx) return (4 - ((ctx->csa.prob.mb_stat_R & 0x00ff00) >> 8)) * sizeof(u32); } -static ssize_t spufs_wbox_info_dump(struct spu_context *ctx, - struct coredump_params *cprm) -{ - return spufs_dump_emit(cprm, &ctx->csa.spu_mailbox_data, - spufs_wbox_info_cnt(ctx)); -} - static ssize_t spufs_wbox_info_read(struct file *file, char __user *buf, size_t len, loff_t *pos) { @@ -2059,15 +1987,6 @@ static void spufs_get_dma_info(struct spu_context *ctx, } } -static ssize_t spufs_dma_info_dump(struct spu_context *ctx, - struct coredump_params *cprm) -{ - struct spu_dma_info info; - - spufs_get_dma_info(ctx, &info); - return spufs_dump_emit(cprm, &info, sizeof(info)); -} - static ssize_t spufs_dma_info_read(struct file *file, char __user *buf, size_t len, loff_t *pos) { @@ -2112,15 +2031,6 @@ static void spufs_get_proxydma_info(struct spu_context *ctx, } } -static ssize_t spufs_proxydma_info_dump(struct spu_context *ctx, - struct coredump_params *cprm) -{ - struct spu_proxydma_info info; - - spufs_get_proxydma_info(ctx, &info); - return spufs_dump_emit(cprm, &info, sizeof(info)); -} - static ssize_t spufs_proxydma_info_read(struct file *file, char __user *buf, size_t len, loff_t *pos) { @@ -2580,27 +2490,3 @@ const struct spufs_tree_descr spufs_dir_debug_contents[] = { { ".ctx", &spufs_ctx_fops, 0444, }, {}, }; - -const struct spufs_coredump_reader spufs_coredump_read[] = { - { "regs", spufs_regs_dump, NULL, sizeof(struct spu_reg128[128])}, - { "fpcr", spufs_fpcr_dump, NULL, sizeof(struct spu_reg128) }, - { "lslr", NULL, spufs_lslr_get, 19 }, - { "decr", NULL, spufs_decr_get, 19 }, - { "decr_status", NULL, spufs_decr_status_get, 19 }, - { "mem", spufs_mem_dump, NULL, LS_SIZE, }, - { "signal1", spufs_signal1_dump, NULL, sizeof(u32) }, - { "signal1_type", NULL, spufs_signal1_type_get, 19 }, - { "signal2", spufs_signal2_dump, NULL, sizeof(u32) }, - { "signal2_type", NULL, spufs_signal2_type_get, 19 }, - { "event_mask", NULL, spufs_event_mask_get, 19 }, - { "event_status", NULL, spufs_event_status_get, 19 }, - { "mbox_info", spufs_mbox_info_dump, NULL, sizeof(u32) }, - { "ibox_info", spufs_ibox_info_dump, NULL, sizeof(u32) }, - { "wbox_info", spufs_wbox_info_dump, NULL, 4 * sizeof(u32)}, - { "dma_info", spufs_dma_info_dump, NULL, sizeof(struct spu_dma_info)}, - { "proxydma_info", spufs_proxydma_info_dump, - NULL, sizeof(struct spu_proxydma_info)}, - { "object-id", NULL, spufs_object_id_get, 19 }, - { "npc", NULL, spufs_npc_get, 19 }, - { NULL }, -}; diff --git a/arch/powerpc/platforms/cell/spufs/spufs.h b/arch/powerpc/platforms/cell/spufs/spufs.h index d33787c57c39a2..612b5075d0ecb8 100644 --- a/arch/powerpc/platforms/cell/spufs/spufs.h +++ b/arch/powerpc/platforms/cell/spufs/spufs.h @@ -232,13 +232,9 @@ extern const struct spufs_tree_descr spufs_dir_debug_contents[]; /* system call implementation */ extern struct spufs_calls spufs_calls; -struct coredump_params; long spufs_run_spu(struct spu_context *ctx, u32 *npc, u32 *status); long spufs_create(const struct path *nd, struct dentry *dentry, unsigned int flags, umode_t mode, struct file *filp); -/* ELF coredump callbacks for writing SPU ELF notes */ -extern int spufs_coredump_extra_notes_size(void); -extern int spufs_coredump_extra_notes_write(struct coredump_params *cprm); extern const struct file_operations spufs_context_fops; @@ -335,14 +331,6 @@ void spufs_stop_callback(struct spu *spu, int irq); void spufs_mfc_callback(struct spu *spu); void spufs_dma_callback(struct spu *spu, int type); -struct spufs_coredump_reader { - char *name; - ssize_t (*dump)(struct spu_context *ctx, struct coredump_params *cprm); - u64 (*get)(struct spu_context *ctx); - size_t size; -}; -extern const struct spufs_coredump_reader spufs_coredump_read[]; - extern int spu_init_csa(struct spu_state *csa); extern void spu_fini_csa(struct spu_state *csa); extern int spu_save(struct spu_state *prev, struct spu *spu); diff --git a/arch/powerpc/platforms/cell/spufs/syscalls.c b/arch/powerpc/platforms/cell/spufs/syscalls.c index ea4ba1b6ce6a96..b6de37150e734d 100644 --- a/arch/powerpc/platforms/cell/spufs/syscalls.c +++ b/arch/powerpc/platforms/cell/spufs/syscalls.c @@ -82,8 +82,4 @@ struct spufs_calls spufs_calls = { .spu_run = do_spu_run, .notify_spus_active = do_notify_spus_active, .owner = THIS_MODULE, -#ifdef CONFIG_COREDUMP - .coredump_extra_notes_size = spufs_coredump_extra_notes_size, - .coredump_extra_notes_write = spufs_coredump_extra_notes_write, -#endif }; From ed6f1947ddbcaa4048e2bdceb367310d529acea1 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Wed, 26 Aug 2026 18:06:41 +0200 Subject: [PATCH 0336/1352] coredump: stop unsharing the file descriptor table We currently unshare the file descriptor table before we write the coredump. The only code that ever looked at the file descriptor table was the cell spufs coredump support. It walked all file descriptors to find the spufs contexts it would have to dump as extra elf notes. Now that we killed spufs coredumping stop doing that and update the comments referencing spufs as they ave become stale. Link: https://patch.msgid.link/20260826-work-spufs-coredump-v1-2-579e72a7ab66@kernel.org Acked-by: Arnd Bergmann Signed-off-by: Christian Brauner (Amutable) --- fs/binfmt_elf.c | 4 ++-- fs/coredump.c | 5 ----- 2 files changed, 2 insertions(+), 7 deletions(-) diff --git a/fs/binfmt_elf.c b/fs/binfmt_elf.c index 06d0df1053823e..db32bb40a86704 100644 --- a/fs/binfmt_elf.c +++ b/fs/binfmt_elf.c @@ -2029,7 +2029,7 @@ static int elf_core_dump(struct coredump_params *cprm) { size_t sz = info.size; - /* For cell spufs and x86 xstate */ + /* For x86 xstate */ sz += elf_coredump_extra_notes_size(); phdr4note = kmalloc_obj(*phdr4note); @@ -2093,7 +2093,7 @@ static int elf_core_dump(struct coredump_params *cprm) if (!write_note_info(&info, cprm)) goto end_coredump; - /* For cell spufs and x86 xstate */ + /* For x86 xstate */ if (elf_coredump_extra_notes_write(cprm)) goto end_coredump; diff --git a/fs/coredump.c b/fs/coredump.c index 6114839f5178b0..3e78941f281e36 100644 --- a/fs/coredump.c +++ b/fs/coredump.c @@ -1118,11 +1118,6 @@ static void do_coredump(struct core_name *cn, struct coredump_params *cprm, if (cn->mask & COREDUMP_REJECT) return; - /* get us an unshared descriptor table; almost always a no-op */ - /* The cell spufs coredump code reads the file descriptor tables */ - if (unshare_files()) - return; - if ((cn->mask & COREDUMP_KERNEL) && !coredump_write(cn, cprm, binfmt)) return; From 045dacb2fcf425e9a6356880a214bfa6e1f7f86e Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 10 Sep 2026 17:47:59 +0200 Subject: [PATCH 0337/1352] fs: don't open-code file_close_fd() in close_fd() close_fd() takes the lock, calls file_close_fd_locked() and drops the lock, which is exactly what file_close_fd() does. Use it. No functional changes. Link: https://patch.msgid.link/20260910-work-coredump-unlock-self-v4-1-a5c1800dc930@kernel.org Reviewed-by: NeilBrown Reviewed-by: Oleg Nesterov Signed-off-by: Christian Brauner (Amutable) --- fs/file.c | 7 ++----- 1 file changed, 2 insertions(+), 5 deletions(-) diff --git a/fs/file.c b/fs/file.c index 628ca07dc4b179..59673547de9029 100644 --- a/fs/file.c +++ b/fs/file.c @@ -732,16 +732,13 @@ struct file *file_close_fd_locked(struct files_struct *files, unsigned fd) int close_fd(unsigned fd) { - struct files_struct *files = current->files; struct file *file; - spin_lock(&files->file_lock); - file = file_close_fd_locked(files, fd); - spin_unlock(&files->file_lock); + file = file_close_fd(fd); if (!file) return -EBADF; - return filp_close(file, files); + return filp_close(file, current->files); } EXPORT_SYMBOL(close_fd); From 50a26fecde5e65387f006c7f8915115243565afd Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 10 Sep 2026 17:48:00 +0200 Subject: [PATCH 0338/1352] fs: add switch_files_struct() Add switch_files_struct() to install another table on a task. It consumes the reference to the new table and puts the old one. Convert every place that switches a descriptor table except unshare_files(). No functional changes. Link: https://patch.msgid.link/20260910-work-coredump-unlock-self-v4-2-a5c1800dc930@kernel.org Reviewed-by: NeilBrown Reviewed-by: Oleg Nesterov Signed-off-by: Christian Brauner (Amutable) --- fs/file.c | 23 +++++++++++------------ include/linux/fdtable.h | 1 + kernel/fork.c | 6 ++---- 3 files changed, 14 insertions(+), 16 deletions(-) diff --git a/fs/file.c b/fs/file.c index 59673547de9029..345011dad47283 100644 --- a/fs/file.c +++ b/fs/file.c @@ -515,16 +515,18 @@ void put_files_struct(struct files_struct *files) } } -void exit_files(struct task_struct *tsk) +/* Install @files on @tsk, consuming the reference, and put the old table. */ +void switch_files_struct(struct task_struct *tsk, struct files_struct *files) { - struct files_struct * files = tsk->files; + scoped_guard(task_lock, tsk) + swap(tsk->files, files); + put_files_struct(files); +} - if (files) { - task_lock(tsk); - tsk->files = NULL; - task_unlock(tsk); - put_files_struct(files); - } +void exit_files(struct task_struct *tsk) +{ + if (tsk->files) + switch_files_struct(tsk, NULL); } struct files_struct init_files = { @@ -855,10 +857,7 @@ SYSCALL_DEFINE3(close_range, unsigned int, fd, unsigned int, max_fd, * We're done closing the files we were supposed to. Time to install * the new file descriptor table and drop the old one. */ - task_lock(me); - me->files = cur_fds; - task_unlock(me); - put_files_struct(fds); + switch_files_struct(me, cur_fds); } return 0; diff --git a/include/linux/fdtable.h b/include/linux/fdtable.h index c45306a9f00723..9614c6ecd47734 100644 --- a/include/linux/fdtable.h +++ b/include/linux/fdtable.h @@ -100,6 +100,7 @@ static inline bool close_on_exec(unsigned int fd, const struct files_struct *fil struct task_struct; void put_files_struct(struct files_struct *fs); +void switch_files_struct(struct task_struct *tsk, struct files_struct *files); int unshare_files(void); struct fd_range { unsigned int from, to; diff --git a/kernel/fork.c b/kernel/fork.c index a5934a3176346b..f09ca97411a2b6 100644 --- a/kernel/fork.c +++ b/kernel/fork.c @@ -3323,10 +3323,8 @@ int ksys_unshare(unsigned long unshare_flags) if (new_fs) new_fs = switch_fs_struct(new_fs); - if (new_fd) { - guard(task_lock)(current); - swap(current->files, new_fd); - } + if (new_fd) + switch_files_struct(current, no_free_ptr(new_fd)); if (new_cred) { /* Install the new user namespace */ From 82271be0cf9a217f4fc6e89eebb504db40a93ec7 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 10 Sep 2026 17:48:01 +0200 Subject: [PATCH 0339/1352] fs: move unshare_fd() to fs/file.c Move unshare_fd() where the rest of the descriptor table lifecycle helpers live. No functional changes. Link: https://patch.msgid.link/20260910-work-coredump-unlock-self-v4-3-a5c1800dc930@kernel.org Reviewed-by: NeilBrown Reviewed-by: Oleg Nesterov Signed-off-by: Christian Brauner (Amutable) --- fs/file.c | 18 ++++++++++++++++++ include/linux/fdtable.h | 1 + kernel/fork.c | 18 ------------------ 3 files changed, 19 insertions(+), 18 deletions(-) diff --git a/fs/file.c b/fs/file.c index 345011dad47283..636a87e527b36c 100644 --- a/fs/file.c +++ b/fs/file.c @@ -471,6 +471,24 @@ struct files_struct *dup_fd(struct files_struct *oldf, struct fd_range *punch_ho return newf; } +/* + * Unshare file descriptor table if it is being shared + */ +int unshare_fd(unsigned long unshare_flags, struct files_struct **new_fdp) +{ + struct files_struct *fd = current->files; + + if ((unshare_flags & CLONE_FILES) && + (fd && atomic_read(&fd->count) > 1)) { + fd = dup_fd(fd, NULL); + if (IS_ERR(fd)) + return PTR_ERR(fd); + *new_fdp = fd; + } + + return 0; +} + static struct fdtable *close_files(struct files_struct * files) { /* diff --git a/include/linux/fdtable.h b/include/linux/fdtable.h index 9614c6ecd47734..4ee1598848bb1f 100644 --- a/include/linux/fdtable.h +++ b/include/linux/fdtable.h @@ -102,6 +102,7 @@ struct task_struct; void put_files_struct(struct files_struct *fs); void switch_files_struct(struct task_struct *tsk, struct files_struct *files); int unshare_files(void); +int unshare_fd(unsigned long unshare_flags, struct files_struct **new_fdp); struct fd_range { unsigned int from, to; }; diff --git a/kernel/fork.c b/kernel/fork.c index f09ca97411a2b6..9daf6e94bd5d3e 100644 --- a/kernel/fork.c +++ b/kernel/fork.c @@ -3210,24 +3210,6 @@ static int unshare_fs(unsigned long unshare_flags, struct fs_struct **new_fsp) return 0; } -/* - * Unshare file descriptor table if it is being shared - */ -static int unshare_fd(unsigned long unshare_flags, struct files_struct **new_fdp) -{ - struct files_struct *fd = current->files; - - if ((unshare_flags & CLONE_FILES) && - (fd && atomic_read(&fd->count) > 1)) { - fd = dup_fd(fd, NULL); - if (IS_ERR(fd)) - return PTR_ERR(fd); - *new_fdp = fd; - } - - return 0; -} - /* * unshare allows a process to 'unshare' part of the process * context which was originally shared using clone. copy_* From acdb63ba8fb71cd1ae4583c95a0a176b650fd5e1 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 10 Sep 2026 17:48:02 +0200 Subject: [PATCH 0340/1352] fs: remove unshare_files() exec is the only caller left since commit 433967cab51e ("coredump: stop unsharing the file descriptor table"). All it does is call unshare_fd() with CLONE_FILES and install the copy. Kill the pointless helper and open-code it. No functional changes. Link: https://patch.msgid.link/20260910-work-coredump-unlock-self-v4-4-a5c1800dc930@kernel.org Reviewed-by: NeilBrown Reviewed-by: Oleg Nesterov Signed-off-by: Christian Brauner (Amutable) --- fs/exec.c | 5 ++++- include/linux/fdtable.h | 1 - kernel/fork.c | 24 ------------------------ 3 files changed, 4 insertions(+), 26 deletions(-) diff --git a/fs/exec.c b/fs/exec.c index d3081c8f7c10c0..977778f44cfc33 100644 --- a/fs/exec.c +++ b/fs/exec.c @@ -1124,6 +1124,7 @@ static struct file *bprm_identity_file(const struct linux_binprm *bprm) int begin_new_exec(struct linux_binprm * bprm) { struct task_struct *me = current; + struct files_struct *files = NULL; int retval; /* A pending PT_INTERP substitution this format cannot consume. */ @@ -1160,9 +1161,11 @@ int begin_new_exec(struct linux_binprm * bprm) io_uring_task_cancel(); /* Ensure the files table is not shared. */ - retval = unshare_files(); + retval = unshare_fd(CLONE_FILES, &files); if (retval) goto out; + if (files) + switch_files_struct(me, files); /* * We have to apply CLOEXEC before we change whether the process is diff --git a/include/linux/fdtable.h b/include/linux/fdtable.h index 4ee1598848bb1f..666808a1caf547 100644 --- a/include/linux/fdtable.h +++ b/include/linux/fdtable.h @@ -101,7 +101,6 @@ struct task_struct; void put_files_struct(struct files_struct *fs); void switch_files_struct(struct task_struct *tsk, struct files_struct *files); -int unshare_files(void); int unshare_fd(unsigned long unshare_flags, struct files_struct **new_fdp); struct fd_range { unsigned int from, to; diff --git a/kernel/fork.c b/kernel/fork.c index 9daf6e94bd5d3e..10be4a0ecb3f19 100644 --- a/kernel/fork.c +++ b/kernel/fork.c @@ -3339,30 +3339,6 @@ SYSCALL_DEFINE1(unshare, unsigned long, unshare_flags) return ksys_unshare(unshare_flags); } -/* - * Helper to unshare the files of the current task. - * We don't want to expose copy_files internals to - * the exec layer of the kernel. - */ - -int unshare_files(void) -{ - struct task_struct *task = current; - struct files_struct *old, *copy = NULL; - int error; - - error = unshare_fd(CLONE_FILES, ©); - if (error || !copy) - return error; - - old = task->files; - task_lock(task); - task->files = copy; - task_unlock(task); - put_files_struct(old); - return 0; -} - static int sysctl_max_threads(const struct ctl_table *table, int write, void *buffer, size_t *lenp, loff_t *ppos) { From e2ebce18ca4bc9a555dc140fbdcf09d55d62ad15 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 10 Sep 2026 17:48:03 +0200 Subject: [PATCH 0341/1352] fs: add filp_close_sync() Currently close() already does a synchronous release of the last reference since the task is about to return to userspace and the deferral through task work buys nothing. Add a filp_close_sync() helper. We'll use that in the next patches. No functional changes. Link: https://patch.msgid.link/20260910-work-coredump-unlock-self-v4-5-a5c1800dc930@kernel.org Reviewed-by: NeilBrown Reviewed-by: Oleg Nesterov Signed-off-by: Christian Brauner (Amutable) --- fs/internal.h | 1 + fs/open.c | 17 ++++++++++++++--- 2 files changed, 15 insertions(+), 3 deletions(-) diff --git a/fs/internal.h b/fs/internal.h index c658c8a5ebd569..8812de210d3f18 100644 --- a/fs/internal.h +++ b/fs/internal.h @@ -198,6 +198,7 @@ extern struct file *do_file_open_root(const struct path *, extern struct open_how build_open_how(int flags, umode_t mode); extern int build_open_flags(const struct open_how *how, struct open_flags *op); struct file *file_close_fd_locked(struct files_struct *files, unsigned fd); +int filp_close_sync(struct file *filp, fl_owner_t id); int do_ftruncate(struct file *file, loff_t length, unsigned int flags); int chmod_common(const struct path *path, umode_t mode); diff --git a/fs/open.c b/fs/open.c index 6b1c14e684a93b..998e42ac319abf 100644 --- a/fs/open.c +++ b/fs/open.c @@ -1537,6 +1537,19 @@ int filp_close(struct file *filp, fl_owner_t id) } EXPORT_SYMBOL(filp_close); +/* Like filp_close() but the last reference is put right here. */ +int filp_close_sync(struct file *filp, fl_owner_t id) +{ + int retval; + + /* Kernel threads must never put their final reference here. */ + VFS_WARN_ON_ONCE(current->flags & PF_KTHREAD); + retval = filp_flush(filp, id); + fput_close_sync(filp); + + return retval; +} + /* * Careful here! We test whether the file pointer is NULL before * releasing the fd. This ensures that one clone task can't release @@ -1551,13 +1564,11 @@ SYSCALL_DEFINE1(close, unsigned int, fd) if (!file) return -EBADF; - retval = filp_flush(file, current->files); - /* * We're returning to user space. Don't bother * with any delayed fput() cases. */ - fput_close_sync(file); + retval = filp_close_sync(file, current->files); if (likely(retval == 0)) return 0; From b234893aa0d447c58731aa8c28b851d3a53af50c Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 10 Sep 2026 17:48:04 +0200 Subject: [PATCH 0342/1352] fs: make close_files() synchronous When the last reference to a descriptor table is dropped close_files() closes every file but punts the actual work to task work. For an exiting task that task work only runs in exit_task_work(). Before commit 4a9d4b024a31 ("switch fput to task_work_add") fput() was synchronous everywhere and exit released its files in exit_files(). The deferral made fput() safe from any context. And exit_files() offloaded to task work as a side-effect. And that has downsides. Oleg and Neil noticed that some time ago. A task that exits with a big descriptor table ends up queueing a very large number of files on task work. That leaves a list for any later task_work_cancel() to search under ->pi_lock and costs a lot of atomics too. Let close_files() close right away. Flush and put each file inline the way close(2) does. The final __fput() runs during the table walk now instead of from task_work_run() in exit_task_work(). One difference is the order: task work ran the final __fput()s in reverse and now they run in table order. Every put of a dying table is synchronous now: - exit_files() - copy_process() - close_range(CLOSE_RANGE_UNSHARE) - unshare(2) - exec Kernel threads don't own a file descriptor table and exec already splats were they to exec. kthreadd and every kthread share init_files and init_task pins that forever. Link: https://patch.msgid.link/20260910-work-coredump-unlock-self-v4-6-a5c1800dc930@kernel.org Reviewed-by: NeilBrown Reviewed-by: Oleg Nesterov Signed-off-by: Christian Brauner (Amutable) --- fs/file.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/file.c b/fs/file.c index 636a87e527b36c..b2b466dce7fbe7 100644 --- a/fs/file.c +++ b/fs/file.c @@ -489,7 +489,7 @@ int unshare_fd(unsigned long unshare_flags, struct files_struct **new_fdp) return 0; } -static struct fdtable *close_files(struct files_struct * files) +static struct fdtable *close_files(struct files_struct *files) { /* * It is safe to dereference the fd table without RCU or @@ -509,7 +509,7 @@ static struct fdtable *close_files(struct files_struct * files) if (set & 1) { struct file *file = fdt->fd[i]; if (file) { - filp_close(file, files); + filp_close_sync(file, files); cond_resched(); } } From c1ed65ea167bd064325d99528078ae45836b918c Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 10 Sep 2026 17:48:05 +0200 Subject: [PATCH 0343/1352] fs: make close_range() synchronous __range_close() closes through filp_close() so every file the caller held the last reference to is punted to task work. That costs one cmpxchg per file plus a list entry for any later task_work_cancel() to search under ->pi_lock. close_range(2) exists to close many descriptors in one go fast. So convert it to the same synchronous treatment as close(2) and close_files(). Flush and put each file inline while ->file_lock is dropped. close_range(2) now behaves like close(2). Link: https://patch.msgid.link/20260910-work-coredump-unlock-self-v4-7-a5c1800dc930@kernel.org Reviewed-by: NeilBrown Reviewed-by: Oleg Nesterov Signed-off-by: Christian Brauner (Amutable) --- fs/file.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/file.c b/fs/file.c index b2b466dce7fbe7..178c8cb9da0964 100644 --- a/fs/file.c +++ b/fs/file.c @@ -807,7 +807,7 @@ static inline void __range_close(struct files_struct *files, unsigned int fd, file = file_close_fd_locked(files, fd); if (file) { spin_unlock(&files->file_lock); - filp_close(file, files); + filp_close_sync(file, files); cond_resched(); spin_lock(&files->file_lock); fdt = files_fdtable(files); From fb81dbaa1feea4942941339f2e36498365b80454 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 10 Sep 2026 17:48:06 +0200 Subject: [PATCH 0344/1352] fs: rename do_close_on_exec() to close_cloexec_files() Rename the helper and align it with close_files(). No functional changes. Link: https://patch.msgid.link/20260910-work-coredump-unlock-self-v4-8-a5c1800dc930@kernel.org Reviewed-by: NeilBrown Reviewed-by: Oleg Nesterov Signed-off-by: Christian Brauner (Amutable) --- fs/exec.c | 2 +- fs/file.c | 2 +- include/linux/fdtable.h | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/fs/exec.c b/fs/exec.c index 977778f44cfc33..1d9163155d16e2 100644 --- a/fs/exec.c +++ b/fs/exec.c @@ -1179,7 +1179,7 @@ int begin_new_exec(struct linux_binprm * bprm) * This must happen after the point of no return, and after unsharing * the FD table. */ - do_close_on_exec(me->files); + close_cloexec_files(me->files); /* * Must be called _before_ exec_mmap() as bprm->mm is diff --git a/fs/file.c b/fs/file.c index 178c8cb9da0964..b0490566719cdc 100644 --- a/fs/file.c +++ b/fs/file.c @@ -901,7 +901,7 @@ struct file *file_close_fd(unsigned int fd) return file; } -void do_close_on_exec(struct files_struct *files) +void close_cloexec_files(struct files_struct *files) { unsigned i; struct fdtable *fdt; diff --git a/include/linux/fdtable.h b/include/linux/fdtable.h index 666808a1caf547..2965acd120bcb3 100644 --- a/include/linux/fdtable.h +++ b/include/linux/fdtable.h @@ -106,7 +106,7 @@ struct fd_range { unsigned int from, to; }; struct files_struct *dup_fd(struct files_struct *, struct fd_range *) __latent_entropy; -void do_close_on_exec(struct files_struct *); +void close_cloexec_files(struct files_struct *); int iterate_fd(struct files_struct *, unsigned, int (*)(const void *, struct file *, unsigned), const void *); From 65226dca4c08441e63ad7830bf2947176eb833b0 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 10 Sep 2026 17:48:07 +0200 Subject: [PATCH 0345/1352] fs: make close_cloexec_files() synchronous Punting file closing to task work during exec slows down exec significantly when its done with a bunch of file descriptors. We can do this in-band instead. Flush already runs synchronous. Jann moved close-on-exec in e780259b54e6 ("exec: do_close_on_exec() before taking exec_update_lock") outside of exec_update_lock. The only lock that's still held now is cred_guard_mutex. It's deprecated and has five takers (1) exec (2) ptrace_attach() (3) seccomp() with SECCOMP_FILTER_FLAG_TSYNC (4) writes to /proc//attr/* (5) lsm_set_self_attr() Four of them take the task's own cred_guard_mutex. When close_cloexec_files() runs, de_thread() ensured that the calling task is the only one alive in its thread-group. That leaves ptrace() waiting on cred_guard_mutex of the tracee going through exec. exec already sleeps under cred_guard_mutex in de_thread() when it reads binary and interpreter. So while we add wait-time to an attaching ptracer no new lock dependency is added. vfork() als waits but that's a dup_fd() copy of the fdtable and rarely holds the last reference. If that's an issue we can always change that later. Link: https://lore.kernel.org/CAGudoHEsGP1P+sAWaw_tbh1NesJhSeww8869uzmaqtgk8F43=Q@mail.gmail.com Link: https://patch.msgid.link/20260910-work-coredump-unlock-self-v4-9-a5c1800dc930@kernel.org Reviewed-by: NeilBrown Reviewed-by: Oleg Nesterov Signed-off-by: Christian Brauner (Amutable) --- fs/exec.c | 6 +++--- fs/file.c | 2 +- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/fs/exec.c b/fs/exec.c index 1d9163155d16e2..075a744421e140 100644 --- a/fs/exec.c +++ b/fs/exec.c @@ -1173,9 +1173,9 @@ int begin_new_exec(struct linux_binprm * bprm) * trying to access the should-be-closed file descriptors of a process * undergoing exec(2). * - * This can block on filesystem ->flush() handlers, including waiting - * for FUSE daemons, so do it before exec_mmap takes the - * exec_update_lock. + * This can block on filesystem ->flush() and ->release() handlers, + * including waiting for FUSE daemons, so do it before exec_mmap + * takes the exec_update_lock. * This must happen after the point of no return, and after unsharing * the FD table. */ diff --git a/fs/file.c b/fs/file.c index b0490566719cdc..76e328edf6305e 100644 --- a/fs/file.c +++ b/fs/file.c @@ -928,7 +928,7 @@ void close_cloexec_files(struct files_struct *files) rcu_assign_pointer(fdt->fd[fd], NULL); __put_unused_fd(files, fd); spin_unlock(&files->file_lock); - filp_close(file, files); + filp_close_sync(file, files); cond_resched(); spin_lock(&files->file_lock); } From a23ba77b0061994b33ca34658c1fe09ca7978ddc Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 10 Sep 2026 17:48:08 +0200 Subject: [PATCH 0346/1352] coredump: drop core_state->dumper The core_state->dumper field isn't used anymore. Only its ->next pointer is. The current task is always the dumping thread and the ->task pointer is never read. Replace it with a plain pointer to the list of parked threads. Historically, core_state->dumper was used. Its ->task pointer was read. by fill_note_info() started at &core_state->dumper to ensure that the dumping thread came first in the ELF thread notes. That changed in commit 4b0e21d64253 ("[elf][regset] simplify thread list handling in fill_note_info()"). The first iteration was taken out of the loop. So it's been unused ever since. No functional changes. Suggested-by: NeilBrown Link: https://lore.kernel.org/178900159210.207413.8292125177519817528@noble.neil.brown.name Link: https://patch.msgid.link/20260910-work-coredump-unlock-self-v4-10-a5c1800dc930@kernel.org Reviewed-by: NeilBrown Reviewed-by: Oleg Nesterov Signed-off-by: Christian Brauner (Amutable) --- fs/binfmt_elf.c | 2 +- fs/binfmt_elf_fdpic.c | 2 +- fs/coredump.c | 7 +++---- include/linux/sched/signal.h | 2 +- kernel/exit.c | 4 ++-- 5 files changed, 8 insertions(+), 9 deletions(-) diff --git a/fs/binfmt_elf.c b/fs/binfmt_elf.c index db32bb40a86704..e4bba9b6fecb8d 100644 --- a/fs/binfmt_elf.c +++ b/fs/binfmt_elf.c @@ -1875,7 +1875,7 @@ static int fill_note_info(struct elfhdr *elf, int phdrs, return 0; info->thread->task = dump_task; - for (ct = dump_task->signal->core_state->dumper.next; ct; ct = ct->next) { + for (ct = dump_task->signal->core_state->tasks; ct; ct = ct->next) { t = kzalloc_flex(*t, notes, info->thread_notes); if (unlikely(!t)) return 0; diff --git a/fs/binfmt_elf_fdpic.c b/fs/binfmt_elf_fdpic.c index 068c46875c7442..a2633330c59b9c 100644 --- a/fs/binfmt_elf_fdpic.c +++ b/fs/binfmt_elf_fdpic.c @@ -1504,7 +1504,7 @@ static int elf_fdpic_core_dump(struct coredump_params *cprm) if (!psinfo) goto end_coredump; - for (ct = current->signal->core_state->dumper.next; + for (ct = current->signal->core_state->tasks; ct; ct = ct->next) { tmp = elf_dump_thread_status(cprm->siginfo->si_signo, ct->task, &thread_status_size); diff --git a/fs/coredump.c b/fs/coredump.c index 3e78941f281e36..80ba8a403d7beb 100644 --- a/fs/coredump.c +++ b/fs/coredump.c @@ -526,8 +526,7 @@ static int coredump_wait(int exit_code, struct core_state *core_state) int core_waiters = -EBUSY; init_completion(&core_state->startup); - core_state->dumper.task = tsk; - core_state->dumper.next = NULL; + core_state->tasks = NULL; core_waiters = zap_threads(tsk, core_state, exit_code); if (core_waiters > 0) { @@ -540,7 +539,7 @@ static int coredump_wait(int exit_code, struct core_state *core_state) * all the thread context (extended register state, like * fpu etc) gets copied to the memory. */ - ptr = core_state->dumper.next; + ptr = core_state->tasks; while (ptr != NULL) { wait_task_inactive(ptr->task, TASK_ANY); ptr = ptr->next; @@ -558,7 +557,7 @@ static void coredump_finish(bool core_dumped) spin_lock_irq(¤t->sighand->siglock); if (core_dumped && !__fatal_signal_pending(current)) current->signal->group_exit_code |= 0x80; - next = current->signal->core_state->dumper.next; + next = current->signal->core_state->tasks; current->signal->core_state = NULL; spin_unlock_irq(¤t->sighand->siglock); diff --git a/include/linux/sched/signal.h b/include/linux/sched/signal.h index d45a5476b97dec..14b55d00d6059d 100644 --- a/include/linux/sched/signal.h +++ b/include/linux/sched/signal.h @@ -80,7 +80,7 @@ struct core_thread { struct core_state { atomic_t nr_threads; - struct core_thread dumper; + struct core_thread *tasks; struct completion startup; }; diff --git a/kernel/exit.c b/kernel/exit.c index 4e028f15759788..3df1fffc6674e2 100644 --- a/kernel/exit.c +++ b/kernel/exit.c @@ -435,12 +435,12 @@ static void coredump_task_exit(struct task_struct *tsk, self.task = tsk; if (self.task->flags & PF_SIGNALED) - self.next = xchg(&core_state->dumper.next, &self); + self.next = xchg(&core_state->tasks, &self); else self.task = NULL; /* * Implies mb(), the result of xchg() must be visible - * to core_state->dumper. + * to the dumper. */ if (atomic_dec_and_test(&core_state->nr_threads)) complete(&core_state->startup); From 4f986e209709c97778b526fe899ef21338941ec9 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 10 Sep 2026 17:48:09 +0200 Subject: [PATCH 0347/1352] sched: add wait_var_event_state() All wait_var_event() sleep in a fixed task state. For coredumps we need a variant that takes the state from the caller the way wait_event_state() does. This allows us to continue sleeping with TASK_FREEZABLE. That's certainly also a useful addition for other places. Link: https://patch.msgid.link/20260910-work-coredump-unlock-self-v4-11-a5c1800dc930@kernel.org Reviewed-by: NeilBrown Reviewed-by: Oleg Nesterov Signed-off-by: Christian Brauner (Amutable) --- include/linux/wait_bit.h | 26 ++++++++++++++++++++++++++ 1 file changed, 26 insertions(+) diff --git a/include/linux/wait_bit.h b/include/linux/wait_bit.h index 553d7b23e3adfd..af077ed4caf6a1 100644 --- a/include/linux/wait_bit.h +++ b/include/linux/wait_bit.h @@ -432,6 +432,32 @@ do { \ __ret; \ }) +/** + * wait_var_event_state - wait for a variable to be updated and notified + * @var: the address of variable being waited on + * @condition: the condition to wait for + * @state: the task state to sleep in, %TASK_UNINTERRUPTIBLE etc. + * + * Wait for a @condition to be true, only re-checking when a wake up is + * received for the given @var (an arbitrary kernel address which need + * not be directly related to the given condition, but usually is). + * + * Returns 0 if the condition became true, or %-ERESTARTSYS if a signal + * arrived which @state allows to interrupt. + * + * The condition should normally use smp_load_acquire() or a similarly + * ordered access to ensure that any changes to memory made before the + * condition became true will be visible after the wait completes. + */ +#define wait_var_event_state(var, condition, state) \ +({ \ + int __ret = 0; \ + might_sleep(); \ + if (!(condition)) \ + __ret = ___wait_var_event(var, condition, (state), 0, 0, schedule()); \ + __ret; \ +}) + /** * wait_var_event_any_lock - wait for a variable to be updated under a lock * @var: the address of the variable being waited on From f75691d901bd01eb20093919e2b3748ac1b4ef51 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 10 Sep 2026 17:48:10 +0200 Subject: [PATCH 0348/1352] coredump: replace the startup completion with a thread count coredump_wait() sets core_state->nr_threads to the number of tasks killed and waits for the last thread to enter coredump_task_exit() to signal completion. Let's just wait on the count directly. The exiting tasks can use atomic_dec_and_wake_up() and the dumping task sleeps in wait_var_event_state(). The dumping task must remain freezable since commit f5d39b020809 ("freezer,sched: Rewrite core freezer logic"). So keep the wait TASK_UNINTERRUPTIBLE|TASK_FREEZABLE. Drop the completion and rename nr_threads to threads_remaining. No functional changes. Suggested-by: NeilBrown Link: https://lore.kernel.org/178899497961.207413.10554121774377911612@noble.neil.brown.name Link: https://patch.msgid.link/20260910-work-coredump-unlock-self-v4-12-a5c1800dc930@kernel.org Reviewed-by: NeilBrown Reviewed-by: Oleg Nesterov Signed-off-by: Christian Brauner (Amutable) --- fs/coredump.c | 9 +++++---- include/linux/sched/signal.h | 4 ++-- kernel/exit.c | 6 +++--- 3 files changed, 10 insertions(+), 9 deletions(-) diff --git a/fs/coredump.c b/fs/coredump.c index 80ba8a403d7beb..11c18be497b326 100644 --- a/fs/coredump.c +++ b/fs/coredump.c @@ -39,6 +39,7 @@ #include #include #include +#include #include #include #include @@ -514,7 +515,7 @@ static int zap_threads(struct task_struct *tsk, nr = zap_process(signal, exit_code); clear_tsk_thread_flag(tsk, TIF_SIGPENDING); tsk->flags |= PF_DUMPCORE; - atomic_set(&core_state->nr_threads, nr); + atomic_set(&core_state->threads_remaining, nr); } spin_unlock_irq(&tsk->sighand->siglock); return nr; @@ -525,15 +526,15 @@ static int coredump_wait(int exit_code, struct core_state *core_state) struct task_struct *tsk = current; int core_waiters = -EBUSY; - init_completion(&core_state->startup); core_state->tasks = NULL; core_waiters = zap_threads(tsk, core_state, exit_code); if (core_waiters > 0) { struct core_thread *ptr; - wait_for_completion_state(&core_state->startup, - TASK_UNINTERRUPTIBLE|TASK_FREEZABLE); + wait_var_event_state(&core_state->threads_remaining, + !atomic_read_acquire(&core_state->threads_remaining), + TASK_UNINTERRUPTIBLE|TASK_FREEZABLE); /* * Wait for all the threads to become inactive, so that * all the thread context (extended register state, like diff --git a/include/linux/sched/signal.h b/include/linux/sched/signal.h index 14b55d00d6059d..e039e29cd8c589 100644 --- a/include/linux/sched/signal.h +++ b/include/linux/sched/signal.h @@ -79,9 +79,9 @@ struct core_thread { }; struct core_state { - atomic_t nr_threads; + /* Threads the dumper still waits for. */ + atomic_t threads_remaining; struct core_thread *tasks; - struct completion startup; }; /* diff --git a/kernel/exit.c b/kernel/exit.c index 3df1fffc6674e2..55dbea3b242e1c 100644 --- a/kernel/exit.c +++ b/kernel/exit.c @@ -17,6 +17,7 @@ #include #include #include +#include #include #include #include @@ -442,8 +443,7 @@ static void coredump_task_exit(struct task_struct *tsk, * Implies mb(), the result of xchg() must be visible * to the dumper. */ - if (atomic_dec_and_test(&core_state->nr_threads)) - complete(&core_state->startup); + atomic_dec_and_wake_up(&core_state->threads_remaining); for (;;) { set_current_state(TASK_IDLE|TASK_FREEZABLE); @@ -917,7 +917,7 @@ static void synchronize_group_exit(struct task_struct *tsk, long code) * Serialize with any possible pending coredump. * We must hold siglock around checking core_state * and setting PF_POSTCOREDUMP. The core-inducing thread - * will increment ->nr_threads for each thread in the + * will increment ->threads_remaining for each thread in the * group without PF_POSTCOREDUMP set. */ tsk->flags |= PF_POSTCOREDUMP; From c7dbf14e1b7d92fd25319e1b8c730c69f34f7b78 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Thu, 10 Sep 2026 17:48:11 +0200 Subject: [PATCH 0349/1352] coredump: factor out coredump_wait_inactive() Factor out a new coredump_wait_inactive() helper that COREDUMP_CLOSE_FILES can consume in a bit. No functional changes. Link: https://patch.msgid.link/20260910-work-coredump-unlock-self-v4-13-a5c1800dc930@kernel.org Reviewed-by: NeilBrown Reviewed-by: Oleg Nesterov Signed-off-by: Christian Brauner (Amutable) --- fs/coredump.c | 35 ++++++++++++++++++----------------- 1 file changed, 18 insertions(+), 17 deletions(-) diff --git a/fs/coredump.c b/fs/coredump.c index 11c18be497b326..6f4c6a900dc594 100644 --- a/fs/coredump.c +++ b/fs/coredump.c @@ -521,6 +521,22 @@ static int zap_threads(struct task_struct *tsk, return nr; } +static void coredump_wait_inactive(struct core_state *core_state) +{ + struct core_thread *ptr; + + wait_var_event_state(&core_state->threads_remaining, + !atomic_read_acquire(&core_state->threads_remaining), + TASK_UNINTERRUPTIBLE | TASK_FREEZABLE); + /* + * Wait for all the threads to become inactive, so that + * all the thread context (extended register state, like + * fpu etc) gets copied to the memory. + */ + for (ptr = core_state->tasks; ptr; ptr = ptr->next) + wait_task_inactive(ptr->task, TASK_ANY); +} + static int coredump_wait(int exit_code, struct core_state *core_state) { struct task_struct *tsk = current; @@ -529,23 +545,8 @@ static int coredump_wait(int exit_code, struct core_state *core_state) core_state->tasks = NULL; core_waiters = zap_threads(tsk, core_state, exit_code); - if (core_waiters > 0) { - struct core_thread *ptr; - - wait_var_event_state(&core_state->threads_remaining, - !atomic_read_acquire(&core_state->threads_remaining), - TASK_UNINTERRUPTIBLE|TASK_FREEZABLE); - /* - * Wait for all the threads to become inactive, so that - * all the thread context (extended register state, like - * fpu etc) gets copied to the memory. - */ - ptr = core_state->tasks; - while (ptr != NULL) { - wait_task_inactive(ptr->task, TASK_ANY); - ptr = ptr->next; - } - } + if (core_waiters > 0) + coredump_wait_inactive(core_state); return core_waiters; } From d37b51a8c35c1a4aee099d91bd2017181aaafe1b Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Mon, 21 Sep 2026 15:44:56 +0200 Subject: [PATCH 0350/1352] fork: release the files of a failed fork after sched_cancel_fork() Reorder copy_process() cleanup so sched_fork() taking scx_fork_rwsem scx_pre_fork() is safe and isn't held around exiting files. Right now, a fork that fails while another thread closed the bpf link fd will deadlock against its own read side. It also blocks every fork on the system: copy_process() holds scx_fork_rwsem for read exit_files() close_files() filp_close_sync() bpf_scx_unreg() kthread_flush_work() waits for the disable work scx_root_disable() percpu_down_write(&scx_fork_rwsem) Simply release the child's files after sched_cancel_fork() dropped the lock. The task starts out as a copy of its parent so p->files points to the parent's table until copy_files() replaces it. exit_files() must not run for a fork that failed before that. Clear p->files up front so exit_files() is a no-op for those and make copy_files() set it explicitly for CLONE_FILES. Reported-by: Chris Mason Link: https://patch.msgid.link/20260921-work-coredump-fixes-v3-7-8e4adb1619e6@kernel.org Signed-off-by: Christian Brauner (Amutable) --- kernel/fork.c | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/kernel/fork.c b/kernel/fork.c index 10be4a0ecb3f19..50f5b3e2ca87d9 100644 --- a/kernel/fork.c +++ b/kernel/fork.c @@ -1676,6 +1676,7 @@ static int copy_files(u64 clone_flags, struct task_struct *tsk, if (clone_flags & CLONE_FILES) { atomic_inc(&oldf->count); + tsk->files = oldf; return 0; } @@ -2199,6 +2200,8 @@ __latent_entropy struct task_struct *copy_process( INIT_LIST_HEAD(&p->sibling); rcu_copy_process(p); p->vfork_done = NULL; + /* Set by copy_files(), exit_files() on the error path skips NULL. */ + p->files = NULL; spin_lock_init(&p->alloc_lock); init_sigpending(&p->pending); @@ -2300,7 +2303,7 @@ __latent_entropy struct task_struct *copy_process( goto bad_fork_cleanup_semundo; retval = copy_fs(clone_flags, p, args->umh); if (retval) - goto bad_fork_cleanup_files; + goto bad_fork_cleanup_semundo; retval = copy_sighand(clone_flags, p); if (retval) goto bad_fork_cleanup_fs; @@ -2613,8 +2616,6 @@ __latent_entropy struct task_struct *copy_process( __cleanup_sighand(p->sighand); bad_fork_cleanup_fs: exit_fs(p); /* blocking */ -bad_fork_cleanup_files: - exit_files(p); /* blocking */ bad_fork_cleanup_semundo: exit_sem(p); bad_fork_cleanup_security: @@ -2625,6 +2626,8 @@ __latent_entropy struct task_struct *copy_process( perf_event_free_task(p); bad_fork_sched_cancel_fork: sched_cancel_fork(p); + /* ->release() of a file may need scx_fork_rwsem for write. */ + exit_files(p); /* blocking */ bad_fork_cleanup_policy: lockdep_free_task(p); #ifdef CONFIG_NUMA From b6bbec603c4bc1ed74da83d323b647d54cc22ca5 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Mon, 21 Sep 2026 15:44:57 +0200 Subject: [PATCH 0351/1352] exit: hang up the tty before closing the files do_exit() closes the task's files in exit_files() and hangs up the controlling tty of a session leader in disassociate_ctty(1) after that. Since commit d99d38540bf0 ("fs: make close_files() synchronous") the final __fput() of every file runs inside exit_files(). So when the session leader holds the last open of its tty the tty is released before disassociate_ctty() runs. tty_release() clears signal->tty for the whole session in session_clear_tty() once the count drops to zero and sends no signal doing so. disassociate_ctty(1) then finds neither a tty nor a tty_old_pgrp and does nothing. The foreground process group loses its SIGHUP: do_exit() exit_files() close_files() tty_release() tty->count == 0 session_clear_tty() signal->tty = NULL, no signal disassociate_ctty(1) get_current_tty() NULL signal->tty_old_pgrp NULL, nothing sent That only affects real ttys. For a pty the master's open keeps the slave's count above zero. And it only affects a foreground job that holds no descriptor to the tty anymore while its session leader exits. Everything else is unchanged. The DTR drop on the last close happens in tty_port_shutdown() regardless, stopped jobs get their SIGHUP from kill_orphaned_pgrp() and signal->tty is cleared either way. Before that commit the final __fput() ran from exit_task_work() which comes after disassociate_ctty(). That order isn't old. Until v3.14 exit_task_work() came right after exit_files() and before v3.6 fput() was synchronous, so the tty was always released first. Commit c39df5fa37b0 ("exit: call disassociate_ctty() before exit_task_namespaces()") moved disassociate_ctty() up to fix a pppd crash and in front of exit_task_work() as a side effect. The hangup in this case has worked since then and that's eleven years of userspace being able to rely on it. Hang the tty up before closing the files. This is the ordinary hangup with the file still open: __tty_hangup() swaps in hung_up_tty_fops and tty_release() runs from the close afterwards as it does when a modem drops the line. disassociate_ctty() stays in front of exit_task_namespaces() which the pppd fix needs. Reported-by: Chris Mason Link: https://patch.msgid.link/20260921-work-coredump-fixes-v3-8-8e4adb1619e6@kernel.org Acked-by: Oleg Nesterov Signed-off-by: Christian Brauner (Amutable) --- kernel/exit.c | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/kernel/exit.c b/kernel/exit.c index 55dbea3b242e1c..9ed5eb03d0e199 100644 --- a/kernel/exit.c +++ b/kernel/exit.c @@ -1003,10 +1003,11 @@ void __noreturn do_exit(long code) exit_sem(tsk); exit_shm(tsk); - exit_files(tsk); - exit_fs(tsk); + /* Hang the tty up before the last close of it can clear the session. */ if (group_dead) disassociate_ctty(1); + exit_files(tsk); + exit_fs(tsk); exit_nsproxy_namespaces(tsk); exit_task_work(tsk); exit_thread(tsk); From f729467840595d3e2cad6568344afdb212fc5602 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Mon, 21 Sep 2026 15:45:06 +0200 Subject: [PATCH 0352/1352] fs: close files from the highest descriptor down close_files(), __range_close() and close_cloexec_files() walk the descriptor table from the lowest descriptor up. They used to call filp_close(), which left the final __fput() to task work. task work is a LIFO list, so the releases ran after the walk had finished and in the opposite direction, highest descriptor first. This dumb ordering is relevant for a bunch of broken but long-standing cases. It matters whenever the ->flush() or ->release() of one file waits for something that only the release of another file of the same table provides. Then one of the two orders deadlocks and the other one doesn't: exit, fd 3 is one end of a pipe peer ------------------------------- ---- splice(socket -> pipe) pipe_lock() waits for data or EOF close_files() fd 3: pipe_release() mutex_lock(&pipe->mutex) held by the peer fd 5: the socket, not reached would be the peer's EOF Programs create the thing that guards or wakes another thing first. Hence, it gets the lower descriptor which is the layout that breaks: (1) a tap device released before the AF_LLC socket that holds a reference to it (2) an unlinked fsdax file evicted before the pipe that holds its vmspliced pages (3) an overlayfs directory whose release queues up behind an unlink that waits for a splice into the same directory The exiting task is unkillable in all of them. All of that crap can obviously also become a bug if you reorder the file descriptors. Continue walking all three tables from the highest descriptor down. That restores the order the deferred puts had. ->flush() moves with the release. So it now runs highest descriptor first as well. None of this fixes the underlying defects. For every one of these pairs the mirrored layout deadlocked before and deadlocks again now: (1') splice() holding pipe->mutex across unbounded socket and tty I/O (2') AF_LLC keeping a netdev reference without a NETDEV_UNREGISTER handler (3') uninterruptible wait in dax_break_layout_final() (4') ovl_splice_write() sleeping under the inode lock It all predates the synchronous close and each should really get fixed. Reported-by: Chris Mason Link: https://patch.msgid.link/20260921-work-coredump-fixes-v3-17-8e4adb1619e6@kernel.org Signed-off-by: Christian Brauner (Amutable) --- fs/file.c | 55 +++++++++++++++++++++++++++++-------------------------- 1 file changed, 29 insertions(+), 26 deletions(-) diff --git a/fs/file.c b/fs/file.c index 76e328edf6305e..7f8d0afd8807a7 100644 --- a/fs/file.c +++ b/fs/file.c @@ -497,24 +497,21 @@ static struct fdtable *close_files(struct files_struct *files) * files structure. */ struct fdtable *fdt = rcu_dereference_raw(files->fdt); - unsigned int i, j = 0; + unsigned int j = fdt->max_fds / BITS_PER_LONG; + + /* Highest fd first, the order the deferred puts ran in. */ + while (j--) { + unsigned long set = fdt->open_fds[j]; - for (;;) { - unsigned long set; - i = j * BITS_PER_LONG; - if (i >= fdt->max_fds) - break; - set = fdt->open_fds[j++]; while (set) { - if (set & 1) { - struct file *file = fdt->fd[i]; - if (file) { - filp_close_sync(file, files); - cond_resched(); - } + unsigned int bit = __fls(set); + struct file *file = fdt->fd[j * BITS_PER_LONG + bit]; + + set ^= 1UL << bit; + if (file) { + filp_close_sync(file, files); + cond_resched(); } - i++; - set >>= 1; } } @@ -801,10 +798,14 @@ static inline void __range_close(struct files_struct *files, unsigned int fd, n = last_fd(fdt); max_fd = min(max_fd, n); - for (fd = find_next_bit(fdt->open_fds, max_fd + 1, fd); - fd <= max_fd; - fd = find_next_bit(fdt->open_fds, max_fd + 1, fd + 1)) { - file = file_close_fd_locked(files, fd); + /* Highest fd first, see close_files(). */ + for (n = max_fd + 1; n > fd; ) { + unsigned int cur = find_last_bit(fdt->open_fds, n); + + if (cur >= n || cur < fd) + break; + n = cur; + file = file_close_fd_locked(files, cur); if (file) { spin_unlock(&files->file_lock); filp_close_sync(file, files); @@ -908,20 +909,22 @@ void close_cloexec_files(struct files_struct *files) /* exec unshares first */ spin_lock(&files->file_lock); - for (i = 0; ; i++) { + fdt = files_fdtable(files); + /* Highest fd first, see close_files(). */ + for (i = fdt->max_fds / BITS_PER_LONG; i--; ) { unsigned long set; - unsigned fd = i * BITS_PER_LONG; + fdt = files_fdtable(files); - if (fd >= fdt->max_fds) - break; set = fdt->close_on_exec[i]; if (!set) continue; fdt->close_on_exec[i] = 0; - for ( ; set ; fd++, set >>= 1) { + while (set) { + unsigned int bit = __fls(set); + unsigned fd = i * BITS_PER_LONG + bit; struct file *file; - if (!(set & 1)) - continue; + + set ^= 1UL << bit; file = fdt->fd[fd]; if (!file) continue; From d439f95953abbb84d7ad8b5ded77b068b742cbaf Mon Sep 17 00:00:00 2001 From: Ethan Nelson-Moore Date: Thu, 17 Sep 2026 19:40:37 -0700 Subject: [PATCH 0353/1352] netfs: remove unused fscache_internal.h header fs/netfs/fscache_internal.h has been unused since commit 915cd30cdea8 ("netfs, fscache: Combine fscache with netfs"). Remove it. Signed-off-by: Ethan Nelson-Moore Link: https://patch.msgid.link/20260918024040.338737-1-enelsonmoore@gmail.com Signed-off-by: Christian Brauner (Amutable) --- fs/netfs/fscache_internal.h | 14 -------------- 1 file changed, 14 deletions(-) delete mode 100644 fs/netfs/fscache_internal.h diff --git a/fs/netfs/fscache_internal.h b/fs/netfs/fscache_internal.h deleted file mode 100644 index a09b948fcef212..00000000000000 --- a/fs/netfs/fscache_internal.h +++ /dev/null @@ -1,14 +0,0 @@ -/* SPDX-License-Identifier: GPL-2.0-or-later */ -/* Internal definitions for FS-Cache - * - * Copyright (C) 2021 Red Hat, Inc. All Rights Reserved. - * Written by David Howells (dhowells@redhat.com) - */ - -#include "internal.h" - -#ifdef pr_fmt -#undef pr_fmt -#endif - -#define pr_fmt(fmt) "FS-Cache: " fmt From f23399dbf2d131fac6824e6ae503c2a80eaff908 Mon Sep 17 00:00:00 2001 From: Sebastian Andrzej Siewior Date: Wed, 23 Sep 2026 13:15:20 +0200 Subject: [PATCH 0354/1352] drm/i915/gt: Use spin_lock_irq() instead of local_irq_disable() + spin_lock() execlists_dequeue() is invoked from a function which uses local_irq_disable() to disable interrupts so the spin_lock() behaves like spin_lock_irq(). This breaks PREEMPT_RT because local_irq_disable() + spin_lock() is not the same as spin_lock_irq(). execlists_dequeue_irq() and execlists_dequeue() has each one caller only. If intel_engine_cs::active::lock is acquired and released with the _irq suffix then it behaves almost as if execlists_dequeue() would be invoked with disabled interrupts. The difference is the last part of the function which is then invoked with enabled interrupts. I can't tell if this makes a difference. From looking at it, it might work to move the last unlock at the end of the function as I didn't find anything that would acquire the lock again. Reported-by: Clark Williams Signed-off-by: Sebastian Andrzej Siewior Reviewed-by: Maarten Lankhorst Link: https://patch.msgid.link/20260923111518.310430-9-dev@lankhorst.se Signed-off-by: Maarten Lankhorst --- .../drm/i915/gt/intel_execlists_submission.c | 17 +++++------------ 1 file changed, 5 insertions(+), 12 deletions(-) diff --git a/drivers/gpu/drm/i915/gt/intel_execlists_submission.c b/drivers/gpu/drm/i915/gt/intel_execlists_submission.c index e693b0c9d2a3e2..841f7193267565 100644 --- a/drivers/gpu/drm/i915/gt/intel_execlists_submission.c +++ b/drivers/gpu/drm/i915/gt/intel_execlists_submission.c @@ -1300,7 +1300,7 @@ static void execlists_dequeue(struct intel_engine_cs *engine) * and context switches) submission. */ - spin_lock(&sched_engine->lock); + spin_lock_irq(&sched_engine->lock); /* * If the queue is higher priority than the last @@ -1400,7 +1400,7 @@ static void execlists_dequeue(struct intel_engine_cs *engine) * Even if ELSP[1] is occupied and not worthy * of timeslices, our queue might be. */ - spin_unlock(&sched_engine->lock); + spin_unlock_irq(&sched_engine->lock); return; } } @@ -1426,7 +1426,7 @@ static void execlists_dequeue(struct intel_engine_cs *engine) if (last && !can_merge_rq(last, rq)) { spin_unlock(&ve->base.sched_engine->lock); - spin_unlock(&engine->sched_engine->lock); + spin_unlock_irq(&engine->sched_engine->lock); return; /* leave this for another sibling */ } @@ -1588,7 +1588,7 @@ static void execlists_dequeue(struct intel_engine_cs *engine) */ sched_engine->queue_priority_hint = queue_prio(sched_engine); i915_sched_engine_reset_on_empty(sched_engine); - spin_unlock(&sched_engine->lock); + spin_unlock_irq(&sched_engine->lock); /* * We can skip poking the HW if we ended up with exactly the same set @@ -1614,13 +1614,6 @@ static void execlists_dequeue(struct intel_engine_cs *engine) } } -static void execlists_dequeue_irq(struct intel_engine_cs *engine) -{ - local_irq_disable(); /* Suspend interrupts across request submission */ - execlists_dequeue(engine); - local_irq_enable(); /* flush irq_work (e.g. breadcrumb enabling) */ -} - static void clear_ports(struct i915_request **ports, int count) { memset_p((void **)ports, NULL, count); @@ -2475,7 +2468,7 @@ static void execlists_submission_tasklet(struct tasklet_struct *t) } if (!engine->execlists.pending[0]) { - execlists_dequeue_irq(engine); + execlists_dequeue(engine); start_timeslice(engine); } From a0cd18290718c770cd635b7f6b6516f13c442be4 Mon Sep 17 00:00:00 2001 From: Sebastian Andrzej Siewior Date: Wed, 23 Sep 2026 13:15:21 +0200 Subject: [PATCH 0355/1352] drm/i915: Drop the irqs_disabled() check The !irqs_disabled() check triggers on PREEMPT_RT even with i915_sched_engine::lock acquired. The reason is the lock is transformed into a sleeping lock on PREEMPT_RT and does not disable interrupts. There is no need to check for disabled interrupts. The lockdep annotation below already check if the lock has been acquired by the caller and will yell if the interrupts are not disabled. Remove the !irqs_disabled() check. Reported-by: Maarten Lankhorst Acked-by: Tvrtko Ursulin Signed-off-by: Sebastian Andrzej Siewior Reviewed-by: Rodrigo Vivi Link: https://patch.msgid.link/20260923111518.310430-10-dev@lankhorst.se Signed-off-by: Maarten Lankhorst --- drivers/gpu/drm/i915/i915_request.c | 2 -- 1 file changed, 2 deletions(-) diff --git a/drivers/gpu/drm/i915/i915_request.c b/drivers/gpu/drm/i915/i915_request.c index d2c7b1090df081..f66f8efc70629a 100644 --- a/drivers/gpu/drm/i915/i915_request.c +++ b/drivers/gpu/drm/i915/i915_request.c @@ -610,7 +610,6 @@ bool __i915_request_submit(struct i915_request *request) RQ_TRACE(request, "\n"); - GEM_BUG_ON(!irqs_disabled()); lockdep_assert_held(&engine->sched_engine->lock); /* @@ -719,7 +718,6 @@ void __i915_request_unsubmit(struct i915_request *request) */ RQ_TRACE(request, "\n"); - GEM_BUG_ON(!irqs_disabled()); lockdep_assert_held(&engine->sched_engine->lock); /* From 7890c29092c8ed55a58f84f9f0cd71ac730db1af Mon Sep 17 00:00:00 2001 From: Sebastian Andrzej Siewior Date: Wed, 23 Sep 2026 13:15:22 +0200 Subject: [PATCH 0356/1352] drm/i915/guc: Consider also RCU depth in busy loop. intel_guc_send_busy_loop() looks at in_atomic() and irqs_disabled() to decide if it should busy-spin while waiting or if it may sleep. Both checks will report false on PREEMPT_RT if sleeping spinlocks are acquired leading to RCU splats while the function sleeps. Check also if RCU has been disabled. Reported-by: "John B. Wyatt IV" Reviewed-by: Rodrigo Vivi Signed-off-by: Sebastian Andrzej Siewior Link: https://patch.msgid.link/20260923111518.310430-11-dev@lankhorst.se Signed-off-by: Maarten Lankhorst --- drivers/gpu/drm/i915/gt/uc/intel_guc.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/gpu/drm/i915/gt/uc/intel_guc.h b/drivers/gpu/drm/i915/gt/uc/intel_guc.h index 053780f562c1af..b25fa8f4dc4bde 100644 --- a/drivers/gpu/drm/i915/gt/uc/intel_guc.h +++ b/drivers/gpu/drm/i915/gt/uc/intel_guc.h @@ -362,7 +362,7 @@ static inline int intel_guc_send_busy_loop(struct intel_guc *guc, { int err; unsigned int sleep_period_ms = 1; - bool not_atomic = !in_atomic() && !irqs_disabled(); + bool not_atomic = !in_atomic() && !irqs_disabled() && !rcu_preempt_depth(); /* * FIXME: Have caller pass in if we are in an atomic context to avoid From 70e3cf8a64413d7a4f617c2f95d369e7e501376a Mon Sep 17 00:00:00 2001 From: Maarten Lankhorst Date: Wed, 23 Sep 2026 13:15:23 +0200 Subject: [PATCH 0357/1352] drm/i915/gt: Fix selftests on PREEMPT_RT The engine->busyness() callbacks called from the selftests are on PREEMPT_RT not safe with preemption disabled, because all spinlock_t locks becomes sleeping locks on PREEMPT_RT and must not be acquired with disabled preemption. This is also a problem for perf events, where we disable the busyness events on PREEMPT_RT, as they're run from hardirq context. Previous attempts to fix this failed, so convert the selftest code to read engine->busyness() with migrate_disable() instead of preempt_disable() to prevent selftest failures on PREEMPT_RT. By disabling migration, we prevent moving the selftests between cores, and should decrease the jitter in both the idle cases and busy cases, compared to no prevention at all. Since interrupts were not disabled, some jitter may still occur, but it should hopefully be less with migration disabled. Reviewed-by: Sebastian Andrzej Siewior Link: https://patch.msgid.link/20260923111518.310430-12-dev@lankhorst.se Signed-off-by: Maarten Lankhorst --- drivers/gpu/drm/i915/gt/selftest_engine_pm.c | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/drivers/gpu/drm/i915/gt/selftest_engine_pm.c b/drivers/gpu/drm/i915/gt/selftest_engine_pm.c index 10e556a7eac455..c1eff9edd8a5e9 100644 --- a/drivers/gpu/drm/i915/gt/selftest_engine_pm.c +++ b/drivers/gpu/drm/i915/gt/selftest_engine_pm.c @@ -277,11 +277,11 @@ static int live_engine_busy_stats(void *arg) st_engine_heartbeat_disable(engine); ENGINE_TRACE(engine, "measuring idle time\n"); - preempt_disable(); + migrate_disable(); de = intel_engine_get_busy_time(engine, &t[0]); udelay(100); de = ktime_sub(intel_engine_get_busy_time(engine, &t[1]), de); - preempt_enable(); + migrate_enable(); dt = ktime_sub(t[1], t[0]); if (de < 0 || de > 10) { pr_err("%s: reported %lldns [%d%%] busyness while sleeping [for %lldns]\n", @@ -316,11 +316,11 @@ static int live_engine_busy_stats(void *arg) } ENGINE_TRACE(engine, "measuring busy time\n"); - preempt_disable(); + migrate_disable(); de = intel_engine_get_busy_time(engine, &t[0]); mdelay(100); de = ktime_sub(intel_engine_get_busy_time(engine, &t[1]), de); - preempt_enable(); + migrate_enable(); dt = ktime_sub(t[1], t[0]); if (100 * de < 95 * dt || 95 * de > 100 * dt) { pr_err("%s: reported %lldns [%d%%] busyness while spinning [for %lldns]\n", From 031f0cde234dc0ee4a1b8360a407551fe30020e1 Mon Sep 17 00:00:00 2001 From: Maarten Lankhorst Date: Wed, 23 Sep 2026 13:15:24 +0200 Subject: [PATCH 0358/1352] drm/i915/gt: Set stop_timeout() correctly on PREEMPT-RT Also check if RCU is disabled for PREEMPT-RT, which is the case when local_bh_disable() is called. Reviewed-by: Rodrigo Vivi Reviewed-by: Sebastian Andrzej Siewior Link: https://patch.msgid.link/20260923111518.310430-13-dev@lankhorst.se Signed-off-by: Maarten Lankhorst --- drivers/gpu/drm/i915/gt/intel_engine_cs.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/gpu/drm/i915/gt/intel_engine_cs.c b/drivers/gpu/drm/i915/gt/intel_engine_cs.c index c0fd349a4600c6..9dd9665128caa1 100644 --- a/drivers/gpu/drm/i915/gt/intel_engine_cs.c +++ b/drivers/gpu/drm/i915/gt/intel_engine_cs.c @@ -1607,7 +1607,7 @@ u64 intel_engine_get_last_batch_head(const struct intel_engine_cs *engine) static unsigned long stop_timeout(const struct intel_engine_cs *engine) { - if (in_atomic() || irqs_disabled()) /* inside atomic preempt-reset? */ + if (in_atomic() || irqs_disabled() || rcu_preempt_depth()) /* inside atomic preempt-reset? */ return 0; /* From a8c17ccf9eb85c2b60376bb6ab98d75f897073b9 Mon Sep 17 00:00:00 2001 From: Maarten Lankhorst Date: Wed, 23 Sep 2026 13:15:25 +0200 Subject: [PATCH 0359/1352] drm/i915: Use sleeping selftests for igt_atomic on PREEMPT_RT This makes the i915 selftests slightly happier, especially related to GPU reset. All of the i915 code that used to run in BH's, IRQ's, etc is converted to preemptible code when compiled with PREEMPT_RT, so adding coverage for those situations would require. This is a better approach than trying to convert all locks in i915 from spinlock_t to raw_spinlock_t just to increase test coverage for situations that no longer happen. Reviewed-by: Sebastian Andrzej Siewior Link: https://patch.msgid.link/20260923111518.310430-14-dev@lankhorst.se Signed-off-by: Maarten Lankhorst --- drivers/gpu/drm/i915/selftests/igt_atomic.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/drivers/gpu/drm/i915/selftests/igt_atomic.c b/drivers/gpu/drm/i915/selftests/igt_atomic.c index fb506b6990956b..8ae39cf570b766 100644 --- a/drivers/gpu/drm/i915/selftests/igt_atomic.c +++ b/drivers/gpu/drm/i915/selftests/igt_atomic.c @@ -39,7 +39,14 @@ static void __hardirq_end(void) local_irq_enable(); } +static void __maybe_unused __nop(void) +{} + const struct igt_atomic_section igt_atomic_phases[] = { +#if IS_ENABLED(CONFIG_PREEMPT_RT) + { "sleeping", __nop, __nop }, + { }, +#endif { "preempt", __preempt_begin, __preempt_end }, { "softirq", __softirq_begin, __softirq_end }, { "hardirq", __hardirq_begin, __hardirq_end }, From 4ca0f8f4ca084d7942eb6712a4508c8d64e18f30 Mon Sep 17 00:00:00 2001 From: Hemanth Selam Date: Mon, 7 Sep 2026 10:15:45 +0530 Subject: [PATCH 0360/1352] m68k: coldfire: fix typos in reset.c comment Correct "reseting" to "resetting" and "ColdFure" to "ColdFire". v1 fixed only "reseting", which checkpatch had flagged, and left "ColdFure" two words earlier untouched. Josh Juran spotted it. Only touches a comment, no code changes. Assisted-by: Cursor:claude-opus-5 Signed-off-by: Hemanth Selam Signed-off-by: Greg Ungerer --- arch/m68k/coldfire/reset.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/arch/m68k/coldfire/reset.c b/arch/m68k/coldfire/reset.c index 6e5f8ab39f324a..079c92593eff0c 100644 --- a/arch/m68k/coldfire/reset.c +++ b/arch/m68k/coldfire/reset.c @@ -16,7 +16,7 @@ #include /* - * There are 2 common methods amongst the ColdFure parts for reseting + * There are 2 common methods amongst the ColdFire parts for resetting * the CPU. But there are couple of exceptions, the 5272 and the 547x * have something completely special to them, and we let their specific * subarch code handle them. From 5bf328100cd0c64a59dad1f36d67ea0bbafb109f Mon Sep 17 00:00:00 2001 From: Hemanth Selam Date: Fri, 4 Sep 2026 16:52:44 +0530 Subject: [PATCH 0361/1352] m68k: fix typo "eanble" in comment Correct "eanble" to "Enable", reported by scripts/checkpatch.pl using the misspelling list in scripts/spelling.txt. Only touches comments, no code changes. Assisted-by: Cursor:claude-opus-5 Signed-off-by: Hemanth Selam Signed-off-by: Greg Ungerer --- arch/m68k/include/asm/m53xxacr.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/arch/m68k/include/asm/m53xxacr.h b/arch/m68k/include/asm/m53xxacr.h index 692f90e7fecc13..a29ad33aaf0555 100644 --- a/arch/m68k/include/asm/m53xxacr.h +++ b/arch/m68k/include/asm/m53xxacr.h @@ -32,7 +32,7 @@ #define CACR_DCM_PRE 0x00000200 /* Cache inhibited, precise */ #define CACR_DCM_IMPRE 0x00000300 /* Cache inhibited, imprecise */ #define CACR_WPROTECT 0x00000020 /* Write protect*/ -#define CACR_EUSP 0x00000010 /* Eanble separate user a7 */ +#define CACR_EUSP 0x00000010 /* Enable separate user a7 */ /* * Define the Access Control register flags. From f1f3193d32b3b265c54b2b69fe9e97d8e677a7ee Mon Sep 17 00:00:00 2001 From: Ethan Nelson-Moore Date: Thu, 17 Sep 2026 20:42:13 -0700 Subject: [PATCH 0362/1352] m68k: remove unused header arch/m68k/include/asm/quicc_simple.h has been unused since commit a3595962d824 ("m68knommu: remove obsolete 68360 support"). Remove it. Signed-off-by: Ethan Nelson-Moore Signed-off-by: Greg Ungerer --- arch/m68k/include/asm/quicc_simple.h | 53 ---------------------------- 1 file changed, 53 deletions(-) delete mode 100644 arch/m68k/include/asm/quicc_simple.h diff --git a/arch/m68k/include/asm/quicc_simple.h b/arch/m68k/include/asm/quicc_simple.h deleted file mode 100644 index b9e2808b44ac9b..00000000000000 --- a/arch/m68k/include/asm/quicc_simple.h +++ /dev/null @@ -1,53 +0,0 @@ -/* SPDX-License-Identifier: GPL-2.0 */ -/*********************************** - * $Id: quicc_simple.h,v 1.1 2002/03/02 15:01:10 gerg Exp $ - *********************************** - * - *************************************** - * Simple drivers common header - *************************************** - */ - -#ifndef __SIMPLE_H -#define __SIMPLE_H - -/* #include "quicc.h" */ - -#define GLB_SCC_0 0 -#define GLB_SCC_1 1 -#define GLB_SCC_2 2 -#define GLB_SCC_3 3 - -typedef void (int_routine)(unsigned short interrupt_event); -typedef int_routine *int_routine_ptr; -typedef void *(alloc_routine)(int length); -typedef void (free_routine)(int scc_num, int channel_num, void *buf); -typedef void (store_rx_buffer_routine)(int scc_num, int channel_num, void *buff, int length); -typedef int (handle_tx_error_routine)(int scc_num, int channel_num, QUICC_BD *tbd); -typedef void (handle_rx_error_routine)(int scc_num, int channel_num, QUICC_BD *rbd); -typedef void (handle_lost_error_routine)(int scc_num, int channel_num); - -/* user defined functions for global errors */ -typedef void (handle_glob_overrun_routine)(int scc_number); -typedef void (handle_glob_underrun_routine)(int scc_number); -typedef void (glob_intr_q_overflow_routine)(int scc_number); - -/* - * General initialization and command routines - */ -void quicc_issue_cmd (unsigned short cmd, int scc_num); -void quicc_init(void); -void quicc_scc_init(int scc_number, int number_of_rx_buf, int number_of_tx_buf); -void quicc_smc_init(int smc_number, int number_of_rx_buf, int number_of_tx_buf); -void quicc_scc_start(int scc_num); -void quicc_scc_loopback(int scc_num); - -/* Interrupt enable/disable routines for critical pieces of code*/ -unsigned short IntrDis(void); -void IntrEna(unsigned short old_sr); - -/* For debugging */ -void print_rbd(int scc_num); -void print_tbd(int scc_num); - -#endif From 49716f36bc5b4e2e114e787c64381d33781a2c62 Mon Sep 17 00:00:00 2001 From: Christopher Lusk Date: Sun, 13 Sep 2026 09:17:53 -0400 Subject: [PATCH 0363/1352] m68k: Fix seccomp filtering on ColdFire and 68000 m68k selects HAVE_ARCH_SECCOMP_FILTER for all configurations, but the ColdFire and 68000 syscall entry paths only branch to syscall_trace_enter() for TIF_SYSCALL_TRACE. TIF_SECCOMP alone falls through to syscall dispatch, leaving installed filters ineffective. Test the combined TIF_SYSCALL_TRACE and TIF_SECCOMP mask in both paths and route either flag through the existing slow path. Leave the classic MMU syscall entry implementation unchanged. Validated the combined test under QEMU mcf5208evb (ColdFire): filters denying getpid, openat, unlinkat, and reboot each returned -EPERM. A classic-MMU control (q800) denied the filtered getpid syscall both before and after this change. The 68000 path shares the same source-level omission and receives the identical masked test but was not separately emulated. Compile-tested W=1 with CONFIG_SECCOMP_FILTER=y on both m5208evb_defconfig and a custom no-MMU UCSIMM configuration; the latter built arch/m68k/68000/entry.o and vmlinux. Fixes: 6baaade15594 ("m68k: Add kernel seccomp support") Cc: stable@vger.kernel.org # v6.3+ Suggested-by: Andreas Schwab Assisted-by: Claude:claude-opus-4-8 Assisted-by: Codex:gpt-5.6-sol Signed-off-by: Christopher Lusk Signed-off-by: Greg Ungerer --- arch/m68k/68000/entry.S | 3 ++- arch/m68k/coldfire/entry.S | 3 ++- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/arch/m68k/68000/entry.S b/arch/m68k/68000/entry.S index c257cc415c4780..0d3cbededc43c3 100644 --- a/arch/m68k/68000/entry.S +++ b/arch/m68k/68000/entry.S @@ -79,7 +79,8 @@ ENTRY(system_call) /* Doing a trace ? */ getthreadinfo - btst #(TIF_SYSCALL_TRACE%8),%a2@(TINFO_FLAGS+(31-TIF_SYSCALL_TRACE)/8) + moveb %a2@(TINFO_FLAGS+2),%d1 + andl #((_TIF_SYSCALL_TRACE + _TIF_SECCOMP) >> 8),%d1 jne do_trace cmpl #NR_syscalls,%d0 jcc badsys diff --git a/arch/m68k/coldfire/entry.S b/arch/m68k/coldfire/entry.S index 4ea08336e2fb0a..d658abdc520753 100644 --- a/arch/m68k/coldfire/entry.S +++ b/arch/m68k/coldfire/entry.S @@ -72,7 +72,8 @@ ENTRY(system_call) movel %d2,%a0 movel %a0@,%a1 /* save top of frame */ movel %sp,%a1@(TASK_THREAD+THREAD_ESP0) - btst #(TIF_SYSCALL_TRACE%8),%a0@(TINFO_FLAGS+(31-TIF_SYSCALL_TRACE)/8) + moveb %a0@(TINFO_FLAGS+2),%d2 + andl #((_TIF_SYSCALL_TRACE + _TIF_SECCOMP) >> 8),%d2 bnes 1f movel %d3,%a0 From 0fed483a78e2586c39d01d319d14716836028c8c Mon Sep 17 00:00:00 2001 From: Angelo Dureghello Date: Wed, 1 Jul 2026 16:17:24 +0200 Subject: [PATCH 0364/1352] m68k: defconfig: update stmark2 defconfig Update stmark2 defconfig enabling MCF5441X DACs. Signed-off-by: Angelo Dureghello Signed-off-by: Greg Ungerer --- arch/m68k/configs/stmark2_defconfig | 2 ++ 1 file changed, 2 insertions(+) diff --git a/arch/m68k/configs/stmark2_defconfig b/arch/m68k/configs/stmark2_defconfig index cd454369bfe8df..d303b21949a771 100644 --- a/arch/m68k/configs/stmark2_defconfig +++ b/arch/m68k/configs/stmark2_defconfig @@ -76,6 +76,8 @@ CONFIG_DMADEVICES=y CONFIG_MCF_EDMA=y # CONFIG_VIRTIO_MENU is not set # CONFIG_VHOST_MENU is not set +CONFIG_IIO=y +CONFIG_MCF54415_DAC=y CONFIG_EXT2_FS=y CONFIG_EXT2_FS_XATTR=y CONFIG_EXT2_FS_POSIX_ACL=y From 29036c5910df0f0a4c824d7da5b6254c7d4ce500 Mon Sep 17 00:00:00 2001 From: Greg Ungerer Date: Thu, 24 Sep 2026 00:23:33 +1000 Subject: [PATCH 0365/1352] m68knommu: fix compile breakage for 5407 cleopatra board Compiling for the Cleopatra target and specifying the ColdFire 5407 SoC will fail with the following errors: m68k-linux-ld: arch/m68k/coldfire/nettel.o: in function `init_nettel': nettel.c:(.init.text+0x74): undefined reference to `ppdata' m68k-linux-ld: nettel.c:(.init.text+0x9c): undefined reference to `ppdata' m68k-linux-ld: nettel.c:(.init.text+0x6a): undefined reference to `ppdata' m68k-linux-ld: nettel.c:(.init.text+0x7a): undefined reference to `ppdata' m68k-linux-ld: nettel.c:(.init.text+0x92): undefined reference to `ppdata' The fix is to define "ppdata" in the same way as is done for the 5307 and 5272 - it is the same register setup in the 5407 as those other SoC parts. Signed-off-by: Greg Ungerer --- arch/m68k/coldfire/m5407.c | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/arch/m68k/coldfire/m5407.c b/arch/m68k/coldfire/m5407.c index 23a14416241c86..1f3e30b3339826 100644 --- a/arch/m68k/coldfire/m5407.c +++ b/arch/m68k/coldfire/m5407.c @@ -22,6 +22,11 @@ /***************************************************************************/ +/* + * Some platforms need software versions of the GPIO data registers. + */ +unsigned short ppdata; + DEFINE_CLK(pll, "pll.0", MCF_CLK); DEFINE_CLK(sys, "sys.0", MCF_BUSCLK); From 6dc9fb1516aa4930329ef561f078c7de9257abf2 Mon Sep 17 00:00:00 2001 From: Andreas Hindborg Date: Mon, 8 Jun 2026 14:32:00 +0200 Subject: [PATCH 0366/1352] rust: configfs: fix data offset calculation for subsystem callbacks `get_group_data()` chooses between `Group::::container_of()` and `Subsystem::::container_of()` based on whether `this` represents the root group of a configfs subsystem. It detects this by checking `(*this).cg_subsys.is_null()`, but `link_group()` in `fs/configfs/dir.c` unconditionally sets `cg_subsys` for every `config_group` attached anywhere in a registered subsystem, including the subsystem's own `su_group`. The only `config_group` with a NULL `cg_subsys` is the configfs root, on which userspace cannot trigger callbacks. The check is therefore always false at runtime, and the `Subsystem` branch is dead. Subsystem-level callbacks reach the `Group` branch and may read `data` at the wrong offset. Thus change the `is_root` check to correctly identify whether a group is a root group by comparing `this` to `subsys.su_group`. Cc: stable@vger.kernel.org Fixes: 446cafc295bf ("rust: configfs: introduce rust support for configfs") Link: https://msgid.link/20260608-configfs-fix-offset-v1-1-7f01b8fb9e5f@kernel.org Signed-off-by: Andreas Hindborg --- rust/kernel/configfs.rs | 42 +++++++++++++++++++++++++---------------- 1 file changed, 26 insertions(+), 16 deletions(-) diff --git a/rust/kernel/configfs.rs b/rust/kernel/configfs.rs index ec295ab965da17..40dee77dc3e02d 100644 --- a/rust/kernel/configfs.rs +++ b/rust/kernel/configfs.rs @@ -305,26 +305,36 @@ unsafe impl HasGroup for Group { /// /// `this` must be a valid pointer. /// -/// If `this` does not represent the root group of a configfs subsystem, -/// `this` must be a pointer to a `bindings::config_group` embedded in a -/// `Group`. +/// If `this` is the `su_group` field of a `bindings::configfs_subsystem`, that +/// `configfs_subsystem` must be embedded in a `Subsystem`. /// -/// Otherwise, `this` must be a pointer to a `bindings::config_group` that -/// is embedded in a `bindings::configfs_subsystem` that is embedded in a -/// `Subsystem`. +/// Otherwise, `this` must be a pointer to a `bindings::config_group` embedded +/// in a `Group`. unsafe fn get_group_data<'a, Parent>(this: *mut bindings::config_group) -> &'a Parent { // SAFETY: `this` is a valid pointer. - let is_root = unsafe { (*this).cg_subsys.is_null() }; - - if !is_root { - // SAFETY: By C API contact,`this` was returned from a call to - // `make_group`. The pointer is known to be embedded within a - // `Group`. - unsafe { &(*Group::::container_of(this)).data } - } else { - // SAFETY: By C API contract, `this` is a pointer to the - // `bindings::config_group` field within a `Subsystem`. + let subsys = unsafe { (*this).cg_subsys }; + // `link_group()` in `fs/configfs/dir.c` assigns `cg_subsys` for every + // `config_group` attached anywhere in a registered subsystem, including + // the subsystem's own `su_group` (which gets a pointer to itself). The + // only `config_group` with a NULL `cg_subsys` is the configfs root, and + // userspace cannot trigger callbacks on it. The group is therefore the + // subsystem's `su_group` iff it equals `&cg_subsys->su_group`. + // + // SAFETY: For every `config_group` the configfs core dispatches a + // callback on, `cg_subsys` was set by `link_group()` at registration time + // and points to a valid `configfs_subsystem` that outlives the callback. + let is_root = !subsys.is_null() + && core::ptr::eq(this.cast_const(), unsafe { &raw const (*subsys).su_group }); + + if is_root { + // SAFETY: By the above, `this` is the `su_group` field of a + // `configfs_subsystem` that, by function safety requirements, is + // embedded in a `Subsystem`. unsafe { &(*Subsystem::container_of(this)).data } + } else { + // SAFETY: By function safety requirements, `this` is a + // `config_group` embedded in a `Group`. + unsafe { &(*Group::::container_of(this)).data } } } From c9e608d1247cd4c286a54d7f26188c41e8f0a6c2 Mon Sep 17 00:00:00 2001 From: Ankit Nautiyal Date: Thu, 24 Sep 2026 07:49:50 +0530 Subject: [PATCH 0367/1352] drm/i915/display: Update the CMN_SDP_TL in fastset path Commit bfa597238073 ("drm/i915/dip: Enable Common SDP Transmission line") enabled the CMN_SDP_TL for the modeset path, but missed to add it to the fastset path. Since dip.cmn_sdp_tl is already skipped from the strict pipe-config comparison during fastsets, it was always intended to get updated without full modeset, but the update to the register was missed. During a fastset, if CMN_SDP_TL changes, update the register similar to the EMP_AS_SDP_TL, which gets updated in intel_vrr_set_transcoder_timings(). v2: Update cmn_sdp_changed() to also compare gmp_sdp_tl, pps_sdp_tl and vsc_ext_sdp_tl, not just cmn_sdp_tl. (Sashiko) Fixes: bfa597238073 ("drm/i915/dip: Enable Common SDP Transmission line") Cc: Suraj Kandpal Signed-off-by: Ankit Nautiyal Reviewed-by: Suraj Kandpal Link: https://patch.msgid.link/20260924021950.350207-1-ankit.k.nautiyal@intel.com --- drivers/gpu/drm/i915/display/intel_display.c | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/drivers/gpu/drm/i915/display/intel_display.c b/drivers/gpu/drm/i915/display/intel_display.c index e3e0d2f8faefdc..d5d4453f283bb0 100644 --- a/drivers/gpu/drm/i915/display/intel_display.c +++ b/drivers/gpu/drm/i915/display/intel_display.c @@ -69,6 +69,7 @@ #include "intel_cx0_phy.h" #include "intel_ddi.h" #include "intel_de.h" +#include "intel_dip.h" #include "intel_display_driver.h" #include "intel_display_power.h" #include "intel_display_regs.h" @@ -999,6 +1000,15 @@ static bool cmrr_params_changed(const struct intel_crtc_state *old_crtc_state, old_crtc_state->vrr.cmrr.cmrr_n != new_crtc_state->vrr.cmrr.cmrr_n; } +static bool cmn_sdp_changed(const struct intel_crtc_state *old_crtc_state, + const struct intel_crtc_state *new_crtc_state) +{ + return old_crtc_state->dip.cmn_sdp_tl != new_crtc_state->dip.cmn_sdp_tl || + old_crtc_state->dip.gmp_sdp_tl != new_crtc_state->dip.gmp_sdp_tl || + old_crtc_state->dip.pps_sdp_tl != new_crtc_state->dip.pps_sdp_tl || + old_crtc_state->dip.vsc_ext_sdp_tl != new_crtc_state->dip.vsc_ext_sdp_tl; +} + static bool intel_crtc_vrr_enabling(struct intel_atomic_state *state, struct intel_crtc *crtc) { @@ -6924,6 +6934,9 @@ static void intel_pre_update_crtc(struct intel_atomic_state *state, if (vrr_params_changed(old_crtc_state, new_crtc_state) || cmrr_params_changed(old_crtc_state, new_crtc_state)) intel_vrr_set_transcoder_timings(new_crtc_state); + + if (cmn_sdp_changed(old_crtc_state, new_crtc_state)) + intel_dip_cmn_sdp_transmission_line_enable(new_crtc_state); } intel_fbc_update(state, crtc); From 385e5adb9e5d28b48c20a4ec25be917d2943e5ea Mon Sep 17 00:00:00 2001 From: "Masami Hiramatsu (Google)" Date: Wed, 30 Sep 2026 00:09:41 +0900 Subject: [PATCH 0368/1352] selftests/ftrace: Add generic boot tracing test framework Add a generic test framework under tools/testing/selftests/ftrace/boottime/ to verify kernel boot tracing configurations (bootconfig, kernel command-line tracing options, and persistent ring buffers) during early boot. The test runner (run_boottime_test.sh) sources ktap_helpers.sh to generate output in TAP version 13 format. It builds a minimal initramfs with busybox, dynamically resolves test specifications (.bconf for bootconfig, .cmdline for kernel boot parameters, .qemuopts for QEMU options, and # REBOOT: 1 for multi-boot crash/reboot tests with doubled timeout), boots QEMU, and runs corresponding tracefs checker scripts. Link: https://lore.kernel.org/all/178649541846.438282.672153252242971310.stgit@devnote2/ Assisted-by: Antigravity:gemini-3.6-flash Signed-off-by: Masami Hiramatsu (Google) --- tools/testing/selftests/ftrace/Makefile | 4 +- tools/testing/selftests/ftrace/boottime-ktap | 6 + .../selftests/ftrace/boottime/Makefile | 10 + .../testing/selftests/ftrace/boottime/README | 74 ++++ .../ftrace/boottime/run_boottime_test.sh | 403 ++++++++++++++++++ tools/testing/selftests/ftrace/config | 6 + 6 files changed, 501 insertions(+), 2 deletions(-) create mode 100755 tools/testing/selftests/ftrace/boottime-ktap create mode 100644 tools/testing/selftests/ftrace/boottime/Makefile create mode 100644 tools/testing/selftests/ftrace/boottime/README create mode 100755 tools/testing/selftests/ftrace/boottime/run_boottime_test.sh diff --git a/tools/testing/selftests/ftrace/Makefile b/tools/testing/selftests/ftrace/Makefile index 7c12263f82602a..3d41f9545683b5 100644 --- a/tools/testing/selftests/ftrace/Makefile +++ b/tools/testing/selftests/ftrace/Makefile @@ -2,8 +2,8 @@ all: TEST_PROGS_EXTENDED := ftracetest -TEST_PROGS := ftracetest-ktap -TEST_FILES := test.d settings +TEST_PROGS := ftracetest-ktap boottime-ktap +TEST_FILES := test.d settings boottime EXTRA_CLEAN := $(OUTPUT)/logs/* TEST_GEN_FILES := poll diff --git a/tools/testing/selftests/ftrace/boottime-ktap b/tools/testing/selftests/ftrace/boottime-ktap new file mode 100755 index 00000000000000..1fed68c8a94463 --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime-ktap @@ -0,0 +1,6 @@ +#!/bin/sh -e +# SPDX-License-Identifier: GPL-2.0-only +# +# boottime-ktap: Wrapper to integrate boottime tracing test framework with kselftest runner + +exec ./boottime/run_boottime_test.sh "$@" diff --git a/tools/testing/selftests/ftrace/boottime/Makefile b/tools/testing/selftests/ftrace/boottime/Makefile new file mode 100644 index 00000000000000..fd0975350f10ca --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/Makefile @@ -0,0 +1,10 @@ +# SPDX-License-Identifier: GPL-2.0 + +all: + +run_tests: + @./run_boottime_test.sh + +clean: + +.PHONY: all run_tests clean diff --git a/tools/testing/selftests/ftrace/boottime/README b/tools/testing/selftests/ftrace/boottime/README new file mode 100644 index 00000000000000..1eb0191d75d47f --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/README @@ -0,0 +1,74 @@ +Boot Tracing Test Suite +======================= + +This directory contains an automated test framework to verify kernel boot +tracing configurations (bootconfig, kernel command-line tracing options, and +persistent ring buffers) during early boot. + +Overview +-------- +The test runner (`run_boottime_test.sh`) constructs a lightweight initramfs with +busybox and a tracefs checker script from `tests/`, attaches optional bootconfigs +from `bootconfigs/` using the `bootconfig` tool, applies optional command-line +parameters from `cmdlines/`, boots the target kernel image under QEMU, and reports +results according to TAP version 13 format. + +Test Specification Format +------------------------- +For each test script `tests/.sh`, the harness dynamically detects: +- `bootconfigs/.bconf` or `.bconf`: Bootconfig configuration file. +- `cmdlines/.cmdline` or `# CMDLINE: ` header in script: Kernel boot parameters. +- `qemuopts/.qemuopts` or `# QEMUOPTS: ` header in script: Additional QEMU CLI arguments. +- `# APPLETS: ` header in script: Additional BusyBox applet symlinks to create in initramfs. +- `# REBOOT: 1` header in script: Multi-boot reboot/crash test mode (omits `-no-reboot` and doubles timeout). + +Required Tools +-------------- +- Host Architecture: x86 (x86_64/i686/i386/x86; tests skip on non-x86 hosts) +- `qemu-system-x86_64` (or custom QEMU executable) +- `busybox` (host binary at `/usr/bin/busybox` or custom path) +- `bootconfig` (compiled from `tools/bootconfig/bootconfig` or in PATH) +- `file` (standard file architecture identifier) +- `cpio` (standard archiver) +- `timeout` (standard coreutils execution timer) + +Required Kernel Configurations +------------------------------ +The kernel binary under test must be compiled with: +- `CONFIG_BOOT_CONFIG=y` +- `CONFIG_BOOTTIME_TRACING=y` +- `CONFIG_MAGIC_SYSRQ=y` (for guest reboot and poweroff) + +Additional feature-specific config options enable respective testcases: +- `CONFIG_KPROBE_EVENTS=y` (for kprobe tests) +- `CONFIG_SYNTH_EVENTS=y` (for synthetic event tests) +- `CONFIG_EPROBE_EVENTS=y` (for eprobe tests) +- `CONFIG_FPROBE_EVENTS=y` (for fprobe and tracepoint probe tests) +- `CONFIG_FUNCTION_TRACER=y` (for tracer options) +- `CONFIG_RESERVE_MEM=y` (for persistent ring buffer tests) + +Usage +----- +To run all test cases with a compiled kernel image: + + $ ./run_boottime_test.sh -k /path/to/bzImage + +To run a specific test case (e.g. `01-kprobe`): + + $ ./run_boottime_test.sh -k /path/to/bzImage -t 01-kprobe + +Customizing Tool Paths: + + $ ./run_boottime_test.sh -b /path/to/bootconfig \ + -k /path/to/bzImage \ + -q /path/to/qemu-system-x86_64 \ + -B /path/to/busybox + +Environment Variables +--------------------- +Tool locations and kernel paths can also be set via environment variables: +- `BOOTCONFIG`: Path to the `bootconfig` tool +- `KERNEL`: Path to the `bzImage` kernel binary +- `QEMU`: Path to the QEMU binary +- `BUSYBOX`: Path to the `busybox` binary +- `LOGDIR`: Directory path to preserve QEMU execution log files diff --git a/tools/testing/selftests/ftrace/boottime/run_boottime_test.sh b/tools/testing/selftests/ftrace/boottime/run_boottime_test.sh new file mode 100755 index 00000000000000..6e8593d5a2c549 --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/run_boottime_test.sh @@ -0,0 +1,403 @@ +#!/bin/bash +# SPDX-License-Identifier: GPL-2.0 +# Copyright (C) 2026, Google LLC. +# +# Generic Boot Tracing Test Harness +# Builds a lightweight initramfs, applies bootconfigs and kernel cmdline parameters, +# boots QEMU, and verifies tracefs configuration using test scripts. + +set -e + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +if [ -d "$SCRIPT_DIR/boottime" ]; then + SCRIPT_DIR="$SCRIPT_DIR/boottime" +fi +KERNEL_SRC="$(cd "$SCRIPT_DIR/../../../../.." && pwd)" + +# Source KTAP helpers +if [ -f "$SCRIPT_DIR/../../kselftest/ktap_helpers.sh" ]; then + KSELFTEST_DIR="$(cd "$SCRIPT_DIR/../../kselftest" && pwd)" +elif [ -f "$SCRIPT_DIR/../kselftest/ktap_helpers.sh" ]; then + KSELFTEST_DIR="$(cd "$SCRIPT_DIR/../kselftest" && pwd)" +else + echo "Error: ktap_helpers.sh not found relative to $SCRIPT_DIR" >&2 + exit 1 +fi +. "$KSELFTEST_DIR/ktap_helpers.sh" + +BOOTCONFIG="${BOOTCONFIG:-"$KERNEL_SRC/tools/bootconfig/bootconfig"}" +KERNEL="${KERNEL:-"$KERNEL_SRC/arch/x86/boot/bzImage"}" +QEMU="${QEMU:-"qemu-system-x86_64"}" +BUSYBOX="${BUSYBOX:-"$(which busybox 2>/dev/null || echo "/usr/bin/busybox")"}" +TIMEOUT="${TIMEOUT:-60s}" +LOGDIR="${LOGDIR:-""}" +TARGET_TEST="all" + +BUSYBOX_APPLETS=( + sh + cat + mount + umount + echo + sync + reboot + poweroff + cpio + grep + sleep + sed +) + +usage() { + echo "Usage: $0 [options]" + echo "Options:" + echo " -b, --bootconfig PATH Path to bootconfig tool (default: $BOOTCONFIG)" + echo " -k, --kernel PATH Path to kernel image (default: $KERNEL)" + echo " -q, --qemu PATH Path to QEMU executable (default: $QEMU)" + echo " -B, --busybox PATH Path to busybox binary (default: $BUSYBOX)" + echo " -T, --timeout DURATION QEMU execution timeout (default: $TIMEOUT)" + echo " -l, --logdir DIR Directory to save QEMU log files" + echo " -t, --test NAME Run specific test (e.g. 01-kprobe) or 'all'" + echo " -h, --help Show this help message" + exit 1 +} + +while [ $# -gt 0 ]; do + case "$1" in + -b|--bootconfig) + BOOTCONFIG="$2" + shift 2 + ;; + -k|--kernel) + KERNEL="$2" + shift 2 + ;; + -q|--qemu) + QEMU="$2" + shift 2 + ;; + -B|--busybox) + BUSYBOX="$2" + shift 2 + ;; + -T|--timeout) + TIMEOUT="$2" + shift 2 + ;; + -l|--logdir) + LOGDIR="$2" + shift 2 + ;; + -t|--test) + TARGET_TEST="$2" + shift 2 + ;; + -h|--help) + usage + ;; + *) + echo "Unknown option: $1" + usage + ;; + esac +done + +# Check host architecture +HOST_ARCH="$(uname -m)" +case "$HOST_ARCH" in + x86_64|i686|i386|x86) + ;; + *) + if [ "$KERNEL" = "$KERNEL_SRC/arch/x86/boot/bzImage" ]; then + ktap_print_header + ktap_skip_all \ + "Default boot tracing test is x86 only ($HOST_ARCH)" + exit 0 + fi + ;; +esac + +# Ensure bootconfig tool is built if Makefile exists +if [ ! -x "$BOOTCONFIG" ]; then + if [ -f "$KERNEL_SRC/tools/bootconfig/Makefile" ]; then + make -C "$KERNEL_SRC/tools/bootconfig" > /dev/null 2>&1 || true + fi + if [ ! -x "$BOOTCONFIG" ]; then + BOOTCONFIG="$(which bootconfig 2>/dev/null || true)" + fi +fi + +if [ ! -x "$BOOTCONFIG" ]; then + ktap_print_header + ktap_skip_all "bootconfig tool not found at $BOOTCONFIG" + exit 0 +fi + +if [ ! -f "$KERNEL" ]; then + ktap_print_header + ktap_skip_all "Kernel image not found at $KERNEL" + exit 0 +fi + +if ! command -v "$QEMU" >/dev/null 2>&1; then + ktap_print_header + ktap_skip_all "QEMU binary not found: $QEMU" + exit 0 +fi + +if [ ! -x "$BUSYBOX" ]; then + ktap_print_header + ktap_skip_all "busybox binary not found at $BUSYBOX" + exit 0 +fi + +# Verify file command existence and architecture compatibility +if ! command -v file >/dev/null 2>&1; then + ktap_print_header + ktap_skip_all "file tool not found" + exit 0 +fi + +KERNEL_INFO="$(file -bL "$KERNEL" 2>/dev/null || true)" +case "$KERNEL_INFO" in + *x86*) + ;; + *) + ktap_print_header + ktap_skip_all "Kernel architecture is not x86 ($KERNEL_INFO)" + exit 0 + ;; +esac + +BUSYBOX_INFO="$(file -bL "$BUSYBOX" 2>/dev/null || true)" +case "$BUSYBOX_INFO" in + *x86-64*|*x86_64*|*80386*|*i386*) + ;; + *) + ktap_print_header + ktap_skip_all \ + "busybox binary architecture is incompatible with x86 kernel ($BUSYBOX_INFO)" + exit 0 + ;; +esac + +if ! command -v cpio >/dev/null 2>&1; then + ktap_print_header + ktap_skip_all "cpio tool not found" + exit 0 +fi + +if ! command -v timeout >/dev/null 2>&1; then + ktap_print_header + ktap_skip_all "timeout tool not found" + exit 0 +fi + +double_timeout() { + local val="$1" + if [[ "$val" =~ ^([0-9]+)([a-z]*)$ ]]; then + local num="${BASH_REMATCH[1]}" + local unit="${BASH_REMATCH[2]}" + echo "$((num * 2))$unit" + else + echo "120s" + fi +} + +run_single_test() { + local test_script="$1" + local name + local dir + name="$(basename "$test_script" .sh)" + dir="$(dirname "$test_script")" + + local bconf_file="" + if [ -f "$SCRIPT_DIR/bootconfigs/$name.bconf" ]; then + bconf_file="$SCRIPT_DIR/bootconfigs/$name.bconf" + elif [ -f "$dir/$name.bconf" ]; then + bconf_file="$dir/$name.bconf" + fi + + local test_cmdline="" + if [ -f "$SCRIPT_DIR/cmdlines/$name.cmdline" ]; then + test_cmdline="$(cat "$SCRIPT_DIR/cmdlines/$name.cmdline")" + elif [ -f "$dir/$name.cmdline" ]; then + test_cmdline="$(cat "$dir/$name.cmdline")" + fi + + local inline_cmdline + inline_cmdline="$(grep -E '^# *CMDLINE:' "$test_script" | sed 's/^# *CMDLINE://' | xargs || true)" + if [ -n "$inline_cmdline" ]; then + test_cmdline="$test_cmdline $inline_cmdline" + fi + + local test_qemuopts="" + if [ -f "$SCRIPT_DIR/qemuopts/$name.qemuopts" ]; then + test_qemuopts="$(cat "$SCRIPT_DIR/qemuopts/$name.qemuopts")" + elif [ -f "$dir/$name.qemuopts" ]; then + test_qemuopts="$(cat "$dir/$name.qemuopts")" + fi + + local inline_qemuopts + inline_qemuopts="$(grep -E '^# *QEMUOPTS:' "$test_script" | sed 's/^# *QEMUOPTS://' | xargs || true)" + if [ -n "$inline_qemuopts" ]; then + test_qemuopts="$test_qemuopts $inline_qemuopts" + fi + + local test_timeout="$TIMEOUT" + local inline_timeout + inline_timeout="$(grep -E '^# *TIMEOUT:' "$test_script" | sed 's/^# *TIMEOUT://' | xargs || true)" + if [ -n "$inline_timeout" ]; then + test_timeout="$inline_timeout" + fi + + local is_reboot=0 + if grep -q -E '^# *REBOOT: *1' "$test_script" || [ -f "$dir/$name.reboot" ]; then + is_reboot=1 + fi + + local effective_timeout="$test_timeout" + if [ "$is_reboot" -eq 1 ]; then + effective_timeout="$(double_timeout "$test_timeout")" + fi + + local workdir + workdir="$(mktemp -d)" + if [ -z "$workdir" ] || [ ! -d "$workdir" ]; then + ktap_test_fail "$name (failed to create temporary directory)" + return 0 + fi + trap 'rm -rf "$workdir"' EXIT + + local test_applets=("${BUSYBOX_APPLETS[@]}") + local inline_applets + inline_applets="$(grep -E '^# *APPLETS:' "$test_script" | sed 's/^# *APPLETS://' | xargs || true)" + if [ -n "$inline_applets" ]; then + read -r -a extra_applets <<< "$inline_applets" || true + test_applets+=("${extra_applets[@]}") + fi + + local rootfs="$workdir/rootfs" + mkdir -p "$rootfs"/{bin,sbin,etc,proc,sys,dev,tmp} + + cp -L "$BUSYBOX" "$rootfs/bin/busybox" + chmod +x "$rootfs/bin/busybox" + (cd "$rootfs/bin" && for applet in "${test_applets[@]}"; do ln -sf busybox "$applet"; done) + + if command -v ldd >/dev/null 2>&1; then + for lib in $(ldd "$BUSYBOX" 2>/dev/null | grep -o '/[^ ]*' || true); do + if [ -f "$lib" ]; then + mkdir -p "$rootfs$(dirname "$lib")" + cp -L "$lib" "$rootfs$lib" 2>/dev/null || true + fi + done + fi + + cat << 'EOF' > "$rootfs/init" +#!/bin/sh +mount -t proc proc /proc 2>/dev/null +mount -t sysfs sys /sys 2>/dev/null +mount -t tracefs nodev /sys/kernel/tracing 2>/dev/null + +/bin/check_test.sh +RET=$? + +if [ $RET -eq 0 ]; then + echo "TEST RESULT: PASS" +else + echo "TEST RESULT: FAIL" +fi +sync +echo o > /proc/sysrq-trigger 2>/dev/null || true +poweroff -f 2>/dev/null || reboot -f 2>/dev/null || true +while true; do sleep 100 2>/dev/null || break; done +EOF + chmod +x "$rootfs/init" + + cp "$test_script" "$rootfs/bin/check_test.sh" + chmod +x "$rootfs/bin/check_test.sh" + + local initramfs="$workdir/initramfs.cpio" + (cd "$rootfs" && find . | cpio -o -H newc --quiet) > "$initramfs" + + if [ -n "$bconf_file" ] && [ -f "$bconf_file" ]; then + "$BOOTCONFIG" -a "$bconf_file" "$initramfs" > /dev/null + fi + + + local logfile + local base_cmdline="bootconfig console=tty0 console=ttyS0 panic=-1" + if [ -n "$test_cmdline" ]; then + base_cmdline="$base_cmdline $test_cmdline" + fi + + if [ -n "$LOGDIR" ]; then + mkdir -p "$LOGDIR" + logfile="$LOGDIR/$name.log" + base_cmdline="$base_cmdline dump_bconf" + else + logfile="$workdir/qemu.log" + base_cmdline="quiet $base_cmdline" + fi + + local qemu_args=() + if [ -n "$test_qemuopts" ]; then + read -r -d '' -a qemu_args <<< "$test_qemuopts" || true + fi + + if [ "$is_reboot" -ne 1 ]; then + qemu_args+=("-no-reboot") + fi + + timeout "$effective_timeout" "$QEMU" -kernel "$KERNEL" \ + -initrd "$initramfs" \ + -append "$base_cmdline" \ + -display none \ + -serial stdio \ + "${qemu_args[@]}" < /dev/null > "$logfile" 2>&1 || true + + if grep -q "TEST RESULT: PASS" "$logfile"; then + ktap_test_pass "$name" + else + ktap_test_fail "$name" + ktap_print_msg "QEMU Console Output for $name:" + while IFS= read -r line; do + ktap_print_msg "$line" + done < "$logfile" + fi + + rm -rf "$workdir" + trap - EXIT + return 0 +} + +TEST_SCRIPTS=() +if [ -d "$SCRIPT_DIR/tests" ]; then + while IFS= read -r f; do + [ -n "$f" ] && TEST_SCRIPTS+=("$f") + done < <(find "$SCRIPT_DIR/tests" -type f -name "*.sh" | sort) +fi + +TOTAL_TESTS=0 +for test_script in "${TEST_SCRIPTS[@]}"; do + name="$(basename "$test_script" .sh)" + if [ "$TARGET_TEST" != "all" ] && [ "$TARGET_TEST" != "$name" ]; then + continue + fi + TOTAL_TESTS=$((TOTAL_TESTS + 1)) +done + +ktap_print_header +ktap_set_plan "$TOTAL_TESTS" + +for test_script in "${TEST_SCRIPTS[@]}"; do + name="$(basename "$test_script" .sh)" + + if [ "$TARGET_TEST" != "all" ] && [ "$TARGET_TEST" != "$name" ]; then + continue + fi + + run_single_test "$test_script" +done + +ktap_finished diff --git a/tools/testing/selftests/ftrace/config b/tools/testing/selftests/ftrace/config index 544de0db5f58f1..29f05f7c6e2ce2 100644 --- a/tools/testing/selftests/ftrace/config +++ b/tools/testing/selftests/ftrace/config @@ -1,3 +1,5 @@ +CONFIG_BOOT_CONFIG=y +CONFIG_BOOTTIME_TRACING=y CONFIG_BPF_SYSCALL=y CONFIG_DEBUG_INFO_BTF=y CONFIG_DEBUG_INFO_DWARF4=y @@ -8,22 +10,26 @@ CONFIG_FTRACE=y CONFIG_FTRACE_SYSCALLS=y CONFIG_FUNCTION_GRAPH_RETVAL=y CONFIG_FUNCTION_PROFILER=y +CONFIG_FUNCTION_TRACER=y CONFIG_HIST_TRIGGERS=y CONFIG_IRQSOFF_TRACER=y CONFIG_KALLSYMS_ALL=y CONFIG_KPROBES=y CONFIG_KPROBE_EVENTS=y +CONFIG_MAGIC_SYSRQ=y CONFIG_MODULES=y CONFIG_MODULE_UNLOAD=y CONFIG_PREEMPTIRQ_DELAY_TEST=m CONFIG_PREEMPT_TRACER=y CONFIG_PROBE_EVENTS_BTF_ARGS=y +CONFIG_RESERVE_MEM=y CONFIG_SAMPLES=y CONFIG_SAMPLE_FTRACE_DIRECT=m CONFIG_SAMPLE_TRACE_EVENTS=m CONFIG_SAMPLE_TRACE_PRINTK=m CONFIG_SCHED_TRACER=y CONFIG_STACK_TRACER=y +CONFIG_SYNTH_EVENTS=y CONFIG_TRACER_SNAPSHOT=y CONFIG_UPROBES=y CONFIG_UPROBE_EVENTS=y From e82a45df4095d2b23ca84c8dfa9e7887436554c4 Mon Sep 17 00:00:00 2001 From: "Masami Hiramatsu (Google)" Date: Wed, 30 Sep 2026 00:09:41 +0900 Subject: [PATCH 0369/1352] selftests/ftrace: Add boot-time tracing testcases Add test cases for boot-time tracing configuration via bootconfig (see Documentation/trace/boottime-trace.rst). These test cases verify: - 01-kprobe: Dynamic kprobe events via bootconfig - 02-synth: Synthetic events via bootconfig - 03-eprobe: Event probes (eprobes) via bootconfig - 04-fprobe: Function probes (fprobes) via bootconfig - 05-tprobe: Tracepoint probes via bootconfig - 06-instance: Custom trace instances via bootconfig Link: https://lore.kernel.org/all/178649542809.438282.16167291066989252262.stgit@devnote2/ Assisted-by: Antigravity:gemini-3.6-flash Signed-off-by: Masami Hiramatsu (Google) --- .../boottime/bootconfigs/01-kprobe.bconf | 4 +++ .../boottime/bootconfigs/02-synth.bconf | 4 +++ .../boottime/bootconfigs/03-eprobe.bconf | 4 +++ .../boottime/bootconfigs/04-fprobe.bconf | 4 +++ .../boottime/bootconfigs/05-tprobe.bconf | 4 +++ .../boottime/bootconfigs/06-instance.bconf | 5 ++++ .../ftrace/boottime/tests/01-kprobe.sh | 25 ++++++++++++++++ .../ftrace/boottime/tests/02-synth.sh | 25 ++++++++++++++++ .../ftrace/boottime/tests/03-eprobe.sh | 25 ++++++++++++++++ .../ftrace/boottime/tests/04-fprobe.sh | 25 ++++++++++++++++ .../ftrace/boottime/tests/05-tprobe.sh | 25 ++++++++++++++++ .../ftrace/boottime/tests/06-instance.sh | 30 +++++++++++++++++++ 12 files changed, 180 insertions(+) create mode 100644 tools/testing/selftests/ftrace/boottime/bootconfigs/01-kprobe.bconf create mode 100644 tools/testing/selftests/ftrace/boottime/bootconfigs/02-synth.bconf create mode 100644 tools/testing/selftests/ftrace/boottime/bootconfigs/03-eprobe.bconf create mode 100644 tools/testing/selftests/ftrace/boottime/bootconfigs/04-fprobe.bconf create mode 100644 tools/testing/selftests/ftrace/boottime/bootconfigs/05-tprobe.bconf create mode 100644 tools/testing/selftests/ftrace/boottime/bootconfigs/06-instance.bconf create mode 100644 tools/testing/selftests/ftrace/boottime/tests/01-kprobe.sh create mode 100644 tools/testing/selftests/ftrace/boottime/tests/02-synth.sh create mode 100644 tools/testing/selftests/ftrace/boottime/tests/03-eprobe.sh create mode 100644 tools/testing/selftests/ftrace/boottime/tests/04-fprobe.sh create mode 100644 tools/testing/selftests/ftrace/boottime/tests/05-tprobe.sh create mode 100644 tools/testing/selftests/ftrace/boottime/tests/06-instance.sh diff --git a/tools/testing/selftests/ftrace/boottime/bootconfigs/01-kprobe.bconf b/tools/testing/selftests/ftrace/boottime/bootconfigs/01-kprobe.bconf new file mode 100644 index 00000000000000..13d30d1e51a639 --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/bootconfigs/01-kprobe.bconf @@ -0,0 +1,4 @@ +ftrace.event.kprobes.vfs_read { + probes = "vfs_read $arg1" + enable +} diff --git a/tools/testing/selftests/ftrace/boottime/bootconfigs/02-synth.bconf b/tools/testing/selftests/ftrace/boottime/bootconfigs/02-synth.bconf new file mode 100644 index 00000000000000..b1f37713f33cd3 --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/bootconfigs/02-synth.bconf @@ -0,0 +1,4 @@ +ftrace.event.synthetic.boot_lat { + fields = "unsigned long id", "u64 delta" + enable +} diff --git a/tools/testing/selftests/ftrace/boottime/bootconfigs/03-eprobe.bconf b/tools/testing/selftests/ftrace/boottime/bootconfigs/03-eprobe.bconf new file mode 100644 index 00000000000000..c1a8138e605db8 --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/bootconfigs/03-eprobe.bconf @@ -0,0 +1,4 @@ +ftrace.event.eprobes.ep_read { + probes = "sched/sched_switch" + enable +} diff --git a/tools/testing/selftests/ftrace/boottime/bootconfigs/04-fprobe.bconf b/tools/testing/selftests/ftrace/boottime/bootconfigs/04-fprobe.bconf new file mode 100644 index 00000000000000..82ae2ac0629f5d --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/bootconfigs/04-fprobe.bconf @@ -0,0 +1,4 @@ +ftrace.event.fprobes.fp_read { + probes = "vfs_read" + enable +} diff --git a/tools/testing/selftests/ftrace/boottime/bootconfigs/05-tprobe.bconf b/tools/testing/selftests/ftrace/boottime/bootconfigs/05-tprobe.bconf new file mode 100644 index 00000000000000..02a2803c770a3d --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/bootconfigs/05-tprobe.bconf @@ -0,0 +1,4 @@ +ftrace.event.tracepoints.tp_sched { + probes = "sched_switch" + enable +} diff --git a/tools/testing/selftests/ftrace/boottime/bootconfigs/06-instance.bconf b/tools/testing/selftests/ftrace/boottime/bootconfigs/06-instance.bconf new file mode 100644 index 00000000000000..3fdb6717afe201 --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/bootconfigs/06-instance.bconf @@ -0,0 +1,5 @@ +ftrace.instance.foo { + event.sched.sched_switch { + enable + } +} diff --git a/tools/testing/selftests/ftrace/boottime/tests/01-kprobe.sh b/tools/testing/selftests/ftrace/boottime/tests/01-kprobe.sh new file mode 100644 index 00000000000000..6ed7caca4c5f44 --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/tests/01-kprobe.sh @@ -0,0 +1,25 @@ +#!/bin/sh +# SPDX-License-Identifier: GPL-2.0 +# Copyright (C) 2026, Google LLC. +# Check kprobe bootconfig settings on tracefs +TRACEDIR="/sys/kernel/tracing" + +if [ -f /proc/bootconfig ] && grep -q "dump_bconf" /proc/cmdline 2>/dev/null; then + echo "=== /proc/bootconfig ===" + cat /proc/bootconfig + echo "========================" +fi + +if [ ! -d "$TRACEDIR/events/kprobes/vfs_read" ]; then + echo "FAIL: kprobe event kprobes/vfs_read does not exist" + exit 1 +fi + +ENABLE=$(cat "$TRACEDIR/events/kprobes/vfs_read/enable") +if [ "$ENABLE" != "1" ]; then + echo "FAIL: kprobe event kprobes/vfs_read is not enabled ($ENABLE)" + exit 1 +fi + +echo "PASS: 01-kprobe" +exit 0 diff --git a/tools/testing/selftests/ftrace/boottime/tests/02-synth.sh b/tools/testing/selftests/ftrace/boottime/tests/02-synth.sh new file mode 100644 index 00000000000000..745c328dde158d --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/tests/02-synth.sh @@ -0,0 +1,25 @@ +#!/bin/sh +# SPDX-License-Identifier: GPL-2.0 +# Copyright (C) 2026, Google LLC. +# Check synthetic event bootconfig settings on tracefs +TRACEDIR="/sys/kernel/tracing" + +if [ -f /proc/bootconfig ] && grep -q "dump_bconf" /proc/cmdline 2>/dev/null; then + echo "=== /proc/bootconfig ===" + cat /proc/bootconfig + echo "========================" +fi + +if [ ! -d "$TRACEDIR/events/synthetic/boot_lat" ]; then + echo "FAIL: synthetic event synthetic/boot_lat does not exist" + exit 1 +fi + +ENABLE=$(cat "$TRACEDIR/events/synthetic/boot_lat/enable") +if [ "$ENABLE" != "1" ]; then + echo "FAIL: synthetic event synthetic/boot_lat is not enabled ($ENABLE)" + exit 1 +fi + +echo "PASS: 02-synth" +exit 0 diff --git a/tools/testing/selftests/ftrace/boottime/tests/03-eprobe.sh b/tools/testing/selftests/ftrace/boottime/tests/03-eprobe.sh new file mode 100644 index 00000000000000..9b3b7ab23c69e0 --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/tests/03-eprobe.sh @@ -0,0 +1,25 @@ +#!/bin/sh +# SPDX-License-Identifier: GPL-2.0 +# Copyright (C) 2026, Google LLC. +# Check eprobe event bootconfig settings on tracefs +TRACEDIR="/sys/kernel/tracing" + +if [ -f /proc/bootconfig ] && grep -q "dump_bconf" /proc/cmdline 2>/dev/null; then + echo "=== /proc/bootconfig ===" + cat /proc/bootconfig + echo "========================" +fi + +if [ ! -d "$TRACEDIR/events/eprobes/ep_read" ]; then + echo "FAIL: eprobe event eprobes/ep_read does not exist" + exit 1 +fi + +ENABLE=$(cat "$TRACEDIR/events/eprobes/ep_read/enable") +if [ "$ENABLE" != "1" ]; then + echo "FAIL: eprobe event eprobes/ep_read is not enabled ($ENABLE)" + exit 1 +fi + +echo "PASS: 03-eprobe" +exit 0 diff --git a/tools/testing/selftests/ftrace/boottime/tests/04-fprobe.sh b/tools/testing/selftests/ftrace/boottime/tests/04-fprobe.sh new file mode 100644 index 00000000000000..27e090c28a4be6 --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/tests/04-fprobe.sh @@ -0,0 +1,25 @@ +#!/bin/sh +# SPDX-License-Identifier: GPL-2.0 +# Copyright (C) 2026, Google LLC. +# Check fprobe event bootconfig settings on tracefs +TRACEDIR="/sys/kernel/tracing" + +if [ -f /proc/bootconfig ] && grep -q "dump_bconf" /proc/cmdline 2>/dev/null; then + echo "=== /proc/bootconfig ===" + cat /proc/bootconfig + echo "========================" +fi + +if [ ! -d "$TRACEDIR/events/fprobes/fp_read" ]; then + echo "FAIL: fprobe event fprobes/fp_read does not exist" + exit 1 +fi + +ENABLE=$(cat "$TRACEDIR/events/fprobes/fp_read/enable") +if [ "$ENABLE" != "1" ]; then + echo "FAIL: fprobe event fprobes/fp_read is not enabled ($ENABLE)" + exit 1 +fi + +echo "PASS: 04-fprobe" +exit 0 diff --git a/tools/testing/selftests/ftrace/boottime/tests/05-tprobe.sh b/tools/testing/selftests/ftrace/boottime/tests/05-tprobe.sh new file mode 100644 index 00000000000000..810d34055f947d --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/tests/05-tprobe.sh @@ -0,0 +1,25 @@ +#!/bin/sh +# SPDX-License-Identifier: GPL-2.0 +# Copyright (C) 2026, Google LLC. +# Check tracepoint probe event bootconfig settings on tracefs +TRACEDIR="/sys/kernel/tracing" + +if [ -f /proc/bootconfig ] && grep -q "dump_bconf" /proc/cmdline 2>/dev/null; then + echo "=== /proc/bootconfig ===" + cat /proc/bootconfig + echo "========================" +fi + +if [ ! -d "$TRACEDIR/events/tracepoints/tp_sched" ]; then + echo "FAIL: tracepoint probe event tracepoints/tp_sched does not exist" + exit 1 +fi + +ENABLE=$(cat "$TRACEDIR/events/tracepoints/tp_sched/enable") +if [ "$ENABLE" != "1" ]; then + echo "FAIL: tracepoint probe event tracepoints/tp_sched is not enabled ($ENABLE)" + exit 1 +fi + +echo "PASS: 05-tprobe" +exit 0 diff --git a/tools/testing/selftests/ftrace/boottime/tests/06-instance.sh b/tools/testing/selftests/ftrace/boottime/tests/06-instance.sh new file mode 100644 index 00000000000000..f6e8519912e4e4 --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/tests/06-instance.sh @@ -0,0 +1,30 @@ +#!/bin/sh +# SPDX-License-Identifier: GPL-2.0 +# Copyright (C) 2026, Google LLC. +# Check instance bootconfig settings on tracefs +TRACEDIR="/sys/kernel/tracing" + +if [ -f /proc/bootconfig ] && grep -q "dump_bconf" /proc/cmdline 2>/dev/null; then + echo "=== /proc/bootconfig ===" + cat /proc/bootconfig + echo "========================" +fi + +if [ ! -d "$TRACEDIR/instances/foo" ]; then + echo "FAIL: trace instance foo does not exist" + exit 1 +fi + +if [ ! -d "$TRACEDIR/instances/foo/events/sched/sched_switch" ]; then + echo "FAIL: event sched_switch does not exist in instance foo" + exit 1 +fi + +ENABLE=$(cat "$TRACEDIR/instances/foo/events/sched/sched_switch/enable") +if [ "$ENABLE" != "1" ]; then + echo "FAIL: event sched_switch is not enabled in instance foo ($ENABLE)" + exit 1 +fi + +echo "PASS: 06-instance" +exit 0 From 417ec2ce6b3b9a47390614ba696ae3e92d77513d Mon Sep 17 00:00:00 2001 From: "Masami Hiramatsu (Google)" Date: Wed, 30 Sep 2026 00:09:41 +0900 Subject: [PATCH 0370/1352] selftests/ftrace: Add kernel cmdline tracing testcases Add test cases for kernel command-line tracing options. It verifies various parameters like trace_buf_size and trace_options by reading the output of tracing files in sysfs. Link: https://lore.kernel.org/all/178649543741.438282.2638887907664217934.stgit@devnote2/ Assisted-by: Antigravity:gemini-3.6-flash Signed-off-by: Masami Hiramatsu (Google) --- .../cmdlines/cmdline-01-ftrace.cmdline | 1 + .../cmdlines/cmdline-02-trace-event.cmdline | 1 + .../cmdline-03-trace-buf-size.cmdline | 1 + .../cmdlines/cmdline-04-trace-options.cmdline | 1 + .../cmdlines/cmdline-05-trace-clock.cmdline | 1 + .../cmdline-06-trace-instance.cmdline | 1 + .../boottime/tests/cmdline-01-ftrace.sh | 19 ++++++++++++ .../boottime/tests/cmdline-02-trace-event.sh | 26 +++++++++++++++++ .../tests/cmdline-03-trace-buf-size.sh | 29 +++++++++++++++++++ .../tests/cmdline-04-trace-options.sh | 23 +++++++++++++++ .../boottime/tests/cmdline-05-trace-clock.sh | 19 ++++++++++++ .../tests/cmdline-06-trace-instance.sh | 24 +++++++++++++++ 12 files changed, 146 insertions(+) create mode 100644 tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-01-ftrace.cmdline create mode 100644 tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-02-trace-event.cmdline create mode 100644 tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-03-trace-buf-size.cmdline create mode 100644 tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-04-trace-options.cmdline create mode 100644 tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-05-trace-clock.cmdline create mode 100644 tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-06-trace-instance.cmdline create mode 100644 tools/testing/selftests/ftrace/boottime/tests/cmdline-01-ftrace.sh create mode 100644 tools/testing/selftests/ftrace/boottime/tests/cmdline-02-trace-event.sh create mode 100644 tools/testing/selftests/ftrace/boottime/tests/cmdline-03-trace-buf-size.sh create mode 100644 tools/testing/selftests/ftrace/boottime/tests/cmdline-04-trace-options.sh create mode 100644 tools/testing/selftests/ftrace/boottime/tests/cmdline-05-trace-clock.sh create mode 100644 tools/testing/selftests/ftrace/boottime/tests/cmdline-06-trace-instance.sh diff --git a/tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-01-ftrace.cmdline b/tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-01-ftrace.cmdline new file mode 100644 index 00000000000000..4f6fc54240c8a8 --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-01-ftrace.cmdline @@ -0,0 +1 @@ +ftrace=function diff --git a/tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-02-trace-event.cmdline b/tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-02-trace-event.cmdline new file mode 100644 index 00000000000000..66bcd29b14cb15 --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-02-trace-event.cmdline @@ -0,0 +1 @@ +trace_event=sched:sched_switch,kmem:kmalloc diff --git a/tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-03-trace-buf-size.cmdline b/tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-03-trace-buf-size.cmdline new file mode 100644 index 00000000000000..e5d1c4a8ca895a --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-03-trace-buf-size.cmdline @@ -0,0 +1 @@ +trace_buf_size=2048K diff --git a/tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-04-trace-options.cmdline b/tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-04-trace-options.cmdline new file mode 100644 index 00000000000000..51d82f8e948d2d --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-04-trace-options.cmdline @@ -0,0 +1 @@ +trace_options=sym-addr,verbose diff --git a/tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-05-trace-clock.cmdline b/tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-05-trace-clock.cmdline new file mode 100644 index 00000000000000..f72a4ac808a35a --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-05-trace-clock.cmdline @@ -0,0 +1 @@ +trace_clock=global diff --git a/tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-06-trace-instance.cmdline b/tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-06-trace-instance.cmdline new file mode 100644 index 00000000000000..9618e3e2b88bfb --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-06-trace-instance.cmdline @@ -0,0 +1 @@ +trace_instance=bar,sched:sched_switch diff --git a/tools/testing/selftests/ftrace/boottime/tests/cmdline-01-ftrace.sh b/tools/testing/selftests/ftrace/boottime/tests/cmdline-01-ftrace.sh new file mode 100644 index 00000000000000..23a0b15f2dd3de --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/tests/cmdline-01-ftrace.sh @@ -0,0 +1,19 @@ +#!/bin/sh +# SPDX-License-Identifier: GPL-2.0 +# Copyright (C) 2026, Google LLC. +# Check ftrace= kernel command-line tracer setting +TRACEDIR="/sys/kernel/tracing" + +if [ ! -f "$TRACEDIR/current_tracer" ]; then + echo "FAIL: current_tracer does not exist" + exit 1 +fi + +read -r TRACER _ < "$TRACEDIR/current_tracer" +if [ "$TRACER" != "function" ]; then + echo "FAIL: current_tracer is '$TRACER', expected 'function'" + exit 1 +fi + +echo "PASS: cmdline-01-ftrace" +exit 0 diff --git a/tools/testing/selftests/ftrace/boottime/tests/cmdline-02-trace-event.sh b/tools/testing/selftests/ftrace/boottime/tests/cmdline-02-trace-event.sh new file mode 100644 index 00000000000000..b793907c48cd21 --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/tests/cmdline-02-trace-event.sh @@ -0,0 +1,26 @@ +#!/bin/sh +# SPDX-License-Identifier: GPL-2.0 +# Copyright (C) 2026, Google LLC. +# Check trace_event= kernel command-line setting +TRACEDIR="/sys/kernel/tracing" + +if [ ! -d "$TRACEDIR/events/sched/sched_switch" ]; then + echo "FAIL: event sched:sched_switch does not exist" + exit 1 +fi + +if [ ! -d "$TRACEDIR/events/kmem/kmalloc" ]; then + echo "FAIL: event kmem:kmalloc does not exist" + exit 1 +fi + +ENABLE1=$(cat "$TRACEDIR/events/sched/sched_switch/enable") +ENABLE2=$(cat "$TRACEDIR/events/kmem/kmalloc/enable") + +if [ "$ENABLE1" != "1" ] || [ "$ENABLE2" != "1" ]; then + echo "FAIL: events not enabled (sched_switch=$ENABLE1, kmalloc=$ENABLE2)" + exit 1 +fi + +echo "PASS: cmdline-02-trace-event" +exit 0 diff --git a/tools/testing/selftests/ftrace/boottime/tests/cmdline-03-trace-buf-size.sh b/tools/testing/selftests/ftrace/boottime/tests/cmdline-03-trace-buf-size.sh new file mode 100644 index 00000000000000..4b936f84d4d72b --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/tests/cmdline-03-trace-buf-size.sh @@ -0,0 +1,29 @@ +#!/bin/sh +# SPDX-License-Identifier: GPL-2.0 +# Copyright (C) 2026, Google LLC. +# APPLETS: awk +# Check trace_buf_size= kernel command-line setting +TRACEDIR="/sys/kernel/tracing" + +if [ ! -f "$TRACEDIR/buffer_size_kb" ]; then + echo "FAIL: buffer_size_kb does not exist" + exit 1 +fi + +BUF_RAW=$(cat "$TRACEDIR/buffer_size_kb") +case "$BUF_RAW" in + *"expanded:"*) + BUFSIZE=$(echo "$BUF_RAW" | sed -n 's/.*expanded: *\([0-9]*\).*/\1/p') + ;; + *) + BUFSIZE=$(echo "$BUF_RAW" | awk '{print $1}') + ;; +esac + +if [ -z "$BUFSIZE" ] || [ "$BUFSIZE" -lt 2048 ]; then + echo "FAIL: buffer_size_kb is '$BUF_RAW', expected >= 2048" + exit 1 +fi + +echo "PASS: cmdline-03-trace-buf-size" +exit 0 diff --git a/tools/testing/selftests/ftrace/boottime/tests/cmdline-04-trace-options.sh b/tools/testing/selftests/ftrace/boottime/tests/cmdline-04-trace-options.sh new file mode 100644 index 00000000000000..53a65b55f7a70b --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/tests/cmdline-04-trace-options.sh @@ -0,0 +1,23 @@ +#!/bin/sh +# SPDX-License-Identifier: GPL-2.0 +# Copyright (C) 2026, Google LLC. +# Check trace_options= kernel command-line setting +TRACEDIR="/sys/kernel/tracing" + +if [ ! -f "$TRACEDIR/trace_options" ]; then + echo "FAIL: trace_options file does not exist" + exit 1 +fi + +if ! grep -qw "sym-addr" "$TRACEDIR/trace_options"; then + echo "FAIL: sym-addr option is not set in trace_options" + exit 1 +fi + +if ! grep -qw "verbose" "$TRACEDIR/trace_options"; then + echo "FAIL: verbose option is not set in trace_options" + exit 1 +fi + +echo "PASS: cmdline-04-trace-options" +exit 0 diff --git a/tools/testing/selftests/ftrace/boottime/tests/cmdline-05-trace-clock.sh b/tools/testing/selftests/ftrace/boottime/tests/cmdline-05-trace-clock.sh new file mode 100644 index 00000000000000..d6de9f6e2090f2 --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/tests/cmdline-05-trace-clock.sh @@ -0,0 +1,19 @@ +#!/bin/sh +# SPDX-License-Identifier: GPL-2.0 +# Copyright (C) 2026, Google LLC. +# Check trace_clock= kernel command-line setting +TRACEDIR="/sys/kernel/tracing" + +if [ ! -f "$TRACEDIR/trace_clock" ]; then + echo "FAIL: trace_clock file does not exist" + exit 1 +fi + +if ! grep -q '\[global\]' "$TRACEDIR/trace_clock"; then + CLOCK=$(cat "$TRACEDIR/trace_clock") + echo "FAIL: trace_clock is not set to global ($CLOCK)" + exit 1 +fi + +echo "PASS: cmdline-05-trace-clock" +exit 0 diff --git a/tools/testing/selftests/ftrace/boottime/tests/cmdline-06-trace-instance.sh b/tools/testing/selftests/ftrace/boottime/tests/cmdline-06-trace-instance.sh new file mode 100644 index 00000000000000..aae7f0a86d81ea --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/tests/cmdline-06-trace-instance.sh @@ -0,0 +1,24 @@ +#!/bin/sh +# SPDX-License-Identifier: GPL-2.0 +# Copyright (C) 2026, Google LLC. +# Check trace_instance= kernel command-line setting +TRACEDIR="/sys/kernel/tracing" + +if [ ! -d "$TRACEDIR/instances/bar" ]; then + echo "FAIL: trace instance bar does not exist" + exit 1 +fi + +if [ ! -d "$TRACEDIR/instances/bar/events/sched/sched_switch" ]; then + echo "FAIL: event sched_switch does not exist in instance bar" + exit 1 +fi + +ENABLE=$(cat "$TRACEDIR/instances/bar/events/sched/sched_switch/enable") +if [ "$ENABLE" != "1" ]; then + echo "FAIL: event sched_switch is not enabled in instance bar ($ENABLE)" + exit 1 +fi + +echo "PASS: cmdline-06-trace-instance" +exit 0 From 393b1c52b0c7f57b20f08bc043c3cd4ade26944f Mon Sep 17 00:00:00 2001 From: "Masami Hiramatsu (Google)" Date: Wed, 30 Sep 2026 00:09:41 +0900 Subject: [PATCH 0371/1352] selftests/ftrace: Add persistent ring buffer testcases Add test cases for persistent ring buffer (reserve_mem= combined with trace_instance=) and backup instance (trace_instance=backup=boot_map). These test cases verify: - persistent-01-reserve-mem: Memory reservation via reserve_mem= and trace_instance=boot_map@trace. Verifies trace data retention in the boot_map instance across guest crash/reboot via sysrq-trigger. - persistent-02-backup-instance: Backup instance creation via trace_instance=backup=boot_map. Verifies that previous boot trace log is preserved in the backup instance on the subsequent boot. These test cases specify '# REBOOT: 1' to enable multi-boot guest crash/reboot testing with doubled QEMU timeout. Link: https://lore.kernel.org/all/178649544683.438282.14573895196260214231.stgit@devnote2/ Assisted-by: Antigravity:gemini-3.6-flash Signed-off-by: Masami Hiramatsu (Google) --- .../persistent-01-reserve-mem.cmdline | 1 + .../persistent-02-backup-instance.cmdline | 1 + .../tests/persistent-01-reserve-mem.sh | 28 +++++++++++++ .../tests/persistent-02-backup-instance.sh | 40 +++++++++++++++++++ 4 files changed, 70 insertions(+) create mode 100644 tools/testing/selftests/ftrace/boottime/cmdlines/persistent-01-reserve-mem.cmdline create mode 100644 tools/testing/selftests/ftrace/boottime/cmdlines/persistent-02-backup-instance.cmdline create mode 100644 tools/testing/selftests/ftrace/boottime/tests/persistent-01-reserve-mem.sh create mode 100644 tools/testing/selftests/ftrace/boottime/tests/persistent-02-backup-instance.sh diff --git a/tools/testing/selftests/ftrace/boottime/cmdlines/persistent-01-reserve-mem.cmdline b/tools/testing/selftests/ftrace/boottime/cmdlines/persistent-01-reserve-mem.cmdline new file mode 100644 index 00000000000000..c69eeb27deb6fe --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/cmdlines/persistent-01-reserve-mem.cmdline @@ -0,0 +1 @@ +reserve_mem=12M:32M:trace trace_instance=boot_map@trace diff --git a/tools/testing/selftests/ftrace/boottime/cmdlines/persistent-02-backup-instance.cmdline b/tools/testing/selftests/ftrace/boottime/cmdlines/persistent-02-backup-instance.cmdline new file mode 100644 index 00000000000000..43217c986ca1af --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/cmdlines/persistent-02-backup-instance.cmdline @@ -0,0 +1 @@ +reserve_mem=12M:32M:trace trace_instance=boot_map@trace trace_instance=backup=boot_map diff --git a/tools/testing/selftests/ftrace/boottime/tests/persistent-01-reserve-mem.sh b/tools/testing/selftests/ftrace/boottime/tests/persistent-01-reserve-mem.sh new file mode 100644 index 00000000000000..49abaf518481f4 --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/tests/persistent-01-reserve-mem.sh @@ -0,0 +1,28 @@ +#!/bin/sh +# SPDX-License-Identifier: GPL-2.0 +# Copyright (C) 2026, Google LLC. +# Check persistent ring buffer across guest crash/reboot +# REBOOT: 1 +TRACEDIR="/sys/kernel/tracing" + +if [ ! -f "$TRACEDIR/instances/boot_map/trace" ]; then + echo "FAIL: persistent trace instance boot_map/trace file missing" + exit 1 +fi + +# Check if Boot 1 marker was already written +if grep -q "BOOT1_MARKER" "$TRACEDIR/instances/boot_map/trace" 2>/dev/null; then + # Second boot: verify persistent ring buffer content from first boot + echo "PASS: persistent-01-reserve-mem" + exit 0 +fi + +# First boot: write marker to persistent buffer and trigger kernel crash/reboot +echo "BOOT1_MARKER" > "$TRACEDIR/instances/boot_map/trace_marker" +sync + +# Trigger reboot to restart into second boot +echo b > /proc/sysrq-trigger 2>/dev/null || echo c > /proc/sysrq-trigger 2>/dev/null || true +sleep 5 +echo "FAIL: reboot trigger failed on first boot" +exit 1 diff --git a/tools/testing/selftests/ftrace/boottime/tests/persistent-02-backup-instance.sh b/tools/testing/selftests/ftrace/boottime/tests/persistent-02-backup-instance.sh new file mode 100644 index 00000000000000..0b2a2f03ce4f2a --- /dev/null +++ b/tools/testing/selftests/ftrace/boottime/tests/persistent-02-backup-instance.sh @@ -0,0 +1,40 @@ +#!/bin/sh +# SPDX-License-Identifier: GPL-2.0 +# Copyright (C) 2026, Google LLC. +# Check persistent backup instance across guest crash/reboot +# REBOOT: 1 +TRACEDIR="/sys/kernel/tracing" + +# Ensure both boot_map/trace and backup/trace files exist +if [ ! -f "$TRACEDIR/instances/boot_map/trace" ]; then + echo "FAIL: boot_map/trace file missing" + exit 1 +fi + +if [ ! -f "$TRACEDIR/instances/backup/trace" ]; then + echo "FAIL: backup/trace file missing" + exit 1 +fi + +# Check if BOOT1_MARKER is in boot_map/trace (indicates second boot) +if grep -q "BOOT1_MARKER" "$TRACEDIR/instances/boot_map/trace" 2>/dev/null; then + # Second boot: verify BOOT1_MARKER was copied into backup/trace from Boot 1 + if grep -q "BOOT1_MARKER" "$TRACEDIR/instances/backup/trace" 2>/dev/null; then + echo "PASS: persistent-02-backup-instance" + exit 0 + else + echo "FAIL: BOOT1_MARKER found in boot_map/trace" \ + "but missing from backup/trace on second boot" + exit 1 + fi +fi + +# First boot: write BOOT1_MARKER to boot_map and reboot via sysrq-trigger +echo "BOOT1_MARKER" > "$TRACEDIR/instances/boot_map/trace_marker" +sync + +# Trigger reboot to restart into second boot +echo b > /proc/sysrq-trigger 2>/dev/null || echo c > /proc/sysrq-trigger 2>/dev/null || true +sleep 5 +echo "FAIL: reboot trigger failed on first boot" +exit 1 From cba6d63174fd9f2f08c96d38a304509f4466c0c3 Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Wed, 30 Sep 2026 00:09:42 +0900 Subject: [PATCH 0372/1352] kprobes: use DEFINE_DEBUGFS_ATTRIBUTE for the enabled knob This enabled knob is a disgusting terrible hack. It has rolled its own read/write pair since 2007, writing '1' or '0' into a three byte buffer by hand just to print a single character, with an XXX comment begging debugfs for write callbacks on bool files. DEFINE_DEBUGFS_ATTRIBUTE showed up in 2016 and does exactly that, so the disgusting terrible hack has outlived its excuse for nine years. Kill it, and the stale comment with it. The behavior does not change, except the write only accepts 0/1 now instead of y/n/on/off, and nothing uses anything else. Link: https://lore.kernel.org/all/20260818015318.24103-1-include@grrlz.net/ Signed-off-by: Bradley Morgan Signed-off-by: Masami Hiramatsu (Google) --- kernel/kprobes.c | 43 +++++++++---------------------------------- 1 file changed, 9 insertions(+), 34 deletions(-) diff --git a/kernel/kprobes.c b/kernel/kprobes.c index 4edd8ca5c65782..7b6baa4c6fdaf3 100644 --- a/kernel/kprobes.c +++ b/kernel/kprobes.c @@ -3025,47 +3025,22 @@ static int disarm_all_kprobes(void) return ret; } -/* - * XXX: The debugfs bool file interface doesn't allow for callbacks - * when the bool state is switched. We can reuse that facility when - * available - */ -static ssize_t read_enabled_file_bool(struct file *file, - char __user *user_buf, size_t count, loff_t *ppos) +static int kprobes_enabled_set(void *data, u64 val) { - char buf[3]; + if (val) + return arm_all_kprobes(); - if (!kprobes_all_disarmed) - buf[0] = '1'; - else - buf[0] = '0'; - buf[1] = '\n'; - buf[2] = 0x00; - return simple_read_from_buffer(user_buf, count, ppos, buf, 2); + return disarm_all_kprobes(); } -static ssize_t write_enabled_file_bool(struct file *file, - const char __user *user_buf, size_t count, loff_t *ppos) +static int kprobes_enabled_get(void *data, u64 *val) { - bool enable; - int ret; - - ret = kstrtobool_from_user(user_buf, count, &enable); - if (ret) - return ret; - - ret = enable ? arm_all_kprobes() : disarm_all_kprobes(); - if (ret) - return ret; - - return count; + *val = !kprobes_all_disarmed; + return 0; } -static const struct file_operations fops_kp = { - .read = read_enabled_file_bool, - .write = write_enabled_file_bool, - .llseek = default_llseek, -}; +DEFINE_DEBUGFS_ATTRIBUTE(fops_kp, kprobes_enabled_get, + kprobes_enabled_set, "%llu\n"); static int __init debugfs_kprobe_init(void) { From b15df7805dc52555e7c867d33a496edf5be11289 Mon Sep 17 00:00:00 2001 From: Kees Cook Date: Wed, 30 Sep 2026 00:09:42 +0900 Subject: [PATCH 0373/1352] tracing/probes: Add const to new_argv allocation type In preparation for converting the kmalloc family of allocators to the type-aware kmalloc_obj family, we need to make sure that the returned type from the allocation matches the type of the variable being assigned. (The kmalloc family returns "void *", which can be implicitly cast to any pointer type.) The assigned type is "const char **", but the converted allocation type would be "char **", which is the same type without the const qualifier. As there is no general way to safely add const qualifiers, take the size from the assignment target instead. No change in allocation size results. Build tested ARCH=x86_64 allmodconfig with GCC 16.2.0: kernel/trace/trace_probe.o Link: https://lore.kernel.org/all/20260917211022.i.677-kees@kernel.org/ Assisted-by: LLM coccinelle Signed-off-by: Kees Cook Signed-off-by: Masami Hiramatsu (Google) --- kernel/trace/trace_probe.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/kernel/trace/trace_probe.c b/kernel/trace/trace_probe.c index 804442b2f7d2bd..cfb710095f6502 100644 --- a/kernel/trace/trace_probe.c +++ b/kernel/trace/trace_probe.c @@ -2316,7 +2316,7 @@ const char **traceprobe_expand_meta_args(int argc, const char *argv[], else *new_argc = argc; - new_argv = kcalloc(*new_argc, sizeof(char *), GFP_KERNEL); + new_argv = kcalloc(*new_argc, sizeof(*new_argv), GFP_KERNEL); if (!new_argv) return ERR_PTR(-ENOMEM); From a84c1f08a793ae58c3936d0b63389f9f0ecc9bd0 Mon Sep 17 00:00:00 2001 From: Len Brown Date: Thu, 13 Aug 2026 15:27:18 -0400 Subject: [PATCH 0374/1352] tools/power/turbostat: Warn, but don't Err on out-dated perf systems Some systems in the field have perf enabled, but their kernel perf support lacks the latest L2 stats. Warn, but don't Err on those systems, and suggest how to avoid the Warn. Fixes: dd23bfe4c317a ("tools/power turbostat: Add L2 cache statistics") Reported-by: David Arcari Signed-off-by: Len Brown --- tools/power/x86/turbostat/turbostat.c | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/tools/power/x86/turbostat/turbostat.c b/tools/power/x86/turbostat/turbostat.c index 920694c3c1ec94..6d3a58df678dc5 100644 --- a/tools/power/x86/turbostat/turbostat.c +++ b/tools/power/x86/turbostat/turbostat.c @@ -9471,8 +9471,10 @@ void perf_l2_init(void) free_fd_l2_percpu(); return; } - } else - err(-1, "%s: cpu%d: type %d", __func__, cpu, cpus[cpu].type); + } else { + warn("%s: cpu%d: type %d: Update kernel perf support or use \"--hide L2MRPS,L2%%hit\" or \"--hide cache\" or \"--no-perf\"", __func__, cpu, cpus[cpu].type); + return; + } } BIC_PRESENT(BIC_L2_MRPS); BIC_PRESENT(BIC_L2_HIT); From 1bd0b0e49ec13a501c056931ccdf777f25eba51f Mon Sep 17 00:00:00 2001 From: Len Brown Date: Mon, 22 Jun 2026 15:29:34 -0400 Subject: [PATCH 0375/1352] tools/power turbostat: Sanity check GFX%rc6 and SAM%mc6 percentages When measuring over a suspend cycle, GRX%rc6 and SAM%mc6 are observed to print as numbers much greater than 100%. Those numbers are not useful, and they mess up the columns. Detect when they are out of range, and print them as NaN. Signed-off-by: Len Brown --- tools/power/x86/turbostat/turbostat.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tools/power/x86/turbostat/turbostat.c b/tools/power/x86/turbostat/turbostat.c index 6d3a58df678dc5..8fe6464e18d34f 100644 --- a/tools/power/x86/turbostat/turbostat.c +++ b/tools/power/x86/turbostat/turbostat.c @@ -3636,7 +3636,7 @@ int format_counters(PER_THREAD_PARAMS) if (p->gfx_rc6_ms == -1) { /* detect GFX counter reset */ outp += sprintf(outp, "%s**.**", (printed++ ? delim : "")); } else { - outp += sprintf(outp, "%s%.2f", (printed++ ? delim : ""), p->gfx_rc6_ms / 10.0 / interval_float); + outp += sprintf(outp, "%s%.2f", (printed++ ? delim : ""), pct(p->gfx_rc6_ms / 10.0, interval_float)); } } @@ -3653,7 +3653,7 @@ int format_counters(PER_THREAD_PARAMS) if (p->sam_mc6_ms == -1) { /* detect GFX counter reset */ outp += sprintf(outp, "%s**.**", (printed++ ? delim : "")); } else { - outp += sprintf(outp, "%s%.2f", (printed++ ? delim : ""), p->sam_mc6_ms / 10.0 / interval_float); + outp += sprintf(outp, "%s%.2f", (printed++ ? delim : ""), pct(p->sam_mc6_ms / 10.0, interval_float)); } } From 6b0027ef96d42e4a9fa0ed421e4c6db52009ef07 Mon Sep 17 00:00:00 2001 From: Len Brown Date: Thu, 30 Apr 2026 21:28:19 -0400 Subject: [PATCH 0376/1352] tools/power turbostat: Fix fd_perf leak on reset When turbostat notices a topology change, it frees perf file descriptors for each CPU to prepare to re-initialize. Fix an off-by-one error in that per-cpu loop that resulted in a file descriptor leak of the file for the highest numbered CPU in the system. Fixes: 67bab430f4e7 ("tools/power turbostat: Group SMI counter with APERF and MPERF") Assisted-by: GitHub Copilot (claude-sonnet-4.6) Signed-off-by: Len Brown --- tools/power/x86/turbostat/turbostat.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tools/power/x86/turbostat/turbostat.c b/tools/power/x86/turbostat/turbostat.c index 8fe6464e18d34f..3bd53ce827c305 100644 --- a/tools/power/x86/turbostat/turbostat.c +++ b/tools/power/x86/turbostat/turbostat.c @@ -5915,7 +5915,7 @@ void free_fd_msr(void) if (!msr_counter_info) return; - for (int cpu = 0; cpu < topo.max_cpu_num; ++cpu) { + for (int cpu = 0; cpu <= topo.max_cpu_num; ++cpu) { if (msr_counter_info[cpu].fd_perf != -1) close(msr_counter_info[cpu].fd_perf); } From 9c500b17d76fbd303128d8f39b87ab39c0496578 Mon Sep 17 00:00:00 2001 From: Len Brown Date: Thu, 30 Apr 2026 21:40:36 -0400 Subject: [PATCH 0377/1352] tools/power turbostat: Fix CWF PMT off-by-one Fix off-by-one error causing last CPU on CWF to never get a pmt counter allocated. Fixes: 5ce1e9bbb2a1 ("tools/power turbostat: Add CPU%c1e BIC for CWF") Assisted-by: GitHub Copilot (claude-sonnet-4.6) Signed-off-by: Len Brown --- tools/power/x86/turbostat/turbostat.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tools/power/x86/turbostat/turbostat.c b/tools/power/x86/turbostat/turbostat.c index 3bd53ce827c305..f29e76d47d5942 100644 --- a/tools/power/x86/turbostat/turbostat.c +++ b/tools/power/x86/turbostat/turbostat.c @@ -10441,7 +10441,7 @@ void pmt_init(void) mod_num = 0; /* Relative module number for current PMT file. */ /* Open the counter for each CPU. */ - for (cpu_num = 0; cpu_num < topo.max_cpu_num;) { + for (cpu_num = 0; cpu_num <= topo.max_cpu_num;) { if (cpu_is_not_allowed(cpu_num)) goto next_loop_iter; From 8de7d3a3eabe0c885330ff5e5d60828cd457f51b Mon Sep 17 00:00:00 2001 From: Len Brown Date: Sun, 3 May 2026 15:23:25 -0400 Subject: [PATCH 0378/1352] tools/power turbostat: Fix PMT error path fd handling It is possible to leak a file descriptor if a diretory open succeeds, but the scan fails. Close the fd on error. Also, on cleanup, check for NULL fd. Assisted-by: Claude: Claude Sonnet 4.6 Fixes: 4265a86582eaa2 ("tools/power turbostat: Add PMT directory iterator helper") Signed-off-by: Len Brown --- tools/power/x86/turbostat/turbostat.c | 11 ++++++++--- 1 file changed, 8 insertions(+), 3 deletions(-) diff --git a/tools/power/x86/turbostat/turbostat.c b/tools/power/x86/turbostat/turbostat.c index f29e76d47d5942..8eddc9072f56e9 100644 --- a/tools/power/x86/turbostat/turbostat.c +++ b/tools/power/x86/turbostat/turbostat.c @@ -2034,8 +2034,11 @@ const struct dirent *pmt_diriter_begin(struct pmt_diriter_t *iter, const char *p return NULL; num_names = scandir(pmt_root_path, &iter->namelist, pmt_telemdir_filter, pmt_telemdir_sort); - if (num_names == -1) + if (num_names == -1) { + closedir(iter->dir); + iter->dir = NULL; return NULL; + } } iter->current_name_idx = 0; @@ -2063,8 +2066,10 @@ void pmt_diriter_remove(struct pmt_diriter_t *iter) iter->num_names = 0; iter->current_name_idx = 0; - closedir(iter->dir); - iter->dir = NULL; + if (iter->dir) { + closedir(iter->dir); + iter->dir = NULL; + } } unsigned int pmt_counter_get_width(const struct pmt_counter *p) From 95d4a3f996073d76ed18f74f8d1ac10c3b580b13 Mon Sep 17 00:00:00 2001 From: Len Brown Date: Sun, 3 May 2026 14:17:54 -0400 Subject: [PATCH 0379/1352] tools/power turbostat: Fix add_counter issue if > 24 counters. We allocate space for added counters ahead of time, and we arbitrarily chose 24 added counters of each type. If a future code flow hits this limit and continues, and then removes counters, we do a phantom increment and lose track of the number of added counters. Increment the count only if there is no error. Fixes: 388e9c8134be6b ("tools/power turbostat: Make extensible via the --add parameter") Assisted-by: Claude: Claude Sonnet 4.6 Signed-off-by: Len Brown --- tools/power/x86/turbostat/turbostat.c | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/tools/power/x86/turbostat/turbostat.c b/tools/power/x86/turbostat/turbostat.c index 8eddc9072f56e9..2faae55e4eecd8 100644 --- a/tools/power/x86/turbostat/turbostat.c +++ b/tools/power/x86/turbostat/turbostat.c @@ -10675,10 +10675,11 @@ int add_counter(unsigned int msr_num, char *path, char *name, fprintf(stderr, "%s: %s FOUND\n", __func__, name); break; } - if (sys.added_thread_counters++ >= MAX_ADDED_THREAD_COUNTERS) { + if (sys.added_thread_counters >= MAX_ADDED_THREAD_COUNTERS) { warnx("ignoring thread counter %s", name); return -1; } + sys.added_thread_counters++; break; case SCOPE_CORE: msrp = find_msrp_by_name(sys.cp, name); @@ -10687,10 +10688,11 @@ int add_counter(unsigned int msr_num, char *path, char *name, fprintf(stderr, "%s: %s FOUND\n", __func__, name); break; } - if (sys.added_core_counters++ >= MAX_ADDED_CORE_COUNTERS) { + if (sys.added_core_counters >= MAX_ADDED_CORE_COUNTERS) { warnx("ignoring core counter %s", name); return -1; } + sys.added_core_counters++; break; case SCOPE_PACKAGE: msrp = find_msrp_by_name(sys.pp, name); @@ -10699,10 +10701,11 @@ int add_counter(unsigned int msr_num, char *path, char *name, fprintf(stderr, "%s: %s FOUND\n", __func__, name); break; } - if (sys.added_package_counters++ >= MAX_ADDED_PACKAGE_COUNTERS) { + if (sys.added_package_counters >= MAX_ADDED_PACKAGE_COUNTERS) { warnx("ignoring package counter %s", name); return -1; } + sys.added_package_counters++; break; default: warnx("ignoring counter %s with unknown scope", name); From ccdfcb7e7ab84573046061ece61bfd2368577f1e Mon Sep 17 00:00:00 2001 From: Len Brown Date: Sun, 24 May 2026 17:03:47 -0400 Subject: [PATCH 0380/1352] turbostat 2026.09.28 Signed-off-by: Len Brown --- tools/power/x86/turbostat/turbostat.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tools/power/x86/turbostat/turbostat.c b/tools/power/x86/turbostat/turbostat.c index 2faae55e4eecd8..6d27074327bf37 100644 --- a/tools/power/x86/turbostat/turbostat.c +++ b/tools/power/x86/turbostat/turbostat.c @@ -10615,7 +10615,7 @@ int get_and_dump_counters(void) void print_version() { - fprintf(outf, "turbostat version 2026.04.21 - Len Brown \n"); + fprintf(outf, "turbostat version 2026.09.28 - Len Brown \n"); } #define COMMAND_LINE_SIZE 2048 From ab36951527b57cea3d1b78cf426fc21d23a516ed Mon Sep 17 00:00:00 2001 From: Srish Srinivasan Date: Sat, 12 Sep 2026 11:59:49 +0530 Subject: [PATCH 0381/1352] keys/trusted_keys: return immediately after TPM unseal failure trusted_tpm_unseal() proceeds to pcrlock() when the TPM unseal operation fails. If pcrlock() succeeds, its return value overwrites the unseal error, causing key instantiation to succeed. Return immediately when unseal fails to preserve the original error. Cc: stable@vger.kernel.org # v5.15+ Fixes: 5d0682be3189 ("KEYS: trusted: Add generic trusted keys framework") Signed-off-by: Srish Srinivasan Reviewed-by: Jarkko Sakkinen Link: https://lore.kernel.org/r/20260912062950.279104-2-ssrish@linux.ibm.com Signed-off-by: Jarkko Sakkinen --- security/keys/trusted-keys/trusted_tpm1.c | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/security/keys/trusted-keys/trusted_tpm1.c b/security/keys/trusted-keys/trusted_tpm1.c index bf0bf7f3697052..9cdfeea800a321 100644 --- a/security/keys/trusted-keys/trusted_tpm1.c +++ b/security/keys/trusted-keys/trusted_tpm1.c @@ -923,8 +923,10 @@ static int trusted_tpm_unseal(struct trusted_key_payload *p, char *datablob) ret = tpm2_unseal_trusted(chip, p, options); else ret = key_unseal(p, options); - if (ret < 0) + if (ret < 0) { pr_info("key_unseal failed (%d)\n", ret); + goto out; + } if (options->pcrlock) { ret = pcrlock(options->pcrlock); From 010efc1922ad5ed1f1ea58d164a903cede8341e8 Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Sat, 12 Sep 2026 11:54:40 +0200 Subject: [PATCH 0382/1352] keys: finalize persistent keyring timeout after link attempt When no keyring exists for the requested UID, KEYCTL_GET_PERSISTENT creates and registers one before linking it to the requested destination. The configured timeout is set only after the destination link succeeds. A destination restricted with KEYCTL_RESTRICT_KEYRING makes that link fail with -EPERM. With persistent_keyring_expiry set to 60 seconds, /proc/keys still reports the registered keyring's expiry as "perm". The failed call therefore leaves a quota-exempt keyring in the namespace's hidden register, where it may remain until namespace teardown. Rename the write-locked helper to key_get_or_create_persistent() and return the key reference through a result parameter. Return 0 when it creates a keyring and 1 when the retry finds an existing one. Return a negative error on failure. Another caller may create the keyring between the initial read-locked lookup and the retry under the write lock. Set the timeout after permission checking and linking. Do this on success, or on failure if the call allocated the keyring. This leaves a new keyring collectible after failure without starting its timeout while linking is still in progress. Cc: stable@vger.kernel.org # v5.10+ Fixes: f36f8c75ae2e ("KEYS: Add per-user_namespace registers for persistent per-UID kerberos caches") Assisted-by: LLM Signed-off-by: Karl Mehltretter Link: https://lore.kernel.org/r/20260912095440.79864-1-kmehltretter@gmail.com Reviewed-by: Jarkko Sakkinen Signed-off-by: Jarkko Sakkinen --- security/keys/persistent.c | 55 ++++++++++++++++++++++---------------- 1 file changed, 32 insertions(+), 23 deletions(-) diff --git a/security/keys/persistent.c b/security/keys/persistent.c index 97af230aa4b22b..bdab9953932613 100644 --- a/security/keys/persistent.c +++ b/security/keys/persistent.c @@ -33,25 +33,31 @@ static int key_create_persistent_register(struct user_namespace *ns) } /* - * Create the persistent keyring for the specified user. + * Get or create the persistent keyring for the specified user. * * Called with the namespace's sem locked for writing. + * + * Returns a boolean indicating whether the keyring already existed, + * or a negative error. On success, *persistent_ref holds a reference + * to the keyring. */ -static key_ref_t key_create_persistent(struct user_namespace *ns, kuid_t uid, - struct keyring_index_key *index_key) +static int key_get_or_create_persistent(struct user_namespace *ns, kuid_t uid, + struct keyring_index_key *index_key, + key_ref_t *persistent_ref) { struct key *persistent; - key_ref_t reg_ref, persistent_ref; + key_ref_t reg_ref; if (!ns->persistent_keyring_register) { - long err = key_create_persistent_register(ns); + int err = key_create_persistent_register(ns); + if (err < 0) - return ERR_PTR(err); + return err; } else { reg_ref = make_key_ref(ns->persistent_keyring_register, true); - persistent_ref = find_key_to_update(reg_ref, index_key); - if (persistent_ref) - return persistent_ref; + *persistent_ref = find_key_to_update(reg_ref, index_key); + if (*persistent_ref) + return 1; } persistent = keyring_alloc(index_key->description, @@ -61,9 +67,10 @@ static key_ref_t key_create_persistent(struct user_namespace *ns, kuid_t uid, KEY_ALLOC_NOT_IN_QUOTA, NULL, ns->persistent_keyring_register); if (IS_ERR(persistent)) - return ERR_CAST(persistent); + return PTR_ERR(persistent); - return make_key_ref(persistent, true); + *persistent_ref = make_key_ref(persistent, true); + return 0; } /* @@ -78,6 +85,7 @@ static long key_get_persistent(struct user_namespace *ns, kuid_t uid, key_ref_t reg_ref, persistent_ref; char buf[32]; long ret; + bool created = false; /* Look in the register if it exists */ memset(&index_key, 0, sizeof(index_key)); @@ -100,23 +108,24 @@ static long key_get_persistent(struct user_namespace *ns, kuid_t uid, * also need to create the register. */ down_write(&ns->keyring_sem); - persistent_ref = key_create_persistent(ns, uid, &index_key); + ret = key_get_or_create_persistent(ns, uid, &index_key, + &persistent_ref); up_write(&ns->keyring_sem); - if (!IS_ERR(persistent_ref)) - goto found; - - return PTR_ERR(persistent_ref); + if (ret < 0) + return ret; + created = ret == 0; found: + persistent = key_ref_to_ptr(persistent_ref); ret = key_task_permission(persistent_ref, current_cred(), KEY_NEED_LINK); - if (ret == 0) { - persistent = key_ref_to_ptr(persistent_ref); + if (ret == 0) ret = key_link(key_ref_to_ptr(dest_ref), persistent); - if (ret == 0) { - key_set_timeout(persistent, persistent_keyring_expiry); - ret = persistent->serial; - } - } + + if (ret == 0 || created) + key_set_timeout(persistent, persistent_keyring_expiry); + + if (ret == 0) + ret = persistent->serial; key_ref_put(persistent_ref); return ret; From 25d6a2c5965b7dd4f284365a1b7f525d839f86d3 Mon Sep 17 00:00:00 2001 From: Stefano Garzarella Date: Wed, 23 Sep 2026 19:35:06 +0200 Subject: [PATCH 0383/1352] KEYS: trusted: Fix blob allocation size in tpm2_key_decode() tpm2_key_decode() allocates 4 bytes more than needed. The ASN.1 callbacks tpm2_key_priv() and tpm2_key_pub() provide the lengths of TPM2B_PRIVATE and TPM2B_PUBLIC, so ctx.priv_len and ctx.pub_len already account for the 2-byte `size` field each of those structures starts with. I noticed this while reviewing commit 114f00d738f1 ("KEYS: trusted: Fix tpm2_load_cmd() boundary check"), which correctly reports ctx.priv_len + ctx.pub_len as the decoded blob size [1]. Let's allocate exactly that amount, matching the data copied into the blob. [1] https://lore.kernel.org/linux-integrity/apfoKo-BdwaLXtkT@sgarzare-redhat/ Cc: stable@vger.kernel.org # v5.13+ Fixes: f2219745250f ("security: keys: trusted: use ASN.1 TPM2 key format for the blobs") Signed-off-by: Stefano Garzarella Link: https://lore.kernel.org/r/20260923173506.41519-1-sgarzare@redhat.com Reviewed-by: Jarkko Sakkinen Signed-off-by: Jarkko Sakkinen --- security/keys/trusted-keys/trusted_tpm2.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/security/keys/trusted-keys/trusted_tpm2.c b/security/keys/trusted-keys/trusted_tpm2.c index 01f18bb370477e..a9b8a31a637ce8 100644 --- a/security/keys/trusted-keys/trusted_tpm2.c +++ b/security/keys/trusted-keys/trusted_tpm2.c @@ -115,7 +115,7 @@ static int tpm2_key_decode(struct trusted_key_payload *payload, if (ctx.priv_len + ctx.pub_len > MAX_BLOB_SIZE) return -EINVAL; - blob = kmalloc(ctx.priv_len + ctx.pub_len + 4, GFP_KERNEL); + blob = kmalloc(ctx.priv_len + ctx.pub_len, GFP_KERNEL); if (!blob) return -ENOMEM; From 1c5264b465ad0d2706e03dd0ca648cdffd7ccf0a Mon Sep 17 00:00:00 2001 From: Daehyeon Ko <4ncienth@gmail.com> Date: Fri, 25 Sep 2026 18:53:08 +0900 Subject: [PATCH 0384/1352] KEYS: trusted: Reject short TPM2 public areas tpm2_load_cmd() reads TPMA_OBJECT with get_unaligned_be32(pub + 4), but does not require public_len to cover that field. A new-format blob with public_len zero makes the read begin at the end of the B + 4-byte decoded allocation, causing a four-byte heap out-of-bounds read before the TPM command is transmitted. This is reachable from an unprivileged add_key() call when TPM trusted keys and a TPM2 device are available. This affects v5.13-rc1 and later kernels built with CONFIG_TRUSTED_KEYS=y and CONFIG_TRUSTED_KEYS_TPM=y when a TPM2 device is present. A KASAN run as UID 1000 with no effective capabilities reported: BUG: KASAN: slab-out-of-bounds in tpm2_unseal_trusted Read of size 4 at addr ffff888106273b48 by task exploit/160 CPU: 1 UID: 1000 PID: 160 Comm: exploit The buggy address belongs to the object at ffff888106273b40 which belongs to the cache kmalloc-8 of size 8 The buggy address is located 0 bytes to the right of allocated 8-byte region [ffff888106273b40, ffff888106273b48) TPMT_PUBLIC starts with the two-byte type, two-byte nameAlg and four-byte objectAttributes fields. Require public_len to cover all eight bytes before reading the attributes. The exact input produced the KASAN read in 3/3 fresh boots. The fixed build rejected it with -E2BIG and no KASAN report in 3/3 boots; valid new- and old-format trusted-key loads continued to succeed. Cc: stable@vger.kernel.org # v5.15+ Fixes: e5fb5d2c5a03 ("security: keys: trusted: Make sealed key properly interoperable") Assisted-by: LLM Signed-off-by: Daehyeon Ko <4ncienth@gmail.com> Link: https://lore.kernel.org/r/20260925095308.3248297-1-4ncienth@gmail.com Signed-off-by: Jarkko Sakkinen --- security/keys/trusted-keys/trusted_tpm2.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/security/keys/trusted-keys/trusted_tpm2.c b/security/keys/trusted-keys/trusted_tpm2.c index a9b8a31a637ce8..c176ceba7859f5 100644 --- a/security/keys/trusted-keys/trusted_tpm2.c +++ b/security/keys/trusted-keys/trusted_tpm2.c @@ -413,6 +413,8 @@ static int tpm2_load_cmd(struct tpm_chip *chip, public_len = get_unaligned_be16(blob + 2 + private_len); if (private_len + 2 + public_len + 2 > blob_len) return -E2BIG; + if (public_len < 8) + return -E2BIG; pub = blob + 2 + private_len + 2; /* key attributes are always at offset 4 */ From 8474b0d97f2c136f9870b85852e5b76bd4e4103e Mon Sep 17 00:00:00 2001 From: Jan Kara Date: Wed, 30 Sep 2026 12:28:56 +0200 Subject: [PATCH 0385/1352] isofs: Simplify isofs_read_level3_size() After recent changes it is possible to further simplify isofs_read_level3_size() by reorganizing the loop a bit and removing empty_blocks, block_saved, offset_saved variables. Signed-off-by: Jan Kara --- fs/isofs/inode.c | 21 ++++++++------------- 1 file changed, 8 insertions(+), 13 deletions(-) diff --git a/fs/isofs/inode.c b/fs/isofs/inode.c index e884618c0c5357..c8f0f8803a7b13 100644 --- a/fs/isofs/inode.c +++ b/fs/isofs/inode.c @@ -1173,9 +1173,8 @@ static int isofs_read_level3_size(struct inode *inode) unsigned long bufsize = ISOFS_BUFFER_SIZE(inode); int high_sierra = ISOFS_SB(inode->i_sb)->s_high_sierra; struct buffer_head *bh = NULL; - unsigned long block, offset, block_saved, offset_saved; + unsigned long block, offset; int i = 0; - int empty_blocks = 0; int more_entries = 0; struct iso_inode_info *ei = ISOFS_I(inode); @@ -1209,7 +1208,7 @@ static int isofs_read_level3_size(struct inode *inode) * chain of empty blocks could be walked without bound. */ if (offset >= bufsize || de->length[0] == 0) { - if (offset == 0 && ++empty_blocks + i > 100) + if (offset == 0 && ++i > 100) goto out_toomany; brelse(bh); bh = NULL; @@ -1225,21 +1224,17 @@ static int isofs_read_level3_size(struct inode *inode) return -EIO; } + /* Save the first continuation directory entry in the inode */ + if (more_entries && !ei->i_next_section_block) { + ei->i_next_section_block = block; + ei->i_next_section_offset = offset; + } de_len = de->length[0]; - block_saved = block; - offset_saved = offset; offset += de_len; - inode->i_size += isonum_733(de->size); - if (i == 1) { - ei->i_next_section_block = block_saved; - ei->i_next_section_offset = offset_saved; - } - more_entries = de->flags[-high_sierra] & 0x80; - i++; - if (i + empty_blocks > 100) + if (++i > 100) goto out_toomany; } while (more_entries); out: From 272011bdd9147524b50effea12052299d35c7b06 Mon Sep 17 00:00:00 2001 From: Matthias Goergens Date: Sat, 26 Sep 2026 12:29:13 +0800 Subject: [PATCH 0386/1352] isofs: return long Joliet names whole get_joliet_filename() writes into a buffer allocated by its caller but is not told how big that buffer is. It hardcodes PAGE_SIZE as the output limit for utf16s_to_utf8s() and gives uni16_to_x8() no limit at all. PAGE_SIZE was never the right bound. Before commit b2eb2e288604 ("isofs: Drop support of directory entries straddling blocks") both callers allocated a page, but only its first 1024 bytes were for the name, the rest holding a copy of a directory record that straddled a block. It could not overflow only because a directory record holds at most 222 bytes of name (255 bytes, 33 of them fixed), which no converter expands beyond 1024 bytes. It also keeps the converted length in an unsigned char. 222 bytes of name are 111 UTF-16 units, and the UTF-8 converter, used for iocharset=utf8 and when CONFIG_NLS_DEFAULT is "utf8", needs three bytes for each CJK, Thai or Devanagari character, so such a name can take up to 333 bytes. Past 255 the length wraps. readdir then reports a short name that usually ends in the middle of a UTF-8 sequence, and lookup finds the file only under that name, never under its real one; a name that wraps to exactly 0 bytes is not listed at all. Such names are out of spec, since Joliet allows 64 units, but common tools write them: PowerISO 9.5 and UltraISO 9.76 both keep up to 110 units by default. Windows, for which Joliet was made, returns them whole, including 111-unit names in records that leave out the padding byte. Return them whole here too. Pass the buffer size down and have both converters respect it. utf16s_to_utf8s() stops before a character that does not fit, and uni16_to_x8() now hands uni2char() the space that actually remains and stops on -ENAMETOOLONG, as fs/hfsplus/unicode.c does. Both callers pass JOLIET_NAME_MAX + 1, room for 111 units at three bytes each plus the terminator; no character set needs more than three bytes for a UTF-16 unit, and isofs_dir_record_valid() already rejects a record whose name_len claims more than the record holds, so no name is cut. Make the length an int, which is what both converters return. These names are longer than NAME_MAX. POSIX lets the limit vary by filesystem, reported by pathconf(_PC_NAME_MAX), and the VFS limits a name only by PATH_MAX (verify_dirent_name() in fs/readdir.c). vfat, exfat, hfsplus and ntfs3 already return names of up to 255 UTF-16 units, 765 bytes of UTF-8, and such names break the same things there: copying the file to a filesystem with a 255-byte limit, such as ext4 or tmpfs, fails with ENAMETOOLONG, and cp, tar, rsync and Python's shutil report that file and carry on with the rest, so the copy is incomplete but says so. glibc's readdir() returns the names, but the deprecated readdir_r() skips them and fails with ENAMETOOLONG, and an inotify reader with the buffer size inotify(7) suggests gets EINVAL. In the kernel, fanotify reports events on such a file without its name and warns once in fanotify_info_copy_name(), and a directory with such a name cannot be reconnected from a file handle, because the generic get_name() in fs/exportfs, which isofs uses, skips names longer than NAME_MAX. Cutting at NAME_MAX would avoid these, but would list names that Windows does not show and that can coincide within a directory. Every name of at most 255 bytes is returned exactly as before. Fixes: 1da177e4c3f4 ("Linux-2.6.12-rc2") Signed-off-by: Matthias Goergens Link: https://patch.msgid.link/20260926042916.3277409-2-matthias.goergens@gmail.com Signed-off-by: Jan Kara --- fs/isofs/dir.c | 4 +++- fs/isofs/isofs.h | 11 ++++++++++- fs/isofs/joliet.c | 27 ++++++++++++++++++++------- fs/isofs/namei.c | 3 ++- 4 files changed, 35 insertions(+), 10 deletions(-) diff --git a/fs/isofs/dir.c b/fs/isofs/dir.c index c7ca7603e97a12..f4243fcbb69790 100644 --- a/fs/isofs/dir.c +++ b/fs/isofs/dir.c @@ -199,7 +199,9 @@ static int do_isofs_readdir(struct inode *inode, struct file *file, if (map) { #ifdef CONFIG_JOLIET if (sbi->s_joliet_level) { - len = get_joliet_filename(de, tmpname, inode); + len = get_joliet_filename(de, tmpname, + JOLIET_NAME_MAX + 1, + inode); p = tmpname; } else #endif diff --git a/fs/isofs/isofs.h b/fs/isofs/isofs.h index 79ca0256843aa5..47c43a3c6a614f 100644 --- a/fs/isofs/isofs.h +++ b/fs/isofs/isofs.h @@ -121,7 +121,16 @@ bool isofs_dir_record_valid(struct iso_directory_record *de, unsigned long offset, unsigned long bufsize); -int get_joliet_filename(struct iso_directory_record *, unsigned char *, struct inode *); +/* + * The longest name the Joliet converter returns, in bytes. A directory + * record is at most 255 bytes long, which leaves room for 111 UTF-16 units + * of name, and no character set needs more than three bytes for one unit. + */ +#define JOLIET_NAME_MAX \ + ((255 - sizeof(struct iso_directory_record)) / 2 * 3) + +int get_joliet_filename(struct iso_directory_record *de, unsigned char *outname, + int outsize, struct inode *inode); int get_acorn_filename(struct iso_directory_record *, char *, struct inode *); extern struct dentry *isofs_lookup(struct inode *, struct dentry *, unsigned int flags); diff --git a/fs/isofs/joliet.c b/fs/isofs/joliet.c index c0f04a1e7f695f..d37e67c5e36f42 100644 --- a/fs/isofs/joliet.c +++ b/fs/isofs/joliet.c @@ -15,19 +15,26 @@ * Convert Unicode 16 to UTF-8 or ASCII. */ static int -uni16_to_x8(unsigned char *ascii, __be16 *uni, int len, struct nls_table *nls) +uni16_to_x8(unsigned char *ascii, __be16 *uni, int len, struct nls_table *nls, + int outsize) { __be16 *ip, ch; - unsigned char *op; + unsigned char *op, *end; ip = uni; op = ascii; + end = ascii + outsize - 1; /* leave room for the terminator */ while ((ch = get_unaligned(ip)) && len) { int llen; - llen = nls->uni2char(be16_to_cpu(ch), op, NLS_MAX_CHARSET_SIZE); + + if (op >= end) + break; + llen = nls->uni2char(be16_to_cpu(ch), op, end - op); if (llen > 0) op += llen; + else if (llen == -ENAMETOOLONG) + break; else *op++ = '?'; ip++; @@ -38,21 +45,27 @@ uni16_to_x8(unsigned char *ascii, __be16 *uni, int len, struct nls_table *nls) return (op - ascii); } +/* + * Convert the Joliet name of @de into @outname, a buffer of @outsize bytes. + * The result is at most @outsize - 1 bytes long; a longer name is cut at a + * character boundary. + */ int -get_joliet_filename(struct iso_directory_record * de, unsigned char *outname, struct inode * inode) +get_joliet_filename(struct iso_directory_record *de, unsigned char *outname, + int outsize, struct inode *inode) { struct nls_table *nls; - unsigned char len = 0; + int len = 0; nls = ISOFS_SB(inode->i_sb)->s_nls_iocharset; if (!nls) { len = utf16s_to_utf8s((const wchar_t *) de->name, de->name_len[0] >> 1, UTF16_BIG_ENDIAN, - outname, PAGE_SIZE); + outname, outsize - 1); } else { len = uni16_to_x8(outname, (__be16 *) de->name, - de->name_len[0] >> 1, nls); + de->name_len[0] >> 1, nls, outsize); } if ((len > 2) && (outname[len-2] == ';') && (outname[len-1] == '1')) len -= 2; diff --git a/fs/isofs/namei.c b/fs/isofs/namei.c index 010682f5901a49..ccf6f01cb3bbbf 100644 --- a/fs/isofs/namei.c +++ b/fs/isofs/namei.c @@ -107,7 +107,8 @@ isofs_find_entry(struct inode *dir, struct dentry *dentry, dpnt = tmpname; #ifdef CONFIG_JOLIET } else if (sbi->s_joliet_level) { - dlen = get_joliet_filename(de, tmpname, dir); + dlen = get_joliet_filename(de, tmpname, + JOLIET_NAME_MAX + 1, dir); dpnt = tmpname; #endif } else if (sbi->s_mapping == 'a') { From 5d5cc259718dcfadd5b42f53a2d28d3302289a85 Mon Sep 17 00:00:00 2001 From: Matthias Goergens Date: Sat, 26 Sep 2026 12:29:14 +0800 Subject: [PATCH 0387/1352] isofs: report the Joliet name length in statfs isofs_statfs() reports NAME_MAX as the longest name, but a Joliet name can now be up to JOLIET_NAME_MAX bytes long. glibc answers pathconf(_PC_NAME_MAX) from f_namelen, so a program sizing a buffer from it should get room for every name readdir returns. vfat has reported 255 times NLS_MAX_CHARSET_SIZE, 1530, since commit f68e542f3478 ("fat: Fix statfs->f_namelen"), which followed a report of readdir_r() overflowing a buffer sized that way, and exfat reports the same. Windows reports a maximum component length of 110 for CDFS. Report JOLIET_NAME_MAX on Joliet mounts. Rock Ridge and plain ISO 9660 mounts keep NAME_MAX. Signed-off-by: Matthias Goergens Link: https://patch.msgid.link/20260926042916.3277409-3-matthias.goergens@gmail.com Signed-off-by: Jan Kara --- fs/isofs/inode.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/fs/isofs/inode.c b/fs/isofs/inode.c index c8f0f8803a7b13..4824e42995ccfd 100644 --- a/fs/isofs/inode.c +++ b/fs/isofs/inode.c @@ -1012,7 +1012,8 @@ static int isofs_statfs (struct dentry *dentry, struct kstatfs *buf) buf->f_files = ISOFS_SB(sb)->s_ninodes; buf->f_ffree = 0; buf->f_fsid = u64_to_fsid(id); - buf->f_namelen = NAME_MAX; + buf->f_namelen = ISOFS_SB(sb)->s_joliet_level ? + JOLIET_NAME_MAX : NAME_MAX; return 0; } From 83cead2e1a5c4a2303e5cb495756c6eb5cfc87b0 Mon Sep 17 00:00:00 2001 From: Matthias Goergens Date: Sat, 26 Sep 2026 12:29:15 +0800 Subject: [PATCH 0388/1352] isofs: pass the name buffer size to get_rock_ridge_filename() get_rock_ridge_filename() assembles a name from one or more NM entries into a buffer allocated by its caller. It is not told how big that buffer is, and instead stops adding NM entries once the name would exceed NAME_MAX, relying on the callers' buffer being larger. Pass the size in and check against it. Both callers pass NAME_MAX + 1, so the behaviour is unchanged: an NM entry that would take the name past 255 bytes is dropped, together with any that follow, and the entries before it are returned. Signed-off-by: Matthias Goergens Link: https://patch.msgid.link/20260926042916.3277409-4-matthias.goergens@gmail.com Signed-off-by: Jan Kara --- fs/isofs/dir.c | 3 ++- fs/isofs/isofs.h | 3 ++- fs/isofs/namei.c | 3 ++- fs/isofs/rock.c | 8 ++++++-- 4 files changed, 12 insertions(+), 5 deletions(-) diff --git a/fs/isofs/dir.c b/fs/isofs/dir.c index f4243fcbb69790..ed8a8dc41fb271 100644 --- a/fs/isofs/dir.c +++ b/fs/isofs/dir.c @@ -190,7 +190,8 @@ static int do_isofs_readdir(struct inode *inode, struct file *file, map = 1; if (sbi->s_rock) { - len = get_rock_ridge_filename(de, tmpname, inode); + len = get_rock_ridge_filename(de, tmpname, NAME_MAX + 1, + inode); if (len != 0) { /* may be -1 */ p = tmpname; map = 0; diff --git a/fs/isofs/isofs.h b/fs/isofs/isofs.h index 47c43a3c6a614f..a2d28a23e8924a 100644 --- a/fs/isofs/isofs.h +++ b/fs/isofs/isofs.h @@ -115,7 +115,8 @@ struct timespec64 iso_date(u8 *p, int flags); struct inode; /* To make gcc happy */ extern int parse_rock_ridge_inode(struct iso_directory_record *, struct inode *, int relocated); -extern int get_rock_ridge_filename(struct iso_directory_record *, char *, struct inode *); +int get_rock_ridge_filename(struct iso_directory_record *de, char *retname, + int retnamesize, struct inode *inode); extern int isofs_name_translate(struct iso_directory_record *, char *, struct inode *); bool isofs_dir_record_valid(struct iso_directory_record *de, unsigned long offset, diff --git a/fs/isofs/namei.c b/fs/isofs/namei.c index ccf6f01cb3bbbf..7ee1a2e1e38b37 100644 --- a/fs/isofs/namei.c +++ b/fs/isofs/namei.c @@ -102,7 +102,8 @@ isofs_find_entry(struct inode *dir, struct dentry *dentry, dpnt = de->name; if (sbi->s_rock && - ((i = get_rock_ridge_filename(de, tmpname, dir)))) { + ((i = get_rock_ridge_filename(de, tmpname, NAME_MAX + 1, + dir)))) { dlen = i; /* possibly -1 */ dpnt = tmpname; #ifdef CONFIG_JOLIET diff --git a/fs/isofs/rock.c b/fs/isofs/rock.c index 84e0d764c2100f..5a5984b7222478 100644 --- a/fs/isofs/rock.c +++ b/fs/isofs/rock.c @@ -209,10 +209,14 @@ static int rock_check_overflow(struct rock_state *rs, int sig) } /* + * Build the Rock Ridge name of @de in @retname, a buffer of @retnamesize + * bytes. From the first NM entry that does not fit along with the + * terminator, the rest of the name is dropped. + * * return length of name field; 0: not found, -1: to be ignored */ int get_rock_ridge_filename(struct iso_directory_record *de, - char *retname, struct inode *inode) + char *retname, int retnamesize, struct inode *inode) { struct rock_state rs; struct rock_ridge *rr; @@ -287,7 +291,7 @@ int get_rock_ridge_filename(struct iso_directory_record *de, break; } len = rr->len - 5; - if (retnamlen + len > NAME_MAX) { + if (retnamlen + len >= retnamesize) { truncate = 1; break; } From 4b0b4f23c30ee8cc5350e8e0d3e7f42e874f022a Mon Sep 17 00:00:00 2001 From: Matthias Goergens Date: Sat, 26 Sep 2026 12:29:16 +0800 Subject: [PATCH 0389/1352] isofs: shrink the name conversion buffer isofs_readdir() and isofs_lookup() allocate 1024 bytes for the name converters. Now that get_rock_ridge_filename() and get_joliet_filename() are told the buffer size, nothing writes beyond JOLIET_NAME_MAX + 1 bytes: get_joliet_filename() is passed that size, get_rock_ridge_filename() is passed NAME_MAX + 1, isofs_name_translate() copies at most the 222 bytes of name a directory record can hold, and get_acorn_filename() appends at most five bytes to that. Allocate JOLIET_NAME_MAX + 1 bytes, matching what the Joliet callers pass. No functional change. Signed-off-by: Matthias Goergens Link: https://patch.msgid.link/20260926042916.3277409-5-matthias.goergens@gmail.com Signed-off-by: Jan Kara --- fs/isofs/dir.c | 7 ++++++- fs/isofs/namei.c | 2 +- 2 files changed, 7 insertions(+), 2 deletions(-) diff --git a/fs/isofs/dir.c b/fs/isofs/dir.c index ed8a8dc41fb271..310b74b08b5e6a 100644 --- a/fs/isofs/dir.c +++ b/fs/isofs/dir.c @@ -240,7 +240,12 @@ static int isofs_readdir(struct file *file, struct dir_context *ctx) char *tmpname; struct inode *inode = file_inode(file); - tmpname = kmalloc(1024, GFP_KERNEL); + /* + * Rockridge can produce names of NAME_MAX size, Acorn extensions at + * most 227 chars. + */ + BUILD_BUG_ON(JOLIET_NAME_MAX < NAME_MAX || JOLIET_NAME_MAX < 227); + tmpname = kmalloc(JOLIET_NAME_MAX + 1, GFP_KERNEL); if (tmpname == NULL) return -ENOMEM; diff --git a/fs/isofs/namei.c b/fs/isofs/namei.c index 7ee1a2e1e38b37..19128de0c3a7be 100644 --- a/fs/isofs/namei.c +++ b/fs/isofs/namei.c @@ -155,7 +155,7 @@ struct dentry *isofs_lookup(struct inode *dir, struct dentry *dentry, unsigned i struct inode *inode; char *tmpname; - tmpname = kmalloc(1024, GFP_USER); + tmpname = kmalloc(JOLIET_NAME_MAX + 1, GFP_USER); if (!tmpname) return ERR_PTR(-ENOMEM); From 4e8bc6b5bbe6305e530e2f024cf983506d72f657 Mon Sep 17 00:00:00 2001 From: Sebastian Andrzej Siewior Date: Fri, 18 Sep 2026 12:37:25 +0200 Subject: [PATCH 0390/1352] drm/i915: Use %p for pointer formatting Commit 2563a4524febe ("drm/i915: restrict kernel address leak in debugfs") introduced the %pK modifier in order not to leak kernel pointer. Since commit ad67b74d2469d ("printk: hash addresses printed with %p") pointers are hashed by default and the behaviour can be controller by `hash_pointers' boot argument. The policy on %p is to not introduce new ones. Keeping the %p allows to assign an id to the object. Removing %pK makes it possible to remove its handling from the library. Use %p instead %pK to allow debugging if needed. Signed-off-by: Sebastian Andrzej Siewior Reviewed-by: Andi Shyti Signed-off-by: Andi Shyti Link: https://patch.msgid.link/20260918103725.cfYzL1Wt@linutronix.de --- drivers/gpu/drm/i915/i915_debugfs.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/gpu/drm/i915/i915_debugfs.c b/drivers/gpu/drm/i915/i915_debugfs.c index a3e27f9e4f476e..93ba7369ba6e2a 100644 --- a/drivers/gpu/drm/i915/i915_debugfs.c +++ b/drivers/gpu/drm/i915/i915_debugfs.c @@ -179,7 +179,7 @@ i915_debugfs_describe_obj(struct seq_file *m, struct drm_i915_gem_object *obj) struct i915_vma *vma; int pin_count = 0; - seq_printf(m, "%pK: %c%c%c %8zdKiB %02x %02x %s%s%s", + seq_printf(m, "%p: %c%c%c %8zdKiB %02x %02x %s%s%s", &obj->base, get_tiling_flag(obj), get_global_flag(obj), From 28ecc9eabe7c7bb4f9ea1d013eab900f7a70943b Mon Sep 17 00:00:00 2001 From: Bjorn Helgaas Date: Wed, 23 Sep 2026 14:57:20 -0500 Subject: [PATCH 0391/1352] PCI: tegra264: Fix Link Capabilities register offset The PCI Express Capability begins at 0x48. Link Capabilities is a 32-bit register at offset 0xc, and Link Status is a 16-bit register at offset 0x12: Link Capabilities is at 0x48 + 0xc = 0x54 Link Status is at 0x48 + 0x12 = 0x5a Previously the driver read Link Capabilities with a 16-bit read from XTL_RC_PCIE_CFG_LINK_CAPS (0x56), which incorrectly read just the upper half of the register. When a hotplug-capable port has no link during probe, tegra264_pcie_icc_set() consequently derives the maximum speed and width from unrelated bits and requests the wrong interconnect bandwidth. Correct the Link Capabilities usage by adding a XTL_RC_PCIE_CAP definition for the base of the PCIe Capability, using the existing PCI_EXP_LNKCAP (0xc) and PCI_EXP_LNKSTA (0x12) offsets so they're easily searchable, and reading the entire 32 bits of Link Capabilities. Fixes: 01c3c27a0ef6 ("PCI: tegra264: Add Tegra264 support") Based-on-patch-by: Linmao Li Link: https://lore.kernel.org/20260827093919.2825467-1-lilinmao@kylinos.cn Signed-off-by: Bjorn Helgaas Link: https://patch.msgid.link/20260923195719.1933175-2-bhelgaas@google.com --- drivers/pci/controller/pcie-tegra264.c | 9 ++++----- 1 file changed, 4 insertions(+), 5 deletions(-) diff --git a/drivers/pci/controller/pcie-tegra264.c b/drivers/pci/controller/pcie-tegra264.c index 653136db401e0e..c97576ab4a316a 100644 --- a/drivers/pci/controller/pcie-tegra264.c +++ b/drivers/pci/controller/pcie-tegra264.c @@ -49,8 +49,7 @@ #define XAL_RC_BAR_CNTL_STANDARD_64B_BAR_EN BIT(2) /* XTL registers */ -#define XTL_RC_PCIE_CFG_LINK_CAPS 0x56 -#define XTL_RC_PCIE_CFG_LINK_STATUS 0x5a +#define XTL_RC_PCIE_CAP 0x48 /* PCIe Capability */ #define XTL_RC_MGMT_PERST_CONTROL 0x218 #define XTL_RC_MGMT_PERST_CONTROL_PERST_O_N BIT(0) @@ -118,11 +117,11 @@ static void tegra264_pcie_icc_set(struct tegra264_pcie *pcie) * possible, so this is as good as it gets for now. */ if (pcie->link_up) { - value = readw(pcie->ecam + XTL_RC_PCIE_CFG_LINK_STATUS); + value = readw(pcie->ecam + XTL_RC_PCIE_CAP + PCI_EXP_LNKSTA); speed = FIELD_GET(PCI_EXP_LNKSTA_CLS, value); width = FIELD_GET(PCI_EXP_LNKSTA_NLW, value); } else { - value = readw(pcie->ecam + XTL_RC_PCIE_CFG_LINK_CAPS); + value = readl(pcie->ecam + XTL_RC_PCIE_CAP + PCI_EXP_LNKCAP); speed = FIELD_GET(PCI_EXP_LNKCAP_SLS, value); width = FIELD_GET(PCI_EXP_LNKCAP_MLW, value); } @@ -263,7 +262,7 @@ static bool tegra264_pcie_supports_hotplug(struct tegra264_pcie *pcie) static bool tegra264_pcie_link_up(struct tegra264_pcie *pcie, enum pci_bus_speed *speed) { - u16 value = readw(pcie->ecam + XTL_RC_PCIE_CFG_LINK_STATUS); + u16 value = readw(pcie->ecam + XTL_RC_PCIE_CAP + PCI_EXP_LNKSTA); if (value & PCI_EXP_LNKSTA_DLLLA) { if (speed) From aa98d2da191dc95b1d49983bad683ce9ccb7bdb6 Mon Sep 17 00:00:00 2001 From: Guanghui Yang <3497809730@qq.com> Date: Sat, 8 Aug 2026 14:13:55 +0800 Subject: [PATCH 0392/1352] btrfs: free unlinked replace target on initialization failure btrfs_init_dev_replace_tgtdev() allocates the replacement target before looking up its dev_t and initializing its zoned device information. If either lookup_bdev() or btrfs_get_dev_zone_info() fails, the device has not been linked into fs_devices->devices yet, but the error path only drops the block device file reference. Free the allocated device on this error path to release its name, allocation state, zone info, and the device itself. The issue was found by a failure-path metadata residual analyzer and verified with targeted failure injection on v6.14. Assisted-by: Codex:gpt-5 Reviewed-by: Qu Wenruo Signed-off-by: Guanghui Yang <3497809730@qq.com> Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/dev-replace.c | 10 +++++++--- fs/btrfs/volumes.c | 2 +- fs/btrfs/volumes.h | 1 + 3 files changed, 9 insertions(+), 4 deletions(-) diff --git a/fs/btrfs/dev-replace.c b/fs/btrfs/dev-replace.c index 0284be0e4e826d..cdfe093e5c5c4f 100644 --- a/fs/btrfs/dev-replace.c +++ b/fs/btrfs/dev-replace.c @@ -235,7 +235,8 @@ static int btrfs_init_dev_replace_tgtdev(struct btrfs_fs_info *fs_info, struct btrfs_device **device_out) { struct btrfs_fs_devices *fs_devices = fs_info->fs_devices; - struct btrfs_device *device; + struct btrfs_device *device = NULL; + struct btrfs_device *tmp_device; struct file *bdev_file; struct block_device *bdev; u64 devid = BTRFS_DEV_REPLACE_DEVID; @@ -264,8 +265,8 @@ static int btrfs_init_dev_replace_tgtdev(struct btrfs_fs_info *fs_info, sync_blockdev(bdev); - list_for_each_entry(device, &fs_devices->devices, dev_list) { - if (device->bdev == bdev) { + list_for_each_entry(tmp_device, &fs_devices->devices, dev_list) { + if (tmp_device->bdev == bdev) { btrfs_err(fs_info, "target device is in the filesystem!"); ret = -EEXIST; @@ -285,6 +286,7 @@ static int btrfs_init_dev_replace_tgtdev(struct btrfs_fs_info *fs_info, device = btrfs_alloc_device(NULL, &devid, NULL, device_path); if (IS_ERR(device)) { ret = PTR_ERR(device); + device = NULL; goto error; } @@ -328,6 +330,8 @@ static int btrfs_init_dev_replace_tgtdev(struct btrfs_fs_info *fs_info, error: /* Undo the open-time freeze deny. */ + if (device) + btrfs_free_device(device); btrfs_release_device_allow_freeze(bdev_file); return ret; } diff --git a/fs/btrfs/volumes.c b/fs/btrfs/volumes.c index 85ea9c5d4536c8..95cfb2664a789f 100644 --- a/fs/btrfs/volumes.c +++ b/fs/btrfs/volumes.c @@ -403,7 +403,7 @@ static struct btrfs_fs_devices *alloc_fs_devices(const u8 *fsid) return fs_devs; } -static void btrfs_free_device(struct btrfs_device *device) +void btrfs_free_device(struct btrfs_device *device) { WARN_ON(!list_empty(&device->post_commit_list)); /* diff --git a/fs/btrfs/volumes.h b/fs/btrfs/volumes.h index 0415d74cad9ba9..337d7007d9e225 100644 --- a/fs/btrfs/volumes.h +++ b/fs/btrfs/volumes.h @@ -799,6 +799,7 @@ void btrfs_rm_dev_replace_remove_srcdev(struct btrfs_device *srcdev); void btrfs_rm_dev_replace_free_srcdev(struct btrfs_device *srcdev); void btrfs_destroy_dev_replace_tgtdev(struct btrfs_device *tgtdev, bool allow_freeze); +void btrfs_free_device(struct btrfs_device *device); unsigned long btrfs_full_stripe_len(struct btrfs_fs_info *fs_info, u64 logical); u64 btrfs_calc_stripe_length(const struct btrfs_chunk_map *map); From fbf8cda1796ed4f945cffdb3af7cd3073b6cbf4c Mon Sep 17 00:00:00 2001 From: Guanghui Yang <3497809730@qq.com> Date: Tue, 11 Aug 2026 09:02:27 +0930 Subject: [PATCH 0393/1352] btrfs: roll back sprout setup after device add failure btrfs_init_new_device() calls btrfs_setup_sprout() before creating the first writable chunks for a seed filesystem. That moves the seed devices out of fs_info->fs_devices, clears the seeding state and installs a new fsid for the sprout filesystem. If a later step fails, the error path removes the new device but leaves fs_info->fs_devices in the partially initialized sprout state. The mounted filesystem can then be left with no open devices after the failed device add. Add the inverse of btrfs_setup_sprout() and use it from the error path so the mounted seed filesystem is restored before the temporary seed_devices copy is released. Fixes: 2b82032c34ec ("Btrfs: Seed device support") Assisted-by: Codex:gpt-5 Reviewed-by: Qu Wenruo Signed-off-by: Guanghui Yang <3497809730@qq.com> [ Fix a conflict with per-profile available space, revert sprout before updating per-profile available space estimation. ] Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/volumes.c | 37 +++++++++++++++++++++++++++++++++++++ 1 file changed, 37 insertions(+) diff --git a/fs/btrfs/volumes.c b/fs/btrfs/volumes.c index 95cfb2664a789f..1fad1aa716debf 100644 --- a/fs/btrfs/volumes.c +++ b/fs/btrfs/volumes.c @@ -2814,6 +2814,41 @@ static void btrfs_setup_sprout(struct btrfs_fs_info *fs_info, btrfs_set_super_flags(disk_super, super_flags); } +static void btrfs_rollback_sprout(struct btrfs_fs_info *fs_info, + struct btrfs_fs_devices *seed_devices) +{ + struct btrfs_fs_devices *fs_devices = fs_info->fs_devices; + struct btrfs_super_block *disk_super = fs_info->super_copy; + struct btrfs_device *device; + u64 super_flags; + + lockdep_assert_held(&uuid_mutex); + lockdep_assert_held(&fs_devices->device_list_mutex); + + list_del_init(&seed_devices->seed_list); + list_splice_init_rcu(&seed_devices->devices, &fs_devices->devices, synchronize_rcu); + list_for_each_entry(device, &fs_devices->devices, dev_list) { + device->fs_devices = fs_devices; + } + + fs_devices->seeding = true; + fs_devices->num_devices = seed_devices->num_devices; + fs_devices->open_devices = seed_devices->open_devices; + fs_devices->missing_devices = seed_devices->missing_devices; + fs_devices->rotating = seed_devices->rotating; + fs_devices->latest_dev = seed_devices->latest_dev; + + memcpy(fs_devices->fsid, seed_devices->fsid, BTRFS_FSID_SIZE); + memcpy(fs_devices->metadata_uuid, seed_devices->metadata_uuid, BTRFS_FSID_SIZE); + memcpy(disk_super->fsid, seed_devices->fsid, BTRFS_FSID_SIZE); + + super_flags = (btrfs_super_flags(disk_super) | BTRFS_SUPER_FLAG_SEEDING); + btrfs_set_super_flags(disk_super, super_flags); + + seed_devices->opened = 0; + free_fs_devices(seed_devices); +} + /* * Store the expected generation for seed devices in device items. */ @@ -3165,6 +3200,8 @@ int btrfs_init_new_device(struct btrfs_fs_info *fs_info, const char *device_path orig_super_total_bytes); btrfs_set_super_num_devices(fs_info->super_copy, orig_super_num_devices); + if (seeding_dev) + btrfs_rollback_sprout(fs_info, seed_devices); btrfs_update_per_profile_avail(fs_info); mutex_unlock(&fs_info->chunk_mutex); mutex_unlock(&fs_info->fs_devices->device_list_mutex); From 03b6463e76c93e4ab21af2986714ef0d3c6452ff Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Mon, 17 Aug 2026 17:00:43 +0930 Subject: [PATCH 0394/1352] btrfs: refactor read_key_bytes() to remove the dest_folio parameter The function read_key_bytes() have 3 call sites: - For BTRFS_VERITY_DESC_ITEM_KEY offset 0 inside btrfs_get_verity_descriptor() - For BTRFS_VERITY_DESC_ITEM_KEY offset 1 inside btrfs_get_verity_descriptor() Those are to read the description items, which are pretty small with fixed item size. Those call sites do not utilize the @dest_folio parameter. - For btrfs_read_merkle_tree_page() This is to read the BTRFS_VERITY_MERKLE_ITEM_KEY, which can be pretty large and split into multiple items. This is the only call site utilizing the @dest_folio parameter. Just for the only btrfs_read_merkle_tree_page() call site, we have a complex scheme for @dest and @dest_folio parameters. Since @dest can be NULL, it means if we pass @dest as NULL, then no matter if @dest_folio is provided, the merkle data will not be loaded into that @dest_folio. This can lead to a bug where a highmem folio is not mapped, then we pass folio_address(folio), which is NULL, into read_key_bytes(), causing no data to be written into @dest_folio. To address the complex scheme between @dest and @dest_folio, remove the @dest_folio parameter completely, and let the only caller to map the folio and pass the mapped kernel address into read_key_bytes() instead. This not only reduces the parameter list, but also make it much clear on the @dest parameter handling. The only downside is a longer duration of locally mapped page, but this should still be fine, as kmap_local_folio() can survive context switch. Reported-by: Hongling Zeng Link: https://lore.kernel.org/linux-btrfs/20260817022012.19658-1-zenghongling@kylinos.cn/ Fixes: 146054090b08 ("btrfs: initial fsverity support") Reviewed-by: David Sterba Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/verity.c | 31 ++++++++++++++----------------- 1 file changed, 14 insertions(+), 17 deletions(-) diff --git a/fs/btrfs/verity.c b/fs/btrfs/verity.c index 600337a84fbee5..83dc7dee14cfb3 100644 --- a/fs/btrfs/verity.c +++ b/fs/btrfs/verity.c @@ -286,21 +286,17 @@ static int write_key_bytes(struct btrfs_inode *inode, u8 key_type, u64 offset, * @dest: Buffer to read into. This parameter has slightly tricky * semantics. If it is NULL, the function will not do any copying * and will just return the size of all the items up to len bytes. - * If dest_page is passed, then the function will kmap_local the - * page and ignore dest, but it must still be non-NULL to avoid the - * counting-only behavior. * @len: length in bytes to read - * @dest_folio: copy into this folio instead of the dest buffer * * Helper function to read items from the btree. This returns the number of * bytes read or < 0 for errors. We can return short reads if the items don't * exist on disk or aren't big enough to fill the desired length. Supports - * reading into a provided buffer (dest) or into the page cache + * reading into a provided buffer (dest). * * Returns number of bytes read or a negative error code on failure. */ static int read_key_bytes(struct btrfs_inode *inode, u8 key_type, u64 offset, - char *dest, u64 len, struct folio *dest_folio) + char *dest, u64 len) { BTRFS_PATH_AUTO_FREE(path); struct btrfs_root *root = inode->root; @@ -320,7 +316,11 @@ static int read_key_bytes(struct btrfs_inode *inode, u8 key_type, u64 offset, if (!path) return -ENOMEM; - if (dest_folio) + /* + * Merkle items can be large and split across multiple items, so enable + * readahead for such cases. + */ + if (key_type == BTRFS_VERITY_MERKLE_ITEM_KEY) path->reada = READA_FORWARD; key.objectid = btrfs_ino(inode); @@ -364,7 +364,7 @@ static int read_key_bytes(struct btrfs_inode *inode, u8 key_type, u64 offset, break; } - /* desc = NULL to just sum all the item lengths */ + /* dest == NULL to just sum all the item lengths */ if (!dest) copy_end = item_end; else @@ -377,16 +377,10 @@ static int read_key_bytes(struct btrfs_inode *inode, u8 key_type, u64 offset, copy_offset = offset - key.offset; if (dest) { - if (dest_folio) - kaddr = kmap_local_folio(dest_folio, 0); - data = btrfs_item_ptr(leaf, path->slots[0], void); read_extent_buffer(leaf, kaddr + dest_offset, (unsigned long)data + copy_offset, copy_bytes); - - if (dest_folio) - kunmap_local(kaddr); } offset += copy_bytes; @@ -677,7 +671,7 @@ int btrfs_get_verity_descriptor(struct inode *inode, void *buf, size_t buf_size) memset(&item, 0, sizeof(item)); ret = read_key_bytes(BTRFS_I(inode), BTRFS_VERITY_DESC_ITEM_KEY, 0, - (char *)&item, sizeof(item), NULL); + (char *)&item, sizeof(item)); if (ret < 0) return ret; @@ -694,7 +688,7 @@ int btrfs_get_verity_descriptor(struct inode *inode, void *buf, size_t buf_size) return -ERANGE; ret = read_key_bytes(BTRFS_I(inode), BTRFS_VERITY_DESC_ITEM_KEY, 1, - buf, buf_size, NULL); + buf, buf_size); if (ret < 0) return ret; if (ret != true_size) @@ -720,6 +714,7 @@ static struct page *btrfs_read_merkle_tree_page(struct inode *inode, struct folio *folio; u64 off = (u64)index << PAGE_SHIFT; loff_t merkle_pos = merkle_file_pos(inode); + void *kaddr; int ret; if (merkle_pos < 0) @@ -763,6 +758,7 @@ static struct page *btrfs_read_merkle_tree_page(struct inode *inode, } read_folio: + kaddr = kmap_local_folio(folio, 0); /* * Merkle item keys are indexed from byte 0 in the merkle tree. * They have the form: @@ -770,7 +766,8 @@ static struct page *btrfs_read_merkle_tree_page(struct inode *inode, * [ inode objectid, BTRFS_MERKLE_ITEM_KEY, offset in bytes ] */ ret = read_key_bytes(BTRFS_I(inode), BTRFS_VERITY_MERKLE_ITEM_KEY, off, - folio_address(folio), PAGE_SIZE, folio); + kaddr, PAGE_SIZE); + kunmap_local(kaddr); if (ret < 0) { folio_unlock(folio); folio_put(folio); From 646c82ee14a23c846ed58f95cabcc2fa35e3dbb5 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Wed, 19 Aug 2026 10:36:16 +0930 Subject: [PATCH 0395/1352] btrfs: replace btrfs_repair_io_failure() to use bio for page iteration Currently btrfs_repair_io_failure() uses a @paddrs[] array to iterate pages. Such a parameter is required for bs > ps cases, as one fs block crosses several pages. However there is a much simpler and existing way to iterate pages: bio and bvec_iter. This changes btrfs_repair_io_failure() by: - Use a const @bvec_iter pointer to locate where the pages are - Extract file offset/logical from the @bbio - Require no @step parameter Above features allow us to shorten the parameter list. - Rename the function to btrfs_repair_bbio_failure() - Change the caller in btrfs_repair_eb_io_failure() to allocate a bbio Unlike the data read path, we do not have a handy bbio in that case. So we need to allocate one just for btrfs_repair_bbio_failure(). - Change the error reporting in btrfs_repair_bbio_failure() to include root id and use inode number directly Now for btree inode we will report a proper inode number (1). Reviewed-by: Boris Burkov Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/bio.c | 61 ++++++++++++++++++++++++++-------------------- fs/btrfs/bio.h | 5 ++-- fs/btrfs/disk-io.c | 25 +++++++++++++------ 3 files changed, 55 insertions(+), 36 deletions(-) diff --git a/fs/btrfs/bio.c b/fs/btrfs/bio.c index cc0bd03048bae6..f8d4c2d550073a 100644 --- a/fs/btrfs/bio.c +++ b/fs/btrfs/bio.c @@ -186,7 +186,6 @@ static void btrfs_end_repair_bio(struct btrfs_bio *repair_bbio, */ struct bvec_iter saved_iter = repair_bbio->saved_iter; const u32 step = min(fs_info->sectorsize, PAGE_SIZE); - const u64 logical = repair_bbio->saved_iter.bi_sector << SECTOR_SHIFT; const u32 nr_steps = repair_bbio->saved_iter.bi_size / step; int mirror = repair_bbio->mirror_num; phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; @@ -220,9 +219,8 @@ static void btrfs_end_repair_bio(struct btrfs_bio *repair_bbio, do { mirror = prev_repair_mirror(fbio, mirror); - btrfs_repair_io_failure(fs_info, btrfs_ino(inode), - repair_bbio->file_offset, fs_info->sectorsize, - logical, paddrs, step, mirror); + btrfs_repair_bbio_failure(repair_bbio, &repair_bbio->saved_iter, + fs_info->sectorsize, mirror); } while (mirror != fbio->bbio->mirror_num); done: @@ -925,21 +923,23 @@ void btrfs_submit_bbio(struct btrfs_bio *bbio, int mirror_num) * The I/O is issued synchronously to block the repair read completion from * freeing the bio. * - * @ino: Offending inode number - * @fileoff: File offset inside the inode + * @bbio: Original bbio where the repair is needed + * @orig_iter: Points to where the repair start is * @length: Length of the repair write - * @logical: Logical address of the range - * @paddrs: Physical address array of the content - * @step: Length of for each paddrs * @mirror_num: Mirror number to write to. Must not be zero */ -int btrfs_repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 fileoff, - u32 length, u64 logical, const phys_addr_t paddrs[], - unsigned int step, int mirror_num) +int btrfs_repair_bbio_failure(struct btrfs_bio *bbio, const struct bvec_iter *orig_iter, + u32 length, int mirror_num) { - const u32 nr_steps = DIV_ROUND_UP_POW2(length, step); + struct btrfs_inode *inode = bbio->inode; + struct btrfs_fs_info *fs_info = inode->root->fs_info; struct btrfs_io_stripe smap = { 0 }; - struct bio *bio = NULL; + struct bvec_iter iter = *orig_iter; + struct bio *repair_bio = NULL; + const u64 logical = iter.bi_sector << SECTOR_SHIFT; + const u64 fileoff = bbio->file_offset + + ((iter.bi_sector - bbio->saved_iter.bi_sector) << SECTOR_SHIFT); + u32 cur = 0; int ret = 0; BUG_ON(!mirror_num); @@ -950,8 +950,9 @@ int btrfs_repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 fileoff, ASSERT(IS_ALIGNED(fileoff, fs_info->sectorsize)); /* Either it's a single data or metadata block. */ ASSERT(length <= BTRFS_MAX_BLOCKSIZE); - ASSERT(step <= length); - ASSERT(is_power_of_2(step)); + + /* Our current iter should not be before the original bbio saved_iter. */ + ASSERT(iter.bi_sector >= bbio->saved_iter.bi_sector); /* * The fs either mounted RO or hit critical errors, no need @@ -979,15 +980,22 @@ int btrfs_repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 fileoff, goto out_counter_dec; } - bio = bio_alloc(smap.dev->bdev, nr_steps, REQ_OP_WRITE | REQ_SYNC, GFP_NOFS); - bio->bi_iter.bi_sector = smap.physical >> SECTOR_SHIFT; - for (int i = 0; i < nr_steps; i++) { - ret = bio_add_page(bio, phys_to_page(paddrs[i]), step, offset_in_page(paddrs[i])); - /* We should have allocated enough slots to contain all the different pages. */ - ASSERT(ret == step); + repair_bio = bio_alloc(smap.dev->bdev, max(1, length >> PAGE_SHIFT), + REQ_OP_WRITE | REQ_SYNC, GFP_NOFS); + repair_bio->bi_iter.bi_sector = smap.physical >> SECTOR_SHIFT; + while (cur < length) { + struct page *page = bio_iter_page(&bbio->bio, iter); + const u32 pg_off = bio_iter_offset(&bbio->bio, iter); + const u32 cur_len = min(bio_iter_len(&bbio->bio, iter), length - cur); + + ret = bio_add_page(repair_bio, page, cur_len, pg_off); + ASSERT(ret == cur_len); + bio_advance_iter_single(&bbio->bio, &iter, cur_len); + cur += cur_len; } - ret = submit_bio_wait(bio); - bio_put(bio); + + ret = submit_bio_wait(repair_bio); + bio_put(repair_bio); if (ret) { /* try to remap that extent elsewhere? */ btrfs_dev_stat_inc_and_print(smap.dev, BTRFS_DEV_STAT_WRITE_ERRS); @@ -995,8 +1003,9 @@ int btrfs_repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 fileoff, } btrfs_info_rl(fs_info, - "read error corrected: ino %llu off %llu (dev %s sector %llu)", - ino, fileoff, btrfs_dev_name(smap.dev), + "read error corrected: root %llu ino %llu off %llu (dev %s sector %llu)", + btrfs_root_id(inode->root), btrfs_ino(inode), fileoff, + btrfs_dev_name(smap.dev), smap.physical >> SECTOR_SHIFT); ret = 0; diff --git a/fs/btrfs/bio.h b/fs/btrfs/bio.h index 303ed6c7103d92..b7bd377a016249 100644 --- a/fs/btrfs/bio.h +++ b/fs/btrfs/bio.h @@ -126,8 +126,7 @@ void btrfs_bio_end_io(struct btrfs_bio *bbio, blk_status_t status); void btrfs_submit_bbio(struct btrfs_bio *bbio, int mirror_num); void btrfs_submit_repair_write(struct btrfs_bio *bbio, int mirror_num, bool dev_replace); -int btrfs_repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 fileoff, - u32 length, u64 logical, const phys_addr_t paddrs[], - unsigned int step, int mirror_num); +int btrfs_repair_bbio_failure(struct btrfs_bio *bbio, const struct bvec_iter *orig_iter, + u32 length, int mirror_num); #endif diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index dc7ad92876c0b3..0f3e8e74b7682e 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -176,19 +176,24 @@ static int btrfs_repair_eb_io_failure(const struct extent_buffer *eb, int mirror_num) { struct btrfs_fs_info *fs_info = eb->fs_info; - const u32 step = min(fs_info->nodesize, PAGE_SIZE); - const u32 nr_steps = eb->len / step; - phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; + struct btrfs_bio *bbio; + int ret; if (sb_rdonly(fs_info->sb)) return -EROFS; + /* + * This bbio is only to queue all pages for btrfs_repair_bbio_failure(). + * Thus it will never get its endio called. + */ + bbio = btrfs_bio_alloc(max(1, fs_info->nodesize >> PAGE_SHIFT), REQ_OP_READ, + BTRFS_I(fs_info->btree_inode), eb->start, NULL, NULL); + bbio->bio.bi_iter.bi_sector = eb->start >> SECTOR_SHIFT; for (int i = 0; i < num_extent_pages(eb); i++) { struct folio *folio = eb->folios[i]; /* No large folio support yet. */ ASSERT(folio_order(folio) == 0); - ASSERT(i < nr_steps); /* * For nodesize < page size, there is just one paddr, with some @@ -197,11 +202,17 @@ static int btrfs_repair_eb_io_failure(const struct extent_buffer *eb, * For nodesize >= page size, it's one or more paddrs, and eb->start * must be aligned to page boundary. */ - paddrs[i] = page_to_phys(&folio->page) + offset_in_page(eb->start); + ret = bio_add_page(&bbio->bio, &folio->page, min(PAGE_SIZE, fs_info->nodesize), + offset_in_page(eb->start)); + ASSERT(ret == min(PAGE_SIZE, fs_info->nodesize)); } + /* Since the bbio is never submitted, we have to save the iter manually. */ + bbio->saved_iter = bbio->bio.bi_iter; - return btrfs_repair_io_failure(fs_info, 0, eb->start, eb->len, - eb->start, paddrs, step, mirror_num); + ret = btrfs_repair_bbio_failure(bbio, &bbio->saved_iter, fs_info->nodesize, + mirror_num); + bio_put(&bbio->bio); + return ret; } /* From 37e78c8a6a5d82756e9f3429ef1d773d5c07228b Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Wed, 19 Aug 2026 10:36:17 +0930 Subject: [PATCH 0396/1352] btrfs: enhance btrfs_data_csum_ok() to use bio for page iteration Currently btrfs_data_csum_ok() requires a @paddr[] array to iterate all possible pages for bs > ps cases. However for all btrfs_data_csum_ok() call sites, we already have a btrfs_bio, and the bio infrastructure has many flexible ways to iterate multiple pages already. Change btrfs_data_csum_ok() to make full use of btrfs_bio by: - Change the parameter list to require a @bvec_iter pointer And remove @bio_offset, which can be calculated through @bvec_iter and bbio->saved_iter. Also remove paddrs[], we will iterate all the pages using bio interfaces. - Make the same parameter changes to repair_one_sector() - Use bio interfaces to iterate pages from a bio - Rename the function to btrfs_bio_data_csum_ok() - Remove on-stack paddrs[] array usage Reviewed-by: Boris Burkov Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/bio.c | 82 +++++++++++++++--------------------------- fs/btrfs/btrfs_inode.h | 4 +-- fs/btrfs/inode.c | 49 +++++++++++++++++++------ 3 files changed, 70 insertions(+), 65 deletions(-) diff --git a/fs/btrfs/bio.c b/fs/btrfs/bio.c index f8d4c2d550073a..19b4855969f536 100644 --- a/fs/btrfs/bio.c +++ b/fs/btrfs/bio.c @@ -180,29 +180,13 @@ static void btrfs_end_repair_bio(struct btrfs_bio *repair_bbio, struct btrfs_failed_bio *fbio = repair_bbio->private; struct btrfs_inode *inode = repair_bbio->inode; struct btrfs_fs_info *fs_info = inode->root->fs_info; - /* - * We can not move forward the saved_iter, as it will be later - * utilized by repair_bbio again. - */ - struct bvec_iter saved_iter = repair_bbio->saved_iter; - const u32 step = min(fs_info->sectorsize, PAGE_SIZE); - const u32 nr_steps = repair_bbio->saved_iter.bi_size / step; int mirror = repair_bbio->mirror_num; - phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; - phys_addr_t paddr; - unsigned int slot = 0; - /* Repair bbio should be eaxctly one block sized. */ + /* Repair bbio should be exactly one block sized. */ ASSERT(repair_bbio->saved_iter.bi_size == fs_info->sectorsize); - btrfs_bio_for_each_block(paddr, &repair_bbio->bio, &saved_iter, step) { - ASSERT(slot < nr_steps); - paddrs[slot] = paddr; - slot++; - } - if (repair_bbio->bio.bi_status || - !btrfs_data_csum_ok(repair_bbio, dev, 0, paddrs)) { + !btrfs_bio_data_csum_ok(repair_bbio, &repair_bbio->saved_iter, dev)) { bio_reset(&repair_bbio->bio, NULL, REQ_OP_READ); repair_bbio->bio.bi_iter = repair_bbio->saved_iter; @@ -236,25 +220,21 @@ static void btrfs_end_repair_bio(struct btrfs_bio *repair_bbio, * read succeeded to restore the redundancy. */ static struct btrfs_failed_bio *repair_one_sector(struct btrfs_bio *failed_bbio, - u32 bio_offset, - phys_addr_t paddrs[], + const struct bvec_iter *orig_iter, struct btrfs_failed_bio *fbio) { struct btrfs_inode *inode = failed_bbio->inode; struct btrfs_fs_info *fs_info = inode->root->fs_info; - const u32 sectorsize = fs_info->sectorsize; - const u32 step = min(fs_info->sectorsize, PAGE_SIZE); - const u32 nr_steps = sectorsize / step; - /* - * For bs > ps cases, the saved_iter can be partially moved forward. - * In that case we should round it down to the block boundary. - */ - const u64 logical = round_down(failed_bbio->saved_iter.bi_sector << SECTOR_SHIFT, - sectorsize); struct btrfs_bio *repair_bbio; struct bio *repair_bio; + struct bvec_iter iter = *orig_iter; + const u32 sectorsize = fs_info->sectorsize; + const u32 bio_offset = ((iter.bi_sector - failed_bbio->saved_iter.bi_sector) << + SECTOR_SHIFT); + const u64 logical = (iter.bi_sector << SECTOR_SHIFT); int num_copies; int mirror; + u32 cur = 0; btrfs_debug(fs_info, "repair read error: read error at %llu", failed_bbio->file_offset + bio_offset); @@ -275,17 +255,21 @@ static struct btrfs_failed_bio *repair_one_sector(struct btrfs_bio *failed_bbio, atomic_inc(&fbio->repair_count); - repair_bio = bio_alloc_bioset(NULL, nr_steps, REQ_OP_READ, GFP_NOFS, - &btrfs_repair_bioset); + repair_bio = bio_alloc_bioset(NULL, max(1, sectorsize >> PAGE_SHIFT), + REQ_OP_READ, GFP_NOFS, &btrfs_repair_bioset); repair_bio->bi_iter.bi_sector = logical >> SECTOR_SHIFT; - for (int i = 0; i < nr_steps; i++) { + while (cur < sectorsize) { + struct page *page = bio_iter_page(&failed_bbio->bio, iter); + const u32 pg_off = bio_iter_offset(&failed_bbio->bio, iter); + const u32 cur_len = min(bio_iter_len(&failed_bbio->bio, iter), + sectorsize - cur); int ret; - ASSERT(offset_in_page(paddrs[i]) + step <= PAGE_SIZE); + ret = bio_add_page(repair_bio, page, cur_len, pg_off); + ASSERT(ret == cur_len); - ret = bio_add_page(repair_bio, phys_to_page(paddrs[i]), step, - offset_in_page(paddrs[i])); - ASSERT(ret == step); + bio_advance_iter_single(&failed_bbio->bio, &iter, cur_len); + cur += cur_len; } repair_bbio = btrfs_bio(repair_bio); @@ -303,18 +287,16 @@ static void btrfs_check_read_bio(struct btrfs_bio *bbio, struct btrfs_device *de struct btrfs_inode *inode = bbio->inode; struct btrfs_fs_info *fs_info = inode->root->fs_info; const u32 sectorsize = fs_info->sectorsize; - const u32 step = min(sectorsize, PAGE_SIZE); - const u32 nr_steps = sectorsize / step; - struct bvec_iter *iter = &bbio->saved_iter; + struct bvec_iter iter; blk_status_t status = bbio->bio.bi_status; struct btrfs_failed_bio *fbio = NULL; - phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; - phys_addr_t paddr; - u32 offset = 0; /* Read-repair requires the inode field to be set by the submitter. */ ASSERT(inode); + /* The original bbio should be sectorsize aligned. */ + ASSERT(IS_ALIGNED(bbio->saved_iter.bi_size, sectorsize)); + /* * Hand off repair bios to the repair code as there is no upper level * submitter for them. @@ -327,16 +309,10 @@ static void btrfs_check_read_bio(struct btrfs_bio *bbio, struct btrfs_device *de /* Clear the I/O error. A failed repair will reset it. */ bbio->bio.bi_status = BLK_STS_OK; - btrfs_bio_for_each_block(paddr, &bbio->bio, iter, step) { - paddrs[(offset / step) % nr_steps] = paddr; - offset += step; - - if (IS_ALIGNED(offset, sectorsize)) { - if (status || - !btrfs_data_csum_ok(bbio, dev, offset - sectorsize, paddrs)) - fbio = repair_one_sector(bbio, offset - sectorsize, - paddrs, fbio); - } + for (iter = bbio->saved_iter; iter.bi_size; + bio_advance_iter(&bbio->bio, &iter, sectorsize)) { + if (status || !btrfs_bio_data_csum_ok(bbio, &iter, dev)) + fbio = repair_one_sector(bbio, &iter, fbio); } if (bbio->csum != bbio->csum_inline) kvfree(bbio->csum); @@ -924,7 +900,7 @@ void btrfs_submit_bbio(struct btrfs_bio *bbio, int mirror_num) * freeing the bio. * * @bbio: Original bbio where the repair is needed - * @orig_iter: Points to where the repair start is + * @orig_iter: Points to where the repair starts * @length: Length of the repair write * @mirror_num: Mirror number to write to. Must not be zero */ diff --git a/fs/btrfs/btrfs_inode.h b/fs/btrfs/btrfs_inode.h index 1082fa92c1457a..171f96bdb8aa76 100644 --- a/fs/btrfs/btrfs_inode.h +++ b/fs/btrfs/btrfs_inode.h @@ -513,8 +513,8 @@ void btrfs_calculate_block_csum_pages(struct btrfs_fs_info *fs_info, const phys_addr_t paddrs[], u8 *dest); int btrfs_check_block_csum(struct btrfs_fs_info *fs_info, phys_addr_t paddr, u8 *csum, const u8 * const csum_expected); -bool btrfs_data_csum_ok(struct btrfs_bio *bbio, struct btrfs_device *dev, - u32 bio_offset, const phys_addr_t paddrs[]); +bool btrfs_bio_data_csum_ok(struct btrfs_bio *bbio, const struct bvec_iter *orig_iter, + struct btrfs_device *dev); noinline int can_nocow_extent(struct btrfs_inode *inode, u64 offset, u64 *len, struct btrfs_file_extent *file_extent, bool nowait); diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 558b4a3f963364..0073e0a58fc1e6 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -3551,27 +3551,31 @@ int btrfs_check_block_csum(struct btrfs_fs_info *fs_info, phys_addr_t paddr, u8 * different noncontiguous pages. * * @bbio: btrfs_io_bio which contains the csum - * @dev: device the sector is on - * @bio_offset: offset to the beginning of the bio (in bytes) - * @paddrs: physical addresses which back the fs block + * @orig_iter: bvec iter pointing to the start of the block + * @dev: device the sector is on (optional) * * Check if the checksum on a data block is valid. When a checksum mismatch is * detected, report the error and fill the corrupted range with zero. * * Return %true if the sector is ok or had no checksum to start with, else %false. */ -bool btrfs_data_csum_ok(struct btrfs_bio *bbio, struct btrfs_device *dev, - u32 bio_offset, const phys_addr_t paddrs[]) +bool btrfs_bio_data_csum_ok(struct btrfs_bio *bbio, + const struct bvec_iter *orig_iter, + struct btrfs_device *dev) { struct btrfs_inode *inode = bbio->inode; struct btrfs_fs_info *fs_info = inode->root->fs_info; + struct bvec_iter iter = *orig_iter; + struct btrfs_csum_ctx cctx; const u32 blocksize = fs_info->sectorsize; - const u32 step = min(blocksize, PAGE_SIZE); - const u32 nr_steps = blocksize / step; + const u32 bio_offset = (iter.bi_sector - bbio->saved_iter.bi_sector) << SECTOR_SHIFT; u64 file_offset = bbio->file_offset + bio_offset; u64 end = file_offset + blocksize - 1; u8 *csum_expected; u8 csum[BTRFS_CSUM_SIZE]; + u32 cur = 0; + + ASSERT(iter.bi_sector >= bbio->saved_iter.bi_sector); if (!bbio->csum) return true; @@ -3587,7 +3591,22 @@ bool btrfs_data_csum_ok(struct btrfs_bio *bbio, struct btrfs_device *dev, csum_expected = bbio->csum + (bio_offset >> fs_info->sectorsize_bits) * fs_info->csum_size; - btrfs_calculate_block_csum_pages(fs_info, paddrs, csum); + btrfs_csum_init(&cctx, fs_info->csum_type); + while (cur < blocksize) { + struct page *page = bio_iter_page(&bbio->bio, iter); + const u32 pg_off = bio_iter_offset(&bbio->bio, iter); + const u32 cur_len = min(bio_iter_len(&bbio->bio, iter), blocksize - cur); + void *kaddr; + + kaddr = kmap_local_page(page) + pg_off; + btrfs_csum_update(&cctx, kaddr, cur_len); + kunmap_local(kaddr); + + bio_advance_iter_single(&bbio->bio, &iter, cur_len); + cur += cur_len; + } + btrfs_csum_final(&cctx, csum); + if (unlikely(memcmp(csum, csum_expected, fs_info->csum_size) != 0)) goto zeroit; return true; @@ -3597,8 +3616,18 @@ bool btrfs_data_csum_ok(struct btrfs_bio *bbio, struct btrfs_device *dev, bbio->mirror_num); if (dev) btrfs_dev_stat_inc_and_print(dev, BTRFS_DEV_STAT_CORRUPTION_ERRS); - for (int i = 0; i < nr_steps; i++) - memzero_page(phys_to_page(paddrs[i]), offset_in_page(paddrs[i]), step); + cur = 0; + iter = *orig_iter; + while (cur < blocksize) { + struct page *page = bio_iter_page(&bbio->bio, iter); + const u32 pg_off = bio_iter_offset(&bbio->bio, iter); + const u32 cur_len = min(bio_iter_len(&bbio->bio, iter), blocksize - cur); + + memzero_page(page, pg_off, cur_len); + + bio_advance_iter_single(&bbio->bio, &iter, cur_len); + cur += cur_len; + } return false; } From aa5cf1defb101618e88e45d3de93821dbfe7cffd Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Wed, 19 Aug 2026 10:36:18 +0930 Subject: [PATCH 0397/1352] btrfs: use a shared helper to calculate data checksum for a bio Since we are already calculating data checksum using bio interface, extract the generation part into btrfs_csum_one_bio_block(), and use that to replace the paddrs[] array based solution in csum_one_bio(). This will reduce 128 bytes on-stack memory usage for csum_one_bio(). Reviewed-by: Boris Burkov Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/btrfs_inode.h | 2 ++ fs/btrfs/file-item.c | 19 +++++------------ fs/btrfs/inode.c | 46 +++++++++++++++++++++++++----------------- 3 files changed, 34 insertions(+), 33 deletions(-) diff --git a/fs/btrfs/btrfs_inode.h b/fs/btrfs/btrfs_inode.h index 171f96bdb8aa76..e137a99151bd07 100644 --- a/fs/btrfs/btrfs_inode.h +++ b/fs/btrfs/btrfs_inode.h @@ -515,6 +515,8 @@ int btrfs_check_block_csum(struct btrfs_fs_info *fs_info, phys_addr_t paddr, u8 const u8 * const csum_expected); bool btrfs_bio_data_csum_ok(struct btrfs_bio *bbio, const struct bvec_iter *orig_iter, struct btrfs_device *dev); +void btrfs_csum_one_bio_block(struct btrfs_fs_info *fs_info, struct bio *bio, + const struct bvec_iter *orig_iter, u8 *csum); noinline int can_nocow_extent(struct btrfs_inode *inode, u64 offset, u64 *len, struct btrfs_file_extent *file_extent, bool nowait); diff --git a/fs/btrfs/file-item.c b/fs/btrfs/file-item.c index cf50fd623f41a8..581ca5653be93a 100644 --- a/fs/btrfs/file-item.c +++ b/fs/btrfs/file-item.c @@ -801,25 +801,16 @@ static void csum_one_bio(struct btrfs_bio *bbio, struct bvec_iter *src) { struct btrfs_inode *inode = bbio->inode; struct btrfs_fs_info *fs_info = inode->root->fs_info; - struct bio *bio = &bbio->bio; struct btrfs_ordered_sum *sums = bbio->sums; - struct bvec_iter iter = *src; - phys_addr_t paddr; + struct bvec_iter iter; const u32 blocksize = fs_info->sectorsize; - const u32 step = min(blocksize, PAGE_SIZE); - const u32 nr_steps = blocksize / step; - phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; - u32 offset = 0; int index = 0; - btrfs_bio_for_each_block(paddr, bio, &iter, step) { - paddrs[(offset / step) % nr_steps] = paddr; - offset += step; + for (iter = *src; iter.bi_size; bio_advance_iter(&bbio->bio, &iter, blocksize)) { + btrfs_csum_one_bio_block(fs_info, &bbio->bio, &iter, + sums->sums + index); - if (IS_ALIGNED(offset, blocksize)) { - btrfs_calculate_block_csum_pages(fs_info, paddrs, sums->sums + index); - index += fs_info->csum_size; - } + index += fs_info->csum_size; } } diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 0073e0a58fc1e6..62a0e626adfa9e 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -3546,6 +3546,32 @@ int btrfs_check_block_csum(struct btrfs_fs_info *fs_info, phys_addr_t paddr, u8 return 0; } +/* Generate data checksum for a single fs block, pointed to by @orig_iter. */ +void btrfs_csum_one_bio_block(struct btrfs_fs_info *fs_info, struct bio *bio, + const struct bvec_iter *orig_iter, u8 *csum) +{ + struct btrfs_csum_ctx cctx; + struct bvec_iter iter = *orig_iter; + const u32 blocksize = fs_info->sectorsize; + u32 cur = 0; + + btrfs_csum_init(&cctx, fs_info->csum_type); + while (cur < blocksize) { + struct page *page = bio_iter_page(bio, iter); + const u32 pg_off = bio_iter_offset(bio, iter); + const u32 cur_len = min(bio_iter_len(bio, iter), blocksize - cur); + void *kaddr; + + kaddr = kmap_local_page(page) + pg_off; + btrfs_csum_update(&cctx, kaddr, cur_len); + kunmap_local(kaddr); + + bio_advance_iter_single(bio, &iter, cur_len); + cur += cur_len; + } + btrfs_csum_final(&cctx, csum); +} + /* * Verify the checksum of a single data sector, which can be scattered at * different noncontiguous pages. @@ -3566,7 +3592,6 @@ bool btrfs_bio_data_csum_ok(struct btrfs_bio *bbio, struct btrfs_inode *inode = bbio->inode; struct btrfs_fs_info *fs_info = inode->root->fs_info; struct bvec_iter iter = *orig_iter; - struct btrfs_csum_ctx cctx; const u32 blocksize = fs_info->sectorsize; const u32 bio_offset = (iter.bi_sector - bbio->saved_iter.bi_sector) << SECTOR_SHIFT; u64 file_offset = bbio->file_offset + bio_offset; @@ -3591,22 +3616,7 @@ bool btrfs_bio_data_csum_ok(struct btrfs_bio *bbio, csum_expected = bbio->csum + (bio_offset >> fs_info->sectorsize_bits) * fs_info->csum_size; - btrfs_csum_init(&cctx, fs_info->csum_type); - while (cur < blocksize) { - struct page *page = bio_iter_page(&bbio->bio, iter); - const u32 pg_off = bio_iter_offset(&bbio->bio, iter); - const u32 cur_len = min(bio_iter_len(&bbio->bio, iter), blocksize - cur); - void *kaddr; - - kaddr = kmap_local_page(page) + pg_off; - btrfs_csum_update(&cctx, kaddr, cur_len); - kunmap_local(kaddr); - - bio_advance_iter_single(&bbio->bio, &iter, cur_len); - cur += cur_len; - } - btrfs_csum_final(&cctx, csum); - + btrfs_csum_one_bio_block(fs_info, &bbio->bio, orig_iter, csum); if (unlikely(memcmp(csum, csum_expected, fs_info->csum_size) != 0)) goto zeroit; return true; @@ -3616,8 +3626,6 @@ bool btrfs_bio_data_csum_ok(struct btrfs_bio *bbio, bbio->mirror_num); if (dev) btrfs_dev_stat_inc_and_print(dev, BTRFS_DEV_STAT_CORRUPTION_ERRS); - cur = 0; - iter = *orig_iter; while (cur < blocksize) { struct page *page = bio_iter_page(&bbio->bio, iter); const u32 pg_off = bio_iter_offset(&bbio->bio, iter); From 84c47359fc80b38be1edf14180605e328bb2969b Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Wed, 19 Aug 2026 10:36:19 +0930 Subject: [PATCH 0398/1352] btrfs: remove on-stack paddrs[] array usage Since the bs > ps support, we have to handle cases where a data block is inside several discontiguous pages. Thus we need a local paddrs[] array to assemble a data block for bs > ps cases. However to handle all possible bs/ps combinations, we have to declare such array using the max block size vs page size, no matter the current block size and page size. This adds 128 bytes on-stack memory usage for several call sites, and also introduced several duplicated helpers to calculate checksum for a data block: - btrfs_calculate_block_csum_folio() - btrfs_calculate_block_csum_pages() - btrfs_check_block_csum() The differences are mostly in how the data is passed. The first one accepts a contiguous paddr range. The second one accepts an array of paddrs[]. The last one is just a simple wrapper of the first one. However the most common interface to iterate a data block is through bio, and we have already converted most callers to use the bio based interface, e.g. btrfs_bio_data_csum_ok() and btrfs_csum_one_bio_block(). Convert the remaining two call sites to address the remaining paddrs[] usage: - btrfs_calculate_block_csum_pages() inside verify_bio_data_sectors() This can be switched to btrfs_csum_one_bio_block(). This removes the 128 bytes on-stack memory usage. - btrfs_calculate_block_csum_pages() inside verify_one_sector() This call site doesn't use on-stack memory for paddrs[], but reuses the existing btrfs_raid_bio::bio_paddrs[] or btrfs_raid_bio::stripe_paddrs[]. So implement a local version called calculate_block_csum_paddrs(). Now there is no fixed on-stack paddrs[] usage anymore. Reviewed-by: Boris Burkov Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/btrfs_inode.h | 6 ---- fs/btrfs/inode.c | 75 ------------------------------------------ fs/btrfs/raid56.c | 46 +++++++++++++++----------- 3 files changed, 27 insertions(+), 100 deletions(-) diff --git a/fs/btrfs/btrfs_inode.h b/fs/btrfs/btrfs_inode.h index e137a99151bd07..89e5e9c0c904f6 100644 --- a/fs/btrfs/btrfs_inode.h +++ b/fs/btrfs/btrfs_inode.h @@ -507,12 +507,6 @@ static inline void btrfs_set_inode_mapping_order(struct btrfs_inode *inode) inode->root->fs_info->block_max_order); } -void btrfs_calculate_block_csum_folio(struct btrfs_fs_info *fs_info, - const phys_addr_t paddr, u8 *dest); -void btrfs_calculate_block_csum_pages(struct btrfs_fs_info *fs_info, - const phys_addr_t paddrs[], u8 *dest); -int btrfs_check_block_csum(struct btrfs_fs_info *fs_info, phys_addr_t paddr, u8 *csum, - const u8 * const csum_expected); bool btrfs_bio_data_csum_ok(struct btrfs_bio *bbio, const struct bvec_iter *orig_iter, struct btrfs_device *dev); void btrfs_csum_one_bio_block(struct btrfs_fs_info *fs_info, struct bio *bio, diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 62a0e626adfa9e..5fd579a94fc1d3 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -3471,81 +3471,6 @@ int btrfs_finish_ordered_io(struct btrfs_ordered_extent *ordered) return btrfs_finish_one_ordered(ordered); } -/* - * Calculate the checksum of an fs block at physical memory address @paddr, - * and save the result to @dest. - * - * The folio containing @paddr must be large enough to contain a full fs block. - */ -void btrfs_calculate_block_csum_folio(struct btrfs_fs_info *fs_info, - const phys_addr_t paddr, u8 *dest) -{ - struct folio *folio = page_folio(phys_to_page(paddr)); - const u32 blocksize = fs_info->sectorsize; - const u32 step = min(blocksize, PAGE_SIZE); - const u32 nr_steps = blocksize / step; - phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; - - /* The full block must be inside the folio. */ - ASSERT(offset_in_folio(folio, paddr) + blocksize <= folio_size(folio)); - - for (int i = 0; i < nr_steps; i++) { - u32 pindex = offset_in_folio(folio, paddr + i * step) >> PAGE_SHIFT; - - /* - * For bs <= ps cases, we will only run the loop once, so the offset - * inside the page will only added to paddrs[0]. - * - * For bs > ps cases, the block must be page aligned, thus offset - * inside the page will always be 0. - */ - paddrs[i] = page_to_phys(folio_page(folio, pindex)) + offset_in_page(paddr); - } - return btrfs_calculate_block_csum_pages(fs_info, paddrs, dest); -} - -/* - * Calculate the checksum of a fs block backed by multiple noncontiguous pages - * at @paddrs[] and save the result to @dest. - * - * The folio containing @paddr must be large enough to contain a full fs block. - */ -void btrfs_calculate_block_csum_pages(struct btrfs_fs_info *fs_info, - const phys_addr_t paddrs[], u8 *dest) -{ - const u32 blocksize = fs_info->sectorsize; - const u32 step = min(blocksize, PAGE_SIZE); - const u32 nr_steps = blocksize / step; - struct btrfs_csum_ctx csum; - - btrfs_csum_init(&csum, fs_info->csum_type); - for (int i = 0; i < nr_steps; i++) { - const phys_addr_t paddr = paddrs[i]; - void *kaddr; - - ASSERT(offset_in_page(paddr) + step <= PAGE_SIZE); - kaddr = kmap_local_page(phys_to_page(paddr)) + offset_in_page(paddr); - btrfs_csum_update(&csum, kaddr, step); - kunmap_local(kaddr); - } - btrfs_csum_final(&csum, dest); -} - -/* - * Verify the checksum for a single sector without any extra action that depend - * on the type of I/O. - * - * @kaddr must be a properly kmapped address. - */ -int btrfs_check_block_csum(struct btrfs_fs_info *fs_info, phys_addr_t paddr, u8 *csum, - const u8 * const csum_expected) -{ - btrfs_calculate_block_csum_folio(fs_info, paddr, csum); - if (unlikely(memcmp(csum, csum_expected, fs_info->csum_size) != 0)) - return -EIO; - return 0; -} - /* Generate data checksum for a single fs block, pointed to by @orig_iter. */ void btrfs_csum_one_bio_block(struct btrfs_fs_info *fs_info, struct bio *bio, const struct bvec_iter *orig_iter, u8 *csum) diff --git a/fs/btrfs/raid56.c b/fs/btrfs/raid56.c index 1ee52a9dcee36f..a5d0ef09d92abb 100644 --- a/fs/btrfs/raid56.c +++ b/fs/btrfs/raid56.c @@ -1652,12 +1652,7 @@ static void verify_bio_data_sectors(struct btrfs_raid_bio *rbio, struct bio *bio) { struct btrfs_fs_info *fs_info = rbio->bioc->fs_info; - const u32 step = min(fs_info->sectorsize, PAGE_SIZE); - const u32 nr_steps = rbio->sector_nsteps; int total_sector_nr = get_bio_sector_nr(rbio, bio); - u32 offset = 0; - phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; - phys_addr_t paddr; /* No data csum for the whole stripe, no need to verify. */ if (!rbio->csum_bitmap || !rbio->csum_buf) @@ -1667,28 +1662,20 @@ static void verify_bio_data_sectors(struct btrfs_raid_bio *rbio, if (total_sector_nr >= rbio->nr_data * rbio->stripe_nsectors) return; - btrfs_bio_for_each_block_all(paddr, bio, step) { + for (struct bvec_iter iter = init_bvec_iter_for_bio(bio); + iter.bi_size; + bio_advance_iter(bio, &iter, fs_info->sectorsize), total_sector_nr++) { u8 csum_buf[BTRFS_CSUM_SIZE]; u8 *expected_csum; - paddrs[(offset / step) % nr_steps] = paddr; - offset += step; - - /* Not yet covering the full fs block, continue to the next step. */ - if (!IS_ALIGNED(offset, fs_info->sectorsize)) - continue; - /* No csum for this sector, skip to the next sector. */ - if (!test_bit(total_sector_nr, rbio->csum_bitmap)) { - total_sector_nr++; + if (!test_bit(total_sector_nr, rbio->csum_bitmap)) continue; - } expected_csum = rbio->csum_buf + total_sector_nr * fs_info->csum_size; - btrfs_calculate_block_csum_pages(fs_info, paddrs, csum_buf); + btrfs_csum_one_bio_block(fs_info, bio, &iter, csum_buf); if (unlikely(memcmp(csum_buf, expected_csum, fs_info->csum_size) != 0)) set_bit(total_sector_nr, rbio->error_bitmap); - total_sector_nr++; } } @@ -1879,6 +1866,27 @@ void raid56_parity_write(struct bio *bio, struct btrfs_io_context *bioc) start_async_work(rbio, rmw_rbio_work); } +static void calculate_block_csum_paddrs(struct btrfs_fs_info *fs_info, + const phys_addr_t paddrs[], u8 *dest) +{ + const u32 blocksize = fs_info->sectorsize; + const u32 step = min(blocksize, PAGE_SIZE); + const u32 nr_steps = blocksize / step; + struct btrfs_csum_ctx csum; + + btrfs_csum_init(&csum, fs_info->csum_type); + for (int i = 0; i < nr_steps; i++) { + const phys_addr_t paddr = paddrs[i]; + void *kaddr; + + ASSERT(offset_in_page(paddr) + step <= PAGE_SIZE); + kaddr = kmap_local_page(phys_to_page(paddr)) + offset_in_page(paddr); + btrfs_csum_update(&csum, kaddr, step); + kunmap_local(kaddr); + } + btrfs_csum_final(&csum, dest); +} + static int verify_one_sector(struct btrfs_raid_bio *rbio, int stripe_nr, int sector_nr) { @@ -1906,7 +1914,7 @@ static int verify_one_sector(struct btrfs_raid_bio *rbio, csum_expected = rbio->csum_buf + (stripe_nr * rbio->stripe_nsectors + sector_nr) * fs_info->csum_size; - btrfs_calculate_block_csum_pages(fs_info, paddrs, csum_buf); + calculate_block_csum_paddrs(fs_info, paddrs, csum_buf); if (unlikely(memcmp(csum_buf, csum_expected, fs_info->csum_size) != 0)) return -EIO; return 0; From 24bb0ce51985ec253383b59f7133ed1057af394a Mon Sep 17 00:00:00 2001 From: Breno Leitao Date: Mon, 24 Aug 2026 04:54:45 -0700 Subject: [PATCH 0399/1352] btrfs: skip extent tree lock in the shrinker for inodes without extent maps find_first_inode_to_shrink() takes inode->extent_tree.lock in write mode on every inode it walks, only to find out whether that inode has any extent maps. Most inodes have none, so the lock is taken and dropped again without any work being done. Check whether the tree is empty before taking the lock. tree->root is only modified with the tree lock held for write, so the unlocked read is a harmless race: a false empty just defers the inode to a later scan, which already happens whenever the write_trylock() below fails, and a false non-empty falls through to the existing check under the lock. Across the Meta production fleet the extent map shrinker is ~0.35% of non-idle kernel CPU. Attributing callees to their caller, find_first_inode_to_shrink() is ~65% of that, and the write_trylock() it does is ~30% of the whole shrinker. Micro benchmark: a 6 GiB btrfs on a loop device, 100000 empty files kept open, plus 200 1 MiB files created last so they get the highest inode numbers and every scan has to walk all the empty ones first. Each round drops the page cache, re-reads the data files to recreate the extent maps, then triggers the shrinker with "echo 2 > /proc/sys/vm/drop_caches". 15 rounds per run on ARM64 (Neoverse V2), 8 CPUs, no lock debugging. Cost of find_first_inode_to_shrink() from the ftrace function profiler, in ns per inode walked, median of runs: base patched delta idle 46.4 40.1 -13.6% 4 concurrent readers 47.8 38.4 -19.7% A separate build with CONFIG_LOCK_STAT, same test, for the extent map tree rwlock. The shrinker is not the only user of that lock, every extent map insert and lookup takes it too, which is why the acquisition count drops by two thirds rather than to nothing: base patched delta write acquisitions 628016 228000 -63.7% hold time total (us) 47512 22717 -52.2% acq cacheline bounces 1574 1288 -18.2% Signed-off-by: Breno Leitao Reviewed-by: Filipe Manana Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/extent_map.c | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/fs/btrfs/extent_map.c b/fs/btrfs/extent_map.c index 6ad7b39ae358b2..86d9c6f5ff4bd2 100644 --- a/fs/btrfs/extent_map.c +++ b/fs/btrfs/extent_map.c @@ -1219,6 +1219,14 @@ static struct btrfs_inode *find_first_inode_to_shrink(struct btrfs_root *root, tree = &inode->extent_tree; + /* + * Most inodes have no extent maps, so check without the lock. + * The race is harmless: a false empty just defers the inode to + * a later scan, and a false non-empty is caught under the lock. + */ + if (data_race(RB_EMPTY_ROOT(&tree->root))) + goto next; + /* * We want to be fast so if the lock is busy we don't want to * spend time waiting for it (some task is about to do IO for From 0e9d525c9decb26190c05abfe6ff95d3cce6e42c Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Tue, 25 Aug 2026 13:42:30 +0930 Subject: [PATCH 0400/1352] btrfs: remove unused variable flags from btrfs_read_qgroup_config() Since commit e562a8bdf652 ("btrfs: introduce BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN"), that @flags variable is no longer utilized. Just remove it. Reviewed-by: Johannes Thumshirn Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/qgroup.c | 2 -- 1 file changed, 2 deletions(-) diff --git a/fs/btrfs/qgroup.c b/fs/btrfs/qgroup.c index f68b696b4bf720..f3685cbb8f2e39 100644 --- a/fs/btrfs/qgroup.c +++ b/fs/btrfs/qgroup.c @@ -426,7 +426,6 @@ int btrfs_read_qgroup_config(struct btrfs_fs_info *fs_info) struct extent_buffer *l; int slot; int ret = 0; - u64 flags = 0; u64 rescan_progress = 0; if (!fs_info->quota_root) @@ -609,7 +608,6 @@ int btrfs_read_qgroup_config(struct btrfs_fs_info *fs_info) } out: btrfs_free_path(path); - fs_info->qgroup_flags |= flags; if (ret >= 0) { if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_ON) set_bit(BTRFS_FS_QUOTA_ENABLED, &fs_info->flags); From e3ff62f7e4e8f56dfcb488d76b627dc3053fa0e4 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Tue, 25 Aug 2026 13:42:31 +0930 Subject: [PATCH 0401/1352] btrfs: qgroup: use atomic operations for btrfs_fs_info::qgroup_flags Currently we define btrfs_fs_info::qgroup_flags as u64, to match the on-disk qgroup status item's flag. But for now we have only 4 bits utilized for that flag, and since it's u64 we have no way to properly use the existing atomic bit operations (requires an unsigned long pointer). This results in a lot of non-atomic operations inside qgroup code. Some maybe fine as other locks are involved, but still it's not a good practice. Remove those non-atomic operations by: - Re-define btrfs_fs_info::qgroup_flags as unsigned long - Define BTRFS_QGROUP_STATUS_BIT_* and BTRFS_QGROUP_RUNTIME_BIT_* Instead of the old value define the bit number. - Use set_bit()/clear_bit()/test_bit() to replace open-coded bit operations - Add one extra check at qgroup status item read time To make sure the on-disk flag is still inside ULONG_MAX. Otherwise reject the status item and disable qgroup. - Get rid of unnecessary spinlock when checking a single bit Reviewed-by: Johannes Thumshirn Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/fs.h | 2 +- fs/btrfs/ioctl.c | 2 +- fs/btrfs/qgroup.c | 97 ++++++++++++++++----------------- fs/btrfs/qgroup.h | 4 +- fs/btrfs/sysfs.c | 8 +-- include/uapi/linux/btrfs_tree.h | 21 ++++--- 6 files changed, 67 insertions(+), 67 deletions(-) diff --git a/fs/btrfs/fs.h b/fs/btrfs/fs.h index 10e15a319b93de..3eba8438593cde 100644 --- a/fs/btrfs/fs.h +++ b/fs/btrfs/fs.h @@ -811,7 +811,7 @@ struct btrfs_fs_info { struct btrfs_discard_ctl discard_ctl; /* Is qgroup tracking in a consistent state? */ - u64 qgroup_flags; + unsigned long qgroup_flags; /* Holds configuration and tracking. Protected by qgroup_lock. */ struct rb_root qgroup_tree; diff --git a/fs/btrfs/ioctl.c b/fs/btrfs/ioctl.c index e4b2da31a0d5de..54960351fbd171 100644 --- a/fs/btrfs/ioctl.c +++ b/fs/btrfs/ioctl.c @@ -3881,7 +3881,7 @@ static long btrfs_ioctl_quota_rescan_status(struct btrfs_fs_info *fs_info, if (!capable(CAP_SYS_ADMIN)) return -EPERM; - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_RESCAN) { + if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) { qsa.flags = 1; qsa.progress = fs_info->qgroup_rescan_progress.objectid; } diff --git a/fs/btrfs/qgroup.c b/fs/btrfs/qgroup.c index f3685cbb8f2e39..2c2ac0f16b1e5b 100644 --- a/fs/btrfs/qgroup.c +++ b/fs/btrfs/qgroup.c @@ -34,7 +34,7 @@ enum btrfs_qgroup_mode btrfs_qgroup_mode(const struct btrfs_fs_info *fs_info) { if (!test_bit(BTRFS_FS_QUOTA_ENABLED, &fs_info->flags)) return BTRFS_QGROUP_MODE_DISABLED; - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE) + if (test_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags)) return BTRFS_QGROUP_MODE_SIMPLE; return BTRFS_QGROUP_MODE_FULL; } @@ -384,14 +384,14 @@ static bool squota_check_parent_usage(struct btrfs_fs_info *fs_info, struct btrf __printf(2, 3) static void qgroup_mark_inconsistent(struct btrfs_fs_info *fs_info, const char *fmt, ...) { - const u64 old_flags = fs_info->qgroup_flags; + const unsigned long old_flags = fs_info->qgroup_flags; if (btrfs_qgroup_mode(fs_info) == BTRFS_QGROUP_MODE_SIMPLE) return; - fs_info->qgroup_flags |= (BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT | - BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN | - BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING); - if (!(old_flags & BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT)) { + set_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); + set_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags); + set_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, &fs_info->qgroup_flags); + if (!test_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &old_flags)) { struct va_format vaf; va_list args; @@ -472,8 +472,12 @@ int btrfs_read_qgroup_config(struct btrfs_fs_info *fs_info) "old qgroup version, quota disabled"); goto out; } - fs_info->qgroup_flags = btrfs_qgroup_status_flags(l, ptr); - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE) + if (btrfs_qgroup_status_flags(l, ptr) > ULONG_MAX) { + btrfs_err(fs_info, "invalid qgroup status flags, quota disabled"); + goto out; + } + fs_info->qgroup_flags = (unsigned long)btrfs_qgroup_status_flags(l, ptr); + if (test_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags)) qgroup_read_enable_gen(fs_info, l, slot, ptr); else if (btrfs_qgroup_status_generation(l, ptr) != fs_info->generation) qgroup_mark_inconsistent(fs_info, "qgroup generation mismatch"); @@ -609,12 +613,12 @@ int btrfs_read_qgroup_config(struct btrfs_fs_info *fs_info) out: btrfs_free_path(path); if (ret >= 0) { - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_ON) + if (test_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags)) set_bit(BTRFS_FS_QUOTA_ENABLED, &fs_info->flags); - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_RESCAN) + if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) ret = qgroup_rescan_init(fs_info, rescan_progress, 0); } else { - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_RESCAN; + clear_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags); btrfs_sysfs_del_qgroups(fs_info); } @@ -1099,9 +1103,9 @@ int btrfs_quota_enable(struct btrfs_fs_info *fs_info, struct btrfs_qgroup_status_item); btrfs_set_qgroup_status_generation(leaf, ptr, trans->transid); btrfs_set_qgroup_status_version(leaf, ptr, BTRFS_QGROUP_STATUS_VERSION); - fs_info->qgroup_flags = BTRFS_QGROUP_STATUS_FLAG_ON; + fs_info->qgroup_flags = (1UL << BTRFS_QGROUP_STATUS_BIT_ON); if (simple) { - fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE; + set_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags); btrfs_set_fs_incompat(fs_info, SIMPLE_QUOTA); /* * Set the enable generation to the next transaction, as we cannot @@ -1111,7 +1115,7 @@ int btrfs_quota_enable(struct btrfs_fs_info *fs_info, */ btrfs_set_qgroup_status_enable_gen(leaf, ptr, trans->transid + 1); } else { - fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT; + set_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); } btrfs_set_qgroup_status_flags(leaf, ptr, fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAGS_MASK); @@ -1401,8 +1405,8 @@ int btrfs_quota_disable(struct btrfs_fs_info *fs_info) spin_lock(&fs_info->qgroup_lock); quota_root = fs_info->quota_root; fs_info->quota_root = NULL; - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_ON; - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE; + clear_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags); + clear_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags); fs_info->qgroup_drop_subtree_thres = BTRFS_QGROUP_DROP_SUBTREE_THRES_DEFAULT; spin_unlock(&fs_info->qgroup_lock); @@ -1552,7 +1556,7 @@ static int quick_update_accounting(struct btrfs_fs_info *fs_info, } out: if (ret) - fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT; + set_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); return ret; } @@ -1873,7 +1877,7 @@ int btrfs_remove_qgroup(struct btrfs_trans_handle *trans, u64 qgroupid) * very frequently. */ if (btrfs_qgroup_mode(fs_info) == BTRFS_QGROUP_MODE_FULL && - !(fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT)) { + !test_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags)) { if (unlikely(qgroup->rfer || qgroup->excl || qgroup->rfer_cmpr || qgroup->excl_cmpr)) { DEBUG_WARN(); @@ -2118,7 +2122,7 @@ int btrfs_qgroup_trace_extent_post(struct btrfs_trans_handle *trans, */ ASSERT(trans != NULL); - if (fs_info->qgroup_flags & BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING) + if (test_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, &fs_info->qgroup_flags)) return 0; ret = btrfs_find_all_roots(&ctx, true); @@ -2959,7 +2963,7 @@ int btrfs_qgroup_account_extent(struct btrfs_trans_handle *trans, u64 bytenr, * we can't just exit here. */ if (!btrfs_qgroup_full_accounting(fs_info) || - fs_info->qgroup_flags & BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING) + test_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, &fs_info->qgroup_flags)) goto out_free; if (new_roots) { @@ -2981,7 +2985,7 @@ int btrfs_qgroup_account_extent(struct btrfs_trans_handle *trans, u64 bytenr, num_bytes, nr_old_roots, nr_new_roots); mutex_lock(&fs_info->qgroup_rescan_lock); - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_RESCAN) { + if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) { if (fs_info->qgroup_rescan_progress.objectid <= bytenr) { mutex_unlock(&fs_info->qgroup_rescan_lock); ret = 0; @@ -3042,8 +3046,8 @@ int btrfs_qgroup_account_extents(struct btrfs_trans_handle *trans) num_dirty_extents++; trace_btrfs_qgroup_account_extents(fs_info, record, bytenr); - if (!ret && !(fs_info->qgroup_flags & - BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING)) { + if (!ret && !test_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, + &fs_info->qgroup_flags)) { struct btrfs_backref_walk_ctx ctx = { 0 }; ctx.bytenr = bytenr; @@ -3150,9 +3154,9 @@ int btrfs_run_qgroups(struct btrfs_trans_handle *trans) spin_lock(&fs_info->qgroup_lock); } if (btrfs_qgroup_enabled(fs_info)) - fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_ON; + set_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags); else - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_ON; + clear_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags); spin_unlock(&fs_info->qgroup_lock); ret = update_qgroup_status_item(trans); @@ -3842,7 +3846,7 @@ static bool rescan_should_stop(struct btrfs_fs_info *fs_info) return true; if (!btrfs_qgroup_enabled(fs_info)) return true; - if (fs_info->qgroup_flags & BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN) + if (test_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags)) return true; return false; } @@ -3892,12 +3896,10 @@ static void btrfs_qgroup_rescan_worker(struct btrfs_work *work) btrfs_free_path(path); mutex_lock(&fs_info->qgroup_rescan_lock); - if (ret > 0 && - fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT) { - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT; - } else if (ret < 0 || stopped) { - fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT; - } + if (ret > 0) + clear_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); + else if (ret < 0 || stopped) + set_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); mutex_unlock(&fs_info->qgroup_rescan_lock); /* @@ -3921,9 +3923,9 @@ static void btrfs_qgroup_rescan_worker(struct btrfs_work *work) } mutex_lock(&fs_info->qgroup_rescan_lock); - if (!stopped || - fs_info->qgroup_flags & BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN) - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_RESCAN; + if (!stopped || test_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, + &fs_info->qgroup_flags)) + clear_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags); if (trans) { int ret2 = update_qgroup_status_item(trans); @@ -3933,7 +3935,7 @@ static void btrfs_qgroup_rescan_worker(struct btrfs_work *work) } } fs_info->qgroup_rescan_running = false; - fs_info->qgroup_flags &= ~BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN; + clear_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags); complete_all(&fs_info->qgroup_rescan_completion); mutex_unlock(&fs_info->qgroup_rescan_lock); @@ -3944,7 +3946,7 @@ static void btrfs_qgroup_rescan_worker(struct btrfs_work *work) if (stopped) { btrfs_info(fs_info, "qgroup scan paused"); - } else if (fs_info->qgroup_flags & BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN) { + } else if (test_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags)) { btrfs_info(fs_info, "qgroup scan cancelled"); } else if (ret >= 0) { btrfs_info(fs_info, "qgroup scan completed%s", @@ -3971,13 +3973,11 @@ qgroup_rescan_init(struct btrfs_fs_info *fs_info, u64 progress_objectid, if (!init_flags) { /* we're resuming qgroup rescan at mount time */ - if (!(fs_info->qgroup_flags & - BTRFS_QGROUP_STATUS_FLAG_RESCAN)) { + if (!(test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags))) { btrfs_debug(fs_info, "qgroup rescan init failed, qgroup rescan is not queued"); ret = -EINVAL; - } else if (!(fs_info->qgroup_flags & - BTRFS_QGROUP_STATUS_FLAG_ON)) { + } else if (!(test_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags))) { btrfs_debug(fs_info, "qgroup rescan init failed, qgroup is not enabled"); ret = -ENOTCONN; @@ -3990,10 +3990,9 @@ qgroup_rescan_init(struct btrfs_fs_info *fs_info, u64 progress_objectid, mutex_lock(&fs_info->qgroup_rescan_lock); if (init_flags) { - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_RESCAN) { + if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) { ret = -EINPROGRESS; - } else if (!(fs_info->qgroup_flags & - BTRFS_QGROUP_STATUS_FLAG_ON)) { + } else if (!test_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags)) { btrfs_debug(fs_info, "qgroup rescan init failed, qgroup is not enabled"); ret = -ENOTCONN; @@ -4006,13 +4005,13 @@ qgroup_rescan_init(struct btrfs_fs_info *fs_info, u64 progress_objectid, mutex_unlock(&fs_info->qgroup_rescan_lock); return ret; } - fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_RESCAN; + set_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags); } memset(&fs_info->qgroup_rescan_progress, 0, sizeof(fs_info->qgroup_rescan_progress)); - fs_info->qgroup_flags &= ~(BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN | - BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING); + clear_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags); + clear_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, &fs_info->qgroup_flags); fs_info->qgroup_rescan_progress.objectid = progress_objectid; init_completion(&fs_info->qgroup_rescan_completion); mutex_unlock(&fs_info->qgroup_rescan_lock); @@ -4063,7 +4062,7 @@ btrfs_qgroup_rescan(struct btrfs_fs_info *fs_info) ret = btrfs_commit_current_transaction(fs_info->fs_root); if (ret) { - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_RESCAN; + clear_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags); return ret; } @@ -4116,7 +4115,7 @@ int btrfs_qgroup_wait_for_completion(struct btrfs_fs_info *fs_info, void btrfs_qgroup_rescan_resume(struct btrfs_fs_info *fs_info) { - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_RESCAN) { + if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) { mutex_lock(&fs_info->qgroup_rescan_lock); fs_info->qgroup_rescan_running = true; btrfs_queue_work(fs_info->qgroup_rescan_workers, diff --git a/fs/btrfs/qgroup.h b/fs/btrfs/qgroup.h index 80dd2dacd56db4..b3aaad5e617d51 100644 --- a/fs/btrfs/qgroup.h +++ b/fs/btrfs/qgroup.h @@ -121,8 +121,8 @@ struct btrfs_qgroup_swapped_blocks; * To minimize the chance of collision with new persisted status flags, these * count backwards from the MSB. */ -#define BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN (1ULL << 63) -#define BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING (1ULL << 62) +#define BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN (BITS_PER_LONG - 1) +#define BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING (BITS_PER_LONG - 2) #define BTRFS_QGROUP_DROP_SUBTREE_THRES_DEFAULT (3) diff --git a/fs/btrfs/sysfs.c b/fs/btrfs/sysfs.c index 39cb01ee441ab8..1df6340a71234f 100644 --- a/fs/btrfs/sysfs.c +++ b/fs/btrfs/sysfs.c @@ -2359,9 +2359,7 @@ static ssize_t qgroup_enabled_show(struct kobject *qgroups_kobj, struct btrfs_fs_info *fs_info = to_fs_info(qgroups_kobj->parent); bool enabled; - spin_lock(&fs_info->qgroup_lock); - enabled = fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_ON; - spin_unlock(&fs_info->qgroup_lock); + enabled = test_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags); return sysfs_emit(buf, "%d\n", enabled); } @@ -2401,9 +2399,7 @@ static ssize_t qgroup_inconsistent_show(struct kobject *qgroups_kobj, struct btrfs_fs_info *fs_info = to_fs_info(qgroups_kobj->parent); bool inconsistent; - spin_lock(&fs_info->qgroup_lock); - inconsistent = (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT); - spin_unlock(&fs_info->qgroup_lock); + inconsistent = test_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); return sysfs_emit(buf, "%d\n", inconsistent); } diff --git a/include/uapi/linux/btrfs_tree.h b/include/uapi/linux/btrfs_tree.h index cc3b9f7dccafa2..b6ccaf848e4b3e 100644 --- a/include/uapi/linux/btrfs_tree.h +++ b/include/uapi/linux/btrfs_tree.h @@ -1255,13 +1255,16 @@ static inline __u16 btrfs_qgroup_level(__u64 qgroupid) } /* - * is subvolume quota turned on? - */ -#define BTRFS_QGROUP_STATUS_FLAG_ON (1ULL << 0) -/* - * RESCAN is set during the initialization phase + * The following BTRFS_QGROUP_STATUS_BIT_* are for * btrfs_qgroup_status_item::flags. + * + * Is subvolume quota turned on? */ -#define BTRFS_QGROUP_STATUS_FLAG_RESCAN (1ULL << 1) +#define BTRFS_QGROUP_STATUS_BIT_ON (0) +#define BTRFS_QGROUP_STATUS_FLAG_ON (1UL << BTRFS_QGROUP_STATUS_BIT_ON) + +/* RESCAN is set during the initialization phase */ +#define BTRFS_QGROUP_STATUS_BIT_RESCAN (1) +#define BTRFS_QGROUP_STATUS_FLAG_RESCAN (1UL << BTRFS_QGROUP_STATUS_BIT_RESCAN) /* * Some qgroup entries are known to be out of date, * either because the configuration has changed in a way that @@ -1269,14 +1272,16 @@ static inline __u16 btrfs_qgroup_level(__u64 qgroupid) * with a non-qgroup-aware version. * Turning qouta off and on again makes it inconsistent, too. */ -#define BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT (1ULL << 2) +#define BTRFS_QGROUP_STATUS_BIT_INCONSISTENT (2) +#define BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT (1UL << BTRFS_QGROUP_STATUS_BIT_INCONSISTENT) /* * Whether or not this filesystem is using simple quotas. Not exactly the * incompat bit, because we support using simple quotas, disabling it, then * going back to full qgroup quotas. */ -#define BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE (1ULL << 3) +#define BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE (3) +#define BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE (1UL << BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE) #define BTRFS_QGROUP_STATUS_FLAGS_MASK (BTRFS_QGROUP_STATUS_FLAG_ON | \ BTRFS_QGROUP_STATUS_FLAG_RESCAN | \ From dc395dd7b79656a3dac8590bdb15abde7568caba Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Thu, 27 Aug 2026 16:25:29 +0930 Subject: [PATCH 0402/1352] btrfs: reject new qgroup rescan during subvolume dropping Commit 011b46c30476 ("btrfs: skip subtree scan if it's too high to avoid low stall in btrfs_commit_transaction()") introduced a threshold to skip huge subtree scan during subvolume dropping. But that's not covering all cases, e.g. rescan can still be started immediately after that huge subtree skipping. This will cause rescan to do the same accounting for that subtree anyway, still causing a long stall during transaction commit. Introduce a new runtime qgroup flag, BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN, so that during cleanup of a subvolume, no new qgroup rescan can be initiated. The rejection uses the same -EINPROGRESS, as if there is already a running qgroup rescan. And since we have the extra bit, we can no longer allow plain assignment in btrfs_quota_enable(), as the plain assignment will override the REJECT_RESCAN bit. To co-operate this new flag: - Make btrfs_quota_enable() to only set BTRFS_QGROUP_STATUS_BIT_ON So it won't override the existing BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN bit. - Make btrfs_quota_disable() to clear every non-rescan bit This includes: * BTRFS_QGROUP_STATUS_BIT_ON * BTRFS_QGROUP_STATUS_BIT_INCONSISTENT * BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING For rescan related bits, they are either cleared by the rescan thread, or by the caller who rejects rescan. Reviewed-by: Boris Burkov Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/disk-io.c | 2 ++ fs/btrfs/qgroup.c | 13 +++++++++++-- fs/btrfs/qgroup.h | 11 +++++++++++ 3 files changed, 24 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index 0f3e8e74b7682e..94a7e9059a7b3f 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -1496,7 +1496,9 @@ static int cleaner_kthread(void *arg) btrfs_run_delayed_iputs(fs_info); + set_bit(BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN, &fs_info->qgroup_flags); again = btrfs_clean_one_deleted_snapshot(fs_info); + clear_bit(BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN, &fs_info->qgroup_flags); mutex_unlock(&fs_info->cleaner_mutex); /* diff --git a/fs/btrfs/qgroup.c b/fs/btrfs/qgroup.c index 2c2ac0f16b1e5b..e01b31aa0b1bf0 100644 --- a/fs/btrfs/qgroup.c +++ b/fs/btrfs/qgroup.c @@ -1103,7 +1103,7 @@ int btrfs_quota_enable(struct btrfs_fs_info *fs_info, struct btrfs_qgroup_status_item); btrfs_set_qgroup_status_generation(leaf, ptr, trans->transid); btrfs_set_qgroup_status_version(leaf, ptr, BTRFS_QGROUP_STATUS_VERSION); - fs_info->qgroup_flags = (1UL << BTRFS_QGROUP_STATUS_BIT_ON); + set_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags); if (simple) { set_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags); btrfs_set_fs_incompat(fs_info, SIMPLE_QUOTA); @@ -1405,8 +1405,14 @@ int btrfs_quota_disable(struct btrfs_fs_info *fs_info) spin_lock(&fs_info->qgroup_lock); quota_root = fs_info->quota_root; fs_info->quota_root = NULL; + /* + * Clear all on-disk and runtime bits, except RESCAN related ones, that + * are either handled by rescan thread, or the caller who rejects rescan. + */ clear_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags); clear_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags); + clear_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); + clear_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, &fs_info->qgroup_flags); fs_info->qgroup_drop_subtree_thres = BTRFS_QGROUP_DROP_SUBTREE_THRES_DEFAULT; spin_unlock(&fs_info->qgroup_lock); @@ -3990,7 +3996,10 @@ qgroup_rescan_init(struct btrfs_fs_info *fs_info, u64 progress_objectid, mutex_lock(&fs_info->qgroup_rescan_lock); if (init_flags) { - if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) { + if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, + &fs_info->qgroup_flags) || + test_bit(BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN, + &fs_info->qgroup_flags)) { ret = -EINPROGRESS; } else if (!test_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags)) { btrfs_debug(fs_info, diff --git a/fs/btrfs/qgroup.h b/fs/btrfs/qgroup.h index b3aaad5e617d51..090ba536787269 100644 --- a/fs/btrfs/qgroup.h +++ b/fs/btrfs/qgroup.h @@ -124,6 +124,17 @@ struct btrfs_qgroup_swapped_blocks; #define BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN (BITS_PER_LONG - 1) #define BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING (BITS_PER_LONG - 2) +/* + * No new rescan allowed when set. + * + * During huge subtree dropping, qgroup will be marked inconsistent, and skip + * all future accounting to avoid long stall. But, an immediate rescan will + * re-enable qgroup and still stall the system. + * + * This bit is to avoid such rescan during the duration of a subvolume dropping. + */ +#define BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN (BITS_PER_LONG - 3) + #define BTRFS_QGROUP_DROP_SUBTREE_THRES_DEFAULT (3) /* From d3e5caaa71ec13a267f39244bcc4ca73a221c57c Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Thu, 27 Aug 2026 16:25:30 +0930 Subject: [PATCH 0403/1352] btrfs: avoid long stall when dropping a non-shared large subvolume Commit 011b46c30476 ("btrfs: skip subtree scan if it's too high to avoid low stall in btrfs_commit_transaction()") introduced a mechanism to skip large subtree during snapshot dropping. But even for a subvolume without any shared tree blocks, we can still queue quite a lot of qgroup records into one transaction, and cause a long qgroup related stall. So also add a check against the subvolume root level, to determine if we need to mark qgroup inconsistent. Reviewed-by: Boris Burkov Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/extent-tree.c | 10 ++++++++++ fs/btrfs/qgroup.c | 18 ++++++++++++++++++ fs/btrfs/qgroup.h | 1 + 3 files changed, 29 insertions(+) diff --git a/fs/btrfs/extent-tree.c b/fs/btrfs/extent-tree.c index d6a4390ee34ac9..a0d5ab03aae264 100644 --- a/fs/btrfs/extent-tree.c +++ b/fs/btrfs/extent-tree.c @@ -6315,6 +6315,16 @@ int btrfs_drop_snapshot(struct btrfs_root *root, bool update_ref, bool for_reloc set_bit(BTRFS_ROOT_DELETING, &root->state); unfinished_drop = test_bit(BTRFS_ROOT_UNFINISHED_DROP, &root->state); + /* + * For subvolume dropping, check if the subvolume is large enough so + * that we need to mark qgroup inconsistent to avoid long qgroup stall. + * + * Even for a subvolume without any snapshot, there can still be + * a lot of qgroup records queued into one transaction. + */ + if (!for_reloc) + btrfs_qgroup_check_tree_drop(fs_info, rootid, + btrfs_header_level(root->node)); if (btrfs_disk_key_objectid(&root_item->drop_progress) == 0) { level = btrfs_header_level(root->node); path->nodes[level] = btrfs_lock_root_node(root); diff --git a/fs/btrfs/qgroup.c b/fs/btrfs/qgroup.c index e01b31aa0b1bf0..05e35eb126dc5b 100644 --- a/fs/btrfs/qgroup.c +++ b/fs/btrfs/qgroup.c @@ -2748,6 +2748,24 @@ int btrfs_qgroup_trace_subtree(struct btrfs_trans_handle *trans, return 0; } +void btrfs_qgroup_check_tree_drop(struct btrfs_fs_info *fs_info, u64 rootid, u8 level) +{ + u8 drop_subtree_thres; + + if (btrfs_qgroup_mode(fs_info) != BTRFS_QGROUP_MODE_FULL) + return; + + if (!btrfs_is_fstree(rootid)) + return; + + spin_lock(&fs_info->qgroup_lock); + drop_subtree_thres = fs_info->qgroup_drop_subtree_thres; + spin_unlock(&fs_info->qgroup_lock); + + if (level >= drop_subtree_thres) + qgroup_mark_inconsistent(fs_info, "subtree level reached threshold"); +} + static void qgroup_iterator_nested_add(struct list_head *head, struct btrfs_qgroup *qgroup) { if (!list_empty(&qgroup->nested_iterator)) diff --git a/fs/btrfs/qgroup.h b/fs/btrfs/qgroup.h index 090ba536787269..c64b26b09c22f2 100644 --- a/fs/btrfs/qgroup.h +++ b/fs/btrfs/qgroup.h @@ -376,6 +376,7 @@ int btrfs_qgroup_trace_leaf_items(struct btrfs_trans_handle *trans, int btrfs_qgroup_trace_subtree(struct btrfs_trans_handle *trans, struct extent_buffer *root_eb, u64 root_gen, int root_level); +void btrfs_qgroup_check_tree_drop(struct btrfs_fs_info *fs_info, u64 rootid, u8 level); int btrfs_qgroup_account_extent(struct btrfs_trans_handle *trans, u64 bytenr, u64 num_bytes, struct ulist *old_roots, struct ulist *new_roots); From 965877eb94c8e6c065292bfd1065f522fde1017e Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Fri, 21 Aug 2026 19:42:10 +0930 Subject: [PATCH 0404/1352] btrfs: remove runtime tweakable feature sysfs interface There are 2 features that are marked runtime tweakable inside /sys/fs/btrfs/features/ - acl Which is a mount option, and it will not show up in /sys/fs/btrfs//features/ directory anyway. - extended_iref This feature can only be enabled, but not disabled at runtime. Furthermore it's already the default behavior since 3.12. So it means this feature is always enabled and cannot be disabled for modern btrfs. So there is no need to maintain the ability to modify btrfs' runtime features through sysfs. And furthermore, the existing btrfs_feature_attr_store() is race-prone, it relies on fs_info->transaction_kthread, but our sysfs interfaces are enabled before transaction_kthread. Meaning at mount time a sysfs write can trigger NULL pointer dereference if the transaction_kthread is not yet initialized. The opposite is also possible during unmount. Thankfully that race is not possible in the real world, as the only supported feature is already enabled. But it also means we do not really need to keep the race-prone infrastructure, so just remove it completely, and make the per-module and per-mount features files to be completely read-only. Even with the sysfs tweakable features removed, we can still enable extended_iref feature through ioctl. Reviewed-by: Boris Burkov Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/sysfs.c | 121 ++--------------------------------------------- 1 file changed, 4 insertions(+), 117 deletions(-) diff --git a/fs/btrfs/sysfs.c b/fs/btrfs/sysfs.c index 1df6340a71234f..c5bb1c7eac6afe 100644 --- a/fs/btrfs/sysfs.c +++ b/fs/btrfs/sysfs.c @@ -83,8 +83,7 @@ struct raid_kobject { #define BTRFS_FEAT_ATTR(_name, _feature_set, _feature_prefix, _feature_bit) \ static struct btrfs_feature_attr btrfs_attr_features_##_name = { \ .kobj_attr = __INIT_KOBJ_ATTR(_name, S_IRUGO, \ - btrfs_feature_attr_show, \ - btrfs_feature_attr_store), \ + btrfs_feature_attr_show, NULL), \ .feature_set = _feature_set, \ .feature_bit = _feature_prefix ##_## _feature_bit, \ } @@ -130,130 +129,20 @@ static u64 get_features(struct btrfs_fs_info *fs_info, return btrfs_super_incompat_flags(disk_super); } -static void set_features(struct btrfs_fs_info *fs_info, - enum btrfs_feature_set set, u64 features) -{ - struct btrfs_super_block *disk_super = fs_info->super_copy; - if (set == FEAT_COMPAT) - btrfs_set_super_compat_flags(disk_super, features); - else if (set == FEAT_COMPAT_RO) - btrfs_set_super_compat_ro_flags(disk_super, features); - else - btrfs_set_super_incompat_flags(disk_super, features); -} - -static int can_modify_feature(struct btrfs_feature_attr *fa) -{ - int val = 0; - u64 set, clear; - switch (fa->feature_set) { - case FEAT_COMPAT: - set = BTRFS_FEATURE_COMPAT_SAFE_SET; - clear = BTRFS_FEATURE_COMPAT_SAFE_CLEAR; - break; - case FEAT_COMPAT_RO: - set = BTRFS_FEATURE_COMPAT_RO_SAFE_SET; - clear = BTRFS_FEATURE_COMPAT_RO_SAFE_CLEAR; - break; - case FEAT_INCOMPAT: - set = BTRFS_FEATURE_INCOMPAT_SAFE_SET; - clear = BTRFS_FEATURE_INCOMPAT_SAFE_CLEAR; - break; - default: - btrfs_warn(NULL, "sysfs: unknown feature set %d", fa->feature_set); - return 0; - } - - if (set & fa->feature_bit) - val |= 1; - if (clear & fa->feature_bit) - val |= 2; - - return val; -} - static ssize_t btrfs_feature_attr_show(struct kobject *kobj, struct kobj_attribute *a, char *buf) { int val = 0; struct btrfs_fs_info *fs_info = to_fs_info(kobj); struct btrfs_feature_attr *fa = to_btrfs_feature_attr(a); + if (fs_info) { u64 features = get_features(fs_info, fa->feature_set); if (features & fa->feature_bit) val = 1; - } else - val = can_modify_feature(fa); - - return sysfs_emit(buf, "%d\n", val); -} - -static ssize_t btrfs_feature_attr_store(struct kobject *kobj, - struct kobj_attribute *a, - const char *buf, size_t count) -{ - struct btrfs_fs_info *fs_info; - struct btrfs_feature_attr *fa = to_btrfs_feature_attr(a); - u64 features, set, clear; - unsigned long val; - int ret; - - fs_info = to_fs_info(kobj); - if (!fs_info) - return -EPERM; - - if (sb_rdonly(fs_info->sb)) - return -EROFS; - - ret = kstrtoul(skip_spaces(buf), 0, &val); - if (ret) - return ret; - - if (fa->feature_set == FEAT_COMPAT) { - set = BTRFS_FEATURE_COMPAT_SAFE_SET; - clear = BTRFS_FEATURE_COMPAT_SAFE_CLEAR; - } else if (fa->feature_set == FEAT_COMPAT_RO) { - set = BTRFS_FEATURE_COMPAT_RO_SAFE_SET; - clear = BTRFS_FEATURE_COMPAT_RO_SAFE_CLEAR; - } else { - set = BTRFS_FEATURE_INCOMPAT_SAFE_SET; - clear = BTRFS_FEATURE_INCOMPAT_SAFE_CLEAR; } - features = get_features(fs_info, fa->feature_set); - - /* Nothing to do */ - if ((val && (features & fa->feature_bit)) || - (!val && !(features & fa->feature_bit))) - return count; - - if ((val && !(set & fa->feature_bit)) || - (!val && !(clear & fa->feature_bit))) { - btrfs_info(fs_info, - "%sabling feature %s on mounted fs is not supported.", - val ? "En" : "Dis", fa->kobj_attr.attr.name); - return -EPERM; - } - - btrfs_info(fs_info, "%s %s feature flag", - val ? "Setting" : "Clearing", fa->kobj_attr.attr.name); - - spin_lock(&fs_info->super_lock); - features = get_features(fs_info, fa->feature_set); - if (val) - features |= fa->feature_bit; - else - features &= ~fa->feature_bit; - set_features(fs_info, fa->feature_set, features); - spin_unlock(&fs_info->super_lock); - - /* - * We don't want to do full transaction commit from inside sysfs - */ - set_bit(BTRFS_FS_NEED_TRANS_COMMIT, &fs_info->flags); - wake_up_process(fs_info->transaction_kthread); - - return count; + return sysfs_emit(buf, "%d\n", val); } static umode_t btrfs_feature_visible(struct kobject *kobj, @@ -269,9 +158,7 @@ static umode_t btrfs_feature_visible(struct kobject *kobj, fa = attr_to_btrfs_feature_attr(attr); features = get_features(fs_info, fa->feature_set); - if (can_modify_feature(fa)) - mode |= S_IWUSR; - else if (!(features & fa->feature_bit)) + if (!(features & fa->feature_bit)) mode = 0; } From 74f0c0edcd07b18d82de5e76e59e979e69816664 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Tue, 1 Sep 2026 09:31:01 +0930 Subject: [PATCH 0405/1352] btrfs: use ordered extent to grab the logical address for submission In submit_one_sector() we call btrfs_get_extent() to grab the IO extent map so that we know where the logical location to submit the block. However there is no guarantee that there is an IO extent map for the block, and if there is no IO extent map nor ordered extent, btrfs_get_extent() can grab the file extent from on-disk metadata. That's why we have ASSERT()s to reject holes and compressed file extents. On the other hand, for the write range we should have both an IO extent map and an ordered extent, so there is no reason not to grab the ordered extent instead. There is some minor advantages: - No hole ordered extent So no need to rely on ASSERT()s to reject hole extents. And the ASSERT()s are depending on the kernel config, without CONFIG_BTRFS_ASSERT those ASSERT()s won't even trigger. - No IO errors Unlike btrfs_get_extent() which can return IO error when doing the metadata tree search, btrfs_lookup_ordered_extent() will either return an OE or not found. - Cached OE in bio_ctrl->bbio At bbio allocation we have already did an OE lookup, and we have a high chance that the current block also belongs to that OE. Use that cached OE can reduce the frequency to do an rb-tree search. - Smaller rb-tree Unlike extent-map-tree, which can contain cached extent maps, the life span of ordered extents are much shorter, they get removed from the ordered tree after the file extent item is inserted into the subvolume tree. So doing ordered extent tree search can be a tiny faster. And since we're here, also address some minor points: - Add error message for every EUCLEAN error - Remove a dead comment on btrfs_folio_clear_dirty() We no longer call folio_clear_dirty_for_io() since commit 095be159f3eb ("btrfs: unify folio dirty flag clearing"), so the folio flag is still dirty, and the folio dirty flag will be cleared by the last dirty block. Reviewed-by: Boris Burkov Reviewed-by: Johannes Thumshirn Reviewed-by: Daniel Vacek Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/extent_io.c | 64 +++++++++++++++++++++++++++----------------- 1 file changed, 40 insertions(+), 24 deletions(-) diff --git a/fs/btrfs/extent_io.c b/fs/btrfs/extent_io.c index d7600e5fa3d95d..a221b63bdb205d 100644 --- a/fs/btrfs/extent_io.c +++ b/fs/btrfs/extent_io.c @@ -1808,6 +1808,22 @@ static noinline_for_stack int writepage_delalloc(struct btrfs_inode *inode, return 0; } +static struct btrfs_ordered_extent *get_oe_from_bbio(const struct btrfs_bio *bbio, + u64 filepos) +{ + struct btrfs_ordered_extent *oe; + + if (!bbio || !bbio->ordered) + return NULL; + + oe = bbio->ordered; + if (!in_range(filepos, oe->file_offset, oe->num_bytes)) + return NULL; + + refcount_inc(&oe->refs); + return oe; +} + /* * Return 0 if we have submitted or queued the sector for submission. * Return <0 for critical errors, and the involved sector will be cleaned up. @@ -1820,11 +1836,10 @@ static int submit_one_sector(struct btrfs_inode *inode, loff_t i_size) { struct btrfs_fs_info *fs_info = inode->root->fs_info; - struct extent_map *em; + struct btrfs_ordered_extent *oe; u64 block_start; u64 disk_bytenr; u64 extent_offset; - u64 em_end; const u32 sectorsize = fs_info->sectorsize; unsigned int queued; @@ -1833,8 +1848,11 @@ static int submit_one_sector(struct btrfs_inode *inode, /* @filepos >= i_size case should be handled by the caller. */ ASSERT(filepos < i_size); - em = btrfs_get_extent(inode, NULL, filepos, sectorsize); - if (IS_ERR(em)) { + /* Try to reuse the existing OE from bbio first. */ + oe = get_oe_from_bbio(bio_ctrl->bbio, filepos); + if (!oe) + oe = btrfs_lookup_ordered_extent(inode, filepos); + if (unlikely(!oe)) { /* * bio_ctrl may contain a bio crossing several folios. * Submit it immediately so that the bio has a chance @@ -1857,31 +1875,25 @@ static int submit_one_sector(struct btrfs_inode *inode, */ btrfs_mark_ordered_io_finished(inode, filepos, fs_info->sectorsize, false); - return PTR_ERR(em); + btrfs_err_rl(fs_info, + "no ordered extent for root %lld ino %llu filepos %llu", + btrfs_root_id(inode->root), btrfs_ino(inode), + filepos); + return -EUCLEAN; } - extent_offset = filepos - em->start; - em_end = btrfs_extent_map_end(em); - ASSERT(filepos <= em_end); - ASSERT(IS_ALIGNED(em->start, sectorsize)); - ASSERT(IS_ALIGNED(em->len, sectorsize)); - - block_start = btrfs_extent_map_block_start(em); - disk_bytenr = btrfs_extent_map_block_start(em) + extent_offset; + extent_offset = filepos - oe->file_offset; + ASSERT(filepos < oe->file_offset + oe->num_bytes); + ASSERT(IS_ALIGNED(oe->file_offset, sectorsize)); + ASSERT(IS_ALIGNED(oe->num_bytes, sectorsize)); + ASSERT(oe->compress_type == BTRFS_COMPRESS_NONE); + ASSERT(!test_bit(BTRFS_ORDERED_COMPRESSED, &oe->flags)); - ASSERT(!btrfs_extent_map_is_compressed(em)); - ASSERT(block_start != EXTENT_MAP_HOLE); - ASSERT(block_start != EXTENT_MAP_INLINE); + block_start = oe->disk_bytenr + oe->offset; + disk_bytenr = block_start + extent_offset; - btrfs_free_extent_map(em); - em = NULL; + btrfs_put_ordered_extent(oe); - /* - * Although the PageDirty bit is cleared before entering this - * function, subpage dirty bit is not cleared. - * So clear subpage dirty bit here so next time we won't submit - * a folio for a range already written to disk. - */ btrfs_folio_clear_dirty(fs_info, folio, filepos, sectorsize); btrfs_folio_set_writeback(fs_info, folio, filepos, sectorsize); /* @@ -1898,6 +1910,10 @@ static int submit_one_sector(struct btrfs_inode *inode, btrfs_folio_clear_writeback(fs_info, folio, filepos, sectorsize); btrfs_mark_ordered_io_finished(inode, filepos, fs_info->sectorsize, false); + btrfs_err_rl(fs_info, + "failed to queue sector for root %lld ino %llu filepos %llu", + btrfs_root_id(inode->root), + btrfs_ino(inode), filepos); return -EUCLEAN; } return 0; From 644ec5de5acef3c13506f062459e33a78e50b00c Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Tue, 1 Sep 2026 10:01:33 +0930 Subject: [PATCH 0406/1352] btrfs: tree-checker: reject file extent items for special files File extent items are only utilized by regular files or symlinks, other files like directory/char/block/FIFO/sock files should not have any file extent item. Previously we were unable to reject such cases, as the inode item may not be in the same leaf. But we already have @prev_key in check_leaf_item(), this means we just need a new way to pass the mode of the previously hit inode item, then we can detect such problems. Introduce a new helper structure, saved_inode_info, to record the inode number and its mode hit in the same leaf, and keep it across the whole leaf. Then if we hit a file extent item, and the inode item is in the same leaf, we can refer to that to determine if we need to reject the file extent item. Now with the following corrupted fs tree, the kernel can safely reject the leaf: item 0 key (256 INODE_ITEM 0) itemoff 16123 itemsize 160 generation 3 transid 9 size 12 nbytes 16384 block group 0 mode 40755 links 1 uid 0 gid 0 rdev 0 sequence 1 flags 0x0(none) item 1 key (256 INODE_REF 256) itemoff 16111 itemsize 12 index 0 namelen 2 name: .. item 2 key (256 DIR_ITEM 496027801) itemoff 16075 itemsize 36 location key (257 INODE_ITEM 0) type FILE transid 9 data_len 0 name_len 6 name: foobar item 3 key (256 DIR_INDEX 2) itemoff 16039 itemsize 36 location key (257 INODE_ITEM 0) type FILE transid 9 data_len 0 name_len 6 name: foobar item 4 key (257 INODE_ITEM 0) itemoff 15879 itemsize 160 generation 9 transid 9 size 8192 nbytes 8192 block group 0 mode 60600 links 1 uid 0 gid 0 rdev 0 ^^ This is BLK type, not REG. sequence 2 flags 0x0(none) item 5 key (257 INODE_REF 256) itemoff 15863 itemsize 16 index 2 namelen 6 name: foobar item 6 key (257 EXTENT_DATA 0) itemoff 15810 itemsize 53 generation 9 type 1 (regular) extent data disk byte 13631488 nr 8192 extent data offset 0 nr 8192 ram 8192 extent compression 0 (none) extent encryption 0 With the patch, kernel will reject it with the following tree-checker errors: BTRFS critical (device loop0): corrupt leaf: root=5 block=30408704 slot=6 ino=257 file_offset=0, unexpected file extent item type 1 for inode mode 060600 BTRFS error (device loop0): read time tree block corruption detected on logical 30408704 mirror 1 Reported-by: ZhengYuan Huang Link: https://lore.kernel.org/linux-btrfs/20260817132051.267646-1-gality369@gmail.com/ Assisted-by: LLM (for generating the corrupted image) Reviewed-by: Johannes Thumshirn Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/tree-checker.c | 66 ++++++++++++++++++++++++++++++++++------- 1 file changed, 55 insertions(+), 11 deletions(-) diff --git a/fs/btrfs/tree-checker.c b/fs/btrfs/tree-checker.c index 4b1e47173c637c..857ded43c3b9bd 100644 --- a/fs/btrfs/tree-checker.c +++ b/fs/btrfs/tree-checker.c @@ -163,6 +163,12 @@ static void dir_item_err(const struct extent_buffer *eb, int slot, va_end(args); } +/* Record info for the last hit inode. */ +struct saved_inode_info { + u64 ino; + u32 mode; +}; + /* * This functions checks prev_key->objectid, to ensure current key and prev_key * share the same objectid as inode number. @@ -204,15 +210,41 @@ static bool check_prev_ino(struct extent_buffer *leaf, prev_key->objectid, key->objectid); return false; } + +static bool can_have_extent_data(struct extent_buffer *leaf, + struct btrfs_key *key, int slot, u8 fi_type, + const struct saved_inode_info *inode_info) +{ + /* No inode item in this leaf. */ + if (inode_info->ino != key->objectid) + return true; + if (S_ISREG(inode_info->mode)) + return true; + if (S_ISLNK(inode_info->mode)) { + /* For a symlink, the file extent item should always be inlined. */ + if (unlikely(fi_type != BTRFS_FILE_EXTENT_INLINE)) + return false; + return true; + } + + /* + * The rest are special files, e.g. block/FIFO files, which cannnot + * have any file extent. + */ + return false; +} + static int check_extent_data_item(struct extent_buffer *leaf, struct btrfs_key *key, int slot, - struct btrfs_key *prev_key) + struct btrfs_key *prev_key, + const struct saved_inode_info *inode_info) { struct btrfs_fs_info *fs_info = leaf->fs_info; struct btrfs_file_extent_item *fi; u32 sectorsize = fs_info->sectorsize; u32 item_size = btrfs_item_size(leaf, slot); u64 extent_end; + u8 fi_type; if (unlikely(!IS_ALIGNED(key->offset, sectorsize))) { file_extent_err(leaf, slot, @@ -243,12 +275,18 @@ static int check_extent_data_item(struct extent_buffer *leaf, SZ_4K); return -EUCLEAN; } - if (unlikely(btrfs_file_extent_type(leaf, fi) >= - BTRFS_NR_FILE_EXTENT_TYPES)) { + fi_type = btrfs_file_extent_type(leaf, fi); + if (unlikely(fi_type >= BTRFS_NR_FILE_EXTENT_TYPES)) { file_extent_err(leaf, slot, "invalid type for file extent, have %u expect range [0, %u]", - btrfs_file_extent_type(leaf, fi), - BTRFS_NR_FILE_EXTENT_TYPES - 1); + fi_type, BTRFS_NR_FILE_EXTENT_TYPES - 1); + return -EUCLEAN; + } + + if (unlikely(!can_have_extent_data(leaf, key, slot, fi_type, inode_info))) { + file_extent_err(leaf, slot, + "unexpected file extent item type %u for inode mode 0%o", + fi_type, inode_info->mode); return -EUCLEAN; } @@ -270,7 +308,8 @@ static int check_extent_data_item(struct extent_buffer *leaf, btrfs_file_extent_encryption(leaf, fi)); return -EUCLEAN; } - if (btrfs_file_extent_type(leaf, fi) == BTRFS_FILE_EXTENT_INLINE) { + + if (fi_type == BTRFS_FILE_EXTENT_INLINE) { /* Inline extent must have 0 as key offset */ if (unlikely(key->offset)) { file_extent_err(leaf, slot, @@ -1206,7 +1245,8 @@ static int check_dev_item(struct extent_buffer *leaf, } static int check_inode_item(struct extent_buffer *leaf, - struct btrfs_key *key, int slot) + struct btrfs_key *key, int slot, + struct saved_inode_info *inode_info) { struct btrfs_fs_info *fs_info = leaf->fs_info; struct btrfs_inode_item *iitem; @@ -1291,6 +1331,8 @@ static int check_inode_item(struct extent_buffer *leaf, ro_flags); return -EUCLEAN; } + inode_info->ino = key->objectid; + inode_info->mode = mode; return 0; } @@ -2348,14 +2390,15 @@ static int check_free_space_bitmap(struct extent_buffer *leaf, static enum btrfs_tree_block_status check_leaf_item(struct extent_buffer *leaf, struct btrfs_key *key, int slot, - struct btrfs_key *prev_key) + struct btrfs_key *prev_key, + struct saved_inode_info *inode_info) { int ret = 0; struct btrfs_chunk *chunk; switch (key->type) { case BTRFS_EXTENT_DATA_KEY: - ret = check_extent_data_item(leaf, key, slot, prev_key); + ret = check_extent_data_item(leaf, key, slot, prev_key, inode_info); break; case BTRFS_EXTENT_CSUM_KEY: ret = check_csum_item(leaf, key, slot, prev_key); @@ -2385,7 +2428,7 @@ static enum btrfs_tree_block_status check_leaf_item(struct extent_buffer *leaf, ret = check_dev_extent_item(leaf, key, slot, prev_key); break; case BTRFS_INODE_ITEM_KEY: - ret = check_inode_item(leaf, key, slot); + ret = check_inode_item(leaf, key, slot, inode_info); break; case BTRFS_ROOT_ITEM_KEY: ret = check_root_item(leaf, key, slot); @@ -2433,6 +2476,7 @@ static enum btrfs_tree_block_status check_leaf_item(struct extent_buffer *leaf, enum btrfs_tree_block_status __btrfs_check_leaf(struct extent_buffer *leaf) { struct btrfs_fs_info *fs_info = leaf->fs_info; + struct saved_inode_info inode_info = { 0 }; /* No valid key type is 0, so all key should be larger than this key */ struct btrfs_key prev_key = {0, 0, 0}; struct btrfs_key key; @@ -2568,7 +2612,7 @@ enum btrfs_tree_block_status __btrfs_check_leaf(struct extent_buffer *leaf) } /* Check if the item size and content meet other criteria. */ - ret = check_leaf_item(leaf, &key, slot, &prev_key); + ret = check_leaf_item(leaf, &key, slot, &prev_key, &inode_info); if (unlikely(ret != BTRFS_TREE_BLOCK_CLEAN)) return ret; From 51ddcd0e62144bc208ad7d53458326e5b1980320 Mon Sep 17 00:00:00 2001 From: Daniel Vacek Date: Wed, 2 Sep 2026 11:20:24 +0200 Subject: [PATCH 0407/1352] btrfs: consume given iter directly instead of copying in csum_one_bio() Avoid copying the iter twice in async case. We already have a copy csum_one_bio() can consume directly. No need to copy it again the second time. We can use this copy also in the sync case and get rid of the parameter. Reviewed-by: Qu Wenruo Signed-off-by: Daniel Vacek Signed-off-by: David Sterba --- fs/btrfs/file-item.c | 16 ++++++++-------- 1 file changed, 8 insertions(+), 8 deletions(-) diff --git a/fs/btrfs/file-item.c b/fs/btrfs/file-item.c index 581ca5653be93a..2c6c14dfe39320 100644 --- a/fs/btrfs/file-item.c +++ b/fs/btrfs/file-item.c @@ -797,18 +797,18 @@ int btrfs_lookup_csums_bitmap(struct btrfs_root *root, struct btrfs_path *path, return ret; } -static void csum_one_bio(struct btrfs_bio *bbio, struct bvec_iter *src) +static void csum_one_bio(struct btrfs_bio *bbio) { struct btrfs_inode *inode = bbio->inode; struct btrfs_fs_info *fs_info = inode->root->fs_info; struct btrfs_ordered_sum *sums = bbio->sums; - struct bvec_iter iter; const u32 blocksize = fs_info->sectorsize; int index = 0; - for (iter = *src; iter.bi_size; bio_advance_iter(&bbio->bio, &iter, blocksize)) { - btrfs_csum_one_bio_block(fs_info, &bbio->bio, &iter, - sums->sums + index); + for (struct bvec_iter *iter = &bbio->csum_saved_iter; + iter->bi_size; + bio_advance_iter(&bbio->bio, iter, blocksize)) { + btrfs_csum_one_bio_block(fs_info, &bbio->bio, iter, sums->sums + index); index += fs_info->csum_size; } @@ -820,7 +820,7 @@ static void csum_one_bio_work(struct work_struct *work) ASSERT(btrfs_op(&bbio->bio) == BTRFS_MAP_WRITE); ASSERT(bbio->async_csum == true); - csum_one_bio(bbio, &bbio->csum_saved_iter); + csum_one_bio(bbio); complete(&bbio->csum_done); } @@ -850,13 +850,13 @@ int btrfs_csum_one_bio(struct btrfs_bio *bbio, bool async) bbio->sums = sums; btrfs_add_ordered_sum(ordered, sums); + bbio->csum_saved_iter = bio->bi_iter; if (!async) { - csum_one_bio(bbio, &bbio->bio.bi_iter); + csum_one_bio(bbio); return 0; } init_completion(&bbio->csum_done); bbio->async_csum = true; - bbio->csum_saved_iter = bbio->bio.bi_iter; INIT_WORK(&bbio->csum_work, csum_one_bio_work); schedule_work(&bbio->csum_work); return 0; From 53b64c42171be51856b48dfcaf586fc6cbc3659a Mon Sep 17 00:00:00 2001 From: Daniel Vacek Date: Wed, 2 Sep 2026 15:56:27 +0200 Subject: [PATCH 0408/1352] btrfs: use bio::remaining for async checksumming synchronization We can use bio::remaining counter to sync the offloaded checksumming. As a result we can slim down the btrfs_bio structure by 24 bytes and simplify the code a bit. Difference in pahole output: - /* size: 328, cachelines: 6, members: 15 */ + /* size: 304, cachelines: 5, members: 14 */ Moreover this will allow us enabling async checksumming with encryption where we need to checksum the bounce bio instead of our regular one embedded in btrfs_bio. And so we need to extend it's lifetime. This is the preferred way to do so. This also fixes a bug in experimental build where the async checksumming was using the system workqueue instead of fs_info::endio_workers. Fixes: dd57c78aec39 ("btrfs: introduce btrfs_bio::async_csum") Reviewed-by: Qu Wenruo Signed-off-by: Daniel Vacek Signed-off-by: David Sterba --- fs/btrfs/bio.c | 4 ---- fs/btrfs/bio.h | 4 ---- fs/btrfs/file-item.c | 8 +++----- 3 files changed, 3 insertions(+), 13 deletions(-) diff --git a/fs/btrfs/bio.c b/fs/btrfs/bio.c index 19b4855969f536..771b7d598aeeeb 100644 --- a/fs/btrfs/bio.c +++ b/fs/btrfs/bio.c @@ -103,7 +103,6 @@ static struct btrfs_bio *btrfs_split_bio(struct btrfs_fs_info *fs_info, bbio->can_use_append = orig_bbio->can_use_append; bbio->is_scrub = orig_bbio->is_scrub; bbio->is_remap = orig_bbio->is_remap; - bbio->async_csum = orig_bbio->async_csum; atomic_inc(&orig_bbio->pending_ios); return bbio; @@ -114,9 +113,6 @@ void btrfs_bio_end_io(struct btrfs_bio *bbio, blk_status_t status) /* Make sure we're already in task context. */ ASSERT(in_task()); - if (bbio->async_csum) - wait_for_completion(&bbio->csum_done); - bbio->bio.bi_status = status; if (bbio->bio.bi_pool == &btrfs_clone_bioset) { struct btrfs_bio *orig_bbio = bbio->private; diff --git a/fs/btrfs/bio.h b/fs/btrfs/bio.h index b7bd377a016249..bbf362b8668bbc 100644 --- a/fs/btrfs/bio.h +++ b/fs/btrfs/bio.h @@ -58,7 +58,6 @@ struct btrfs_bio { struct btrfs_ordered_extent *ordered; struct btrfs_ordered_sum *sums; struct work_struct csum_work; - struct completion csum_done; struct bvec_iter csum_saved_iter; u64 orig_physical; u64 orig_logical; @@ -93,9 +92,6 @@ struct btrfs_bio { /* Whether the bio is coming from copy_remapped_data_io(). */ bool is_remap:1; - /* Whether the csum generation for data write is async. */ - bool async_csum:1; - /* Whether the bio is written using zone append. */ bool can_use_append:1; diff --git a/fs/btrfs/file-item.c b/fs/btrfs/file-item.c index 2c6c14dfe39320..ae1fd4da38d31b 100644 --- a/fs/btrfs/file-item.c +++ b/fs/btrfs/file-item.c @@ -819,9 +819,8 @@ static void csum_one_bio_work(struct work_struct *work) struct btrfs_bio *bbio = container_of(work, struct btrfs_bio, csum_work); ASSERT(btrfs_op(&bbio->bio) == BTRFS_MAP_WRITE); - ASSERT(bbio->async_csum == true); csum_one_bio(bbio); - complete(&bbio->csum_done); + bio_endio(&bbio->bio); } /* @@ -855,10 +854,9 @@ int btrfs_csum_one_bio(struct btrfs_bio *bbio, bool async) csum_one_bio(bbio); return 0; } - init_completion(&bbio->csum_done); - bbio->async_csum = true; + bio_inc_remaining(bio); INIT_WORK(&bbio->csum_work, csum_one_bio_work); - schedule_work(&bbio->csum_work); + queue_work(fs_info->endio_workers, &bbio->csum_work); return 0; } From 6fec67244ac20cb74216ea1f4b30c9f64289165d Mon Sep 17 00:00:00 2001 From: Hemanth Selam Date: Mon, 7 Sep 2026 10:18:01 +0530 Subject: [PATCH 0409/1352] btrfs: fix typos and repeated words in comments Fix misspellings and repeated words in comments, found with scripts/checkpatch.pl and codespell. Only touches comments, no code changes. Assisted-by: Cursor:claude-opus-5 Signed-off-by: Hemanth Selam Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/block-group.h | 2 +- fs/btrfs/extent-io-tree.c | 2 +- fs/btrfs/fs.c | 2 +- fs/btrfs/raid56.c | 11 ++++------- fs/btrfs/send.c | 2 +- fs/btrfs/transaction.h | 2 +- fs/btrfs/tree-checker.c | 2 +- include/uapi/linux/btrfs_tree.h | 4 ++-- 8 files changed, 12 insertions(+), 15 deletions(-) diff --git a/fs/btrfs/block-group.h b/fs/btrfs/block-group.h index 69d56864d4baa0..b349f94cf929ab 100644 --- a/fs/btrfs/block-group.h +++ b/fs/btrfs/block-group.h @@ -135,7 +135,7 @@ struct btrfs_block_group { u64 global_root_id; u64 remap_bytes; u32 identity_remap_count; - /* The last commited identity_remap_count value of this block group. */ + /* The last committed identity_remap_count value of this block group. */ u32 last_identity_remap_count; /* * The last committed used bytes of this block group, if the above @used diff --git a/fs/btrfs/extent-io-tree.c b/fs/btrfs/extent-io-tree.c index d6df11f6088c44..992b8b42bdb4e3 100644 --- a/fs/btrfs/extent-io-tree.c +++ b/fs/btrfs/extent-io-tree.c @@ -751,7 +751,7 @@ int btrfs_clear_extent_bit_changeset(struct extent_io_tree *tree, u64 start, u64 btrfs_split_delalloc_extent(tree->inode, state, start); /* - * Temporarilly ajdust this state's range to match the + * Temporarily ajdust this state's range to match the * range for which we are clearing bits. */ state->start = start; diff --git a/fs/btrfs/fs.c b/fs/btrfs/fs.c index de160d29dde82f..75a1217727a7bb 100644 --- a/fs/btrfs/fs.c +++ b/fs/btrfs/fs.c @@ -79,7 +79,7 @@ void btrfs_csum_init(struct btrfs_csum_ctx *ctx, u16 csum_type) blake2b_init(&ctx->blake2b, 32); break; default: - /* Checksume type is validated at mount time. */ + /* Checksum type is validated at mount time. */ BUG(); } } diff --git a/fs/btrfs/raid56.c b/fs/btrfs/raid56.c index a5d0ef09d92abb..8ec24dbb180f92 100644 --- a/fs/btrfs/raid56.c +++ b/fs/btrfs/raid56.c @@ -953,7 +953,7 @@ static void rbio_orig_end_io(struct btrfs_raid_bio *rbio, blk_status_t status) /* * Clear the data bitmap, as the rbio may be cached for later usage. - * do this before before unlock_stripe() so there will be no new bio + * do this before unlock_stripe() so there will be no new bio * for this bio. */ bitmap_clear(&rbio->dbitmap, 0, rbio->stripe_nsectors); @@ -988,7 +988,7 @@ static void rbio_orig_end_io(struct btrfs_raid_bio *rbio, blk_status_t status) * as possible, and only use stripe_sectors as fallback. * * Return NULL if bio_list_only is set but the specified sector has no - * coresponding bio. + * corresponding bio. */ static phys_addr_t *sector_paddrs_in_rbio(struct btrfs_raid_bio *rbio, int stripe_nr, int sector_nr, @@ -1450,10 +1450,7 @@ static int rmw_assemble_write_bios(struct btrfs_raid_bio *rbio, /* We should have at least one data sector. */ ASSERT(bitmap_weight(&rbio->dbitmap, rbio->stripe_nsectors)); - /* - * Reset errors, as we may have errors inherited from from degraded - * write. - */ + /* Reset errors, as we may have errors inherited from degraded write. */ bitmap_clear(rbio->error_bitmap, 0, rbio->nr_sectors); /* @@ -2632,7 +2629,7 @@ static int alloc_rbio_essential_pages(struct btrfs_raid_bio *rbio) return 0; } -/* Return true if the content of the step matches the caclulated one. */ +/* Return true if the content of the step matches the calculated one. */ static bool verify_one_parity_step(struct btrfs_raid_bio *rbio, void *pointers[], unsigned int sector_nr, unsigned int step_nr) diff --git a/fs/btrfs/send.c b/fs/btrfs/send.c index 5c59b9abedcd71..c523bf950c898f 100644 --- a/fs/btrfs/send.c +++ b/fs/btrfs/send.c @@ -7023,7 +7023,7 @@ static int changed_extent(struct send_ctx *sctx, * get modified or replaced with a new one). Note that deduplication * updates the inode item, but it only changes the iversion (sequence * field in the inode item) of the inode, so if a file is deduplicated - * the same amount of times in both the parent and send snapshots, its + * the same number of times in both the parent and send snapshots, its * iversion becomes the same in both snapshots, whence the inode item is * the same on both snapshots. */ diff --git a/fs/btrfs/transaction.h b/fs/btrfs/transaction.h index 3a57f227b5ed28..89153cd2259678 100644 --- a/fs/btrfs/transaction.h +++ b/fs/btrfs/transaction.h @@ -288,7 +288,7 @@ do { \ * Call btrfs_abort_transaction() as early as possible when an error condition * is detected, that way the exact stack trace is reported for some errors. * - * Error number must be negative as it encodes wheather it's the first abort. + * Error number must be negative as it encodes whether it's the first abort. */ #define btrfs_abort_transaction(trans, error) \ do { \ diff --git a/fs/btrfs/tree-checker.c b/fs/btrfs/tree-checker.c index 857ded43c3b9bd..594266c9c116ff 100644 --- a/fs/btrfs/tree-checker.c +++ b/fs/btrfs/tree-checker.c @@ -228,7 +228,7 @@ static bool can_have_extent_data(struct extent_buffer *leaf, } /* - * The rest are special files, e.g. block/FIFO files, which cannnot + * The rest are special files, e.g. block/FIFO files, which cannot * have any file extent. */ return false; diff --git a/include/uapi/linux/btrfs_tree.h b/include/uapi/linux/btrfs_tree.h index b6ccaf848e4b3e..47ee52859b4589 100644 --- a/include/uapi/linux/btrfs_tree.h +++ b/include/uapi/linux/btrfs_tree.h @@ -230,7 +230,7 @@ * * Stored as an inline ref rather to avoid wasting space on a separate item on * top of the existing extent item. However, unlike the other inline refs, - * there is one one owner ref per extent rather than one per extent. + * there is one owner ref per extent rather than one per extent. * * Because of this, it goes at the front of the list of inline refs, and thus * must have a lower type value than any other inline ref type (to satisfy the @@ -243,7 +243,7 @@ #define BTRFS_EXTENT_DATA_REF_KEY 178 /* - * Obsolete key. Defintion removed in 6.6, value may be reused in the future. + * Obsolete key. Definition removed in 6.6, value may be reused in the future. * * #define BTRFS_EXTENT_REF_V0_KEY 180 */ From c0552f54c3e7842b0a775bb3b0a081044be0645e Mon Sep 17 00:00:00 2001 From: Jeff Layton Date: Tue, 25 Aug 2026 12:04:17 -0400 Subject: [PATCH 0410/1352] btrfs: split btrfs_insert_delayed_dir_index() into prealloc and commit phases Split btrfs_insert_delayed_dir_index() into three functions using a new btrfs_dir_index_prealloc struct to bundle the pre-allocated resources: - btrfs_prealloc_delayed_dir_index(): allocates the struct and performs the two GFP_NOFS allocations (delayed node + delayed item) that can fail with -ENOMEM. Returns the struct, or ERR_PTR on failure. - btrfs_insert_delayed_dir_index_prealloc(): populates the item data, inserts into the rb-tree, and reserves metadata space. Cannot fail with -ENOMEM since all allocations were done in the prealloc step. - btrfs_free_delayed_dir_index_prealloc(): frees pre-allocated resources when the caller's btree insertion fails. Tolerates NULL. The prealloc is returned as a pointer rather than filled into a caller-provided struct, so that a plain NULL means "no prealloc" and callers do not need a separate flag to track whether one exists. It is consumed (and freed) by either the commit or the free helper, so ownership is unambiguous. The original btrfs_insert_delayed_dir_index() is refactored into a thin wrapper that calls the prealloc and commit functions. This split allows callers to move the fallible memory allocations before the point of no return (the DIR_ITEM btree insertion), so that -ENOMEM can be returned cleanly without aborting the transaction. Assisted-by: LLM Reviewed-by: Qu Wenruo Signed-off-by: Jeff Layton Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/delayed-inode.c | 128 +++++++++++++++++++++++++++++++-------- fs/btrfs/delayed-inode.h | 17 ++++++ 2 files changed, 121 insertions(+), 24 deletions(-) diff --git a/fs/btrfs/delayed-inode.c b/fs/btrfs/delayed-inode.c index db2ffab0941a6a..21a2d123df14f1 100644 --- a/fs/btrfs/delayed-inode.c +++ b/fs/btrfs/delayed-inode.c @@ -6,6 +6,7 @@ #include #include +#include #include "ctree.h" #include "fs.h" #include "messages.h" @@ -1469,35 +1470,93 @@ static void btrfs_release_dir_index_item_space(struct btrfs_trans_handle *trans) trans->bytes_reserved -= bytes; } -/* Will return 0, -ENOMEM or -EEXIST (index number collision, unexpected). */ -int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans, - const char *name, int name_len, - struct btrfs_inode *dir, - const struct btrfs_disk_key *disk_key, u8 flags, - u64 index) +/* + * Pre-allocate a delayed node and delayed item for a dir index insertion and + * copy the name into the item. Call this before modifying the btree so that + * ENOMEM can be returned before any on-disk state has changed. + * + * The returned prealloc is consumed by either + * btrfs_insert_delayed_dir_index_prealloc() or + * btrfs_free_delayed_dir_index_prealloc(); it must not be used afterwards. + * + * Returns a prealloc on success, ERR_PTR on allocation failure. + */ +struct btrfs_dir_index_prealloc *btrfs_prealloc_delayed_dir_index(struct btrfs_inode *dir, + const char *name, + int name_len) { + struct btrfs_dir_index_prealloc *prealloc; + struct btrfs_delayed_node *node; + struct btrfs_delayed_item *item; + + prealloc = kzalloc_obj(*prealloc, GFP_NOFS); + if (!prealloc) + return ERR_PTR(-ENOMEM); + + node = btrfs_get_or_create_delayed_node(dir, &prealloc->tracker); + if (IS_ERR(node)) { + kfree(prealloc); + return ERR_CAST(node); + } + + item = btrfs_alloc_delayed_item(sizeof(struct btrfs_dir_item) + name_len, + node, BTRFS_DELAYED_INSERTION_ITEM); + if (!item) { + btrfs_release_delayed_node(node, &prealloc->tracker); + kfree(prealloc); + return ERR_PTR(-ENOMEM); + } + + memcpy(item->data + sizeof(struct btrfs_dir_item), name, name_len); + + prealloc->node = node; + prealloc->item = item; + return prealloc; +} +ALLOW_ERROR_INJECTION(btrfs_prealloc_delayed_dir_index, ERRNO); + +/* + * Free resources from btrfs_prealloc_delayed_dir_index() when the btree + * insertion failed and we will not commit the delayed dir index. Does nothing + * if @prealloc is NULL. + */ +void btrfs_free_delayed_dir_index_prealloc(struct btrfs_trans_handle *trans, + struct btrfs_dir_index_prealloc *prealloc) +{ + if (!prealloc) + return; + + btrfs_release_delayed_item(prealloc->item); + btrfs_release_dir_index_item_space(trans); + btrfs_release_delayed_node(prealloc->node, &prealloc->tracker); + kfree(prealloc); +} + +/* + * Commit a pre-allocated delayed dir index item. @prealloc must have been + * returned by btrfs_prealloc_delayed_dir_index(). This populates the item, + * adds it to the delayed node's rb-tree, and reserves metadata space. It + * cannot fail with ENOMEM. @prealloc is freed here in all cases. + * + * Return 0 or -EEXIST (index number collision, unexpected). + */ +int btrfs_insert_delayed_dir_index_prealloc(struct btrfs_trans_handle *trans, + struct btrfs_inode *dir, + struct btrfs_dir_index_prealloc *prealloc, + const struct btrfs_disk_key *disk_key, + u8 flags, u64 index) +{ + struct btrfs_delayed_node *delayed_node = prealloc->node; + struct btrfs_ref_tracker *tracker = &prealloc->tracker; + struct btrfs_delayed_item *delayed_item = prealloc->item; struct btrfs_fs_info *fs_info = trans->fs_info; const unsigned int leaf_data_size = BTRFS_LEAF_DATA_SIZE(fs_info); - struct btrfs_delayed_node *delayed_node; - struct btrfs_ref_tracker delayed_node_tracker; - struct btrfs_delayed_item *delayed_item; + const int name_len = delayed_item->data_len - sizeof(struct btrfs_dir_item); struct btrfs_dir_item *dir_item; bool reserve_leaf_space; u32 data_len; int ret; - delayed_node = btrfs_get_or_create_delayed_node(dir, &delayed_node_tracker); - if (IS_ERR(delayed_node)) - return PTR_ERR(delayed_node); - - delayed_item = btrfs_alloc_delayed_item(sizeof(*dir_item) + name_len, - delayed_node, - BTRFS_DELAYED_INSERTION_ITEM); - if (!delayed_item) { - ret = -ENOMEM; - goto release_node; - } - delayed_item->index = index; dir_item = (struct btrfs_dir_item *)delayed_item->data; @@ -1506,7 +1565,7 @@ int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans, btrfs_set_stack_dir_data_len(dir_item, 0); btrfs_set_stack_dir_name_len(dir_item, name_len); btrfs_set_stack_dir_flags(dir_item, flags); - memcpy((char *)(dir_item + 1), name, name_len); + /* Name was already copied by btrfs_prealloc_delayed_dir_index(). */ data_len = delayed_item->data_len + sizeof(struct btrfs_item); @@ -1524,7 +1583,9 @@ int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans, if (unlikely(ret)) { btrfs_err(trans->fs_info, "error adding delayed dir index item, name: %.*s, index: %llu, root: %llu, dir: %llu, dir->index_cnt: %llu, delayed_node->index_cnt: %llu, error: %pe", - name_len, name, index, btrfs_root_id(delayed_node->root), + name_len, + (const char *)(dir_item + 1), + index, btrfs_root_id(delayed_node->root), delayed_node->inode_id, dir->index_cnt, delayed_node->index_cnt, ERR_PTR(ret)); btrfs_release_delayed_item(delayed_item); @@ -1562,10 +1623,29 @@ int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans, mutex_unlock(&delayed_node->mutex); release_node: - btrfs_release_delayed_node(delayed_node, &delayed_node_tracker); + /* Must release the node before freeing @tracker's containing struct. */ + btrfs_release_delayed_node(delayed_node, tracker); + kfree(prealloc); return ret; } +/* Return 0, -ENOMEM or -EEXIST (index number collision, unexpected). */ +int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans, + const char *name, int name_len, + struct btrfs_inode *dir, + const struct btrfs_disk_key *disk_key, u8 flags, + u64 index) +{ + struct btrfs_dir_index_prealloc *prealloc; + + prealloc = btrfs_prealloc_delayed_dir_index(dir, name, name_len); + if (IS_ERR(prealloc)) + return PTR_ERR(prealloc); + + return btrfs_insert_delayed_dir_index_prealloc(trans, dir, prealloc, + disk_key, flags, index); +} + static bool btrfs_delete_delayed_insertion_item(struct btrfs_delayed_node *node, u64 index) { diff --git a/fs/btrfs/delayed-inode.h b/fs/btrfs/delayed-inode.h index fc752863f89bcd..6d12a145489f2b 100644 --- a/fs/btrfs/delayed-inode.h +++ b/fs/btrfs/delayed-inode.h @@ -121,6 +121,23 @@ int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans, const struct btrfs_disk_key *disk_key, u8 flags, u64 index); +struct btrfs_dir_index_prealloc { + struct btrfs_delayed_node *node; + struct btrfs_ref_tracker tracker; + struct btrfs_delayed_item *item; +}; + +struct btrfs_dir_index_prealloc *btrfs_prealloc_delayed_dir_index(struct btrfs_inode *dir, + const char *name, + int name_len); +void btrfs_free_delayed_dir_index_prealloc(struct btrfs_trans_handle *trans, + struct btrfs_dir_index_prealloc *prealloc); +int btrfs_insert_delayed_dir_index_prealloc(struct btrfs_trans_handle *trans, + struct btrfs_inode *dir, + struct btrfs_dir_index_prealloc *prealloc, + const struct btrfs_disk_key *disk_key, + u8 flags, u64 index); + int btrfs_delete_delayed_dir_index(struct btrfs_trans_handle *trans, struct btrfs_inode *dir, u64 index); From 4c717901c2e23ca36f988270a63801d64c9a70af Mon Sep 17 00:00:00 2001 From: Jeff Layton Date: Tue, 25 Aug 2026 12:04:18 -0400 Subject: [PATCH 0411/1352] btrfs: pre-allocate delayed dir index before btree modification Move the delayed dir index allocation in btrfs_insert_dir_item() before the insert_with_overflow() call that modifies the btree. Previously, the allocations happened after the DIR_ITEM was already inserted, meaning an ENOMEM failure left the btree in a partially-modified state that could only be resolved by aborting the transaction. Add an optional caller-provided btrfs_dir_index_prealloc parameter to btrfs_insert_dir_item(). When non-NULL, ownership of the prealloc transfers to btrfs_insert_dir_item(). When NULL, it allocates internally. All existing callers pass NULL to preserve the current behavior. Since ownership transfers, btrfs_insert_dir_item() must free the prealloc on every path that does not commit it. Route all such exits (including the early path allocation failure) through a common out_free_prealloc label, rather than keying cleanup on need_delayed_index. Remove the btrfs_insert_delayed_dir_index() wrapper, as there are no more callers. Assisted-by: LLM Suggested-by: Qu Wenruo Signed-off-by: Jeff Layton Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/delayed-inode.c | 21 ++------------------ fs/btrfs/delayed-inode.h | 5 ----- fs/btrfs/dir-item.c | 41 ++++++++++++++++++++++++++-------------- fs/btrfs/dir-item.h | 5 +++-- fs/btrfs/inode.c | 2 +- fs/btrfs/transaction.c | 3 +-- 6 files changed, 34 insertions(+), 43 deletions(-) diff --git a/fs/btrfs/delayed-inode.c b/fs/btrfs/delayed-inode.c index 21a2d123df14f1..1f20148c1fd44e 100644 --- a/fs/btrfs/delayed-inode.c +++ b/fs/btrfs/delayed-inode.c @@ -687,7 +687,7 @@ static int btrfs_insert_delayed_item(struct btrfs_trans_handle *trans, /* * For delayed items to insert, we track reserved metadata bytes based * on the number of leaves that we will use. - * See btrfs_insert_delayed_dir_index() and + * See btrfs_insert_delayed_dir_index_prealloc() and * btrfs_delayed_item_reserve_metadata()). */ ASSERT(first_item->bytes_reserved == 0); @@ -1629,23 +1629,6 @@ int btrfs_insert_delayed_dir_index_prealloc(struct btrfs_trans_handle *trans, return ret; } -/* Return 0, -ENOMEM or -EEXIST (index number collision, unexpected). */ -int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans, - const char *name, int name_len, - struct btrfs_inode *dir, - const struct btrfs_disk_key *disk_key, u8 flags, - u64 index) -{ - struct btrfs_dir_index_prealloc *prealloc; - - prealloc = btrfs_prealloc_delayed_dir_index(dir, name, name_len); - if (IS_ERR(prealloc)) - return PTR_ERR(prealloc); - - return btrfs_insert_delayed_dir_index_prealloc(trans, dir, prealloc, - disk_key, flags, index); -} - static bool btrfs_delete_delayed_insertion_item(struct btrfs_delayed_node *node, u64 index) { @@ -1661,7 +1644,7 @@ static bool btrfs_delete_delayed_insertion_item(struct btrfs_delayed_node *node, /* * For delayed items to insert, we track reserved metadata bytes based * on the number of leaves that we will use. - * See btrfs_insert_delayed_dir_index() and + * See btrfs_insert_delayed_dir_index_prealloc() and * btrfs_delayed_item_reserve_metadata()). */ ASSERT(item->bytes_reserved == 0); diff --git a/fs/btrfs/delayed-inode.h b/fs/btrfs/delayed-inode.h index 6d12a145489f2b..57ba96cfaf9cb3 100644 --- a/fs/btrfs/delayed-inode.h +++ b/fs/btrfs/delayed-inode.h @@ -115,11 +115,6 @@ struct btrfs_delayed_item { }; void btrfs_init_delayed_root(struct btrfs_delayed_root *delayed_root); -int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans, - const char *name, int name_len, - struct btrfs_inode *dir, - const struct btrfs_disk_key *disk_key, u8 flags, - u64 index); struct btrfs_dir_index_prealloc { struct btrfs_delayed_node *node; diff --git a/fs/btrfs/dir-item.c b/fs/btrfs/dir-item.c index 84f1c64423d328..3a90736915af6f 100644 --- a/fs/btrfs/dir-item.c +++ b/fs/btrfs/dir-item.c @@ -106,8 +106,11 @@ int btrfs_insert_xattr_item(struct btrfs_trans_handle *trans, * Will return 0 or -ENOMEM */ int btrfs_insert_dir_item(struct btrfs_trans_handle *trans, - const struct fscrypt_str *name, struct btrfs_inode *dir, - const struct btrfs_key *location, u8 type, u64 index) + const struct fscrypt_str *name, + struct btrfs_inode *dir, + const struct btrfs_key *location, u8 type, + u64 index, + struct btrfs_dir_index_prealloc *prealloc) { int ret = 0; int ret2 = 0; @@ -119,17 +122,27 @@ int btrfs_insert_dir_item(struct btrfs_trans_handle *trans, struct btrfs_key key; struct btrfs_disk_key disk_key; u32 data_size; + const bool need_delayed_index = (root != root->fs_info->tree_root); key.objectid = btrfs_ino(dir); key.type = BTRFS_DIR_ITEM_KEY; key.offset = btrfs_name_hash(name->name, name->len); path = btrfs_alloc_path(); - if (!path) - return -ENOMEM; + if (!path) { + ret = -ENOMEM; + goto out_free_prealloc; + } btrfs_cpu_key_to_disk(&disk_key, location); + /* Pre-allocate the delayed dir index before modifying the btree. */ + if (need_delayed_index && !prealloc) { + prealloc = btrfs_prealloc_delayed_dir_index(dir, name->name, name->len); + if (IS_ERR(prealloc)) + return PTR_ERR(prealloc); + } + data_size = sizeof(*dir_item) + name->len; dir_item = insert_with_overflow(trans, root, path, &key, data_size, name->name, name->len); @@ -137,7 +150,7 @@ int btrfs_insert_dir_item(struct btrfs_trans_handle *trans, ret = PTR_ERR(dir_item); if (ret == -EEXIST) goto second_insert; - goto out_free; + goto out_free_prealloc; } if (IS_ENCRYPTED(&dir->vfs_inode)) @@ -154,21 +167,21 @@ int btrfs_insert_dir_item(struct btrfs_trans_handle *trans, write_extent_buffer(leaf, name->name, name_ptr, name->len); second_insert: - /* FIXME, use some real flag for selecting the extra index */ - if (root == root->fs_info->tree_root) { + if (!need_delayed_index) { ret = 0; - goto out_free; + goto out_free_prealloc; } btrfs_release_path(path); - ret2 = btrfs_insert_delayed_dir_index(trans, name->name, name->len, dir, - &disk_key, type, index); -out_free: + ret2 = btrfs_insert_delayed_dir_index_prealloc(trans, dir, prealloc, + &disk_key, type, index); if (ret) return ret; - if (ret2) - return ret2; - return 0; + return ret2; + +out_free_prealloc: + btrfs_free_delayed_dir_index_prealloc(trans, prealloc); + return ret; } static struct btrfs_dir_item *btrfs_lookup_match_dir( diff --git a/fs/btrfs/dir-item.h b/fs/btrfs/dir-item.h index e52174a8baf92c..8d22976a0a7712 100644 --- a/fs/btrfs/dir-item.h +++ b/fs/btrfs/dir-item.h @@ -13,12 +13,14 @@ struct btrfs_path; struct btrfs_inode; struct btrfs_root; struct btrfs_trans_handle; +struct btrfs_dir_index_prealloc; int btrfs_check_dir_item_collision(struct btrfs_root *root, u64 dir_ino, const struct fscrypt_str *name); int btrfs_insert_dir_item(struct btrfs_trans_handle *trans, const struct fscrypt_str *name, struct btrfs_inode *dir, - const struct btrfs_key *location, u8 type, u64 index); + const struct btrfs_key *location, u8 type, u64 index, + struct btrfs_dir_index_prealloc *prealloc); struct btrfs_dir_item *btrfs_lookup_dir_item(struct btrfs_trans_handle *trans, struct btrfs_root *root, struct btrfs_path *path, u64 dir, @@ -53,5 +55,4 @@ static inline u64 btrfs_name_hash(const char *name, int len) { return crc32c((u32)~1, name, len); } - #endif diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 5fd579a94fc1d3..255bdfb5742361 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -6905,7 +6905,7 @@ int btrfs_add_link(struct btrfs_trans_handle *trans, return ret; ret = btrfs_insert_dir_item(trans, name, parent_inode, &key, - btrfs_inode_type(inode), index); + btrfs_inode_type(inode), index, NULL); if (ret == -EEXIST || ret == -EOVERFLOW) goto fail_dir_item; else if (unlikely(ret)) { diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c index 6802b94ed76ff4..ca114235bbe1e1 100644 --- a/fs/btrfs/transaction.c +++ b/fs/btrfs/transaction.c @@ -1890,8 +1890,7 @@ static noinline int create_pending_snapshot(struct btrfs_trans_handle *trans, goto fail; ret = btrfs_insert_dir_item(trans, &fname.disk_name, - parent_inode, &key, BTRFS_FT_DIR, - index); + parent_inode, &key, BTRFS_FT_DIR, index, NULL); if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); goto fail; From 515079a778a42d00def587b8c83ad8a3fc6b6b0f Mon Sep 17 00:00:00 2001 From: Jeff Layton Date: Tue, 25 Aug 2026 12:04:19 -0400 Subject: [PATCH 0412/1352] btrfs: handle ENOMEM from btrfs_insert_dir_item() without aborting Now that btrfs_insert_dir_item() returns -ENOMEM before modifying the btree (thanks to delayed dir index pre-allocation), callers can handle ENOMEM gracefully instead of aborting the transaction. - btrfs_add_link(): add -ENOMEM to the recoverable errors alongside -EEXIST and -EOVERFLOW. - btrfs_create_new_inode(): on -ENOMEM from btrfs_add_link(), orphan the newly-created inode instead of aborting. The inode item was already written with nlink 1, and discard_new_inode() marks it bad so eviction won't delete it. So clear_nlink() alone is not enough: persist nlink 0 via btrfs_update_inode(), otherwise orphan cleanup would see nlink > 0, drop the orphan item, and leak the inode. Fall back to aborting only if that update also fails. This turns a filesystem-killing abort into a graceful -ENOMEM return for create(), mkdir(), mknod(), symlink(), and link() under memory pressure. Assisted-by: LLM Signed-off-by: Jeff Layton Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/inode.c | 24 ++++++++++++++++++++++-- 1 file changed, 22 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 255bdfb5742361..ac6f248d1e7f8b 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -6844,7 +6844,27 @@ int btrfs_create_new_inode(struct btrfs_trans_handle *trans, } else { ret = btrfs_add_link(trans, BTRFS_I(dir), BTRFS_I(inode), name, false, BTRFS_I(inode)->dir_index); - if (unlikely(ret)) { + if (ret == -ENOMEM) { + /* + * Orphan the new inode instead of aborting. The inode + * item was already written with nlink 1, and discard's + * eviction won't delete a bad inode, so nlink 0 must be + * persisted here or orphan cleanup would see nlink > 0, + * drop the orphan item, and leak the inode. + */ + clear_nlink(inode); + /* btrfs_orphan_add() aborts the transaction on failure. */ + ret = btrfs_orphan_add(trans, BTRFS_I(inode)); + if (ret) + goto discard; + ret = btrfs_update_inode(trans, BTRFS_I(inode)); + if (ret) { + btrfs_abort_transaction(trans, ret); + goto discard; + } + ret = -ENOMEM; + goto discard; + } else if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); goto discard; } @@ -6906,7 +6926,7 @@ int btrfs_add_link(struct btrfs_trans_handle *trans, ret = btrfs_insert_dir_item(trans, name, parent_inode, &key, btrfs_inode_type(inode), index, NULL); - if (ret == -EEXIST || ret == -EOVERFLOW) + if (ret == -EEXIST || ret == -EOVERFLOW || ret == -ENOMEM) goto fail_dir_item; else if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); From 39ac2fba40583e8d4c1764b079e4c3572a37f0c1 Mon Sep 17 00:00:00 2001 From: Jeff Layton Date: Tue, 25 Aug 2026 12:04:20 -0400 Subject: [PATCH 0413/1352] btrfs: pre-allocate delayed dir index for non-overwrite rename For rename() without an overwrite target, pre-allocate the delayed dir index before any btree modifications so that ENOMEM can be returned before the source is unlinked from the old directory. Add a prealloc parameter to btrfs_add_link() that allows callers to pass pre-allocated delayed dir index resources. When provided, btrfs_add_link() takes ownership: it either passes the prealloc to btrfs_insert_dir_item() (which commits or frees it), or frees it on early error. All existing callers pass NULL to preserve the current behavior. In btrfs_rename(), when new_inode is NULL (no overwrite), call btrfs_prealloc_delayed_dir_index() before the first btree modification and pass the result through to btrfs_add_link(). If the prealloc fails, -ENOMEM is returned before any btree state has changed. The local prealloc pointer is cleared once ownership passes to btrfs_add_link(), so the out_fail path only frees one we still own. For overwrite rename (new_inode != NULL), the transaction still aborts on ENOMEM since earlier unlink operations have already made irreversible btree modifications. Assisted-by: LLM Signed-off-by: Jeff Layton Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/btrfs_inode.h | 4 +++- fs/btrfs/inode.c | 39 +++++++++++++++++++++++++++++++-------- fs/btrfs/tree-log.c | 4 ++-- 3 files changed, 36 insertions(+), 11 deletions(-) diff --git a/fs/btrfs/btrfs_inode.h b/fs/btrfs/btrfs_inode.h index 89e5e9c0c904f6..46c62f980c24ac 100644 --- a/fs/btrfs/btrfs_inode.h +++ b/fs/btrfs/btrfs_inode.h @@ -32,6 +32,7 @@ struct btrfs_trans_handle; struct btrfs_bio; struct btrfs_file_extent; struct btrfs_delayed_node; +struct btrfs_dir_index_prealloc; /* * Since we search a directory based on f_pos (struct dir_context::pos) we have @@ -523,7 +524,8 @@ int btrfs_unlink_inode(struct btrfs_trans_handle *trans, const struct fscrypt_str *name); int btrfs_add_link(struct btrfs_trans_handle *trans, struct btrfs_inode *parent_inode, struct btrfs_inode *inode, - const struct fscrypt_str *name, bool add_backref, u64 index); + const struct fscrypt_str *name, bool add_backref, u64 index, + struct btrfs_dir_index_prealloc *prealloc); int btrfs_delete_subvolume(struct btrfs_inode *dir, struct dentry *dentry); int btrfs_truncate_block(struct btrfs_inode *inode, u64 offset, u64 start, u64 end); diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index ac6f248d1e7f8b..d2d784e42c9647 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -6843,7 +6843,7 @@ int btrfs_create_new_inode(struct btrfs_trans_handle *trans, } } else { ret = btrfs_add_link(trans, BTRFS_I(dir), BTRFS_I(inode), name, - false, BTRFS_I(inode)->dir_index); + false, BTRFS_I(inode)->dir_index, NULL); if (ret == -ENOMEM) { /* * Orphan the new inode instead of aborting. The inode @@ -6895,7 +6895,8 @@ int btrfs_create_new_inode(struct btrfs_trans_handle *trans, */ int btrfs_add_link(struct btrfs_trans_handle *trans, struct btrfs_inode *parent_inode, struct btrfs_inode *inode, - const struct fscrypt_str *name, bool add_backref, u64 index) + const struct fscrypt_str *name, bool add_backref, u64 index, + struct btrfs_dir_index_prealloc *prealloc) { int ret = 0; struct btrfs_key key; @@ -6921,11 +6922,13 @@ int btrfs_add_link(struct btrfs_trans_handle *trans, } /* Nothing to clean up yet */ - if (ret) + if (ret) { + btrfs_free_delayed_dir_index_prealloc(trans, prealloc); return ret; + } ret = btrfs_insert_dir_item(trans, name, parent_inode, &key, - btrfs_inode_type(inode), index, NULL); + btrfs_inode_type(inode), index, prealloc); if (ret == -EEXIST || ret == -EOVERFLOW || ret == -ENOMEM) goto fail_dir_item; else if (unlikely(ret)) { @@ -7079,7 +7082,7 @@ static int btrfs_link(struct dentry *old_dentry, struct inode *dir, inode_set_ctime_current(inode); ret = btrfs_add_link(trans, BTRFS_I(dir), BTRFS_I(inode), - &fname.disk_name, true, index); + &fname.disk_name, true, index, NULL); if (ret) goto fail; @@ -8493,14 +8496,14 @@ static int btrfs_rename_exchange(struct inode *old_dir, } ret = btrfs_add_link(trans, BTRFS_I(new_dir), BTRFS_I(old_inode), - new_name, false, old_idx); + new_name, false, old_idx, NULL); if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); goto out_fail; } ret = btrfs_add_link(trans, BTRFS_I(old_dir), BTRFS_I(new_inode), - old_name, false, new_idx); + old_name, false, new_idx, NULL); if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); goto out_fail; @@ -8573,6 +8576,7 @@ static int btrfs_rename(struct mnt_idmap *idmap, struct inode *new_inode = d_inode(new_dentry); struct inode *old_inode = d_inode(old_dentry); struct btrfs_rename_ctx rename_ctx; + struct btrfs_dir_index_prealloc *prealloc = NULL; u64 index = 0; int ret; int ret2; @@ -8696,6 +8700,23 @@ static int btrfs_rename(struct mnt_idmap *idmap, if (ret) goto out_fail; + /* + * When not overwriting an existing entry, pre-allocate the delayed dir + * index now so that ENOMEM is returned before any btree modifications. + * For the overwrite case, too many btree changes have already happened + * by the time btrfs_add_link() is called. + */ + if (!new_inode) { + prealloc = btrfs_prealloc_delayed_dir_index(BTRFS_I(new_dir), + new_fname.disk_name.name, + new_fname.disk_name.len); + if (IS_ERR(prealloc)) { + ret = PTR_ERR(prealloc); + prealloc = NULL; + goto out_fail; + } + } + BTRFS_I(old_inode)->dir_index = 0ULL; if (unlikely(old_ino == BTRFS_FIRST_FREE_OBJECTID)) { /* force full log commit if subvolume involved. */ @@ -8791,7 +8812,8 @@ static int btrfs_rename(struct mnt_idmap *idmap, } ret = btrfs_add_link(trans, BTRFS_I(new_dir), BTRFS_I(old_inode), - &new_fname.disk_name, false, index); + &new_fname.disk_name, false, index, prealloc); + prealloc = NULL; if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); goto out_fail; @@ -8816,6 +8838,7 @@ static int btrfs_rename(struct mnt_idmap *idmap, } } out_fail: + btrfs_free_delayed_dir_index_prealloc(trans, prealloc); if (logs_pinned) { btrfs_end_log_trans(root); btrfs_end_log_trans(dest); diff --git a/fs/btrfs/tree-log.c b/fs/btrfs/tree-log.c index a00094604e544e..cfb0b0e9e1248e 100644 --- a/fs/btrfs/tree-log.c +++ b/fs/btrfs/tree-log.c @@ -1683,7 +1683,7 @@ static noinline int add_inode_ref(struct walk_control *wc) } /* insert our name */ - ret = btrfs_add_link(trans, dir, inode, &name, false, ref_index); + ret = btrfs_add_link(trans, dir, inode, &name, false, ref_index, NULL); if (ret) { btrfs_abort_log_replay(wc, ret, "failed to add link for inode %llu in dir %llu ref_index %llu name %.*s root %llu", @@ -2031,7 +2031,7 @@ static noinline int insert_one_name(struct btrfs_trans_handle *trans, return PTR_ERR(dir); } - ret = btrfs_add_link(trans, dir, inode, name, true, index); + ret = btrfs_add_link(trans, dir, inode, name, true, index, NULL); /* FIXME, put inode into FIXUP list */ From fa2357812db8ab30f70f479701d892899247c305 Mon Sep 17 00:00:00 2001 From: Usama Arif Date: Fri, 4 Sep 2026 09:41:48 -0700 Subject: [PATCH 0414/1352] btrfs: zstd: avoid a copy in zstd_decompress_bio() zstd_decompress_bio() gives zstd a sectorsize-sized scratch buffer, and btrfs_decompress_buf2page() then copies the part overlapping the read bio into the destination folios. Every delivered byte is written twice. Instead, choose the output buffer per streaming call. zstd_map_dest() kmaps the current page-bounded segment of the read bio, so zstd writes into the page cache directly. The scratch buffer is kept only for output with no destination: the prefix before a read starting inside a compressed extent, which zstd cannot skip, and gaps left by folios already in the page cache. Varying the output buffer across calls is safe: btrfs uses the default ZSTD_bm_buffered mode, where the sliding window lives in the dstream's internal buffer and the caller's dst is a pure sink. The read bio's iterator must still advance by exactly the bytes delivered, since btrfs_decompress_bio() zero-fills from it; that used to happen inside btrfs_decompress_buf2page() and is now an explicit bio_advance(), made only for output that reached a folio. bio_iter_iovec() exposes at most one base page, so direct output is page-bounded. Compared to the old sectorsize-sized chunks, this can increase stream calls when sectorsize exceeds PAGE_SIZE, but eliminates the extra btrfs copy for output delivered to the read bio; the 64 KiB sectorsize row below shows the copy still wins there. Benchmarked the change in 2-vCPU x86-64 KVM guests (4 KiB pages, RAM disk) using a 64 MiB zstd-compressed file. Results are medians of seven cold-cache reads in each of six interleaved A/B boot pairs; mincore confirmed zero resident pages before every run. Normal sequential reads with readahead produced: sectorsize base patched reduction 4 KiB 8.678 ms 8.004 ms 7.80% 16 KiB 8.216 ms 7.934 ms 3.64% 64 KiB 7.875 ms 7.344 ms 6.88% Random 4 KiB preads at 4 KiB sectorsize, means of six interleaved A/B boot pairs, patched better in all six: base patched gain 264.33 MB/s 272.67 MB/s 3.2% Signed-off-by: Usama Arif Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/zstd.c | 69 +++++++++++++++++++++++++++++++++++++++---------- 1 file changed, 56 insertions(+), 13 deletions(-) diff --git a/fs/btrfs/zstd.c b/fs/btrfs/zstd.c index 58d9ff76fe07bb..280ac5273438b7 100644 --- a/fs/btrfs/zstd.c +++ b/fs/btrfs/zstd.c @@ -589,10 +589,48 @@ int zstd_compress_bio(struct list_head *ws, struct compressed_bio *cb) return ret; } +/* + * Map the destination for the next chunk of output. + * + * @decompressed is the offset of the next output byte inside the fully + * decompressed extent. If that offset has reached the current destination + * segment, its page-bounded bio_vec is kmapped so that zstd can write into the + * page cache directly, and the number of bytes writable there is returned. + * Otherwise @kaddr_ret is set to NULL and the number of bytes to skip before + * that segment is returned. This covers both the initial prefix and gaps in + * the destination bio. + */ +static u32 zstd_map_dest(struct compressed_bio *cb, u32 decompressed, + void **kaddr_ret) +{ + struct bio *orig_bio = &cb->orig_bbio->bio; + struct bio_vec bvec; + u32 bvec_offset; + u32 off; + + bvec = bio_iter_iovec(orig_bio, orig_bio->bi_iter); + /* + * cb->start may underflow, but subtracting that value can still give us + * the correct offset inside the full decompressed extent. + */ + bvec_offset = page_offset(bvec.bv_page) + bvec.bv_offset - cb->start; + + if (decompressed < bvec_offset) { + *kaddr_ret = NULL; + return bvec_offset - decompressed; + } + + off = decompressed - bvec_offset; + ASSERT(off < bvec.bv_len); + *kaddr_ret = bvec_kmap_local(&bvec) + off; + return bvec.bv_len - off; +} + int zstd_decompress_bio(struct list_head *ws, struct compressed_bio *cb) { struct btrfs_fs_info *fs_info = cb_to_fs_info(cb); struct workspace *workspace = list_entry(ws, struct workspace, list); + struct bio *orig_bio = &cb->orig_bbio->bio; struct folio_iter fi; size_t srclen = bio_get_size(&cb->bbio.bio); zstd_dstream *stream; @@ -600,7 +638,6 @@ int zstd_decompress_bio(struct list_head *ws, struct compressed_bio *cb) const unsigned int min_folio_size = btrfs_min_folio_size(fs_info); unsigned long folio_in_index = 0; unsigned long total_folios_in = DIV_ROUND_UP(srclen, min_folio_size); - unsigned long buf_start; unsigned long total_out = 0; bio_first_folio(&fi, &cb->bbio.bio, 0); @@ -624,15 +661,26 @@ int zstd_decompress_bio(struct list_head *ws, struct compressed_bio *cb) workspace->in_buf.pos = 0; workspace->in_buf.size = min_t(size_t, srclen, min_folio_size); - workspace->out_buf.dst = workspace->buf; - workspace->out_buf.pos = 0; - workspace->out_buf.size = fs_info->sectorsize; - - while (1) { + while (orig_bio->bi_iter.bi_size) { size_t ret2; + void *kaddr; + u32 dstlen; + + dstlen = zstd_map_dest(cb, total_out, &kaddr); + if (kaddr) { + workspace->out_buf.dst = kaddr; + workspace->out_buf.size = dstlen; + } else { + workspace->out_buf.dst = workspace->buf; + workspace->out_buf.size = min_t(u32, dstlen, + fs_info->sectorsize); + } + workspace->out_buf.pos = 0; ret2 = zstd_decompress_stream(stream, &workspace->out_buf, &workspace->in_buf); + if (kaddr) + kunmap_local(kaddr); if (unlikely(zstd_is_error(ret2))) { struct btrfs_inode *inode = cb->bbio.inode; @@ -643,14 +691,9 @@ int zstd_decompress_bio(struct list_head *ws, struct compressed_bio *cb) ret = -EIO; goto done; } - buf_start = total_out; total_out += workspace->out_buf.pos; - workspace->out_buf.pos = 0; - - ret = btrfs_decompress_buf2page(workspace->out_buf.dst, - total_out - buf_start, cb, buf_start); - if (ret == 0) - break; + if (kaddr) + bio_advance(orig_bio, workspace->out_buf.pos); if (workspace->in_buf.pos >= srclen) break; From 07b46b7b721d838be5ac0c9d2533f64e141db340 Mon Sep 17 00:00:00 2001 From: Zhen Ni Date: Mon, 22 Dec 2025 11:59:42 +0800 Subject: [PATCH 0415/1352] btrfs: replace is_data_bbio() with is_data_inode() for direct usage After commit 81cea6cd7041 ("btrfs: remove btrfs_bio::fs_info by extracting it from btrfs_bio::inode"), the btrfs_bio::inode field is mandatory for all btrfs_bio allocations. The NULL check is redundant and can be removed. As is_data_bbio() would be a trivial wrapper for is_data_bbio() replace all calls in in bio.c Link: https://lore.kernel.org/linux-btrfs/20251219084316.1164580-1-zhen.ni@easystack.cn Signed-off-by: Zhen Ni Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/bio.c | 18 ++++++------------ 1 file changed, 6 insertions(+), 12 deletions(-) diff --git a/fs/btrfs/bio.c b/fs/btrfs/bio.c index 771b7d598aeeeb..2c6234f6182f89 100644 --- a/fs/btrfs/bio.c +++ b/fs/btrfs/bio.c @@ -27,15 +27,9 @@ struct btrfs_failed_bio { atomic_t repair_count; }; -/* Is this a data path I/O that needs storage layer checksum and repair? */ -static inline bool is_data_bbio(const struct btrfs_bio *bbio) -{ - return bbio->inode && is_data_inode(bbio->inode); -} - static bool bbio_has_ordered_extent(const struct btrfs_bio *bbio) { - return is_data_bbio(bbio) && btrfs_op(&bbio->bio) == BTRFS_MAP_WRITE; + return is_data_inode(bbio->inode) && btrfs_op(&bbio->bio) == BTRFS_MAP_WRITE; } /* @@ -356,7 +350,7 @@ static void simple_end_io_work(struct work_struct *work) if (bio_op(bio) == REQ_OP_READ) { /* Metadata reads are checked and repaired by the submitter. */ - if (is_data_bbio(bbio)) + if (is_data_inode(bbio->inode)) return btrfs_check_read_bio(bbio, bbio->bio.bi_private); return btrfs_bio_end_io(bbio, bbio->bio.bi_status); } @@ -390,7 +384,7 @@ static void btrfs_raid56_end_io(struct bio *bio) btrfs_bio_counter_dec(bioc->fs_info); bbio->mirror_num = bioc->mirror_num; - if (bio_op(bio) == REQ_OP_READ && is_data_bbio(bbio)) + if (bio_op(bio) == REQ_OP_READ && is_data_inode(bbio->inode)) btrfs_check_read_bio(bbio, NULL); else btrfs_bio_end_io(bbio, bbio->bio.bi_status); @@ -753,7 +747,7 @@ static bool btrfs_submit_chunk(struct btrfs_bio *bbio, int mirror_num) * our bio to the physical disk location, so we need to save the * original bytenr so we know what we're checksumming. */ - if (bio_op(bio) == REQ_OP_WRITE && is_data_bbio(bbio)) + if (bio_op(bio) == REQ_OP_WRITE && is_data_inode(bbio->inode)) bbio->orig_logical = logical; bbio->can_use_append = btrfs_use_zone_append(bbio); @@ -779,7 +773,7 @@ static bool btrfs_submit_chunk(struct btrfs_bio *bbio, int mirror_num) * Save the iter for the end_io handler and preload the checksums for * data reads. */ - if (bio_op(bio) == REQ_OP_READ && is_data_bbio(bbio)) { + if (bio_op(bio) == REQ_OP_READ && is_data_inode(bbio->inode)) { bbio->saved_iter = bio->bi_iter; ret = btrfs_lookup_bio_sums(bbio); status = errno_to_blk_status(ret); @@ -788,7 +782,7 @@ static bool btrfs_submit_chunk(struct btrfs_bio *bbio, int mirror_num) } if (btrfs_op(bio) == BTRFS_MAP_WRITE) { - if (is_data_bbio(bbio) && bioc && bioc->use_rst) { + if (is_data_inode(bbio->inode) && bioc && bioc->use_rst) { /* * No locking for the list update, as we only add to * the list in the I/O submission path, and list From e6511e5107e5c400a8305fa29b0ac7c75e2855a2 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Tue, 8 Sep 2026 07:47:39 +0930 Subject: [PATCH 0416/1352] btrfs: use kvmalloc() for b-tree split_item() [BUG] There is a bug report that the kmalloc() call inside split_item() failed with the following call trace, and triggered a transaction abort: kworker/u69:8: page allocation failure: order:4, mode:0x40c40(GFP_NOFS|__GFP_COMP), nodemask=(null) CPU: 3 UID: 0 PID: 1154528 Comm: kworker/u69:8 Not tainted 7.0.2 #1 PREEMPTLAZY Workqueue: events_unbound btrfs_async_reclaim_metadata_space Call Trace: dump_stack_lvl+0x47/0x60 warn_alloc.cold+0x67/0xec __alloc_pages_slowpath.constprop.0+0x9bf/0xed0 __alloc_frozen_pages_noprof+0x1ac/0x1c0 ___kmalloc_large_node+0x9d/0xc0 __kmalloc_noprof+0x17b/0x1f0 split_item+0x9e/0x2e0 btrfs_del_csums+0x285/0x400 __btrfs_free_extent.isra.0+0x6de/0x12b0 __btrfs_run_delayed_refs+0x522/0x10c0 btrfs_run_delayed_refs+0x4d/0x1d0 flush_space+0x34d/0x4e0 do_async_reclaim_metadata_space+0x89/0x1d0 btrfs_async_reclaim_metadata_space+0x44/0x60 process_one_work+0x145/0x230 worker_thread+0x185/0x2e0 kthread+0xca/0x100 ret_from_fork+0x14e/0x200 ret_from_fork_asm+0x11/0x20 BTRFS error (device dm-3 state A): Transaction aborted (error -12) BTRFS: error (device dm-3 state A) in btrfs_del_csums:1053: errno=-12 Out of memory BTRFS info (device dm-3 state EA): forced readonly BTRFS: error (device dm-3 state EA) in do_free_extent_accounting:3168: errno=-12 Out of memory BTRFS error (device dm-3 state EA): failed to run delayed ref for logical 1202913873920 num_bytes 274432 type 184 action 2 ref_mod 1: -12 BTRFS: error (device dm-3 state EA) in btrfs_run_delayed_refs:2247: errno=-12 Out of memory [CAUSE] The kmalloc() call is to allocate a buffer to store the full item. However as shown in the above call trace, the order can be high (4), and since we're using GFP_NOFS, it's impossible to reclaim memory by writing back dirty pages. When there is no physically contiguous memory left, such high order allocation can easily fail, and if such kmalloc() happens in a critical path we can trigger a transaction abort. [FIX] Instead of kmalloc(), which requires physically contiguous pages, use kvmalloc(). There is no special requirement for physically contiguous pages here, we just want virtually contiguous memory as a buffer. Reported-by: xavierbachmeyer182 Link: https://lore.kernel.org/linux-btrfs/250decb0-d940-4fe6-9b54-d06e1b293a1b@suse.com/ Reviewed-by: Johannes Thumshirn Reviewed-by: Daniel Vacek Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/ctree.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/ctree.c b/fs/btrfs/ctree.c index 8fe330d81b8ff3..fef0e49dd91882 100644 --- a/fs/btrfs/ctree.c +++ b/fs/btrfs/ctree.c @@ -3943,7 +3943,7 @@ static noinline int split_item(struct btrfs_trans_handle *trans, orig_offset = btrfs_item_offset(leaf, path->slots[0]); item_size = btrfs_item_size(leaf, path->slots[0]); - buf = kmalloc(item_size, GFP_NOFS); + buf = kvmalloc(item_size, GFP_NOFS); if (!buf) return -ENOMEM; @@ -3981,7 +3981,7 @@ static noinline int split_item(struct btrfs_trans_handle *trans, btrfs_mark_buffer_dirty(trans, leaf); BUG_ON(btrfs_leaf_free_space(leaf) < 0); - kfree(buf); + kvfree(buf); return 0; } From 85c82632b7d2893b1acd4a35fdc2c6f43d9601a2 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Tue, 8 Sep 2026 16:45:39 +0930 Subject: [PATCH 0417/1352] btrfs: tree-log: use kvmalloc() for overwrite_item() The @src_copy buffer utilized inside overwrite_item() can be as large as the nodesize. For an existing btrfs with 64KiB nodesize, it means there is a high chance to fail the kmalloc() call if there is not enough physically contiguous pages. Meanwhile there is really no need for such physically contiguous pages, as we only use that buffer to compare the content of the item. Use kvmalloc() to replace the kmalloc() call. For most cases that kvmalloc() call will be easily fulfilled by regular kmalloc(), but for really large items and large nodes, kvmalloc() will have a much higher chance to get memory allocated. Reviewed-by: Daniel Vacek Reviewed-by: Johannes Thumshirn Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/tree-log.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/tree-log.c b/fs/btrfs/tree-log.c index cfb0b0e9e1248e..7fb476b19eb457 100644 --- a/fs/btrfs/tree-log.c +++ b/fs/btrfs/tree-log.c @@ -503,7 +503,7 @@ static int overwrite_item(struct walk_control *wc) btrfs_release_path(wc->subvol_path); return 0; } - src_copy = kmalloc(item_size, GFP_NOFS); + src_copy = kvmalloc(item_size, GFP_NOFS); if (!src_copy) { btrfs_abort_log_replay(wc, -ENOMEM, "failed to allocate memory for log leaf item"); @@ -514,7 +514,7 @@ static int overwrite_item(struct walk_control *wc) dst_ptr = btrfs_item_ptr_offset(dst_eb, dst_slot); ret = memcmp_extent_buffer(dst_eb, src_copy, dst_ptr, item_size); - kfree(src_copy); + kvfree(src_copy); /* * they have the same contents, just return, this saves * us from cowing blocks in the destination tree and doing From 41a3d31a401597b4f7f7bbdb3cbeae47e6e3e5aa Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Tue, 8 Sep 2026 16:45:40 +0930 Subject: [PATCH 0418/1352] btrfs: use kvmalloc() for uncompress_inline() Although btrfs doesn't support inlined extents larger than PAGE_SIZE for bs > ps cases, it's still possible for the experimental bs > ps support to mount a btrfs created on a system with a much larger page size, thus can still hit an inlined extent that is way larger than the current page size. E.g. a compressed inline extent which has 32K compressed size, is created on 64K page sized ARM64 with 64K sectorsize, then mounted on x86_64 with the experimental bs > ps support. In that case, when reading the compressed inline extent, we need to allocate a buffer that is the same size as the compressed inline extent (32K). That kmalloc() call will request physically contiguous memory for that 32K allocation, and if the system has a very fragmented memory space, such allocation can fail. But there is really no reason that we require such buffer to be physically contiguous, so change it to kvmalloc() to reduce the chance of allocation failure for bs > ps cases. And for all bs <= ps cases, the kvmalloc() call will just be fulfilled by kmalloc() so this will not bring any change to the most common cases. Only bs > ps will get the benefit of less memory allocation failure. Reviewed-by: Daniel Vacek Reviewed-by: Johannes Thumshirn Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/inode.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index d2d784e42c9647..3f65c3dd91e7c9 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -7149,7 +7149,7 @@ static noinline int uncompress_inline(struct btrfs_path *path, compress_type = btrfs_file_extent_compression(leaf, item); max_size = btrfs_file_extent_ram_bytes(leaf, item); inline_size = btrfs_file_extent_inline_item_len(leaf, path->slots[0]); - tmp = kmalloc(inline_size, GFP_NOFS); + tmp = kvmalloc(inline_size, GFP_NOFS); if (!tmp) return -ENOMEM; ptr = btrfs_file_extent_inline_start(item); @@ -7170,7 +7170,7 @@ static noinline int uncompress_inline(struct btrfs_path *path, if (max_size < blocksize) folio_zero_range(folio, max_size, blocksize - max_size); - kfree(tmp); + kvfree(tmp); return ret; } From d7f3b2c4dd8eafc66680034affc2efd1a67329ef Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Tue, 8 Sep 2026 16:45:41 +0930 Subject: [PATCH 0419/1352] btrfs: use kvmalloc() to allocate compression workspace buffer for zlib and zstd With the experimental bs > ps support, the workspace buffer for both zlib and zstd can be as large as 64K, and on 4K page sized systems such kmalloc() calls have a much higher chance to fail, as that requires physically contiguous memory to fulfill such allocation. The same also applies to S390's hardware accelerated path, which requires a buffer size of 4 pages. Meanwhile lzo is already using kvmalloc() for its buffer, and there is no special requirement for any physically contiguous memory anyway. So change the zlib and zstd workspace buffer allocation to use kvmalloc() to reduce the chance of memory allocation failure. Reviewed-by: Daniel Vacek Reviewed-by: Johannes Thumshirn Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/zlib.c | 10 +++++----- fs/btrfs/zstd.c | 4 ++-- 2 files changed, 7 insertions(+), 7 deletions(-) diff --git a/fs/btrfs/zlib.c b/fs/btrfs/zlib.c index 486b52db583ecb..e995d0b1452b92 100644 --- a/fs/btrfs/zlib.c +++ b/fs/btrfs/zlib.c @@ -49,7 +49,7 @@ void zlib_free_workspace(struct list_head *ws) struct workspace *workspace = list_entry(ws, struct workspace, list); kvfree(workspace->strm.workspace); - kfree(workspace->buf); + kvfree(workspace->buf); kfree(workspace); } @@ -84,13 +84,13 @@ struct list_head *zlib_alloc_workspace(struct btrfs_fs_info *fs_info, unsigned i workspace->level = level; workspace->buf = NULL; if (need_special_buffer(fs_info)) { - workspace->buf = kmalloc(ZLIB_DFLTCC_BUF_SIZE, - __GFP_NOMEMALLOC | __GFP_NORETRY | - __GFP_NOWARN | GFP_NOIO); + workspace->buf = kvmalloc(ZLIB_DFLTCC_BUF_SIZE, + __GFP_NOMEMALLOC | __GFP_NORETRY | + __GFP_NOWARN | GFP_NOIO); workspace->buf_size = ZLIB_DFLTCC_BUF_SIZE; } if (!workspace->buf) { - workspace->buf = kmalloc(fs_info->sectorsize, GFP_KERNEL); + workspace->buf = kvmalloc(fs_info->sectorsize, GFP_KERNEL); workspace->buf_size = fs_info->sectorsize; } if (!workspace->strm.workspace || !workspace->buf) diff --git a/fs/btrfs/zstd.c b/fs/btrfs/zstd.c index 280ac5273438b7..cc92d0b1b948cd 100644 --- a/fs/btrfs/zstd.c +++ b/fs/btrfs/zstd.c @@ -373,7 +373,7 @@ void zstd_free_workspace(struct list_head *ws) struct workspace *workspace = list_entry(ws, struct workspace, list); kvfree(workspace->mem); - kfree(workspace->buf); + kvfree(workspace->buf); kfree(workspace); } @@ -391,7 +391,7 @@ struct list_head *zstd_alloc_workspace(struct btrfs_fs_info *fs_info, int level) workspace->req_level = level; workspace->last_used = jiffies; workspace->mem = kvmalloc(workspace->size, GFP_KERNEL | __GFP_NOWARN); - workspace->buf = kmalloc(fs_info->sectorsize, GFP_KERNEL); + workspace->buf = kvmalloc(fs_info->sectorsize, GFP_KERNEL); if (!workspace->mem || !workspace->buf) goto fail; From fdcb76991d976473f084e23e63a68d7b76be9301 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Mon, 7 Sep 2026 16:19:55 -0400 Subject: [PATCH 0420/1352] btrfs: tests: rename process_page_range() to process_folio_range() It already operates on folios. No functional change. Reviewed-by: Qu Wenruo Signed-off-by: Tal Zussman Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/tests/extent-io-tests.c | 20 ++++++++++---------- 1 file changed, 10 insertions(+), 10 deletions(-) diff --git a/fs/btrfs/tests/extent-io-tests.c b/fs/btrfs/tests/extent-io-tests.c index 23459cd4e50388..6eb55bfb2bd4e6 100644 --- a/fs/btrfs/tests/extent-io-tests.c +++ b/fs/btrfs/tests/extent-io-tests.c @@ -18,8 +18,8 @@ #define PROCESS_RELEASE (1U << 1) #define PROCESS_TEST_LOCKED (1U << 2) -static noinline int process_page_range(struct inode *inode, u64 start, u64 end, - unsigned long flags) +static noinline int process_folio_range(struct inode *inode, u64 start, u64 end, + unsigned long flags) { int ret; struct folio_batch fbatch; @@ -221,8 +221,8 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) test_start, max_bytes - 1, start, end); goto out_bits; } - if (process_page_range(inode, start, end, - PROCESS_TEST_LOCKED | PROCESS_UNLOCK)) { + if (process_folio_range(inode, start, end, + PROCESS_TEST_LOCKED | PROCESS_UNLOCK)) { test_err("there were unlocked pages in the range"); goto out_bits; } @@ -276,8 +276,8 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) test_start, total_dirty - 1, start, end); goto out_bits; } - if (process_page_range(inode, start, end, - PROCESS_TEST_LOCKED | PROCESS_UNLOCK)) { + if (process_folio_range(inode, start, end, + PROCESS_TEST_LOCKED | PROCESS_UNLOCK)) { test_err("pages in range were not all locked"); goto out_bits; } @@ -317,8 +317,8 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) test_start, test_start + PAGE_SIZE - 1, start, end); goto out_bits; } - if (process_page_range(inode, start, end, PROCESS_TEST_LOCKED | - PROCESS_UNLOCK)) { + if (process_folio_range(inode, start, end, PROCESS_TEST_LOCKED | + PROCESS_UNLOCK)) { test_err("pages in range were not all locked"); goto out_bits; } @@ -330,8 +330,8 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) out: if (locked_page) put_page(locked_page); - process_page_range(inode, 0, total_dirty - 1, - PROCESS_UNLOCK | PROCESS_RELEASE); + process_folio_range(inode, 0, total_dirty - 1, + PROCESS_UNLOCK | PROCESS_RELEASE); iput(inode); out_root_info: btrfs_free_dummy_root(root); From 0357398b250725628a2c18f04fda27bc2670a759 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Mon, 7 Sep 2026 16:19:56 -0400 Subject: [PATCH 0421/1352] btrfs: tests: convert test_find_delalloc() to use folios This removes the last btrfs callers of find_or_create_page(), find_lock_page(), SetPageDirty(), ClearPageDirty(), and get_page(), and 15 calls to compound_head(). The folio lookups return an ERR_PTR instead of NULL, so adjust the error handling. Update the comments and test messages accordingly. The test still works in PAGE_SIZE units, which relies on the test inode never getting large folios, so assert that the folios are order-0 where that matters. Reviewed-by: Qu Wenruo Signed-off-by: Tal Zussman Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/tests/extent-io-tests.c | 97 ++++++++++++++++---------------- 1 file changed, 49 insertions(+), 48 deletions(-) diff --git a/fs/btrfs/tests/extent-io-tests.c b/fs/btrfs/tests/extent-io-tests.c index 6eb55bfb2bd4e6..ee8eabac47f110 100644 --- a/fs/btrfs/tests/extent-io-tests.c +++ b/fs/btrfs/tests/extent-io-tests.c @@ -112,8 +112,8 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) struct btrfs_root *root = NULL; struct inode *inode = NULL; struct extent_io_tree *tmp; - struct page *page; - struct page *locked_page = NULL; + struct folio *folio; + struct folio *locked_folio = NULL; /* In this test we need at least 2 file extents at its maximum size */ u64 max_bytes = BTRFS_MAX_EXTENT_SIZE; u64 total_dirty = 2 * max_bytes; @@ -152,23 +152,27 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) btrfs_extent_io_tree_init(NULL, tmp, IO_TREE_SELFTEST); /* - * First go through and create and mark all of our pages dirty, we pin - * everything to make sure our pages don't get evicted and screw up our + * First go through and create and mark all of our folios dirty, we pin + * everything to make sure our folios don't get evicted and screw up our * test. */ for (pgoff_t index = 0; index < (total_dirty >> PAGE_SHIFT); index++) { - page = find_or_create_page(inode->i_mapping, index, GFP_KERNEL); - if (!page) { - test_err("failed to allocate test page"); - ret = -ENOMEM; + folio = __filemap_get_folio(inode->i_mapping, index, + FGP_LOCK | FGP_ACCESSED | FGP_CREAT, + GFP_KERNEL); + if (IS_ERR(folio)) { + test_err("failed to allocate test folio"); + ret = PTR_ERR(folio); goto out; } - SetPageDirty(page); + /* The ranges below assume page sized folios. */ + ASSERT(folio_order(folio) == 0); + folio_set_dirty(folio); if (index) { - unlock_page(page); + folio_unlock(folio); } else { - get_page(page); - locked_page = page; + folio_get(folio); + locked_folio = folio; } } @@ -179,8 +183,7 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) btrfs_set_extent_bit(tmp, 0, sectorsize - 1, EXTENT_DELALLOC, NULL); start = 0; end = start + PAGE_SIZE - 1; - found = find_lock_delalloc_range(inode, page_folio(locked_page), &start, - &end); + found = find_lock_delalloc_range(inode, locked_folio, &start, &end); if (!found) { test_err("should have found at least one delalloc"); goto out_bits; @@ -191,8 +194,8 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) goto out_bits; } btrfs_unlock_extent(tmp, start, end, NULL); - unlock_page(locked_page); - put_page(locked_page); + folio_unlock(locked_folio); + folio_put(locked_folio); /* * Test this scenario @@ -201,17 +204,17 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) * |--- search ---| */ test_start = SZ_64M; - locked_page = find_lock_page(inode->i_mapping, - test_start >> PAGE_SHIFT); - if (!locked_page) { - test_err("couldn't find the locked page"); + locked_folio = filemap_lock_folio(inode->i_mapping, test_start >> PAGE_SHIFT); + if (IS_ERR(locked_folio)) { + test_err("couldn't find the locked folio"); + locked_folio = NULL; goto out_bits; } + ASSERT(folio_order(locked_folio) == 0); btrfs_set_extent_bit(tmp, sectorsize, max_bytes - 1, EXTENT_DELALLOC, NULL); start = test_start; end = start + PAGE_SIZE - 1; - found = find_lock_delalloc_range(inode, page_folio(locked_page), &start, - &end); + found = find_lock_delalloc_range(inode, locked_folio, &start, &end); if (!found) { test_err("couldn't find delalloc in our range"); goto out_bits; @@ -223,12 +226,12 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) } if (process_folio_range(inode, start, end, PROCESS_TEST_LOCKED | PROCESS_UNLOCK)) { - test_err("there were unlocked pages in the range"); + test_err("there were unlocked folios in the range"); goto out_bits; } btrfs_unlock_extent(tmp, start, end, NULL); - /* locked_page was unlocked above */ - put_page(locked_page); + /* locked_folio was unlocked above */ + folio_put(locked_folio); /* * Test this scenario @@ -236,16 +239,16 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) * |--- search ---| */ test_start = max_bytes + sectorsize; - locked_page = find_lock_page(inode->i_mapping, test_start >> - PAGE_SHIFT); - if (!locked_page) { - test_err("couldn't find the locked page"); + locked_folio = filemap_lock_folio(inode->i_mapping, test_start >> PAGE_SHIFT); + if (IS_ERR(locked_folio)) { + test_err("couldn't find the locked folio"); + locked_folio = NULL; goto out_bits; } + ASSERT(folio_order(locked_folio) == 0); start = test_start; end = start + PAGE_SIZE - 1; - found = find_lock_delalloc_range(inode, page_folio(locked_page), &start, - &end); + found = find_lock_delalloc_range(inode, locked_folio, &start, &end); if (found) { test_err("found range when we shouldn't have"); goto out_bits; @@ -265,8 +268,7 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) btrfs_set_extent_bit(tmp, max_bytes, total_dirty - 1, EXTENT_DELALLOC, NULL); start = test_start; end = start + PAGE_SIZE - 1; - found = find_lock_delalloc_range(inode, page_folio(locked_page), &start, - &end); + found = find_lock_delalloc_range(inode, locked_folio, &start, &end); if (!found) { test_err("didn't find our range"); goto out_bits; @@ -278,36 +280,35 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) } if (process_folio_range(inode, start, end, PROCESS_TEST_LOCKED | PROCESS_UNLOCK)) { - test_err("pages in range were not all locked"); + test_err("folios in range were not all locked"); goto out_bits; } btrfs_unlock_extent(tmp, start, end, NULL); /* - * Now to test where we run into a page that is no longer dirty in the + * Now to test where we run into a folio that is no longer dirty in the * range we want to find. */ - page = find_get_page(inode->i_mapping, - (max_bytes + SZ_1M) >> PAGE_SHIFT); - if (!page) { - test_err("couldn't find our page"); + folio = filemap_get_folio(inode->i_mapping, (max_bytes + SZ_1M) >> PAGE_SHIFT); + if (IS_ERR(folio)) { + test_err("couldn't find our folio"); goto out_bits; } - ClearPageDirty(page); - put_page(page); + ASSERT(folio_order(folio) == 0); + folio_clear_dirty(folio); + folio_put(folio); /* We unlocked it in the previous test */ - lock_page(locked_page); + folio_lock(locked_folio); start = test_start; end = start + PAGE_SIZE - 1; /* - * Currently if we fail to find dirty pages in the delalloc range we + * Currently if we fail to find dirty folios in the delalloc range we * will adjust max_bytes down to PAGE_SIZE and then re-search. If * this changes at any point in the future we will need to fix this * tests expected behavior. */ - found = find_lock_delalloc_range(inode, page_folio(locked_page), &start, - &end); + found = find_lock_delalloc_range(inode, locked_folio, &start, &end); if (!found) { test_err("didn't find our range"); goto out_bits; @@ -319,7 +320,7 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) } if (process_folio_range(inode, start, end, PROCESS_TEST_LOCKED | PROCESS_UNLOCK)) { - test_err("pages in range were not all locked"); + test_err("folios in range were not all locked"); goto out_bits; } ret = 0; @@ -328,8 +329,8 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) dump_extent_io_tree(tmp); btrfs_clear_extent_bit(tmp, 0, total_dirty - 1, (unsigned)-1, NULL); out: - if (locked_page) - put_page(locked_page); + if (locked_folio) + folio_put(locked_folio); process_folio_range(inode, 0, total_dirty - 1, PROCESS_UNLOCK | PROCESS_RELEASE); iput(inode); From 7b97da893d8acd3a8e4106e4397d2c0890da1d7f Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Mon, 7 Sep 2026 16:19:57 -0400 Subject: [PATCH 0422/1352] btrfs: tests: use eb folio helpers in extent buffer memory checks dump_eb_and_memory_contents() and verify_eb_and_memory() hardcode one page per folio instead of using get_eb_folio_index() and get_eb_offset_in_folio() like the rest of the extent buffer code. Use the helpers and folio_address(). This removes the last struct page usage in the file. No functional change. The tests only run with sectorsize == PAGE_SIZE, and the test extent buffers are backed by order-0 folios. Assisted-by: Claude:claude-fable-5-1 Reviewed-by: Qu Wenruo Signed-off-by: Tal Zussman Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/tests/extent-io-tests.c | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/fs/btrfs/tests/extent-io-tests.c b/fs/btrfs/tests/extent-io-tests.c index ee8eabac47f110..cd045778400dc0 100644 --- a/fs/btrfs/tests/extent-io-tests.c +++ b/fs/btrfs/tests/extent-io-tests.c @@ -672,8 +672,9 @@ static void dump_eb_and_memory_contents(struct extent_buffer *eb, void *memory, const char *test_name) { for (int i = 0; i < eb->len; i++) { - struct page *page = folio_page(eb->folios[i >> PAGE_SHIFT], 0); - void *addr = page_address(page) + offset_in_page(i); + const unsigned long idx = get_eb_folio_index(eb, i); + void *addr = folio_address(eb->folios[idx]) + + get_eb_offset_in_folio(eb, i); if (memcmp(addr, memory + i, 1) != 0) { test_err("%s failed", test_name); @@ -688,9 +689,12 @@ static int verify_eb_and_memory(struct extent_buffer *eb, void *memory, const char *test_name) { for (int i = 0; i < (eb->len >> PAGE_SHIFT); i++) { - void *eb_addr = folio_address(eb->folios[i]); + const unsigned long offset = i << PAGE_SHIFT; + const unsigned long idx = get_eb_folio_index(eb, offset); + void *eb_addr = folio_address(eb->folios[idx]) + + get_eb_offset_in_folio(eb, offset); - if (memcmp(memory + (i << PAGE_SHIFT), eb_addr, PAGE_SIZE) != 0) { + if (memcmp(memory + offset, eb_addr, PAGE_SIZE) != 0) { dump_eb_and_memory_contents(eb, memory, test_name); return -EUCLEAN; } From cb808751b3047381a4785836d838941f3a27fd29 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Mon, 7 Sep 2026 16:19:58 -0400 Subject: [PATCH 0423/1352] btrfs: convert btrfs_compr_pool_scan() to use folios The compression pool holds order-0 folios, but btrfs_compr_pool_scan() walks it as struct page through page->lru. Walk it as folios, matching the other compression pool functions. This removes the last use of page->lru in btrfs and saves a call to compound_head() per freed folio. Reviewed-by: Qu Wenruo Signed-off-by: Tal Zussman Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/compression.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/fs/btrfs/compression.c b/fs/btrfs/compression.c index c62b5148d5ac19..979b2ffbd8fc7e 100644 --- a/fs/btrfs/compression.c +++ b/fs/btrfs/compression.c @@ -168,10 +168,10 @@ static unsigned long btrfs_compr_pool_scan(struct shrinker *sh, struct shrink_co spin_unlock(&compr_pool.lock); list_for_each_safe(tmp, next, &remove) { - struct page *page = list_entry(tmp, struct page, lru); + struct folio *folio = list_entry(tmp, struct folio, lru); - ASSERT(page_ref_count(page) == 1); - put_page(page); + ASSERT(folio_ref_count(folio) == 1); + folio_put(folio); } return freed; From 84e3d269ef2ff5ef376b275902fe38d84c3e6934 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Mon, 7 Sep 2026 16:19:59 -0400 Subject: [PATCH 0424/1352] btrfs: convert heuristic_collect_sample() to use folios Convert the sampling loop to folios. This removes the last caller of find_get_page() in btrfs and saves a call to compound_head() per sampled page. Document that the lookup is not supposed to fail with an ASSERT(). Reviewed-by: Qu Wenruo Signed-off-by: Tal Zussman Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/compression.c | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/fs/btrfs/compression.c b/fs/btrfs/compression.c index 979b2ffbd8fc7e..fa8b92592321ee 100644 --- a/fs/btrfs/compression.c +++ b/fs/btrfs/compression.c @@ -1488,7 +1488,7 @@ static bool sample_repeated_patterns(struct heuristic_ws *ws) static void heuristic_collect_sample(struct inode *inode, u64 start, u64 end, struct heuristic_ws *ws) { - struct page *page; + struct folio *folio; pgoff_t index, index_end; u32 i, curr_sample_pos; u8 *in_data; @@ -1514,8 +1514,10 @@ static void heuristic_collect_sample(struct inode *inode, u64 start, u64 end, curr_sample_pos = 0; while (index < index_end) { - page = find_get_page(inode->i_mapping, index); - in_data = kmap_local_page(page); + folio = filemap_get_folio(inode->i_mapping, index); + ASSERT(!IS_ERR(folio)); + in_data = kmap_local_folio(folio, + offset_in_folio(folio, (u64)index << PAGE_SHIFT)); /* Handle case where the start is not aligned to PAGE_SIZE */ i = start % PAGE_SIZE; while (i < PAGE_SIZE - SAMPLING_READ_SIZE) { @@ -1529,7 +1531,7 @@ static void heuristic_collect_sample(struct inode *inode, u64 start, u64 end, curr_sample_pos += SAMPLING_READ_SIZE; } kunmap_local(in_data); - put_page(page); + folio_put(folio); index++; } From 596b09f209138c759b1e5dec5ea8ab1d594c3250 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Mon, 7 Sep 2026 16:20:00 -0400 Subject: [PATCH 0425/1352] btrfs: fix stale function references in compression comments add_ra_bio_pages() was renamed to add_ra_bio_folios(), and btrfs_compress_filemap_get_folio() wraps filemap_get_folio(), not find_get_page(). Update the comments accordingly. Reviewed-by: Qu Wenruo Signed-off-by: Tal Zussman Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/compression.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/compression.c b/fs/btrfs/compression.c index fa8b92592321ee..20169d02896191 100644 --- a/fs/btrfs/compression.c +++ b/fs/btrfs/compression.c @@ -431,7 +431,7 @@ static noinline int add_ra_bio_folios(struct inode *inode, u64 compressed_end, } /* - * Since add_ra_bio_pages() is always speculative, suppress + * Since add_ra_bio_folios() is always speculative, suppress * allocation warnings. */ masked_constraint_gfp = mapping_gfp_constraint(mapping, constraint_gfp); @@ -960,7 +960,7 @@ bool btrfs_compress_level_valid(unsigned int type, int level) return levels->min_level <= level && level <= levels->max_level; } -/* Wrapper around find_get_page(), with extra error message. */ +/* Wrapper around filemap_get_folio(), with extra error message. */ int btrfs_compress_filemap_get_folio(struct address_space *mapping, u64 start, struct folio **in_folio_ret) { From cb96b5e10a2cc5c844b4617c6f83c96968210f5d Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Mon, 7 Sep 2026 16:20:01 -0400 Subject: [PATCH 0426/1352] btrfs: use folios for reading super blocks from the block device btrfs_read_disk_super() and the zoned super block log comparison go through read_cache_page_gfp() and page_address(), and btrfs_release_disk_super() recovers the page with virt_to_page(). Use mapping_read_folio_gfp(), folio_address(), and virt_to_folio() instead. This removes the last callers of read_cache_page_gfp() and put_page() in btrfs. Compute the super block address with offset_in_folio() as write_dev_supers() does, rather than assuming it is at the start of the page. Assisted-by: Claude:claude-fable-5-1 Reviewed-by: Qu Wenruo Signed-off-by: Tal Zussman Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/volumes.c | 16 +++++++--------- fs/btrfs/zoned.c | 12 ++++++------ 2 files changed, 13 insertions(+), 15 deletions(-) diff --git a/fs/btrfs/volumes.c b/fs/btrfs/volumes.c index 1fad1aa716debf..7c040f22dbc3e5 100644 --- a/fs/btrfs/volumes.c +++ b/fs/btrfs/volumes.c @@ -1358,16 +1358,14 @@ int btrfs_open_devices(struct btrfs_fs_devices *fs_devices, void btrfs_release_disk_super(struct btrfs_super_block *super) { - struct page *page = virt_to_page(super); - - put_page(page); + folio_put(virt_to_folio(super)); } struct btrfs_super_block *btrfs_read_disk_super(struct block_device *bdev, int copy_num, bool drop_cache) { struct btrfs_super_block *super; - struct page *page; + struct folio *folio; u64 bytenr, bytenr_orig; struct address_space *mapping = bdev->bd_mapping; int ret; @@ -1388,7 +1386,7 @@ struct btrfs_super_block *btrfs_read_disk_super(struct block_device *bdev, ASSERT(copy_num == 0); /* - * Drop the page of the primary superblock, so later read will + * Drop the folio of the primary superblock, so later read will * always read from the device. */ invalidate_inode_pages2_range(mapping, bytenr >> PAGE_SHIFT, @@ -1396,12 +1394,12 @@ struct btrfs_super_block *btrfs_read_disk_super(struct block_device *bdev, } filemap_invalidate_lock_shared(mapping); - page = read_cache_page_gfp(mapping, bytenr >> PAGE_SHIFT, GFP_NOFS); + folio = mapping_read_folio_gfp(mapping, bytenr >> PAGE_SHIFT, GFP_NOFS); filemap_invalidate_unlock_shared(mapping); - if (IS_ERR(page)) - return ERR_CAST(page); + if (IS_ERR(folio)) + return ERR_CAST(folio); - super = page_address(page); + super = folio_address(folio) + offset_in_folio(folio, bytenr); if (btrfs_super_magic(super) != BTRFS_MAGIC || btrfs_super_bytenr(super) != bytenr_orig) { btrfs_release_disk_super(super); diff --git a/fs/btrfs/zoned.c b/fs/btrfs/zoned.c index 08a15465a0877d..a1ef8caaacdab4 100644 --- a/fs/btrfs/zoned.c +++ b/fs/btrfs/zoned.c @@ -123,24 +123,24 @@ static int sb_write_pointer(struct block_device *bdev, struct blk_zone *zones, } else if (full[0] && full[1]) { /* Compare two super blocks */ struct address_space *mapping = bdev->bd_mapping; - struct page *page[BTRFS_NR_SB_LOG_ZONES]; struct btrfs_super_block *super[BTRFS_NR_SB_LOG_ZONES]; for (int i = 0; i < BTRFS_NR_SB_LOG_ZONES; i++) { u64 zone_end = (zones[i].start + zones[i].capacity) << SECTOR_SHIFT; u64 bytenr = ALIGN_DOWN(zone_end, BTRFS_SUPER_INFO_SIZE) - BTRFS_SUPER_INFO_SIZE; + struct folio *folio; filemap_invalidate_lock_shared(mapping); - page[i] = read_cache_page_gfp(mapping, - bytenr >> PAGE_SHIFT, GFP_NOFS); + folio = mapping_read_folio_gfp(mapping, bytenr >> PAGE_SHIFT, + GFP_NOFS); filemap_invalidate_unlock_shared(mapping); - if (IS_ERR(page[i])) { + if (IS_ERR(folio)) { if (i == 1) btrfs_release_disk_super(super[0]); - return PTR_ERR(page[i]); + return PTR_ERR(folio); } - super[i] = page_address(page[i]); + super[i] = folio_address(folio) + offset_in_folio(folio, bytenr); } if (btrfs_super_generation(super[0]) > From 91f6fd9efe2599d991bb0382e2ff029c52043529 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Wed, 9 Sep 2026 13:10:48 -0400 Subject: [PATCH 0427/1352] btrfs: use u64 for the page indices in heuristic_collect_sample() index and index_end are derived from the u64 start and end offsets, and index is shifted back into a byte offset for offset_in_folio(), which needs a cast to u64 to be safe on 32-bit. Make them u64 instead so the cast goes away. They still fit pgoff_t where they are passed to filemap_get_folio(), as they came from a valid file offset. Signed-off-by: Tal Zussman Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/compression.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/compression.c b/fs/btrfs/compression.c index 20169d02896191..228d1cdd7c2950 100644 --- a/fs/btrfs/compression.c +++ b/fs/btrfs/compression.c @@ -1489,7 +1489,7 @@ static void heuristic_collect_sample(struct inode *inode, u64 start, u64 end, struct heuristic_ws *ws) { struct folio *folio; - pgoff_t index, index_end; + u64 index, index_end; u32 i, curr_sample_pos; u8 *in_data; @@ -1517,7 +1517,7 @@ static void heuristic_collect_sample(struct inode *inode, u64 start, u64 end, folio = filemap_get_folio(inode->i_mapping, index); ASSERT(!IS_ERR(folio)); in_data = kmap_local_folio(folio, - offset_in_folio(folio, (u64)index << PAGE_SHIFT)); + offset_in_folio(folio, index << PAGE_SHIFT)); /* Handle case where the start is not aligned to PAGE_SIZE */ i = start % PAGE_SIZE; while (i < PAGE_SIZE - SAMPLING_READ_SIZE) { From eac9d298bdc59bbae84ada71a0448340f41fb1a5 Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Wed, 9 Sep 2026 12:26:30 +0100 Subject: [PATCH 0428/1352] btrfs: increment extent count once when logging extents during fast fsync In btrfs_log_changed_extents() we increment the extent count twice, and we then fallback to a transaction commit if the count reaches a threshold of 32K. However we increment the count twice, which is confusing and pointless. So increment the count only once and reduce the threshold to half (16K). Reviewed-by: Qu Wenruo Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/tree-log.c | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/fs/btrfs/tree-log.c b/fs/btrfs/tree-log.c index 7fb476b19eb457..1d9dfb63fc5a4f 100644 --- a/fs/btrfs/tree-log.c +++ b/fs/btrfs/tree-log.c @@ -5350,7 +5350,7 @@ static int btrfs_log_changed_extents(struct btrfs_trans_handle *trans, * have a bunch of extents we just want to commit since it will * be faster. */ - if (++num > 32768) { + if (++num > SZ_16K) { list_del_init(&tree->modified_extents); ret = -EFBIG; goto process; @@ -5368,7 +5368,6 @@ static int btrfs_log_changed_extents(struct btrfs_trans_handle *trans, refcount_inc(&em->refs); em->flags |= EXTENT_FLAG_LOGGING; list_add_tail(&em->list, &extents); - num++; } list_sort(NULL, &extents, extent_cmp); From 747d4cf25961598c6d18fbab5a81cd295513092f Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Fri, 11 Sep 2026 11:56:29 +0100 Subject: [PATCH 0429/1352] btrfs: tree-checker: cache accessor return value in CHECK_FE_ALIGNED() We are calling an accessor for a file extent item multiple times in the CHECK_FE_ALIGNED() macro, even when we don't find a corruption (once in the if statement's expression and then once again in the expression for the return value). This adds extra runtime overhead (which is critical since the tree checker runs against every extent buffer when it's read or before persisting it) and increases the module's size. So cache the accessor's return value in a variable and use it, reducing runtime, object size and making the source code shorter too. Also avoid repeating twice the IS_ALIGNED() computation. Before: $ size fs/btrfs/btrfs.ko text data bss dec hex filename 2073340 217928 15624 2306892 23334c fs/btrfs/btrfs.ko After: $ size fs/btrfs/btrfs.ko text data bss dec hex filename 2073076 217928 15624 2306628 233244 fs/btrfs/btrfs.ko Also running the following fsstress test and capturing the runtime of check_extent_data_item() (the only caller of CHECK_FE_ALIGNED()) in nanoseconds (using bpftrace), showed the following runtime improvements: Test: mkfs.btrfs -f /dev/nullb0 mount /dev/nullb0 /mnt fsstress -w -p 8 -n 5000 -s 12345 -d /mnt umount /mnt Before: Count: 2033671 Range: 0.000 - 1336060.000; Mean: 679.642; Median: 656.000; Stddev: 2031.323 Percentiles: 90th: 850.000; 95th: 909.000; 99th: 1272.000 0.000 - 6.647: 11 | 6.647 - 28.241: 36 | 28.241 - 110.807: 176 | 110.807 - 426.512: 38012 # 426.512 - 1633.663: 1983372 ##################################################### 1633.663 - 6249.399: 8315 | 6249.399 - 23898.422: 3243 | 23898.422 - 91382.340: 153 | 91382.340 - 349418.114: 36 | 349418.114 - 1336060.000: 11 | After: Count: 2092797 Range: 0.000 - 1677284.000; Mean: 627.650; Median: 617.000; Stddev: 1848.010 Percentiles: 90th: 809.000; 95th: 859.000; 99th: 1209.000 0.000 - 6.823: 18 | 6.823 - 29.602: 82 | 29.602 - 118.702: 416 | 118.702 - 467.232: 515774 ################# 467.232 - 1830.549: 1565599 ##################################################### 1830.549 - 7163.339: 5297 | 7163.339 - 28023.237: 4186 | 28023.237 - 109619.427: 166 | 109619.427 - 428793.471: 25 | 428793.471 - 1677284.000: 4 | Reviewed-by: Qu Wenruo Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/tree-checker.c | 11 ++++++----- 1 file changed, 6 insertions(+), 5 deletions(-) diff --git a/fs/btrfs/tree-checker.c b/fs/btrfs/tree-checker.c index 594266c9c116ff..0f4f1b347c9e8e 100644 --- a/fs/btrfs/tree-checker.c +++ b/fs/btrfs/tree-checker.c @@ -108,13 +108,14 @@ static void file_extent_err(const struct extent_buffer *eb, int slot, */ #define CHECK_FE_ALIGNED(leaf, slot, fi, name, alignment) \ ({ \ - if (unlikely(!IS_ALIGNED(btrfs_file_extent_##name((leaf), (fi)), \ - (alignment)))) \ + const u64 val = btrfs_file_extent_##name((leaf), (fi)); \ + const bool not_aligned = !IS_ALIGNED(val, (alignment)); \ + \ + if (unlikely(not_aligned)) \ file_extent_err((leaf), (slot), \ "invalid %s for file extent, have %llu, should be aligned to %u", \ - (#name), btrfs_file_extent_##name((leaf), (fi)), \ - (alignment)); \ - (!IS_ALIGNED(btrfs_file_extent_##name((leaf), (fi)), (alignment))); \ + (#name), val, (alignment)); \ + not_aligned; \ }) static u64 file_extent_end(struct extent_buffer *leaf, From 789bd264cae48413bd57d33e16c43e658d48da51 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Sun, 13 Sep 2026 20:59:15 +0930 Subject: [PATCH 0430/1352] btrfs: cleanup and rename submit_extent_folio() Cleanup submit_extent_folio() by: - Remove @size parameter Since commit b2e743927fdd ("btrfs: make btrfs_do_readpage() to do block-by-block read"), all callers are passing sectorsize as @size, so there is no need for such parameter. Furthermore since we only write one block at a time, there is no need for a while() loop, nor the advance of various local variables. - Update the comments on the parameter list * @disk_bytenr is shared for both read and write * rename @page to @folio - Update the return value to return 0 or error Since we won't queue multiple blocks anyway, there is no point in returning the queued bytes. It makes more sense to return an error code, although the only error code will be -EUCLEAN for writes. - Rename submit_extent_folio() to submit_one_block() - Rename submit_one_sector() to submit_write_sector() - Update the error message to utilize the new returned error code inside submit_write_sector() Reviewed-by: Boris Burkov Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/extent_io.c | 187 +++++++++++++++++++------------------------ 1 file changed, 81 insertions(+), 106 deletions(-) diff --git a/fs/btrfs/extent_io.c b/fs/btrfs/extent_io.c index a221b63bdb205d..55e9144d47595f 100644 --- a/fs/btrfs/extent_io.c +++ b/fs/btrfs/extent_io.c @@ -110,14 +110,14 @@ struct btrfs_bio_ctrl { * make the decision when submitting the bio. * * The pattern between do_readpage(), submit_one_bio() and - * submit_extent_folio() is quite subtle, so tracking this is tricky. + * submit_one_block() is quite subtle, so tracking this is tricky. * * As we process extent E, we might submit a bio with existing built up * extents before adding E to a new bio, or we might just add E to the * bio. As a result, E's generation could apply to the current bio or * to the next one, so we need to be careful to update the bio_ctrl's * generation with E's only when we are sure E is added to bio_ctrl->bbio - * in submit_extent_folio(). + * in submit_one_block(). * * See the comment in btrfs_lookup_bio_sums() for more detail on the * need for this optimization. @@ -797,118 +797,95 @@ static int alloc_new_bio(struct btrfs_inode *inode, } /* - * @disk_bytenr: logical bytenr where the write will be - * @page: page to add to the bio - * @size: portion of page that we want to write to - * @pg_offset: offset of the new bio or to check whether we are adding - * a contiguous page to the previous one + * @disk_bytenr: logical bytenr where the read/write will be + * @folio: the folio the block belongs to + * @pg_offset: the offset inside the folio * @read_em_generation: generation of the extent_map we are submitting * (only used for read) * - * The will either add the page into the existing @bio_ctrl->bbio, or allocate a + * This will either add the block into the existing @bio_ctrl->bbio, or allocate a * new one in @bio_ctrl->bbio. * The mirror number for this IO should already be initialized in * @bio_ctrl->mirror_num. * - * Return the number of bytes that are queued into a bio. - * If the returned bytes is smaller than @size, it means we hit a critical error - * for data write, where there is no ordered extent for the range. + * Return 0 if the block is queued or submitted. + * Return <0 for error. */ -static unsigned int submit_extent_folio(struct btrfs_bio_ctrl *bio_ctrl, - u64 disk_bytenr, struct folio *folio, - size_t size, unsigned long pg_offset, - u64 read_em_generation) +static int submit_one_block(struct btrfs_bio_ctrl *bio_ctrl, + u64 disk_bytenr, struct folio *folio, + unsigned long pg_offset, u64 read_em_generation) { struct btrfs_inode *inode = folio_to_inode(folio); + const struct btrfs_fs_info *fs_info = inode->root->fs_info; + const u32 blocksize = fs_info->sectorsize; loff_t file_offset = folio_pos(folio) + pg_offset; - unsigned int queued = 0; - ASSERT(pg_offset + size <= folio_size(folio)); + ASSERT(pg_offset + blocksize <= folio_size(folio)); ASSERT(bio_ctrl->end_io_func); if (bio_ctrl->bbio && !btrfs_bio_is_contig(bio_ctrl, disk_bytenr, file_offset)) submit_one_bio(bio_ctrl); - do { - u32 len = size; - - /* Allocate new bio if needed */ - if (!bio_ctrl->bbio) { - int ret; - - ret = alloc_new_bio(inode, bio_ctrl, disk_bytenr, file_offset); - if (ret < 0) - break; - } +again: + /* Allocate new bio if needed */ + if (!bio_ctrl->bbio) { + int ret; - /* Cap to the current ordered extent boundary if there is one. */ - if (len > bio_ctrl->len_to_oe_boundary) { - ASSERT(bio_ctrl->compress_type == BTRFS_COMPRESS_NONE); - ASSERT(is_data_inode(inode)); - len = bio_ctrl->len_to_oe_boundary; - } + ret = alloc_new_bio(inode, bio_ctrl, disk_bytenr, file_offset); + if (ret < 0) + return ret; + } - if (!bio_add_folio(&bio_ctrl->bbio->bio, folio, len, pg_offset)) { - /* bio full: move on to a new one */ - submit_one_bio(bio_ctrl); - continue; - } - /* - * Now that the folio is definitely added to the bio, include its - * generation in the max generation calculation. - */ - bio_ctrl->generation = max(bio_ctrl->generation, read_em_generation); - bio_ctrl->next_file_offset += len; + if (!bio_add_folio(&bio_ctrl->bbio->bio, folio, blocksize, pg_offset)) { + /* bio full: move on to a new one */ + submit_one_bio(bio_ctrl); + goto again; + } - if (bio_ctrl->wbc) - wbc_account_cgroup_owner(bio_ctrl->wbc, folio, len); + /* + * Now that the folio is definitely added to the bio, include its + * generation in the max generation calculation. + */ + bio_ctrl->generation = max(bio_ctrl->generation, read_em_generation); + bio_ctrl->next_file_offset += blocksize; - size -= len; - pg_offset += len; - disk_bytenr += len; - file_offset += len; - queued += len; + if (bio_ctrl->wbc) + wbc_account_cgroup_owner(bio_ctrl->wbc, folio, blocksize); - /* - * len_to_oe_boundary defaults to U32_MAX, which isn't folio or - * sector aligned. alloc_new_bio() then sets it to the end of - * our ordered extent for writes into zoned devices. - * - * When len_to_oe_boundary is tracking an ordered extent, we - * trust the ordered extent code to align things properly, and - * the check above to cap our write to the ordered extent - * boundary is correct. - * - * When len_to_oe_boundary is U32_MAX, the cap above would - * result in a 4095 byte IO for the last folio right before - * we hit the bio limit of UINT_MAX. bio_add_folio() has all - * the checks required to make sure we don't overflow the bio, - * and we should just ignore len_to_oe_boundary completely - * unless we're using it to track an ordered extent. - * - * It's pretty hard to make a bio sized U32_MAX, but it can - * happen when the page cache is able to feed us contiguous - * folios for large extents. - */ - if (bio_ctrl->len_to_oe_boundary != U32_MAX) - bio_ctrl->len_to_oe_boundary -= len; - - /* Ordered extent boundary: move on to a new bio. */ - if (bio_ctrl->len_to_oe_boundary == 0) - submit_one_bio(bio_ctrl); - /* - * If we have accumulated decent amount of IO, send it to the - * block layer so that IO can run while we are accumulating - * more folios to write. - */ - else if (bio_ctrl->wbc && - bio_ctrl->bbio->bio.bi_iter.bi_size >= - inode->root->fs_info->writeback_bio_size) - submit_one_bio(bio_ctrl); + /* + * len_to_oe_boundary defaults to U32_MAX, which isn't folio or sector + * aligned. alloc_new_bio() then sets it to the end of our ordered + * extent for writes into zoned devices. + * + * When len_to_oe_boundary is tracking an ordered extent, the + * len_to_oe_boundary should follow that OE and never go beyond the max + * extent size (128MiB). + * + * When len_to_oe_boundary is U32_MAX, decreasing the length by + * blocksize will never make it reach 0, thus skipping the later + * submit_one_bio() call. So if len_to_oe_boundary() is not tracking + * an OE, do not decrease it. + * + * It's pretty hard to make a bio sized U32_MAX, but it can happen when + * the page cache is able to feed us contiguous folios for large + * extents. + */ + if (bio_ctrl->len_to_oe_boundary != U32_MAX) + bio_ctrl->len_to_oe_boundary -= blocksize; - } while (size); - return queued; + /* Ordered extent boundary: move on to a new bio. */ + if (bio_ctrl->len_to_oe_boundary == 0) + submit_one_bio(bio_ctrl); + /* + * If we have accumulated decent amount of IO, send it to the block + * layer so that IO can run while we are accumulating more folios to + * write. + */ + else if (bio_ctrl->wbc && + bio_ctrl->bbio->bio.bi_iter.bi_size >= fs_info->writeback_bio_size) + submit_one_bio(bio_ctrl); + return 0; } static int attach_extent_buffer_folio(struct extent_buffer *eb, @@ -1092,7 +1069,6 @@ static int btrfs_do_readpage(struct folio *folio, struct extent_map **em_cached, u64 disk_bytenr; u64 block_start; u64 em_gen; - unsigned int queued; ASSERT(IS_ALIGNED(cur, fs_info->sectorsize)); if (cur >= last_byte) { @@ -1206,10 +1182,9 @@ static int btrfs_do_readpage(struct folio *folio, struct extent_map **em_cached, if (force_bio_submit) submit_one_bio(bio_ctrl); - queued = submit_extent_folio(bio_ctrl, disk_bytenr, folio, blocksize, - pg_offset, em_gen); + ret = submit_one_block(bio_ctrl, disk_bytenr, folio, pg_offset, em_gen); /* Read submission should not fail. */ - ASSERT(queued == blocksize); + ASSERT(ret == 0); } return 0; } @@ -1830,10 +1805,10 @@ static struct btrfs_ordered_extent *get_oe_from_bbio(const struct btrfs_bio *bbi * * Caller should make sure filepos < i_size and handle filepos >= i_size case. */ -static int submit_one_sector(struct btrfs_inode *inode, - struct folio *folio, - u64 filepos, struct btrfs_bio_ctrl *bio_ctrl, - loff_t i_size) +static int submit_write_sector(struct btrfs_inode *inode, + struct folio *folio, + u64 filepos, struct btrfs_bio_ctrl *bio_ctrl, + loff_t i_size) { struct btrfs_fs_info *fs_info = inode->root->fs_info; struct btrfs_ordered_extent *oe; @@ -1841,7 +1816,7 @@ static int submit_one_sector(struct btrfs_inode *inode, u64 disk_bytenr; u64 extent_offset; const u32 sectorsize = fs_info->sectorsize; - unsigned int queued; + int ret; ASSERT(IS_ALIGNED(filepos, sectorsize)); @@ -1904,17 +1879,17 @@ static int submit_one_sector(struct btrfs_inode *inode, */ ASSERT(folio_test_writeback(folio)); - queued = submit_extent_folio(bio_ctrl, disk_bytenr, folio, - sectorsize, filepos - folio_pos(folio), 0); - if (unlikely(queued < sectorsize)) { + ret = submit_one_block(bio_ctrl, disk_bytenr, folio, + offset_in_folio(folio, filepos), 0); + if (unlikely(ret < 0)) { btrfs_folio_clear_writeback(fs_info, folio, filepos, sectorsize); btrfs_mark_ordered_io_finished(inode, filepos, fs_info->sectorsize, false); btrfs_err_rl(fs_info, - "failed to queue sector for root %lld ino %llu filepos %llu", + "failed to queue sector for root %lld ino %llu filepos %llu: %pe", btrfs_root_id(inode->root), - btrfs_ino(inode), filepos); - return -EUCLEAN; + btrfs_ino(inode), filepos, ERR_PTR(ret)); + return ret; } return 0; } @@ -1999,7 +1974,7 @@ static noinline_for_stack int extent_writepage_io(struct btrfs_inode *inode, btrfs_folio_clear_dirty(fs_info, folio, cur, fs_info->sectorsize); continue; } - ret = submit_one_sector(inode, folio, cur, bio_ctrl, i_size); + ret = submit_write_sector(inode, folio, cur, bio_ctrl, i_size); if (unlikely(ret < 0)) { if (!found_error) found_error = ret; From 2776e8a389246ac34847163acb5d511277517153 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Thu, 10 Sep 2026 11:35:52 +0930 Subject: [PATCH 0431/1352] btrfs: fix off-by-one end related to inode_need_compress() In most cases btrfs uses @end as the inclusive end bytenr for a range, and this applies to inode_need_compress(). However we have several sites not following the inclusive bytenr: - run_delalloc_inline() Which assigned @blocksize as @end for inode_need_compress() This makes inode_need_compress() always skip the disk_i_size check. - heuristic_collect_sample() Which assigned "start + BTRFS_MAX_UNCOMPRESSED" to @end, which is the exclusive bytenr. Neither is really causing any real problem, as heuristic_collect_sample() has proper checks to avoid reading anything beyond @end, and the sampling read size is 16 bytes, so it has enough headroom to handle that off-by-one problem. But still I do not like anything out of the common scheme, so fix the off-by-one @end for both call sites, and add extra ASSERT()s to catch such unaligned parameters. Reviewed-by: Boris Burkov Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/compression.c | 15 +++++++-------- fs/btrfs/inode.c | 5 ++++- 2 files changed, 11 insertions(+), 9 deletions(-) diff --git a/fs/btrfs/compression.c b/fs/btrfs/compression.c index 228d1cdd7c2950..7c1c018dbd6fd7 100644 --- a/fs/btrfs/compression.c +++ b/fs/btrfs/compression.c @@ -1488,11 +1488,14 @@ static bool sample_repeated_patterns(struct heuristic_ws *ws) static void heuristic_collect_sample(struct inode *inode, u64 start, u64 end, struct heuristic_ws *ws) { + const u32 blocksize = BTRFS_I(inode)->root->fs_info->sectorsize; struct folio *folio; u64 index, index_end; u32 i, curr_sample_pos; u8 *in_data; + ASSERT(IS_ALIGNED(start, blocksize) && IS_ALIGNED(end + 1, blocksize)); + /* * Compression handles the input data by chunks of 128KiB * (defined by BTRFS_MAX_UNCOMPRESSED) @@ -1502,18 +1505,14 @@ static void heuristic_collect_sample(struct inode *inode, u64 start, u64 end, * MAX_SAMPLE_SIZE - calculated under assumption that heuristic will * process no more than BTRFS_MAX_UNCOMPRESSED at a time. */ - if (end - start > BTRFS_MAX_UNCOMPRESSED) - end = start + BTRFS_MAX_UNCOMPRESSED; + if (end + 1 - start > BTRFS_MAX_UNCOMPRESSED) + end = start + BTRFS_MAX_UNCOMPRESSED - 1; index = start >> PAGE_SHIFT; index_end = end >> PAGE_SHIFT; - /* Don't miss unaligned end */ - if (!PAGE_ALIGNED(end)) - index_end++; - curr_sample_pos = 0; - while (index < index_end) { + while (index <= index_end) { folio = filemap_get_folio(inode->i_mapping, index); ASSERT(!IS_ERR(folio)); in_data = kmap_local_folio(folio, @@ -1522,7 +1521,7 @@ static void heuristic_collect_sample(struct inode *inode, u64 start, u64 end, i = start % PAGE_SIZE; while (i < PAGE_SIZE - SAMPLING_READ_SIZE) { /* Don't sample any garbage from the last page */ - if (start > end - SAMPLING_READ_SIZE) + if (start > end + 1 - SAMPLING_READ_SIZE) break; memcpy(&ws->sample[curr_sample_pos], &in_data[i], SAMPLING_READ_SIZE); diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 3f65c3dd91e7c9..53f4532593b36c 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -730,6 +730,9 @@ static inline int inode_need_compress(struct btrfs_inode *inode, u64 start, u64 end, bool check_inline) { struct btrfs_fs_info *fs_info = inode->root->fs_info; + const u32 blocksize = fs_info->sectorsize; + + ASSERT(IS_ALIGNED(start, blocksize) && IS_ALIGNED(end + 1, blocksize)); if (unlikely(!btrfs_inode_can_compress(inode))) { DEBUG_WARN("BTRFS: unexpected compression for ino %llu", btrfs_ino(inode)); @@ -2331,7 +2334,7 @@ static int run_delalloc_inline(struct btrfs_inode *inode, struct folio *locked_f btrfs_check_folio_write_protected(locked_folio); if (btrfs_inode_can_compress(inode) && - inode_need_compress(inode, 0, blocksize, true)) { + inode_need_compress(inode, 0, blocksize - 1, true)) { if (inode->defrag_compress > 0 && inode->defrag_compress < BTRFS_NR_COMPRESS_TYPES) { compress_type = inode->defrag_compress; From f0bc2b9a1c9a2f9e392aeb0d6f5be8022333f781 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Thu, 10 Sep 2026 11:35:53 +0930 Subject: [PATCH 0432/1352] btrfs: simplify heuristic_collect_sample() to handle large folios better Currently heuristic_collect_sample() is purely page size based, and it has a lot of extra handling just inside the page. However we already have large folio support, there is no need to look up the same folio repeatedly. Simplify the handling by: - Use @cur as the iterator instead of page index - Handle the sample copying on a per-folio basis Although kmap_local_folio() requires an offset to handle HIGHMEM page mapping, we have rejected large folios for HIGHMEM systems completely. So we can safely handle all sample copying inside the folio in one go. - Remove unnecessary unaligned range handling All the range passed in should be block aligned, thus there is no need to handle cases where sample crosses the block boundary. Reviewed-by: Boris Burkov Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/compression.c | 40 ++++++++++++++++------------------------ 1 file changed, 16 insertions(+), 24 deletions(-) diff --git a/fs/btrfs/compression.c b/fs/btrfs/compression.c index 7c1c018dbd6fd7..ad90e14032dbb3 100644 --- a/fs/btrfs/compression.c +++ b/fs/btrfs/compression.c @@ -1489,10 +1489,8 @@ static void heuristic_collect_sample(struct inode *inode, u64 start, u64 end, struct heuristic_ws *ws) { const u32 blocksize = BTRFS_I(inode)->root->fs_info->sectorsize; - struct folio *folio; - u64 index, index_end; - u32 i, curr_sample_pos; - u8 *in_data; + u64 cur = start; + u32 curr_sample_pos = 0; ASSERT(IS_ALIGNED(start, blocksize) && IS_ALIGNED(end + 1, blocksize)); @@ -1508,33 +1506,27 @@ static void heuristic_collect_sample(struct inode *inode, u64 start, u64 end, if (end + 1 - start > BTRFS_MAX_UNCOMPRESSED) end = start + BTRFS_MAX_UNCOMPRESSED - 1; - index = start >> PAGE_SHIFT; - index_end = end >> PAGE_SHIFT; + while (cur < end) { + struct folio *folio; + void *in_data; + u64 next_pos; - curr_sample_pos = 0; - while (index <= index_end) { - folio = filemap_get_folio(inode->i_mapping, index); + folio = filemap_get_folio(inode->i_mapping, cur >> PAGE_SHIFT); + /* All folios inside the range should exist and be locked. */ ASSERT(!IS_ERR(folio)); - in_data = kmap_local_folio(folio, - offset_in_folio(folio, index << PAGE_SHIFT)); - /* Handle case where the start is not aligned to PAGE_SIZE */ - i = start % PAGE_SIZE; - while (i < PAGE_SIZE - SAMPLING_READ_SIZE) { - /* Don't sample any garbage from the last page */ - if (start > end + 1 - SAMPLING_READ_SIZE) - break; - memcpy(&ws->sample[curr_sample_pos], &in_data[i], - SAMPLING_READ_SIZE); - i += SAMPLING_INTERVAL; - start += SAMPLING_INTERVAL; + next_pos = min_t(u64, end + 1, folio_next_pos(folio)); + in_data = kmap_local_folio(folio, 0); + + for (; cur < next_pos; cur += SAMPLING_INTERVAL) { + memcpy(&ws->sample[curr_sample_pos], + in_data + offset_in_folio(folio, cur), + SAMPLING_READ_SIZE); curr_sample_pos += SAMPLING_READ_SIZE; } kunmap_local(in_data); folio_put(folio); - - index++; + cur = next_pos; } - ws->sample_size = curr_sample_pos; } From d95288aa39bd2bb2caa3cb4868ada0d1d3614812 Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Wed, 16 Sep 2026 17:11:51 +0100 Subject: [PATCH 0433/1352] btrfs: add missing unlikely to a couple error checks during sys chunk array validation It's unexpected to find errors during sys chunk array validation and all checks use the unlikely tag except for two of them, so add it. Reviewed-by: Qu Wenruo Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/disk-io.c | 2 +- fs/btrfs/tree-checker.c | 6 +++--- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index 94a7e9059a7b3f..21b72c90bd90f9 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -2387,7 +2387,7 @@ static int validate_sys_chunk_array(const struct btrfs_fs_info *fs_info, } ret = btrfs_check_chunk_valid(fs_info, NULL, chunk, key.offset, sectorsize); - if (ret < 0) + if (unlikely(ret < 0)) return ret; cur += btrfs_chunk_item_size(num_stripes); } diff --git a/fs/btrfs/tree-checker.c b/fs/btrfs/tree-checker.c index 0f4f1b347c9e8e..9cd79d97b9b5e7 100644 --- a/fs/btrfs/tree-checker.c +++ b/fs/btrfs/tree-checker.c @@ -1121,9 +1121,9 @@ int btrfs_check_chunk_valid(const struct btrfs_fs_info *fs_info, return -EUCLEAN; } - if (!remapped && - !valid_stripe_count(type & BTRFS_BLOCK_GROUP_PROFILE_MASK, - num_stripes, sub_stripes)) { + if (unlikely(!remapped && + !valid_stripe_count(type & BTRFS_BLOCK_GROUP_PROFILE_MASK, + num_stripes, sub_stripes))) { chunk_err(fs_info, leaf, chunk, logical, "invalid num_stripes:sub_stripes %u:%u for profile %llu", num_stripes, sub_stripes, From ed03cc2d619a612bb40a9367c31dce87513ff884 Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Wed, 16 Sep 2026 17:20:52 +0100 Subject: [PATCH 0434/1352] btrfs: remove redundant eb generation check in btrfs_buffer_uptodate() It's pointless to check if the extent buffer's generation does not match the value of 'parent_transid' because if it does, then we have already entered the previous if statement and returned from the function. Reviewed-by: Qu Wenruo Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/disk-io.c | 14 ++++++-------- 1 file changed, 6 insertions(+), 8 deletions(-) diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index 21b72c90bd90f9..161a7b9bb27a5d 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -125,15 +125,13 @@ int btrfs_buffer_uptodate(struct extent_buffer *eb, u64 parent_transid, return 1; } - if (btrfs_header_generation(eb) != parent_transid) { - btrfs_err_rl(eb->fs_info, + btrfs_err_rl(eb->fs_info, "parent transid verify failed on logical %llu mirror %u wanted %llu found %llu", - eb->start, eb->read_mirror, - parent_transid, btrfs_header_generation(eb)); - clear_extent_buffer_uptodate(eb); - return 0; - } - return 1; + eb->start, eb->read_mirror, + parent_transid, btrfs_header_generation(eb)); + clear_extent_buffer_uptodate(eb); + + return 0; } static bool btrfs_supported_super_csum(u16 csum_type) From 137effd10f7a63e31d915e851c2c9fa5a94874cc Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Wed, 16 Sep 2026 17:36:11 +0100 Subject: [PATCH 0435/1352] btrfs: remove duplicate error message when writing super blocks If the total error count is greater the maximum allowed number of errors, we print exactly the same error message twice. Remove one of the messages. Reviewed-by: Qu Wenruo Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/disk-io.c | 2 -- 1 file changed, 2 deletions(-) diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index 161a7b9bb27a5d..a8535e309b5760 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -4228,8 +4228,6 @@ int write_all_supers(struct btrfs_trans_handle *trans) total_errors++; } if (unlikely(total_errors > max_errors)) { - btrfs_err(fs_info, "%d errors while writing supers", - total_errors); mutex_unlock(&fs_info->fs_devices->device_list_mutex); /* FUA is masked off if unsupported and can't be the reason */ From b3f5fe826871e5b7371c38e4ae5d19086df86bb2 Mon Sep 17 00:00:00 2001 From: Boris Burkov Date: Wed, 16 Sep 2026 15:17:24 -0700 Subject: [PATCH 0436/1352] btrfs: keep unused block groups queued when a pass fails Once any block_group sets ret!=0 in the main loop of btrfs_delete_unused_bgs(), the check if (ret || btrfs_mixed_space_info(space_info)) { btrfs_put_block_group(block_group); continue; } skips the rest of the unused bgs while unlinking them from fs_info->unused_bgs. There is no "level triggered" re-queueing of empty block groups onto fs_info->unused_bgs so it is possible to leak quite a bit of space this way and unless we happen to get a balance or re-use/re-empty one of these bgs, they are leaked for good, which can lead to a spurious enospc later. While I have observed such leaked blocked groups that are empty but not on the unused_bgs list on production systems, I have not observed that it is definitely due to this issue. I also reproduced this behavior by injecting an ENOSPC error from btrfs_start_trans_remove_block_group which can also fail with ENOMEM, so this feels like a legitimate injection point. To fix it, instead of checking ret in the loop, just break out of the loop when ret != 0. Also, link the bg to the retry list at the individual failure sites so that the failing bg is not leaked. Assisted-by: LLM (reproducer/error injection) Reviewed-by: Qu Wenruo Signed-off-by: Boris Burkov Signed-off-by: David Sterba --- fs/btrfs/block-group.c | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/fs/btrfs/block-group.c b/fs/btrfs/block-group.c index ee182369254c08..2eb09c9901c9e1 100644 --- a/fs/btrfs/block-group.c +++ b/fs/btrfs/block-group.c @@ -1612,7 +1612,7 @@ void btrfs_delete_unused_bgs(struct btrfs_fs_info *fs_info) space_info = block_group->space_info; - if (ret || btrfs_mixed_space_info(space_info)) { + if (btrfs_mixed_space_info(space_info)) { btrfs_put_block_group(block_group); continue; } @@ -1727,6 +1727,7 @@ void btrfs_delete_unused_bgs(struct btrfs_fs_info *fs_info) ret = inc_block_group_ro(block_group, false); up_write(&space_info->groups_sem); if (ret < 0) { + btrfs_link_bg_list(block_group, &retry_list); ret = 0; goto next; } @@ -1749,6 +1750,7 @@ void btrfs_delete_unused_bgs(struct btrfs_fs_info *fs_info) block_group->start); if (IS_ERR(trans)) { btrfs_dec_block_group_ro(block_group); + btrfs_link_bg_list(block_group, &retry_list); ret = PTR_ERR(trans); goto next; } @@ -1759,6 +1761,7 @@ void btrfs_delete_unused_bgs(struct btrfs_fs_info *fs_info) */ if (!clean_pinned_extents(trans, block_group)) { btrfs_dec_block_group_ro(block_group); + btrfs_link_bg_list(block_group, &retry_list); goto end_trans; } @@ -1845,6 +1848,8 @@ void btrfs_delete_unused_bgs(struct btrfs_fs_info *fs_info) next: btrfs_put_block_group(block_group); spin_lock(&fs_info->unused_bgs_lock); + if (ret) + break; } list_splice_tail(&retry_list, &fs_info->unused_bgs); spin_unlock(&fs_info->unused_bgs_lock); From d8cd855dc3330d003be0d222a082dff572639e08 Mon Sep 17 00:00:00 2001 From: Boris Burkov Date: Tue, 15 Sep 2026 10:58:31 -0700 Subject: [PATCH 0437/1352] btrfs: pre-flush reflink source before taking inode locks Consider the following sketch of a shell script: dd if=/dev/urandom of=/mnt/src bs=1M count=4096 cp --reflink=always /mnt/src /mnt/dst & sleep 0.1 # let the clone reach its flush time dd if=/mnt/src of=/dev/null bs=4K count=1 iflag=direct The current logic in reflink ensures the existence and stability of both the src and destination inode by locking them both and then flushing all dirty pages / ordered_extents under the inode lock. This blocks concurrent usage by even readers of the inode locks, like direct reads or seeks for the duration of writeback on the entirety of the two files. We observe this particular contention frequently in the Meta fleet. It is also a relatively common pattern to write the src file, then reflink it while it is still dirty, so that typical path naturally hits this contention. This workload motivates a relatively simple optimization: trigger the unavoidable src flushing outside the locked region. Therefore, do an optimistic btrfs_wait_ordered_range() without the locks while relying on the calls to btrfs_wait_ordered_range() under the locks in btrfs_remap_file_range_prep() to ensure correctness. In the worst case with concurrent writes while unlocked, we can end up doing the writeback twice, which I think is a reasonable price to pay to avoid victimizing innocent readers in a common case. With and without this patch, the above reproducer runs the reflink in the same ~0.5s on my system. Without the patch, the direct read blocks for basically the full duration of the reflink while with the patch it returns in a few milliseconds. Reviewed-by: Filipe Manana Signed-off-by: Boris Burkov Signed-off-by: David Sterba --- fs/btrfs/reflink.c | 28 ++++++++++++++++++++++++---- 1 file changed, 24 insertions(+), 4 deletions(-) diff --git a/fs/btrfs/reflink.c b/fs/btrfs/reflink.c index d2a4101912bdb2..24c0b226bc3165 100644 --- a/fs/btrfs/reflink.c +++ b/fs/btrfs/reflink.c @@ -833,6 +833,17 @@ static noinline int btrfs_clone_files(struct file *file, struct file *file_src, return 0; } +static u64 calc_remap_wb_len(struct btrfs_inode *src_inode, loff_t off, loff_t len, + unsigned int remap_flags) +{ + const u32 bs = src_inode->root->fs_info->sectorsize; + const loff_t isize = i_size_read(&src_inode->vfs_inode); + + if (len == 0 && !(remap_flags & REMAP_FILE_DEDUP)) + return ALIGN(isize, bs) - ALIGN_DOWN(off, bs); + return ALIGN(len, bs); +} + static int btrfs_remap_file_range_prep(struct file *file_in, loff_t pos_in, struct file *file_out, loff_t pos_out, loff_t *len, unsigned int remap_flags) @@ -876,10 +887,7 @@ static int btrfs_remap_file_range_prep(struct file *file_in, loff_t pos_in, * not for the ordered extents to complete. We need to wait for them * to complete so that new file extent items are in the fs tree. */ - if (*len == 0 && !(remap_flags & REMAP_FILE_DEDUP)) - wb_len = ALIGN(inode_in->vfs_inode.i_size, bs) - ALIGN_DOWN(pos_in, bs); - else - wb_len = ALIGN(*len, bs); + wb_len = calc_remap_wb_len(inode_in, pos_in, *len, remap_flags); /* * Workaround to make sure NOCOW buffered write reach disk as NOCOW. @@ -930,6 +938,7 @@ loff_t btrfs_remap_file_range(struct file *src_file, loff_t off, struct btrfs_inode *src_inode = BTRFS_I(file_inode(src_file)); struct btrfs_inode *dst_inode = BTRFS_I(file_inode(dst_file)); bool same_inode = dst_inode == src_inode; + u64 wb_start, wb_len; int ret; if (btrfs_is_shutdown(src_inode->root->fs_info)) @@ -938,6 +947,17 @@ loff_t btrfs_remap_file_range(struct file *src_file, loff_t off, if (remap_flags & ~(REMAP_FILE_DEDUP | REMAP_FILE_ADVISORY)) return -EINVAL; + /* + * Optimistically write out the src inode before taking locks. + * Consistency is properly ensured by the btrfs_wait_ordered_range() + * inside btrfs_remap_file_range_prep(). + */ + wb_start = ALIGN_DOWN(off, src_inode->root->fs_info->sectorsize); + wb_len = calc_remap_wb_len(src_inode, off, len, remap_flags); + ret = btrfs_wait_ordered_range(src_inode, wb_start, wb_len); + if (ret < 0) + return ret; + if (same_inode) { btrfs_inode_lock(src_inode, BTRFS_ILOCK_MMAP); } else { From 8b81e056e49634d04aa3fbc9d217a7eb4ba8ec82 Mon Sep 17 00:00:00 2001 From: Boris Burkov Date: Tue, 15 Sep 2026 10:58:32 -0700 Subject: [PATCH 0438/1352] btrfs: skip unlocked reflink source flush if the inode has writers The optimistic unlocked flushing is counter-productive if there is an active writer dirtying pages, as it results in doubly writing back any pages that get dirtied again before the locked flush. Therefore, pessimize it slightly and don't do the unlocked flushing if we have outstanding writable file descriptors or memory mappings. Since the inode isn't locked, this is only advisory, any new mapping or fd that sneaks in after the check but before the locking can still cause extra writeback. But it trivially helps quite a bit in simple cases with a long open file descriptor or mapping. I tested this with two similar workloads. One was keeping and fd open and doing continuous writes into the src file while cloning it and doing lseek on it and the other was slightly meaner, constantly opening and closing the fd to try to race this check. Workload 1 shows a huge regression in write amplification and writers and readers who experience userspace wall time stalls of 100ms+ with just the pre-flush, but recovers to normal with this check. Workload 2 is a smaller regression which is mostly helped but not fully resolved with this check. So we must weigh this regression against the benefit of the unlocked flush in the allegedly more common simpler case. Workload 1: fd kept open, 60s of clone/write/lseek looping. kernel clones clone mean s device MiB/clone writer stalls reader stalls for-next 564 0.095 119 10 10 pre-flush (p1) 179 0.338 405 94 93 skip (p1+p2) 540 0.100 116 9 9 Workload 2: same writer, but closing and reopening the fd around each 64 MiB write with some random sleeping to encourage racing. kernel clones clone mean s device MiB/clone writer stalls reader stalls for-next 740 0.070 83 9 9 pre-flush (p1) 255 0.224 246 27 26 skip (p1+p2) 518 0.107 121 11 11 Reviewed-by: Filipe Manana Signed-off-by: Boris Burkov Signed-off-by: David Sterba --- fs/btrfs/reflink.c | 14 ++++++++++---- 1 file changed, 10 insertions(+), 4 deletions(-) diff --git a/fs/btrfs/reflink.c b/fs/btrfs/reflink.c index 24c0b226bc3165..8e37706b98c31c 100644 --- a/fs/btrfs/reflink.c +++ b/fs/btrfs/reflink.c @@ -950,13 +950,19 @@ loff_t btrfs_remap_file_range(struct file *src_file, loff_t off, /* * Optimistically write out the src inode before taking locks. * Consistency is properly ensured by the btrfs_wait_ordered_range() - * inside btrfs_remap_file_range_prep(). + * inside btrfs_remap_file_range_prep(). This optimization causes + * redundant writeback if there is a concurrent writer to the src file + * so try to catch anyone holding the file open for writes and skip + * the optimization in that case. */ wb_start = ALIGN_DOWN(off, src_inode->root->fs_info->sectorsize); wb_len = calc_remap_wb_len(src_inode, off, len, remap_flags); - ret = btrfs_wait_ordered_range(src_inode, wb_start, wb_len); - if (ret < 0) - return ret; + if (!inode_is_open_for_write(&src_inode->vfs_inode) && + !mapping_writably_mapped(src_inode->vfs_inode.i_mapping)) { + ret = btrfs_wait_ordered_range(src_inode, wb_start, wb_len); + if (ret < 0) + return ret; + } if (same_inode) { btrfs_inode_lock(src_inode, BTRFS_ILOCK_MMAP); From 875054e31724462f438d156a001e1ec58eccb84a Mon Sep 17 00:00:00 2001 From: Boris Burkov Date: Tue, 15 Sep 2026 10:58:33 -0700 Subject: [PATCH 0439/1352] btrfs: downgrade the reflink source inode lock for tree walking When doing a reflink, we are not writing to the src inode or modifying it in any way, so it is counter-intuitive to have to lock it exclusively. This has the cost of blocking other shared lock users of the src file for the duration of any new writeback since the optimistic pre-flush and all the actual reflinking. So we would like to downgrade the lock to shared as much as possible. There are two important gotchas to do with direct io that must be addressed to make this idea work. The first gotcha is that dio writes that come in under EOF try to take inode->i_rwsem shared, which would result in messed up views of the extents in the reflinked file. Therefore, add "being used as a reflink src" as another inode flag exception for the dio write shared tests that already exist to check for other dangerous conditions. In order to ensure that this flag is visible to the dio writer, reflink must first take i_rwsem exclusive, then set the flag, then downgrade to shared. The second is that dio reads increment inode->i_dio_count which blocks inode_dio_wait() in __generic_remap_file_range_prep(). Therefore, we must remain exclusive to dio reads through finishing the prep or risk starving the clone on a stream of dio readers. There is still O(extents) work in btrfs_clone() which can be done without holding the lock exclusive, so we should downgrade to shared and admit dio readers (and other shared lock users like lseek) after we are finished with inode_dio_wait(). By running concurrent clones and lseeks or dio reads on a src file with 32k extents on a vm, I can demonstrate 100ms+ stalls on the readers without this patch and they go away entirely with the patch. Reviewed-by: Filipe Manana Signed-off-by: Boris Burkov Signed-off-by: David Sterba --- fs/btrfs/btrfs_inode.h | 6 ++++++ fs/btrfs/direct-io.c | 7 +++++++ fs/btrfs/reflink.c | 17 +++++++++++++++-- 3 files changed, 28 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/btrfs_inode.h b/fs/btrfs/btrfs_inode.h index 46c62f980c24ac..03c00c99099d53 100644 --- a/fs/btrfs/btrfs_inode.h +++ b/fs/btrfs/btrfs_inode.h @@ -99,6 +99,12 @@ enum { * range). */ BTRFS_INODE_COW_WRITE_ERROR, + /* + * Set by reflink while it holds the source's VFS inode lock shared. + * Direct IO writes also take that lock shared; one that finds this bit + * set after acquiring it must retry with the exclusive lock. + */ + BTRFS_INODE_REFLINK_SRC, /* * Indicate this is a directory that points to a subvolume for which * there is no root reference item. That's a case like the following: diff --git a/fs/btrfs/direct-io.c b/fs/btrfs/direct-io.c index 3075d79927130d..ed3914257d3959 100644 --- a/fs/btrfs/direct-io.c +++ b/fs/btrfs/direct-io.c @@ -913,6 +913,13 @@ ssize_t btrfs_direct_write(struct kiocb *iocb, struct iov_iter *from) goto relock; } + if ((ilock_flags & BTRFS_ILOCK_SHARED) && + test_bit(BTRFS_INODE_REFLINK_SRC, &BTRFS_I(inode)->runtime_flags)) { + btrfs_inode_unlock(BTRFS_I(inode), ilock_flags); + ilock_flags &= ~BTRFS_ILOCK_SHARED; + goto relock; + } + ret = generic_write_checks(iocb, from); if (ret <= 0) { btrfs_inode_unlock(BTRFS_I(inode), ilock_flags); diff --git a/fs/btrfs/reflink.c b/fs/btrfs/reflink.c index 8e37706b98c31c..b5d2c094941df6 100644 --- a/fs/btrfs/reflink.c +++ b/fs/btrfs/reflink.c @@ -939,6 +939,7 @@ loff_t btrfs_remap_file_range(struct file *src_file, loff_t off, struct btrfs_inode *dst_inode = BTRFS_I(file_inode(dst_file)); bool same_inode = dst_inode == src_inode; u64 wb_start, wb_len; + bool src_downgraded = false; int ret; if (btrfs_is_shutdown(src_inode->root->fs_info)) @@ -976,6 +977,12 @@ loff_t btrfs_remap_file_range(struct file *src_file, loff_t off, if (ret < 0 || len == 0) goto out_unlock; + if (!same_inode) { + set_bit(BTRFS_INODE_REFLINK_SRC, &src_inode->runtime_flags); + downgrade_write(&src_inode->vfs_inode.i_rwsem); + src_downgraded = true; + } + if (remap_flags & REMAP_FILE_DEDUP) ret = btrfs_extent_same(src_inode, off, len, dst_inode, destoff); else @@ -986,8 +993,14 @@ loff_t btrfs_remap_file_range(struct file *src_file, loff_t off, btrfs_inode_unlock(src_inode, BTRFS_ILOCK_MMAP); } else { btrfs_double_mmap_unlock(src_inode, dst_inode); - unlock_two_nondirectories(&src_inode->vfs_inode, - &dst_inode->vfs_inode); + if (src_downgraded) { + clear_bit(BTRFS_INODE_REFLINK_SRC, &src_inode->runtime_flags); + inode_unlock_shared(&src_inode->vfs_inode); + inode_unlock(&dst_inode->vfs_inode); + } else { + unlock_two_nondirectories(&src_inode->vfs_inode, + &dst_inode->vfs_inode); + } } /* From b3b24c384dd8ecaac1ba0193dec395894eede19b Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Thu, 17 Sep 2026 11:38:04 +0100 Subject: [PATCH 0440/1352] btrfs: fix off by one super block end offset calculation when writing super blocks We are skipping the write of a super block mirror if its end offset matches exactly the device's end offset. This happens because we are using a greater than or equals comparison instead of just a greater than comparison. Fixes: 935e5cc935bc ("Btrfs: fix wrong disk size when writing super blocks") Reviewed-by: Qu Wenruo Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/disk-io.c | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index a8535e309b5760..90cf0647c47e36 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -3909,8 +3909,7 @@ static int write_dev_supers(struct btrfs_device *device, atomic_inc(&device->sb_write_errors); continue; } - if (bytenr + BTRFS_SUPER_INFO_SIZE >= - device->commit_total_bytes) + if (bytenr + BTRFS_SUPER_INFO_SIZE > device->commit_total_bytes) break; btrfs_set_super_bytenr(sb, bytenr_orig); @@ -3988,8 +3987,7 @@ static int wait_dev_supers(struct btrfs_device *device, int max_mirrors) primary_failed = true; continue; } - if (bytenr + BTRFS_SUPER_INFO_SIZE >= - device->commit_total_bytes) + if (bytenr + BTRFS_SUPER_INFO_SIZE > device->commit_total_bytes) break; folio = filemap_get_folio(device->bdev->bd_mapping, From 6f35e18e0a94613e040621eb20accec106e2a8d7 Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Fri, 18 Sep 2026 11:58:00 +0100 Subject: [PATCH 0441/1352] btrfs: clear BTRFS_ROOT_IN_TRANS_SETUP on early exit from record_root_in_trans() If we exit early because the transaction that last used the root already matches the current transaction, we leave the BTRFS_ROOT_IN_TRANS_SETUP bit set in the root (which we just set right before the exit). While this does not cause any functional issue, it makes callers of btrfs_record_root_in_trans() lock fs_info->reloc_mutex and call record_root_in_trans() for nothing, causing unnecessary lock contention, until one of them clears the bit in record_root_in_trans(). One caller of btrfs_record_root_in_trans() is start_transaction(), used to start new transaction or joining an existing one, which is a hot path. So clear BTRFS_ROOT_IN_TRANS_SETUP on early exit. Assisted-by: LLM Reviewed-by: Boris Burkov Reviewed-by: Qu Wenruo Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/transaction.c | 1 + 1 file changed, 1 insertion(+) diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c index ca114235bbe1e1..c1555621ae4e68 100644 --- a/fs/btrfs/transaction.c +++ b/fs/btrfs/transaction.c @@ -432,6 +432,7 @@ static int record_root_in_trans(struct btrfs_trans_handle *trans, spin_lock(&fs_info->fs_roots_radix_lock); if (btrfs_get_root_last_trans(root) == trans->transid && !force) { spin_unlock(&fs_info->fs_roots_radix_lock); + clear_bit(BTRFS_ROOT_IN_TRANS_SETUP, &root->state); return 0; } radix_tree_tag_set(&fs_info->fs_roots_radix, From 1bc0b46531b1f4c639f359a74a6e55da68eeea98 Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Fri, 18 Sep 2026 12:24:37 +0100 Subject: [PATCH 0442/1352] btrfs: fix barrier usage in btrfs_record_root_in_trans() The barrier usage in btrfs_record_root_in_trans() is wrong, as the writer side, in record_root_in_trans(), sets BTRFS_ROOT_IN_TRANS_SETUP, does a write barrier and then sets the root's last transaction. This means the reader side must check the root's last transaction, issue a read barrier and then check for BTRFS_ROOT_IN_TRANS_SETUP. However, currently we issue a read barrier and then check the root's last transaction and the bit BTRFS_ROOT_IN_TRANS_SETUP, which can be problematic because the CPU is free to reorder the checks and the following can happen: 1) Before reading the root's last_trans, it checks that BTRFS_ROOT_IN_TRANS_SETUP is not set. 2) A writer sets BTRFS_ROOT_IN_TRANS_SETUP, does smp_wmb() and updates the root's last_trans. 3) The reader then sees the root's last_trans matches the current transaction and falsely concludes the root setup is completes and returns without waiting for the writer task to complete the setup (calling btrfs_init_reloc_root()). So fix the reading ordered as previously described: check the root's last_trans, issue read barrier and then check BTRFS_ROOT_IN_TRANS_SETUP (the reverse of what the writer side does). Assisted-by: LLM Reviewed-by: Boris Burkov Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/transaction.c | 9 +++++---- 1 file changed, 5 insertions(+), 4 deletions(-) diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c index c1555621ae4e68..13203ea9e1163f 100644 --- a/fs/btrfs/transaction.c +++ b/fs/btrfs/transaction.c @@ -511,10 +511,11 @@ int btrfs_record_root_in_trans(struct btrfs_trans_handle *trans, * see record_root_in_trans for comments about IN_TRANS_SETUP usage * and barriers */ - smp_rmb(); - if (btrfs_get_root_last_trans(root) == trans->transid && - !test_bit(BTRFS_ROOT_IN_TRANS_SETUP, &root->state)) - return 0; + if (btrfs_get_root_last_trans(root) == trans->transid) { + smp_rmb(); + if (!test_bit(BTRFS_ROOT_IN_TRANS_SETUP, &root->state)) + return 0; + } mutex_lock(&fs_info->reloc_mutex); ret = record_root_in_trans(trans, root, false); From 12e0fd2dd9b38c8e8ce574ea386fa8e7232c59e3 Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Fri, 18 Sep 2026 12:46:22 +0100 Subject: [PATCH 0443/1352] btrfs: assert reloc mutex is held in record_root_in_trans() The fs_info->reloc_mutex is supposed to be held when record_root_in_trans() is called and we mention that in a comment inside the function. Add a lockdep assertion to check the lock is held. Reviewed-by: Boris Burkov Reviewed-by: Qu Wenruo Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/transaction.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c index 13203ea9e1163f..0a3913b61f164a 100644 --- a/fs/btrfs/transaction.c +++ b/fs/btrfs/transaction.c @@ -413,6 +413,8 @@ static int record_root_in_trans(struct btrfs_trans_handle *trans, struct btrfs_fs_info *fs_info = root->fs_info; int ret = 0; + lockdep_assert_held(&fs_info->reloc_mutex); + if ((test_bit(BTRFS_ROOT_SHAREABLE, &root->state) && btrfs_get_root_last_trans(root) < trans->transid) || force) { WARN_ON(!force && root->commit_root != root->node); From 8afffb0fb70ef2ef43b8beaf7f509a044bc18a39 Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Fri, 18 Sep 2026 17:41:28 +0100 Subject: [PATCH 0444/1352] btrfs: fix dangling nodes in tree-mod-log after error in btrfs_tree_mod_log_insert_root() If in btrfs_tree_mod_log_insert_root() there is an error in the call to tree_mod_log_insert() (the only possible error is -EEXIST, which means we have a bug or some memory corruption maybe) we free all the nodes in the 'tm_list' array but we don't remove them from the tree-mod-log rbtree, which can result in use-after-free bugs later. So make sure we remove the nodes from the rbtree before freeing them after an error. Fixes: 5de865eebb83 ("Btrfs: fix tree mod logging") Assisted-by: LLM (found the bug) Reviewed-by: Boris Burkov Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/tree-mod-log.c | 13 ++++++++++--- 1 file changed, 10 insertions(+), 3 deletions(-) diff --git a/fs/btrfs/tree-mod-log.c b/fs/btrfs/tree-mod-log.c index a8094928f4c9b8..bfa348d6483f3a 100644 --- a/fs/btrfs/tree-mod-log.c +++ b/fs/btrfs/tree-mod-log.c @@ -483,10 +483,17 @@ int btrfs_tree_mod_log_insert_root(struct extent_buffer *old_root, goto out_unlock; } - if (tm_list) + if (tm_list) { ret = tree_mod_log_free_eb(fs_info, tm_list, nritems); - if (!ret) - ret = tree_mod_log_insert(fs_info, tm); + if (ret) + goto out_unlock; + } + + ret = tree_mod_log_insert(fs_info, tm); + if (ret && tm_list) { + for (i = 0; i < nritems; i++) + rb_erase(&tm_list[i]->node, &fs_info->tree_mod_log); + } out_unlock: write_unlock(&fs_info->tree_mod_log_lock); From 44b3fa4811ab514d5e92dcffbcbbed7960c981ca Mon Sep 17 00:00:00 2001 From: Wentao Liang Date: Wed, 16 Sep 2026 17:16:10 +0000 Subject: [PATCH 0445/1352] btrfs: scrub: fix local_root reference leak in scrub_print_warning_inode() When paths_from_inode() fails, scrub_print_warning_inode() jumps to err without dropping the reference taken by btrfs_get_fs_root(), leaking a reference to the root every time path resolution fails while printing scrub warnings. Every other error and success path of the function drops the reference. Drop the reference on the paths_from_inode() failure path too. Fixes: 558540c17771 ("btrfs scrub: print paths of corrupted files") CC: stable@vger.kernel.org Reviewed-by: Qu Wenruo Signed-off-by: Wentao Liang Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/scrub.c | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/fs/btrfs/scrub.c b/fs/btrfs/scrub.c index c09d4213ad8915..479c0e037015ab 100644 --- a/fs/btrfs/scrub.c +++ b/fs/btrfs/scrub.c @@ -536,8 +536,10 @@ static int scrub_print_warning_inode(u64 inum, u64 offset, u64 num_bytes, } ret = paths_from_inode(inum, ipath); - if (ret < 0) + if (ret < 0) { + btrfs_put_root(local_root); goto err; + } /* * we deliberately ignore the bit ipath might have been too small to From 4ed5343aaa5d3b00fdcff50b72ce82b427887cd0 Mon Sep 17 00:00:00 2001 From: Wentao Liang Date: Wed, 16 Sep 2026 17:17:16 +0000 Subject: [PATCH 0446/1352] btrfs: zoned: fix block group reference leak in btrfs_repair_one_zone() btrfs_repair_one_zone() hands the block group reference taken by btrfs_lookup_block_group() over to relocating_repair_kthread(), which drops it. However the return value of kthread_run() is not checked, so when the kernel thread fails to spawn, nothing executes the put and the block group reference is leaked. Check kthread_run() and drop the reference if the thread failed to start. Fixes: f7ef5287a63d ("btrfs: zoned: relocate block group to repair IO failure in zoned filesystems") CC: stable@vger.kernel.org Reviewed-by: Boris Burkov Signed-off-by: Wentao Liang Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/volumes.c | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/volumes.c b/fs/btrfs/volumes.c index 7c040f22dbc3e5..116d882eecfec2 100644 --- a/fs/btrfs/volumes.c +++ b/fs/btrfs/volumes.c @@ -9006,6 +9006,7 @@ static int relocating_repair_kthread(void *data) bool btrfs_repair_one_zone(struct btrfs_fs_info *fs_info, u64 logical) { struct btrfs_block_group *cache; + struct task_struct *task; if (!btrfs_is_zoned(fs_info)) return false; @@ -9023,8 +9024,9 @@ bool btrfs_repair_one_zone(struct btrfs_fs_info *fs_info, u64 logical) return true; } - kthread_run(relocating_repair_kthread, cache, - "btrfs-relocating-repair"); + task = kthread_run(relocating_repair_kthread, cache, "btrfs-relocating-repair"); + if (IS_ERR(task)) + btrfs_put_block_group(cache); return true; } From 56be0384a653c27fd1515becb9cbbcfd80a661f4 Mon Sep 17 00:00:00 2001 From: Yang Xiuwei Date: Wed, 19 Aug 2026 10:54:32 +0800 Subject: [PATCH 0447/1352] btrfs: always return -EIOCBQUEUED after btrfs_uring_read_extent_endio If all bios finish before btrfs_encoded_read_regular_fill_pages() returns, it calls btrfs_uring_read_extent_endio() and previously returned the I/O status. A negative errno then made btrfs_uring_read_extent() unlock and free while btrfs_uring_read_finished() did the same again. Return -EIOCBQUEUED so only the deferred path cleans up. Reported-by: Yue Sun Closes: https://lore.kernel.org/linux-btrfs/20260630091609.3414-1-samsun1006219@gmail.com/ Suggested-by: Jens Axboe Fixes: 34310c442e17 ("btrfs: add io_uring command for encoded reads (ENCODED_READ ioctl)") Signed-off-by: Yang Xiuwei Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/inode.c | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 53f4532593b36c..2a32072849cbd4 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -9631,7 +9631,6 @@ int btrfs_encoded_read_regular_fill_pages(struct btrfs_inode *inode, struct completion sync_reads; unsigned long i = 0; struct btrfs_bio *bbio; - int ret; /* * Fast path for synchronous reads which completes in this call, io_uring @@ -9678,10 +9677,10 @@ int btrfs_encoded_read_regular_fill_pages(struct btrfs_inode *inode, if (uring_ctx) { if (refcount_dec_and_test(&priv->pending_refs)) { - ret = blk_status_to_errno(READ_ONCE(priv->status)); - btrfs_uring_read_extent_endio(uring_ctx, ret); + int error = blk_status_to_errno(READ_ONCE(priv->status)); + + btrfs_uring_read_extent_endio(uring_ctx, error); kfree(priv); - return ret; } return -EIOCBQUEUED; From 239dbe7e616b6cbb6b07eb437da7d2803e156b7d Mon Sep 17 00:00:00 2001 From: Yang Xiuwei Date: Wed, 19 Aug 2026 10:54:33 +0800 Subject: [PATCH 0448/1352] btrfs: free iov when btrfs_uring_read_extent() fails After btrfs_uring_read_extent(), the caller always jumped to out_acct. That skips kfree(data->iov), which is only correct for -EIOCBQUEUED where the deferred path owns the iov. On failure, fall through to out_free instead. Fixes: 34310c442e17 ("btrfs: add io_uring command for encoded reads (ENCODED_READ ioctl)") Signed-off-by: Yang Xiuwei Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/ioctl.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/ioctl.c b/fs/btrfs/ioctl.c index 54960351fbd171..64ccfa5fefdabf 100644 --- a/fs/btrfs/ioctl.c +++ b/fs/btrfs/ioctl.c @@ -4865,8 +4865,8 @@ static int btrfs_uring_encoded_read(struct io_uring_cmd *cmd, unsigned int issue cached_state, disk_bytenr, disk_io_size, count, data->args.compression, data->iov, cmd); - - goto out_acct; + if (ret == -EIOCBQUEUED) + goto out_acct; } out_free: From 37ae5d05f79487977b64b102ede17431bf970ee3 Mon Sep 17 00:00:00 2001 From: Yang Xiuwei Date: Wed, 19 Aug 2026 10:54:34 +0800 Subject: [PATCH 0449/1352] btrfs: unlock inode and extent in caller when io_uring read extent fails btrfs_uring_read_extent() runs only after btrfs_encoded_read() has taken the inode shared lock and the extent lock. On failure it used to unlock in out_fail, and a pages-array allocation failure returned -ENOMEM without unlocking at all. Unlock in the caller instead on all failure returns, matching the copy_to_user() error path. The deferred -EIOCBQUEUED path still unlocks in btrfs_uring_read_finished(). Fixes: 34310c442e17 ("btrfs: add io_uring command for encoded reads (ENCODED_READ ioctl)") Suggested-by: Qu Wenruo Reviewed-by: Qu Wenruo Signed-off-by: Yang Xiuwei Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/ioctl.c | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/fs/btrfs/ioctl.c b/fs/btrfs/ioctl.c index 64ccfa5fefdabf..588c25d1968879 100644 --- a/fs/btrfs/ioctl.c +++ b/fs/btrfs/ioctl.c @@ -4601,7 +4601,7 @@ static void btrfs_uring_read_finished(struct io_tw_req tw_req, io_tw_token_t tw) size_t page_offset; ssize_t ret; - /* The inode lock has already been acquired in btrfs_uring_read_extent. */ + /* The inode lock has already been acquired in btrfs_encoded_read(). */ btrfs_lockdep_inode_acquire(inode, i_rwsem); if (priv->err) { @@ -4667,7 +4667,6 @@ static int btrfs_uring_read_extent(struct kiocb *iocb, struct iov_iter *iter, struct iovec *iov, struct io_uring_cmd *cmd) { struct btrfs_inode *inode = BTRFS_I(file_inode(iocb->ki_filp)); - struct extent_io_tree *io_tree = &inode->io_tree; struct page **pages = NULL; struct btrfs_uring_priv *priv = NULL; unsigned long nr_pages; @@ -4723,8 +4722,6 @@ static int btrfs_uring_read_extent(struct kiocb *iocb, struct iov_iter *iter, return -EIOCBQUEUED; out_fail: - btrfs_unlock_extent(io_tree, start, lockend, &cached_state); - btrfs_inode_unlock(inode, BTRFS_ILOCK_SHARED); kfree(priv); for (int i = 0; i < nr_pages; i++) { if (pages[i]) @@ -4867,6 +4864,8 @@ static int btrfs_uring_encoded_read(struct io_uring_cmd *cmd, unsigned int issue data->iov, cmd); if (ret == -EIOCBQUEUED) goto out_acct; + btrfs_unlock_extent(io_tree, start, lockend, &cached_state); + btrfs_inode_unlock(inode, BTRFS_ILOCK_SHARED); } out_free: From 268083051933bb93fe7860639943b0dd5f53b18a Mon Sep 17 00:00:00 2001 From: Yang Xiuwei Date: Wed, 19 Aug 2026 10:54:35 +0800 Subject: [PATCH 0450/1352] btrfs: don't stash io_uring encoded data across -EAGAIN Returning -EAGAIN while leaving btrfs_uring_encoded_data in the cmd PDU leaks if the request is cancelled or the ring exits before reissue. io_uring does not free driver PDU allocations on cleanup. Write: io_queue_sqe() always issues with IO_URING_F_NONBLOCK first, so return -EAGAIN before allocating and free data on every exit. Read: free on nowait -EAGAIN too; only -EIOCBQUEUED keeps the allocation for btrfs_uring_read_finished(). Fixes: 34310c442e17 ("btrfs: add io_uring command for encoded reads (ENCODED_READ ioctl)") Fixes: e32dcdb0af9f ("btrfs: add io_uring interface for encoded writes") Signed-off-by: Yang Xiuwei Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/ioctl.c | 20 +++++++++++--------- 1 file changed, 11 insertions(+), 9 deletions(-) diff --git a/fs/btrfs/ioctl.c b/fs/btrfs/ioctl.c index 588c25d1968879..25d9feaa90030f 100644 --- a/fs/btrfs/ioctl.c +++ b/fs/btrfs/ioctl.c @@ -4834,7 +4834,7 @@ static int btrfs_uring_encoded_read(struct io_uring_cmd *cmd, unsigned int issue ret = btrfs_encoded_read(&kiocb, &data->iter, &data->args, &cached_state, &disk_bytenr, &disk_io_size); if (ret == -EAGAIN) - goto out_acct; + goto out_free; if (ret < 0 && ret != -EIOCBQUEUED) goto out_free; @@ -4876,8 +4876,10 @@ static int btrfs_uring_encoded_read(struct io_uring_cmd *cmd, unsigned int issue add_rchar(current, ret); inc_syscr(current); - if (ret != -EIOCBQUEUED && ret != -EAGAIN) + if (ret != -EIOCBQUEUED) { kfree(data); + bc->data = NULL; + } return ret; } @@ -4906,6 +4908,11 @@ static int btrfs_uring_encoded_write(struct io_uring_cmd *cmd, unsigned int issu goto out_acct; } + if (issue_flags & IO_URING_F_NONBLOCK) { + ret = -EAGAIN; + goto out_acct; + } + if (!data) { data = kzalloc_obj(*data, GFP_NOFS); if (!data) { @@ -4974,11 +4981,6 @@ static int btrfs_uring_encoded_write(struct io_uring_cmd *cmd, unsigned int issu } } - if (issue_flags & IO_URING_F_NONBLOCK) { - ret = -EAGAIN; - goto out_acct; - } - pos = data->args.offset; ret = rw_verify_area(WRITE, file, &pos, data->args.len); if (ret < 0) @@ -5004,8 +5006,8 @@ static int btrfs_uring_encoded_write(struct io_uring_cmd *cmd, unsigned int issu add_wchar(current, ret); inc_syscw(current); - if (ret != -EAGAIN) - kfree(data); + kfree(data); + bc->data = NULL; return ret; } From 7775ab7579ca39188072cb384b120150d4f9b88c Mon Sep 17 00:00:00 2001 From: Yang Xiuwei Date: Wed, 19 Aug 2026 10:54:36 +0800 Subject: [PATCH 0451/1352] btrfs: drop unused uring encoded IO REISSUE stash helpers After not keeping state across -EAGAIN, restoring bc->data on REISSUE is dead. Remove it, stop using the cmd PDU on the write path, and fold the read -EAGAIN check into the existing error path. Signed-off-by: Yang Xiuwei Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/ioctl.c | 12 ------------ 1 file changed, 12 deletions(-) diff --git a/fs/btrfs/ioctl.c b/fs/btrfs/ioctl.c index 25d9feaa90030f..52aab510aea0d1 100644 --- a/fs/btrfs/ioctl.c +++ b/fs/btrfs/ioctl.c @@ -4749,9 +4749,6 @@ static int btrfs_uring_encoded_read(struct io_uring_cmd *cmd, unsigned int issue struct io_btrfs_cmd *bc = io_uring_cmd_to_pdu(cmd, struct io_btrfs_cmd); struct btrfs_uring_encoded_data *data = NULL; - if (cmd->flags & IORING_URING_CMD_REISSUE) - data = bc->data; - if (!capable(CAP_SYS_ADMIN)) { ret = -EPERM; goto out_acct; @@ -4833,8 +4830,6 @@ static int btrfs_uring_encoded_read(struct io_uring_cmd *cmd, unsigned int issue ret = btrfs_encoded_read(&kiocb, &data->iter, &data->args, &cached_state, &disk_bytenr, &disk_io_size); - if (ret == -EAGAIN) - goto out_free; if (ret < 0 && ret != -EIOCBQUEUED) goto out_free; @@ -4891,12 +4886,8 @@ static int btrfs_uring_encoded_write(struct io_uring_cmd *cmd, unsigned int issu struct kiocb kiocb; ssize_t ret; void __user *sqe_addr; - struct io_btrfs_cmd *bc = io_uring_cmd_to_pdu(cmd, struct io_btrfs_cmd); struct btrfs_uring_encoded_data *data = NULL; - if (cmd->flags & IORING_URING_CMD_REISSUE) - data = bc->data; - if (!capable(CAP_SYS_ADMIN)) { ret = -EPERM; goto out_acct; @@ -4920,8 +4911,6 @@ static int btrfs_uring_encoded_write(struct io_uring_cmd *cmd, unsigned int issu goto out_acct; } - bc->data = data; - if (issue_flags & IO_URING_F_COMPAT) { #if defined(CONFIG_64BIT) && defined(CONFIG_COMPAT) struct btrfs_ioctl_encoded_io_args_32 args32; @@ -5007,7 +4996,6 @@ static int btrfs_uring_encoded_write(struct io_uring_cmd *cmd, unsigned int issu inc_syscw(current); kfree(data); - bc->data = NULL; return ret; } From d320f08ce090e0702a0decd3f020469fc9dd1680 Mon Sep 17 00:00:00 2001 From: Peng Fan Date: Sun, 20 Sep 2026 10:27:29 +0800 Subject: [PATCH 0452/1352] btrfs: use assign_bit() where applicable Convert open-coded if/else with set_bit/clear_bit and their non-atomic __set_bit/__clear_bit variants to the assign_bit/__assign_bit API. Done with Coccinelle semantic patch: // set_bit -> clear_bit => assign_bit @@ expression cond, bit, addr; @@ -if (cond) - set_bit(bit, addr); -else - clear_bit(bit, addr); +assign_bit(bit, addr, cond); // clear_bit -> set_bit => assign_bit @@ expression cond, bit, addr; @@ -if (cond) - clear_bit(bit, addr); -else - set_bit(bit, addr); +assign_bit(bit, addr, !cond); // __set_bit -> __clear_bit => __assign_bit @@ expression cond, bit, addr; @@ -if (cond) - __set_bit(bit, addr); -else - __clear_bit(bit, addr); +__assign_bit(bit, addr, cond); // __clear_bit -> __set_bit => __assign_bit @@ expression cond, bit, addr; @@ -if (cond) - __clear_bit(bit, addr); -else - __set_bit(bit, addr); +__assign_bit(bit, addr, !cond); Signed-off-by: Peng Fan Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/qgroup.c | 6 ++---- fs/btrfs/sysfs.c | 5 +---- fs/btrfs/volumes.c | 6 ++---- 3 files changed, 5 insertions(+), 12 deletions(-) diff --git a/fs/btrfs/qgroup.c b/fs/btrfs/qgroup.c index 05e35eb126dc5b..bf06962d49f3df 100644 --- a/fs/btrfs/qgroup.c +++ b/fs/btrfs/qgroup.c @@ -3177,10 +3177,8 @@ int btrfs_run_qgroups(struct btrfs_trans_handle *trans) "qgroup limit item update error %d", ret); spin_lock(&fs_info->qgroup_lock); } - if (btrfs_qgroup_enabled(fs_info)) - set_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags); - else - clear_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags); + assign_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags, + btrfs_qgroup_enabled(fs_info)); spin_unlock(&fs_info->qgroup_lock); ret = update_qgroup_status_item(trans); diff --git a/fs/btrfs/sysfs.c b/fs/btrfs/sysfs.c index c5bb1c7eac6afe..3ac1fa608d3b56 100644 --- a/fs/btrfs/sysfs.c +++ b/fs/btrfs/sysfs.c @@ -1117,10 +1117,7 @@ static ssize_t quota_override_store(struct kobject *kobj, if (knob > 1) return -EINVAL; - if (knob) - set_bit(BTRFS_FS_QUOTA_OVERRIDE, &fs_info->flags); - else - clear_bit(BTRFS_FS_QUOTA_OVERRIDE, &fs_info->flags); + assign_bit(BTRFS_FS_QUOTA_OVERRIDE, &fs_info->flags, knob); return len; } diff --git a/fs/btrfs/volumes.c b/fs/btrfs/volumes.c index 116d882eecfec2..d3e6a9b429dd9d 100644 --- a/fs/btrfs/volumes.c +++ b/fs/btrfs/volumes.c @@ -698,10 +698,8 @@ static int btrfs_open_one_device(struct btrfs_fs_devices *fs_devices, clear_bit(BTRFS_DEV_STATE_WRITEABLE, &device->dev_state); fs_devices->seeding = true; } else { - if (bdev_read_only(file_bdev(bdev_file))) - clear_bit(BTRFS_DEV_STATE_WRITEABLE, &device->dev_state); - else - set_bit(BTRFS_DEV_STATE_WRITEABLE, &device->dev_state); + assign_bit(BTRFS_DEV_STATE_WRITEABLE, &device->dev_state, + !bdev_read_only(file_bdev(bdev_file))); } if (bdev_rot(file_bdev(bdev_file))) From 0e2c2d085293e97524e5ff10bd2511bbe28376a6 Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Fri, 25 Sep 2026 13:47:21 +0100 Subject: [PATCH 0453/1352] btrfs: fix xattr replace when multiple xattrs are packed in the same item If we have a btrfs_dir_item item that packs multiple xattrs and then we replace the value of one of them (with the setxattr(2) family of syscalls) with another value of a different size, we end up not having a fully initialized btrfs_dir_item, resulting in a corruption that the tree checker will detect at extent buffer writeback time. This is because in btrfs_setxattr() when we find a btrfs_dir_item with multiple xattrs (due to the crc32c hash of their name being the same) we delete one of the xattr items (btrfs_dir_item) and then insert a new one, but the deletion and insertion results in shifting existing data in the leaf and therefore when the new value of a xattr has a different size, the new btrfs_dir_item is placed in a leaf section that was not initialized and we only copy the value's data and set the value's length in the new btrfs_dir_item, without setting the name, the name's length, the key (which must be all zeroes for xattrs), flags (BTRFS_FT_XATTR) and transaction ID. The following script reproduces the issue: $ cat test.sh #!/bin/bash DEV=/dev/sdi MNT=/mnt/sdi mkfs.btrfs -f $DEV mount $DEV $MNT touch $MNT/testfile # Add two xattrs that, on btrfs, have the same hash (crc32c) for their # name and therefore are packed into the same btrfs_dir_item. setfattr -n user.foobar -v 123 $MNT/testfile setfattr -n user.WvG1c1Td -v qwerty $MNT/testfile # Verify the xattrs are present. echo "xattrs before:" getfattr --absolute-names --dump $MNT/testfile # Now replace the value of the foobar xattr with a significantly larger # value. setfattr -n user.foobar -v abcdefghijklmnopqrstuvwxyz $MNT/testfile # Check the xattrs have the expected values. echo "xattrs after:" getfattr --absolute-names --dump $MNT/testfile umount $MNT Running it: $ ./test.sh (...) xattrs before: # file: /mnt/sdi/testfile user.WvG1c1Td="qwerty" user.foobar="123" xattrs after: # file: /mnt/sdi/testfile user.WvG1c1Td="qwerty" So the "user.foobar" xattr is missing and there was a transaction abort when unmounting the fs with the following traces in dmesg: $ dmesg [869800.159271] BTRFS warning (device sdi): access to eb bytenr 30474240 len 16384 out of range start 16015 len 25964 [869800.159293] ------------[ cut here ]------------ [869800.159296] WARNING: fs/btrfs/extent_io.c:4408 at report_eb_range+0x44/0x60 [btrfs], CPU#8: getfattr/2605179 [869800.168961] Modules linked in: btrfs dm_thin_pool (...) [869800.190527] CPU: 8 UID: 0 PID: 2605179 Comm: getfattr Tainted: G W 7.3.0-rc3-btrfs-next-244+ #1 PREEMPT(full) [869800.193726] Tainted: [W]=WARN [869800.194502] Hardware name: QEMU Standard PC (i440FX + PIIX, 1996), BIOS rel-1.16.2-0-gea1b7a073390-prebuilt.qemu.org 04/01/2014 [869800.197576] RIP: 0010:report_eb_range+0x44/0x60 [btrfs] [869800.198985] Code: 48 8b 7b 18 (...) [869800.203531] RSP: 0018:ffffce4541937d58 EFLAGS: 00010246 [869800.204609] RAX: 0000000000000000 RBX: ffff8dde054b8738 RCX: 0000000000000000 [869800.206137] RDX: 0000000000000000 RSI: 0000000000000001 RDI: ffffffffc04c92a0 [869800.207645] RBP: 0000000000003e8f R08: 0000000000000000 R09: 3fffffffffefffff [869800.226175] R10: ffffce4541937a88 R11: 0000000000000003 R12: 000000000000656c [869800.227594] R13: ffff8dde14be800f R14: 000000000000656c R15: 0000000000000069 [869800.229097] FS: 00007f59023bb780(0000) GS:ffff8de5788ed000(0000) knlGS:0000000000000000 [869800.231137] CS: 0010 DS: 0000 ES: 0000 CR0: 0000000080050033 [869800.232321] CR2: 0000558dc45e6a78 CR3: 0000000765b16004 CR4: 0000000000370ef0 [869800.233783] Call Trace: [869800.234330] [869800.234791] read_extent_buffer+0x4d/0x100 [btrfs] [869800.235898] btrfs_listxattr+0x199/0x240 [btrfs] [869800.236912] vfs_listxattr+0x51/0xa0 [869800.237683] listxattr+0x7e/0x100 [869800.238398] path_listxattrat+0x9e/0x190 [869800.239117] do_syscall_64+0x89/0x470 [869800.239877] entry_SYSCALL_64_after_hwframe+0x76/0x7e [869800.240914] RIP: 0033:0x7f59024cdcb7 [869800.241686] Code: f0 ff ff 73 (...) [869800.245353] RSP: 002b:00007ffd056dcdb8 EFLAGS: 00000246 ORIG_RAX: 00000000000000c2 [869800.246892] RAX: ffffffffffffffda RBX: 00007ffd056df2e2 RCX: 00007f59024cdcb7 [869800.248853] RDX: 0000000000006600 RSI: 0000558dc45e0470 RDI: 00007ffd056df2e2 [869800.250514] RBP: 00007ffd056df2e2 R08: 0000000000006600 R09: 0000000000006600 [869800.252295] R10: 0000000000000004 R11: 0000000000000246 R12: 00000000ffffff9c [869800.254105] R13: 0000558dc45e0470 R14: 0000000000006600 R15: 0000000000000000 [869800.255931] [869800.256529] ---[ end trace 0000000000000000 ]--- [869800.260071] page: refcount:2 mapcount:0 mapping:000000007ccfc77f index:0x1d10 pfn:0x608c69 [869800.260075] memcg:ffff8dde00344d40 [869800.260076] aops:btree_aops [btrfs] ino:1 [869800.260151] flags: 0x17fffc00000402a(uptodate|lru|private|writeback|node=0|zone=2|lastcpupid=0x1ffff) [869800.260154] raw: 017fffc00000402a fffff4adc773d4c8 fffff4adc48f3d88 ffff8de36504ba30 [869800.260155] raw: 0000000000001d10 ffff8dde054b8738 00000002ffffffff ffff8dde00344d40 [869800.260156] page dumped because: eb page dump [869800.260157] BTRFS critical (device sdi): corrupt leaf: root=5 block=30474240 slot=5 ino=257, invalid location key type, have 46, expect 132 or 1 [869800.260161] BTRFS info (device sdi): leaf 30474240 gen 9 total ptrs 6 free space 15629 owner 5 [869800.260163] BTRFS info (device sdi): refs 3 lock_owner 0 current 2550294 [869800.260164] item 0 key (256 INODE_ITEM 0) itemoff 16123 itemsize 160 [869800.260165] inode generation 3 transid 0 size 0 nbytes 16384 [869800.260166] block group 0 mode 40755 links 1 uid 0 gid 0 [869800.260167] rdev 0 sequence 0 flags 0x0 [869800.260168] atime 1790340637.0 [869800.260169] ctime 1790340637.0 [869800.260169] mtime 1790340637.0 [869800.260170] otime 1790340637.0 [869800.260170] item 1 key (256 INODE_REF 256) itemoff 16111 itemsize 12 [869800.260172] index 0 name_len 2 [869800.260172] item 2 key (256 DIR_ITEM 982728850) itemoff 16073 itemsize 38 [869800.260173] location key (257 1 0) type 1 [869800.260174] transid 9 data_len 0 name_len 8 [869800.260175] item 3 key (257 INODE_ITEM 0) itemoff 15913 itemsize 160 [869800.260176] inode generation 9 transid 9 size 0 nbytes 0 [869800.260177] block group 0 mode 100664 links 1 uid 0 gid 0 [869800.260177] rdev 0 sequence 0 flags 0x0 [869800.260178] atime 1790340638.38778502 [869800.260179] ctime 1790340638.38778502 [869800.260179] mtime 1790340638.38778502 [869800.260180] otime 1790340638.38778502 [869800.260180] item 4 key (257 INODE_REF 256) itemoff 15895 itemsize 18 [869800.260181] index 2 name_len 8 [869800.260182] item 5 key (257 XATTR_ITEM 751495445) itemoff 15779 itemsize 116 [869800.260183] location key (0 0 0) type 8 [869800.266052] transid 9 data_len 6 name_len 13 [869800.266053] location key (8243121639454149888 46 7229457603934778967) type 0 [869800.266055] transid 113 data_len 26 name_len 0 [869800.266056] location key (8608196880778817904 120 162425) type 9 [869800.266057] transid 8391162079612502016 data_len 26982 name_len 25964 [869800.266058] BTRFS error (device sdi): block=30474240 write time tree block corruption detected [869800.266091] ------------[ cut here ]------------ [869800.266092] WARNING: fs/btrfs/disk-io.c:336 at btree_csum_one_bio+0x20b/0x220 [btrfs], CPU#7: kworker/u50:7/2550294 [869800.268392] Modules linked in: btrfs dm_thin_pool (...) [869800.365851] CPU: 7 UID: 0 PID: 2550294 Comm: kworker/u50:7 Tainted: G W 7.3.0-rc3-btrfs-next-244+ #1 PREEMPT(full) [869800.368937] Tainted: [W]=WARN [869800.369834] Hardware name: QEMU Standard PC (i440FX + PIIX, 1996), BIOS rel-1.16.2-0-gea1b7a073390-prebuilt.qemu.org 04/01/2014 [869800.372778] Workqueue: writeback wb_workfn (flush-btrfs-3821) [869800.374314] RIP: 0010:btree_csum_one_bio+0x20b/0x220 [btrfs] [869800.375915] Code: 89 44 24 04 (...) [869800.380639] RSP: 0018:ffffce4548e3f7d0 EFLAGS: 00010246 [869800.382008] RAX: 0000000000000000 RBX: ffff8dde054b8738 RCX: 0000000000000000 [869800.383850] RDX: 0000000000000000 RSI: 0000000000000001 RDI: ffff8de091c2ddc0 [869800.385710] RBP: ffff8dde196a2000 R08: 0000000000000000 R09: 3fffffffffefffff [869800.387388] R10: ffffce4548e3f500 R11: 0000000000000003 R12: ffffce4548e3f7d8 [869800.388973] R13: ffff8dde196a2000 R14: ffff8de36504b750 R15: ffff8dde4c497b00 [869800.390414] FS: 0000000000000000(0000) GS:ffff8de5788ad000(0000) knlGS:0000000000000000 [869800.392002] CS: 0010 DS: 0000 ES: 0000 CR0: 0000000080050033 [869800.393160] CR2: 000055cd6e92ad5c CR3: 00000007cb264001 CR4: 0000000000370ef0 [869800.394595] Call Trace: [869800.395113] [869800.395570] btrfs_submit_bbio+0x872/0x890 [btrfs] [869800.397284] write_meta_extent_buffer+0x70/0x80 [btrfs] [869800.398940] btree_writepages+0x141/0x4f0 [btrfs] [869800.400426] ? get_random_u32+0x8a/0xf0 [869800.401417] ? build_slab_freelist+0x47/0x130 [869800.402574] ? preempt_count_add+0x6b/0xa0 [869800.403633] ? _raw_spin_lock_irqsave+0x23/0x50 [869800.404807] ? _raw_spin_unlock_irqrestore+0x22/0x40 [869800.406085] ? alloc_from_new_slab+0x18f/0x330 [869800.407223] do_writepages+0xc6/0x160 [869800.408191] ? refill_objects+0xd8/0x300 [869800.409211] __writeback_single_inode+0x42/0x350 [869800.410408] writeback_sb_inodes+0x231/0x560 [869800.411511] wb_writeback+0x8a/0x300 [869800.412440] wb_workfn+0xbf/0x460 [869800.413291] ? _raw_spin_unlock+0x14/0x30 [869800.414328] ? finish_task_switch.isra.0+0xb9/0x380 [869800.415105] process_one_work+0x1d1/0x3d0 [869800.416633] worker_thread+0x1c4/0x330 [869800.417467] ? __pfx_worker_thread+0x10/0x10 [869800.418452] kthread+0xfc/0x130 [869800.419257] ? __pfx_kthread+0x10/0x10 [869800.420089] ret_from_fork+0x1f7/0x2c0 [869800.420863] ? __pfx_kthread+0x10/0x10 [869800.421654] ret_from_fork_asm+0x1a/0x30 [869800.422484] [869800.422953] ---[ end trace 0000000000000000 ]--- [869800.424005] BTRFS error (device sdi state A): Transaction 9 aborted (-EIO) [869800.424010] BTRFS: error (device sdi state A) in __btrfs_run_delayed_items:1162: errno=-5 IO failure [869800.424011] BTRFS info (device sdi state EA): forced readonly [869800.424013] BTRFS warning (device sdi state EA): Skipping commit of aborted transaction. [869800.424014] BTRFS: error (device sdi state EA) in cleanup_transaction:2076: errno=-5 IO failure Fix this by always setting all fields in the new btrfs_dir_item when we replace an existing xattr. Fixes: 5f5bc6b1e2d5 ("Btrfs: make xattr replace operations atomic") Reviewed-by: Qu Wenruo Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/xattr.c | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/fs/btrfs/xattr.c b/fs/btrfs/xattr.c index ab55d10bd71fdb..e8762e12f3e1e5 100644 --- a/fs/btrfs/xattr.c +++ b/fs/btrfs/xattr.c @@ -161,6 +161,7 @@ int btrfs_setxattr(struct btrfs_trans_handle *trans, struct inode *inode, const u16 old_data_len = btrfs_dir_data_len(leaf, di); const u32 item_size = btrfs_item_size(leaf, slot); const u32 data_size = sizeof(*di) + name_len + size; + unsigned long name_ptr; unsigned long data_ptr; char *ptr; @@ -189,8 +190,16 @@ int btrfs_setxattr(struct btrfs_trans_handle *trans, struct inode *inode, ptr = btrfs_item_ptr(leaf, slot, char); ptr += btrfs_item_size(leaf, slot) - data_size; di = (struct btrfs_dir_item *)ptr; + memzero_extent_buffer(leaf, (unsigned long)ptr + + offsetof(struct btrfs_dir_item, location), + sizeof(struct btrfs_disk_key)); + btrfs_set_dir_flags(leaf, di, BTRFS_FT_XATTR); + btrfs_set_dir_transid(leaf, di, trans->transid); + btrfs_set_dir_name_len(leaf, di, name_len); btrfs_set_dir_data_len(leaf, di, size); + name_ptr = (unsigned long)(di + 1); data_ptr = ((unsigned long)(di + 1)) + name_len; + write_extent_buffer(leaf, name, name_ptr, name_len); write_extent_buffer(leaf, value, data_ptr, size); } else { /* From c803a3956678a2405ff952a56c7a3f160ac5dc30 Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Fri, 25 Sep 2026 16:45:34 +0100 Subject: [PATCH 0454/1352] btrfs: simplify dir item location setup in btrfs_insert_xattr_item() The location (a btrfs_disk_key) of a btrfs_dir_item used for a xattr is always zeroed and we have a complex setup in btrfs_insert_xattr_item() where we declare an on stack btrfs_key, zero it out, convert it into a btrfs_disk_key, also declared on stack, and then pass that btrfs_disk_key to a btrfs_set_dir_item_key() call. We can simplify this, and save stack space (one btrfs_disk_key and one btrfs_key), by simply using memzero_extent_buffer(). Reviewed-by: Qu Wenruo Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/dir-item.c | 9 ++++----- 1 file changed, 4 insertions(+), 5 deletions(-) diff --git a/fs/btrfs/dir-item.c b/fs/btrfs/dir-item.c index 3a90736915af6f..d39f85035b4344 100644 --- a/fs/btrfs/dir-item.c +++ b/fs/btrfs/dir-item.c @@ -62,8 +62,7 @@ int btrfs_insert_xattr_item(struct btrfs_trans_handle *trans, int ret = 0; struct btrfs_dir_item *dir_item; unsigned long name_ptr, data_ptr; - struct btrfs_key key, location; - struct btrfs_disk_key disk_key; + struct btrfs_key key; struct extent_buffer *leaf; u32 data_size; @@ -79,11 +78,11 @@ int btrfs_insert_xattr_item(struct btrfs_trans_handle *trans, name, name_len); if (IS_ERR(dir_item)) return PTR_ERR(dir_item); - memset(&location, 0, sizeof(location)); leaf = path->nodes[0]; - btrfs_cpu_key_to_disk(&disk_key, &location); - btrfs_set_dir_item_key(leaf, dir_item, &disk_key); + memzero_extent_buffer(leaf, (unsigned long)dir_item + + offsetof(struct btrfs_dir_item, location), + sizeof(struct btrfs_disk_key)); btrfs_set_dir_flags(leaf, dir_item, BTRFS_FT_XATTR); btrfs_set_dir_name_len(leaf, dir_item, name_len); btrfs_set_dir_transid(leaf, dir_item, trans->transid); From 0f952bdbdfc750734d876df57bc00938a934dc82 Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Fri, 25 Sep 2026 17:12:45 +0100 Subject: [PATCH 0455/1352] btrfs: fix lost error return value in btrfs_listxattr() If the input buffer does not have enough space to store the current xattr, we set 'iter_ret' to -ERANGE and then do "break", but that only exits the while loop over the xattrs in the current btrfs_dir_item, and then we continue the btrfs_for_each_slot() iteration, which overwrites the value of 'iter_ret' causing us to lose the error return value and proceed as if the buffer has enough space. Fix this by returning -ERANGE directly (the path is automatically freed) instead of breaking from the while loop. Fixes: 184b3d190087 ("btrfs: use btrfs_for_each_slot in btrfs_listxattr") Assisted-by: LLM (found the bug) Reviewed-by: Qu Wenruo Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/xattr.c | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/fs/btrfs/xattr.c b/fs/btrfs/xattr.c index e8762e12f3e1e5..cb001a211f5575 100644 --- a/fs/btrfs/xattr.c +++ b/fs/btrfs/xattr.c @@ -329,10 +329,8 @@ ssize_t btrfs_listxattr(struct dentry *dentry, char *buffer, size_t size) if (!size) goto next; - if (!buffer || (name_len + 1) > size_left) { - iter_ret = -ERANGE; - break; - } + if (!buffer || (name_len + 1) > size_left) + return -ERANGE; read_extent_buffer(leaf, buffer, name_ptr, name_len); buffer[name_len] = '\0'; From ae8acaa0db35482d6768078edaba071068c54d91 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Thu, 24 Sep 2026 15:28:14 +0930 Subject: [PATCH 0456/1352] btrfs: move __TRANS_* and TRANS_* flags out of transaction.h Among all those flags, only __TRANS_DUMMY is used outside of transaction.[ch], and there are only two places using it: - find_parent_nodes() Introduce a helper, btrfs_trans_is_dummy(), for this call site. - btrfs_init_dummy_trans() Move the function into transaction.[ch], and hide it behind CONFIG_BTRFS_FS_RUN_SANITY_TESTS. With the above changes, we can hide __TRANS_* and TRANS_* flags inside transaction.c. This allows us to modify those flags in the future without causing any changes to existing callers. After the flags relocation, now we can also hide __TRANS_DUMMY behind CONFIG_BTRFS_FS_RUN_SANITY_TESTS. Reviewed-by: Filipe Manana Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/backref.c | 2 +- fs/btrfs/tests/btrfs-tests.c | 9 --------- fs/btrfs/tests/btrfs-tests.h | 2 -- fs/btrfs/transaction.c | 36 ++++++++++++++++++++++++++++++++++++ fs/btrfs/transaction.h | 29 +++++++++++------------------ 5 files changed, 48 insertions(+), 30 deletions(-) diff --git a/fs/btrfs/backref.c b/fs/btrfs/backref.c index 1be632c742bdde..9c93c0c0d6f57a 100644 --- a/fs/btrfs/backref.c +++ b/fs/btrfs/backref.c @@ -1432,7 +1432,7 @@ static int find_parent_nodes(struct btrfs_backref_walk_ctx *ctx, goto out; } - if (ctx->trans && likely(ctx->trans->type != __TRANS_DUMMY) && + if (ctx->trans && likely(!btrfs_is_dummy_transaction(ctx->trans)) && ctx->time_seq != BTRFS_SEQ_LAST) { /* * We have a specific time_seq we care about and trans which diff --git a/fs/btrfs/tests/btrfs-tests.c b/fs/btrfs/tests/btrfs-tests.c index 6287d940323d69..cebc9a17b94dbd 100644 --- a/fs/btrfs/tests/btrfs-tests.c +++ b/fs/btrfs/tests/btrfs-tests.c @@ -247,15 +247,6 @@ void btrfs_init_dummy_transaction(struct btrfs_transaction *trans, struct btrfs_ spin_lock_init(&trans->delayed_refs.lock); } -void btrfs_init_dummy_trans(struct btrfs_trans_handle *trans, - struct btrfs_fs_info *fs_info) -{ - memset(trans, 0, sizeof(*trans)); - trans->transid = 1; - trans->type = __TRANS_DUMMY; - trans->fs_info = fs_info; -} - int btrfs_run_sanity_tests(void) { int ret, i; diff --git a/fs/btrfs/tests/btrfs-tests.h b/fs/btrfs/tests/btrfs-tests.h index cea58fe84a6d5f..a1685513810594 100644 --- a/fs/btrfs/tests/btrfs-tests.h +++ b/fs/btrfs/tests/btrfs-tests.h @@ -59,8 +59,6 @@ btrfs_alloc_dummy_block_group(struct btrfs_fs_info *fs_info, unsigned long lengt void btrfs_free_dummy_block_group(struct btrfs_block_group *cache); DEFINE_FREE(btrfs_free_dummy_block_group, struct btrfs_block_group *, btrfs_free_dummy_block_group(_T)); -void btrfs_init_dummy_trans(struct btrfs_trans_handle *trans, - struct btrfs_fs_info *fs_info); void btrfs_init_dummy_transaction(struct btrfs_transaction *trans, struct btrfs_fs_info *fs_info); struct btrfs_device *btrfs_alloc_dummy_device(struct btrfs_fs_info *fs_info); diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c index 0a3913b61f164a..dfab2f7f07760a 100644 --- a/fs/btrfs/transaction.c +++ b/fs/btrfs/transaction.c @@ -38,6 +38,26 @@ static struct kmem_cache *btrfs_trans_handle_cachep; +enum { + ENUM_BIT(__TRANS_FREEZABLE), + ENUM_BIT(__TRANS_START), + ENUM_BIT(__TRANS_ATTACH), + ENUM_BIT(__TRANS_JOIN), + ENUM_BIT(__TRANS_JOIN_NOLOCK), +#ifdef CONFIG_BTRFS_FS_RUN_SANITY_TESTS + ENUM_BIT(__TRANS_DUMMY), +#endif + ENUM_BIT(__TRANS_JOIN_NOSTART), +}; + +#define TRANS_START (__TRANS_START | __TRANS_FREEZABLE) +#define TRANS_ATTACH (__TRANS_ATTACH) +#define TRANS_JOIN (__TRANS_JOIN | __TRANS_FREEZABLE) +#define TRANS_JOIN_NOLOCK (__TRANS_JOIN_NOLOCK) +#define TRANS_JOIN_NOSTART (__TRANS_JOIN_NOSTART) + +#define TRANS_EXTWRITERS (__TRANS_START | __TRANS_ATTACH) + /* * Transaction states and transitions * @@ -139,6 +159,22 @@ static const unsigned int btrfs_blocked_trans_types[TRANS_STATE_MAX] = { __TRANS_JOIN_NOSTART), }; +#ifdef CONFIG_BTRFS_FS_RUN_SANITY_TESTS +bool btrfs_is_dummy_transaction(const struct btrfs_trans_handle *trans) +{ + return trans->type == __TRANS_DUMMY; +} + +void btrfs_init_dummy_trans(struct btrfs_trans_handle *trans, + struct btrfs_fs_info *fs_info) +{ + memset(trans, 0, sizeof(*trans)); + trans->transid = 1; + trans->type = __TRANS_DUMMY; + trans->fs_info = fs_info; +} +#endif + void btrfs_put_transaction(struct btrfs_transaction *transaction) { if (refcount_dec_and_test(&transaction->use_count)) { diff --git a/fs/btrfs/transaction.h b/fs/btrfs/transaction.h index 89153cd2259678..0b2e69293b3b90 100644 --- a/fs/btrfs/transaction.h +++ b/fs/btrfs/transaction.h @@ -119,24 +119,6 @@ struct btrfs_transaction { wait_queue_head_t pending_wait; }; -enum { - ENUM_BIT(__TRANS_FREEZABLE), - ENUM_BIT(__TRANS_START), - ENUM_BIT(__TRANS_ATTACH), - ENUM_BIT(__TRANS_JOIN), - ENUM_BIT(__TRANS_JOIN_NOLOCK), - ENUM_BIT(__TRANS_DUMMY), - ENUM_BIT(__TRANS_JOIN_NOSTART), -}; - -#define TRANS_START (__TRANS_START | __TRANS_FREEZABLE) -#define TRANS_ATTACH (__TRANS_ATTACH) -#define TRANS_JOIN (__TRANS_JOIN | __TRANS_FREEZABLE) -#define TRANS_JOIN_NOLOCK (__TRANS_JOIN_NOLOCK) -#define TRANS_JOIN_NOSTART (__TRANS_JOIN_NOSTART) - -#define TRANS_EXTWRITERS (__TRANS_START | __TRANS_ATTACH) - /* * Number of extent buffers a transaction handle tracks for writeback * inhibition. The CLOCK reference bits pack into a u32 so this must not exceed @@ -305,6 +287,17 @@ do { \ __LINE__, __error); \ } while (0) +#ifdef CONFIG_BTRFS_FS_RUN_SANITY_TESTS +bool btrfs_is_dummy_transaction(const struct btrfs_trans_handle *trans); +void btrfs_init_dummy_trans(struct btrfs_trans_handle *trans, + struct btrfs_fs_info *fs_info); +#else +static inline bool btrfs_is_dummy_transaction(const struct btrfs_trans_handle *trans) +{ + return false; +} +#endif + int btrfs_end_transaction(struct btrfs_trans_handle *trans); struct btrfs_trans_handle *btrfs_start_transaction(struct btrfs_root *root, unsigned int num_items); From 4a6c18b225d67cf5f67d3b898f21d5c622dad069 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Thu, 24 Sep 2026 15:28:15 +0930 Subject: [PATCH 0457/1352] btrfs: remove __TRANS_FREEZABLE Inside transaction.c most TRANS_* flags are just a single bit, but there are 2 exceptions: - TRANS_START Which is (__TRANS_START | __TRANS_FREEZABLE) - TRANS_JOIN Which is (__TRANS_JOIN | __TRANS_FREEZABLE) The extra __TRANS_FREEZABLE flag indicates that those operations need to acquire sb intwrite lock to handle fs freezing. However since there are only two operations requiring sb intwrite lock, there is no need to introduce a dedicated flag for it, we can introduce a new TRANS_SB_INTWRITER_MASK to cover the only two cases, then use that new mask to determine whether the type requires sb intwrite lock. This makes all TRANS_* flags a single bit. Reviewed-by: Filipe Manana Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/transaction.c | 20 ++++++++++++-------- 1 file changed, 12 insertions(+), 8 deletions(-) diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c index dfab2f7f07760a..7e716f5bdf7419 100644 --- a/fs/btrfs/transaction.c +++ b/fs/btrfs/transaction.c @@ -39,7 +39,6 @@ static struct kmem_cache *btrfs_trans_handle_cachep; enum { - ENUM_BIT(__TRANS_FREEZABLE), ENUM_BIT(__TRANS_START), ENUM_BIT(__TRANS_ATTACH), ENUM_BIT(__TRANS_JOIN), @@ -50,12 +49,15 @@ enum { ENUM_BIT(__TRANS_JOIN_NOSTART), }; -#define TRANS_START (__TRANS_START | __TRANS_FREEZABLE) +#define TRANS_START (__TRANS_START) #define TRANS_ATTACH (__TRANS_ATTACH) -#define TRANS_JOIN (__TRANS_JOIN | __TRANS_FREEZABLE) +#define TRANS_JOIN (__TRANS_JOIN) #define TRANS_JOIN_NOLOCK (__TRANS_JOIN_NOLOCK) #define TRANS_JOIN_NOSTART (__TRANS_JOIN_NOSTART) +/* Those types need to hold sb intwrite lock. */ +#define TRANS_SB_INTWRITER_MASK (__TRANS_START | __TRANS_JOIN) + #define TRANS_EXTWRITERS (__TRANS_START | __TRANS_ATTACH) /* @@ -750,6 +752,8 @@ start_transaction(struct btrfs_root *root, unsigned int num_items, } /* + * Only TRANS_START and TRANS_JOIN require sb intwrite lock. + * * If we are JOIN_NOLOCK we're already committing a transaction and * waiting on this guy, so we don't need to do the sb_start_intwrite * because we're already holding a ref. We need this because we could @@ -759,7 +763,7 @@ start_transaction(struct btrfs_root *root, unsigned int num_items, * If we are ATTACH, it means we just want to catch the current * transaction and commit it, so we needn't do sb_start_intwrite(). */ - if (type & __TRANS_FREEZABLE) + if (type & TRANS_SB_INTWRITER_MASK) sb_start_intwrite(fs_info->sb); if (may_wait_transaction(fs_info, type)) @@ -861,7 +865,7 @@ start_transaction(struct btrfs_root *root, unsigned int num_items, return h; join_fail: - if (type & __TRANS_FREEZABLE) + if (type & TRANS_SB_INTWRITER_MASK) sb_end_intwrite(fs_info->sb); kmem_cache_free(btrfs_trans_handle_cachep, h); alloc_fail: @@ -1142,7 +1146,7 @@ static int __btrfs_end_transaction(struct btrfs_trans_handle *trans, btrfs_trans_release_chunk_metadata(trans); - if (trans->type & __TRANS_FREEZABLE) + if (trans->type & TRANS_SB_INTWRITER_MASK) sb_end_intwrite(info->sb); /* @@ -2154,7 +2158,7 @@ static void cleanup_transaction(struct btrfs_trans_handle *trans, int err) fs_info->running_transaction = NULL; spin_unlock(&fs_info->trans_lock); - if (trans->type & __TRANS_FREEZABLE) + if (trans->type & TRANS_SB_INTWRITER_MASK) sb_end_intwrite(fs_info->sb); btrfs_put_transaction(cur_trans); btrfs_put_transaction(cur_trans); @@ -2686,7 +2690,7 @@ int btrfs_commit_transaction(struct btrfs_trans_handle *trans) btrfs_put_transaction(cur_trans); btrfs_put_transaction(cur_trans); - if (trans->type & __TRANS_FREEZABLE) + if (trans->type & TRANS_SB_INTWRITER_MASK) sb_end_intwrite(fs_info->sb); btrfs_scrub_continue(fs_info); From 1e57075aad6b38215412b518403c22c7422ca6b1 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Thu, 24 Sep 2026 15:28:16 +0930 Subject: [PATCH 0458/1352] btrfs: remove __TRANS_* flags After patch "btrfs: remove __TRANS_FREEZABLE", each TRANS_* flag is just the corresponding single-bit __TRANS_* flag. There is no need to split __TRANS_* and TRANS_* flags, just remove __TRANS_* flags and use TRANS_* flags instead. Since every TRANS_* flag is a single bit, do extra cleanups: - Add an ASSERT() in start_transaction() To make sure there is only a single bit set in @type - Rename TRANS_EXTWRITERS to TRANS_EXTWRITERS_MASK To follow the naming scheme that a multi-bit value has the _MASK suffix. Reviewed-by: Filipe Manana Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/transaction.c | 77 ++++++++++++++++++++---------------------- 1 file changed, 37 insertions(+), 40 deletions(-) diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c index 7e716f5bdf7419..cc71ad8f5e82eb 100644 --- a/fs/btrfs/transaction.c +++ b/fs/btrfs/transaction.c @@ -39,26 +39,20 @@ static struct kmem_cache *btrfs_trans_handle_cachep; enum { - ENUM_BIT(__TRANS_START), - ENUM_BIT(__TRANS_ATTACH), - ENUM_BIT(__TRANS_JOIN), - ENUM_BIT(__TRANS_JOIN_NOLOCK), + ENUM_BIT(TRANS_START), + ENUM_BIT(TRANS_ATTACH), + ENUM_BIT(TRANS_JOIN), + ENUM_BIT(TRANS_JOIN_NOLOCK), #ifdef CONFIG_BTRFS_FS_RUN_SANITY_TESTS - ENUM_BIT(__TRANS_DUMMY), + ENUM_BIT(TRANS_DUMMY), #endif - ENUM_BIT(__TRANS_JOIN_NOSTART), + ENUM_BIT(TRANS_JOIN_NOSTART), }; -#define TRANS_START (__TRANS_START) -#define TRANS_ATTACH (__TRANS_ATTACH) -#define TRANS_JOIN (__TRANS_JOIN) -#define TRANS_JOIN_NOLOCK (__TRANS_JOIN_NOLOCK) -#define TRANS_JOIN_NOSTART (__TRANS_JOIN_NOSTART) - /* Those types need to hold sb intwrite lock. */ -#define TRANS_SB_INTWRITER_MASK (__TRANS_START | __TRANS_JOIN) +#define TRANS_SB_INTWRITER_MASK (TRANS_START | TRANS_JOIN) -#define TRANS_EXTWRITERS (__TRANS_START | __TRANS_ATTACH) +#define TRANS_EXTWRITERS_MASK (TRANS_START | TRANS_ATTACH) /* * Transaction states and transitions @@ -139,32 +133,32 @@ enum { static const unsigned int btrfs_blocked_trans_types[TRANS_STATE_MAX] = { [TRANS_STATE_RUNNING] = 0U, [TRANS_STATE_COMMIT_PREP] = 0U, - [TRANS_STATE_COMMIT_START] = (__TRANS_START | __TRANS_ATTACH), - [TRANS_STATE_COMMIT_DOING] = (__TRANS_START | - __TRANS_ATTACH | - __TRANS_JOIN | - __TRANS_JOIN_NOSTART), - [TRANS_STATE_UNBLOCKED] = (__TRANS_START | - __TRANS_ATTACH | - __TRANS_JOIN | - __TRANS_JOIN_NOLOCK | - __TRANS_JOIN_NOSTART), - [TRANS_STATE_SUPER_COMMITTED] = (__TRANS_START | - __TRANS_ATTACH | - __TRANS_JOIN | - __TRANS_JOIN_NOLOCK | - __TRANS_JOIN_NOSTART), - [TRANS_STATE_COMPLETED] = (__TRANS_START | - __TRANS_ATTACH | - __TRANS_JOIN | - __TRANS_JOIN_NOLOCK | - __TRANS_JOIN_NOSTART), + [TRANS_STATE_COMMIT_START] = (TRANS_START | TRANS_ATTACH), + [TRANS_STATE_COMMIT_DOING] = (TRANS_START | + TRANS_ATTACH | + TRANS_JOIN | + TRANS_JOIN_NOSTART), + [TRANS_STATE_UNBLOCKED] = (TRANS_START | + TRANS_ATTACH | + TRANS_JOIN | + TRANS_JOIN_NOLOCK | + TRANS_JOIN_NOSTART), + [TRANS_STATE_SUPER_COMMITTED] = (TRANS_START | + TRANS_ATTACH | + TRANS_JOIN | + TRANS_JOIN_NOLOCK | + TRANS_JOIN_NOSTART), + [TRANS_STATE_COMPLETED] = (TRANS_START | + TRANS_ATTACH | + TRANS_JOIN | + TRANS_JOIN_NOLOCK | + TRANS_JOIN_NOSTART), }; #ifdef CONFIG_BTRFS_FS_RUN_SANITY_TESTS bool btrfs_is_dummy_transaction(const struct btrfs_trans_handle *trans) { - return trans->type == __TRANS_DUMMY; + return trans->type == TRANS_DUMMY; } void btrfs_init_dummy_trans(struct btrfs_trans_handle *trans, @@ -172,7 +166,7 @@ void btrfs_init_dummy_trans(struct btrfs_trans_handle *trans, { memset(trans, 0, sizeof(*trans)); trans->transid = 1; - trans->type = __TRANS_DUMMY; + trans->type = TRANS_DUMMY; trans->fs_info = fs_info; } #endif @@ -261,21 +255,21 @@ static noinline void switch_commit_roots(struct btrfs_trans_handle *trans) static inline void extwriter_counter_inc(struct btrfs_transaction *trans, unsigned int type) { - if (type & TRANS_EXTWRITERS) + if (type & TRANS_EXTWRITERS_MASK) atomic_inc(&trans->num_extwriters); } static inline void extwriter_counter_dec(struct btrfs_transaction *trans, unsigned int type) { - if (type & TRANS_EXTWRITERS) + if (type & TRANS_EXTWRITERS_MASK) atomic_dec(&trans->num_extwriters); } static inline void extwriter_counter_init(struct btrfs_transaction *trans, unsigned int type) { - atomic_set(&trans->num_extwriters, ((type & TRANS_EXTWRITERS) ? 1 : 0)); + atomic_set(&trans->num_extwriters, ((type & TRANS_EXTWRITERS_MASK) ? 1 : 0)); } static inline int extwriter_counter_read(struct btrfs_transaction *trans) @@ -666,11 +660,14 @@ start_transaction(struct btrfs_root *root, unsigned int num_items, bool do_chunk_alloc = false; int ret; + /* The type must be a single TRANS_* bit set. */ + ASSERT(is_power_of_2(type)); + if (unlikely(BTRFS_FS_ERROR(fs_info))) return ERR_PTR(-EROFS); if (current->journal_info) { - WARN_ON(type & TRANS_EXTWRITERS); + WARN_ON(type & TRANS_EXTWRITERS_MASK); h = current->journal_info; refcount_inc(&h->use_count); WARN_ON(refcount_read(&h->use_count) > 2); From 043a43e38e6c62b60d81ffcc4c09039787f98e95 Mon Sep 17 00:00:00 2001 From: Boris Burkov Date: Mon, 28 Sep 2026 17:20:33 -0700 Subject: [PATCH 0459/1352] btrfs: allocate additional SYSTEM space earlier Currently, reserve_chunk_space() allocates a new system chunk when the space left is smaller than the reservation for a single chunk tree update. Therefore, a filesystem in single metadata mode on a very large block device can accumulate tens of thousands (TB) of chunks while never allocating more than the original 4MiB SYSTEM block group created by mkfs. If it then fills up (with large fallocates for example) and then that data is freed, en-masse, we end up trying to delete thousands of block groups in a single transaction in btrfs_delete_unused_bgs(). This process touches most of the nodes and leaves of the chunk tree, essentially trying to allocate roughly double the current usage of the chunk tree for cow. When this runs into needing a fresh system chunk, the fs is full of the empty bgs and we abort with ENOSPC in btrfs_remove_chunk() like: BTRFS error (device loop0 state A): Transaction 30733 aborted (-ENOSPC) space_info SYSTEM (sub-group id 0) has 0 free, is not full space_info total=4194304, used=3276800, pinned=0, reserved=917504, may_use=0, readonly=0 zone_unusable=0 BTRFS: error (device loop0 state A) in btrfs_remove_chunk:3643: errno=-28 No space left This was seen on a 30TiB production filesystem with 10TiB of unused block groups and reproduces on a 30TiB sparse loop device: mkfs with -m single, fallocate 1GiB files until ENOSPC, delete two of every three adjacent files, sync. To fix this, we should allocate the (relatively tiny) system chunks a little bit more eagerly. If we do it when it is half full, we ensure we can delete all of the block groups in one go. To fill up half of a 4MiB system bg requires thousands of block_groups so this should only affect very large fileystems. Assisted-by: LLM Reviewed-by: Filipe Manana Signed-off-by: Boris Burkov Signed-off-by: David Sterba --- fs/btrfs/block-group.c | 42 +++++++++++++++++++++++++++++++++--------- 1 file changed, 33 insertions(+), 9 deletions(-) diff --git a/fs/btrfs/block-group.c b/fs/btrfs/block-group.c index 2eb09c9901c9e1..0a41b16ed27021 100644 --- a/fs/btrfs/block-group.c +++ b/fs/btrfs/block-group.c @@ -1574,6 +1574,19 @@ static bool btrfs_link_bg_list(struct btrfs_block_group *bg, struct list_head *l return added; } +/* + * Compute a rough bound on the bytes needed to modify every leaf in the chunk + * tree. Normally, we could just allocate a new system chunk then, but that can + * fail if there is no free space for a new dev extent. This can happen when + * we fill up a large filesystem then rapidly delete a large portion of it. + */ +static u64 system_space_target(struct btrfs_space_info *sinfo) +{ + lockdep_assert_held(&sinfo->lock); + + return btrfs_space_info_used(sinfo, true) + sinfo->bytes_used; +} + /* * Process the unused_bgs list and remove any that don't have any allocated * space inside of them. @@ -1661,6 +1674,9 @@ void btrfs_delete_unused_bgs(struct btrfs_fs_info *fs_info) if (btrfs_is_block_group_used(block_group) || (block_group->ro && !(block_group->flags & BTRFS_BLOCK_GROUP_REMAPPED)) || list_is_singular(&block_group->list) || + ((block_group->flags & BTRFS_BLOCK_GROUP_SYSTEM) && + space_info->total_bytes - block_group->length < + system_space_target(space_info)) || test_bit(BLOCK_GROUP_FLAG_FULLY_REMAPPED, &block_group->runtime_flags)) { /* * We want to bail if we made new allocations or have @@ -1674,6 +1690,10 @@ void btrfs_delete_unused_bgs(struct btrfs_fs_info *fs_info) * next block group of this type would be created with a * "single" profile (even if we're in a raid fs) because * fs_info->avail_*_alloc_bits would be 0. + * + * Also bail out if this is a system block group that + * system_space_target() relies on to ensure head room for + * mass deletion. */ trace_btrfs_skip_unused_block_group(block_group); spin_unlock(&block_group->lock); @@ -4509,6 +4529,7 @@ static void reserve_chunk_space(struct btrfs_trans_handle *trans, struct btrfs_fs_info *fs_info = trans->fs_info; struct btrfs_space_info *info; u64 left; + bool want_system_chunk, need_system_chunk; int ret = 0; /* @@ -4520,15 +4541,17 @@ static void reserve_chunk_space(struct btrfs_trans_handle *trans, info = btrfs_find_space_info(fs_info, BTRFS_BLOCK_GROUP_SYSTEM); spin_lock(&info->lock); left = info->total_bytes - btrfs_space_info_used(info, true); + want_system_chunk = info->total_bytes < system_space_target(info); + need_system_chunk = left < bytes; spin_unlock(&info->lock); - if (left < bytes && btrfs_test_opt(fs_info, ENOSPC_DEBUG)) { + if (need_system_chunk && btrfs_test_opt(fs_info, ENOSPC_DEBUG)) { btrfs_info(fs_info, "left=%llu, need=%llu, flags=%llu", left, bytes, type); btrfs_dump_space_info(info, 0, false); } - if (left < bytes) { + if (want_system_chunk || need_system_chunk) { u64 flags = btrfs_system_alloc_profile(fs_info); struct btrfs_block_group *bg; struct btrfs_space_info *space_info; @@ -4572,13 +4595,14 @@ static void reserve_chunk_space(struct btrfs_trans_handle *trans, } } - if (!ret) { - ret = btrfs_block_rsv_add(fs_info, - &fs_info->chunk_block_rsv, - bytes, BTRFS_RESERVE_NO_FLUSH); - if (!ret) - trans->chunk_bytes_reserved += bytes; - } + if (ret && need_system_chunk) + return; + + ret = btrfs_block_rsv_add(fs_info, + &fs_info->chunk_block_rsv, + bytes, BTRFS_RESERVE_NO_FLUSH); + if (!ret) + trans->chunk_bytes_reserved += bytes; } /* From 7780115d3115dbf5f953395b291f553809e3691c Mon Sep 17 00:00:00 2001 From: Boris Burkov Date: Mon, 28 Sep 2026 17:20:34 -0700 Subject: [PATCH 0460/1352] btrfs: commit after deleting an unused block group if system space is low If the device is fully allocated and the system space_info has no room for a second copy of the chunk tree, deleting many unused block groups in one transaction aborts it with -ENOSPC in btrfs_remove_chunk(). While we can get ahead of this problem in practice by allocating the system space more aggressively, that doesn't help a system that already got into this state anyway. In particular, this would be a problem on systems that got into this state on old code from before that improvement, but it is also a backstop in case we do mess up and still end up with too little system space. The way we end up needing a lot of system chunk space in practice is if we need to delete many bgs at once, resulting in touching many nodes/leaves in the chunk tree. Therefore, if we detect we are in a condition where we are low on system space, commit a single transaction deleting a single bg, which should not need much system chunk space. After this we have the freed bg and can proceed as usual. Any additional chunk allocations (for any reason) will also check the system space and are very likely to allocate more system space, so this should be a relatively robust mechanism. Assisted-by: LLM Reviewed-by: Filipe Manana Signed-off-by: Boris Burkov Signed-off-by: David Sterba --- fs/btrfs/block-group.c | 17 ++++++++++++++++- 1 file changed, 16 insertions(+), 1 deletion(-) diff --git a/fs/btrfs/block-group.c b/fs/btrfs/block-group.c index 0a41b16ed27021..c148b9d6e597fd 100644 --- a/fs/btrfs/block-group.c +++ b/fs/btrfs/block-group.c @@ -1596,6 +1596,7 @@ void btrfs_delete_unused_bgs(struct btrfs_fs_info *fs_info) LIST_HEAD(retry_list); struct btrfs_block_group *block_group; struct btrfs_space_info *space_info; + struct btrfs_space_info *system_info; struct btrfs_trans_handle *trans; const bool async_trim_enabled = btrfs_test_opt(fs_info, DISCARD_ASYNC); int ret = 0; @@ -1613,10 +1614,13 @@ void btrfs_delete_unused_bgs(struct btrfs_fs_info *fs_info) if (!mutex_trylock(&fs_info->reclaim_bgs_lock)) return; + system_info = btrfs_find_space_info(fs_info, BTRFS_BLOCK_GROUP_SYSTEM); + spin_lock(&fs_info->unused_bgs_lock); while (!list_empty(&fs_info->unused_bgs)) { u64 used; int trimming; + bool low; block_group = list_first_entry(&fs_info->unused_bgs, struct btrfs_block_group, @@ -1864,7 +1868,18 @@ void btrfs_delete_unused_bgs(struct btrfs_fs_info *fs_info) btrfs_get_block_group(block_group); } end_trans: - btrfs_end_transaction(trans); + /* + * If we are low on system space, commit immediately to get back + * unallocated space to greatly improve the chances of + * successfully allocating a system chunk. + */ + spin_lock(&system_info->lock); + low = system_info->total_bytes < system_space_target(system_info); + spin_unlock(&system_info->lock); + if (unlikely(!ret && low)) + ret = btrfs_commit_transaction(trans); + else + btrfs_end_transaction(trans); next: btrfs_put_block_group(block_group); spin_lock(&fs_info->unused_bgs_lock); From 158763c20c573304abe6aa7c732644bb8a2c360c Mon Sep 17 00:00:00 2001 From: Johannes Thumshirn Date: Tue, 29 Sep 2026 16:02:19 +0200 Subject: [PATCH 0461/1352] btrfs: zoned: fix double list add in btrfs_load_block_group_zone_info For DUP and RAID profiles with zones in a mismatching active state, i.e. due to one of the zones of the block-group being on a conventional zone, btrfs_zone_activate() has already put the block group on the list. This leads to a double list add and refcounting problems. Fixes: 265f7237dd25 ("btrfs: zoned: allow DUP on meta-data block groups") Fixes: 568220fa9657 ("btrfs: zoned: support RAID0/1/10 on top of raid stripe tree") Reviewed-by: Daniel Vacek Signed-off-by: Johannes Thumshirn Signed-off-by: David Sterba --- fs/btrfs/zoned.c | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/fs/btrfs/zoned.c b/fs/btrfs/zoned.c index a1ef8caaacdab4..9f562cc34e3836 100644 --- a/fs/btrfs/zoned.c +++ b/fs/btrfs/zoned.c @@ -2005,10 +2005,12 @@ int btrfs_load_block_group_zone_info(struct btrfs_block_group *cache, bool new) if (!ret) { cache->meta_write_pointer = cache->alloc_offset + cache->start; if (test_bit(BLOCK_GROUP_FLAG_ZONE_IS_ACTIVE, &cache->runtime_flags)) { - btrfs_get_block_group(cache); spin_lock(&fs_info->zone_active_bgs_lock); - list_add_tail(&cache->active_bg_list, - &fs_info->zone_active_bgs); + if (list_empty(&cache->active_bg_list)) { + btrfs_get_block_group(cache); + list_add_tail(&cache->active_bg_list, + &fs_info->zone_active_bgs); + } spin_unlock(&fs_info->zone_active_bgs_lock); } } else { From a901bba92957563b7ecd4779f2a117de491032fd Mon Sep 17 00:00:00 2001 From: David Sterba Date: Wed, 21 Feb 2024 15:50:10 +0100 Subject: [PATCH 0462/1352] btrfs: === misc-next on b-for-next === Any commits after this one are for testing and evaluation only. Signed-off-by: David Sterba --- fs/btrfs/Kconfig | 1 + 1 file changed, 1 insertion(+) diff --git a/fs/btrfs/Kconfig b/fs/btrfs/Kconfig index 4b10d78ed99b16..0b8d8905e38e25 100644 --- a/fs/btrfs/Kconfig +++ b/fs/btrfs/Kconfig @@ -1,4 +1,5 @@ # SPDX-License-Identifier: GPL-2.0 +# misc-next marker config BTRFS_FS tristate "Btrfs filesystem support" From 0ceedc1572cec8f8249d9b21e90b33a8292888fa Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Wed, 16 Sep 2026 23:59:57 -0400 Subject: [PATCH 0463/1352] btrfs: stop enabling the v1 space cache from the on-disk state Since commit 545e560a5b0f ("btrfs: disable v1 space cache") the mount options can no longer request the v1 space cache, but a filesystem with an active v1 cache and no free space tree still enables it from cache_generation, and remount does the same. Drop both, so SPACE_CACHE can never be set. btrfs_start_pre_rw_mount() then sees the on-disk cache as active but unwanted and cleans it up, as -o nospace_cache does today. That covers the read-only to read-write remount as well, so drop the toggle in btrfs_remount_cleanup(), which would otherwise start a transaction on remounts of a read-only filesystem with an old cache. The cleanup is now unconditional, and the first read-write mount fails if it fails, as it did with -o nospace_cache. This also lets an old filesystem mount without options when the page size is larger than the sector size, which btrfs_check_features() rejected once SPACE_CACHE was set from the superblock. Assisted-by: Claude:claude-fable-5-1 Reviewed-by: Qu Wenruo Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/super.c | 16 +++------------- 1 file changed, 3 insertions(+), 13 deletions(-) diff --git a/fs/btrfs/super.c b/fs/btrfs/super.c index ddb620ac241b42..213db18c6dfccf 100644 --- a/fs/btrfs/super.c +++ b/fs/btrfs/super.c @@ -759,12 +759,12 @@ void btrfs_set_free_space_cache_settings(struct btrfs_fs_info *fs_info) /* * At this point we don't have explicit options set by the user, set - * them ourselves based on the state of the file system. + * them ourselves based on the state of the file system. An existing + * v1 space cache is no longer used and gets cleaned up once the + * filesystem is mounted read-write. */ if (btrfs_fs_compat_ro(fs_info, FREE_SPACE_TREE)) btrfs_set_opt(fs_info->mount_opt, FREE_SPACE_TREE); - else if (btrfs_free_space_cache_v1_active(fs_info)) - btrfs_set_opt(fs_info->mount_opt, SPACE_CACHE); } static void set_device_specific_options(struct btrfs_fs_info *fs_info) @@ -1264,8 +1264,6 @@ static inline void btrfs_remount_begin(struct btrfs_fs_info *fs_info, static inline void btrfs_remount_cleanup(struct btrfs_fs_info *fs_info, unsigned long long old_opts) { - const bool cache_opt = btrfs_test_opt(fs_info, SPACE_CACHE); - /* * We need to cleanup all defraggable inodes if the autodefragment is * close or the filesystem is read only. @@ -1282,10 +1280,6 @@ static inline void btrfs_remount_cleanup(struct btrfs_fs_info *fs_info, else if (btrfs_raw_test_opt(old_opts, DISCARD_ASYNC) && !btrfs_test_opt(fs_info, DISCARD_ASYNC)) btrfs_discard_cleanup(fs_info); - - /* If we toggled space cache */ - if (cache_opt != btrfs_free_space_cache_v1_active(fs_info)) - btrfs_set_free_space_cache_v1_active(fs_info, cache_opt); } static int btrfs_remount_rw(struct btrfs_fs_info *fs_info) @@ -1535,10 +1529,6 @@ static int btrfs_reconfigure(struct fs_context *fc) btrfs_set_opt(fs_info->mount_opt, FREE_SPACE_TREE); btrfs_clear_opt(fs_info->mount_opt, SPACE_CACHE); } - if (btrfs_free_space_cache_v1_active(fs_info)) { - btrfs_clear_opt(fs_info->mount_opt, FREE_SPACE_TREE); - btrfs_set_opt(fs_info->mount_opt, SPACE_CACHE); - } } ret = 0; From 68ee583b1d04a96135931959ce4fa697a4151cc0 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Wed, 16 Sep 2026 23:59:58 -0400 Subject: [PATCH 0464/1352] btrfs: remove the v1 space cache writeout from the transaction commit Nothing sets SPACE_CACHE anymore, so the dirty block group writers never have a cache to write out or wait for. Remove cache_save_setup(), btrfs_setup_space_cache(), the io_list handling, the io_bgs list and BTRFS_TRANS_CACHE_ENOSPC, and the abort-time cleanup of in-flight cache IO. The -ENOENT retry in btrfs_write_dirty_block_groups() handled a free space endio worker creating a block group during the commit critical section, so drop it too. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/block-group.c | 397 +++-------------------------------------- fs/btrfs/block-group.h | 1 - fs/btrfs/disk-io.c | 44 ----- fs/btrfs/transaction.c | 9 +- fs/btrfs/transaction.h | 18 -- 5 files changed, 26 insertions(+), 443 deletions(-) diff --git a/fs/btrfs/block-group.c b/fs/btrfs/block-group.c index c148b9d6e597fd..059f984e053c72 100644 --- a/fs/btrfs/block-group.c +++ b/fs/btrfs/block-group.c @@ -1194,29 +1194,10 @@ int btrfs_remove_block_group(struct btrfs_trans_handle *trans, goto out; } - /* - * get the inode first so any iput calls done for the io_list - * aren't the final iput (no unlinks allowed now) - */ inode = lookup_free_space_inode(block_group, path); mutex_lock(&trans->transaction->cache_write_mutex); - /* - * Make sure our free space cache IO is done before removing the - * free space inode - */ spin_lock(&trans->transaction->dirty_bgs_lock); - if (!list_empty(&block_group->io_list)) { - list_del_init(&block_group->io_list); - - WARN_ON(!IS_ERR(inode) && inode != block_group->io_ctl.inode); - - spin_unlock(&trans->transaction->dirty_bgs_lock); - btrfs_wait_cache_io(trans, block_group, path); - btrfs_put_block_group(block_group); - spin_lock(&trans->transaction->dirty_bgs_lock); - } - if (!list_empty(&block_group->dirty_list)) { list_del_init(&block_group->dirty_list); remove_rsv = true; @@ -3413,197 +3394,6 @@ static int update_block_group_item(struct btrfs_trans_handle *trans, } -static void cache_save_setup(struct btrfs_block_group *block_group, - struct btrfs_trans_handle *trans, - struct btrfs_path *path) -{ - struct btrfs_fs_info *fs_info = block_group->fs_info; - struct inode *inode = NULL; - struct extent_changeset *data_reserved = NULL; - u64 alloc_hint = 0; - int dcs = BTRFS_DC_ERROR; - u64 cache_size = 0; - int retries = 0; - int ret = 0; - - if (!btrfs_test_opt(fs_info, SPACE_CACHE)) - return; - - /* - * If this block group is smaller than 100 megs don't bother caching the - * block group. - */ - if (block_group->length < (100 * SZ_1M)) { - spin_lock(&block_group->lock); - block_group->disk_cache_state = BTRFS_DC_WRITTEN; - spin_unlock(&block_group->lock); - return; - } - - if (TRANS_ABORTED(trans)) - return; -again: - inode = lookup_free_space_inode(block_group, path); - if (IS_ERR(inode) && PTR_ERR(inode) != -ENOENT) { - ret = PTR_ERR(inode); - btrfs_release_path(path); - goto out; - } - - if (IS_ERR(inode)) { - if (retries) { - ret = PTR_ERR(inode); - btrfs_err(fs_info, - "failed to lookup free space inode after creation for block group %llu: %d", - block_group->start, ret); - goto out_free; - } - retries++; - - if (block_group->ro) - goto out_free; - - ret = create_free_space_inode(trans, block_group, path); - if (ret) - goto out_free; - goto again; - } - - /* - * We want to set the generation to 0, that way if anything goes wrong - * from here on out we know not to trust this cache when we load up next - * time. - */ - BTRFS_I(inode)->generation = 0; - ret = btrfs_update_inode(trans, BTRFS_I(inode)); - if (unlikely(ret)) { - /* - * So theoretically we could recover from this, simply set the - * super cache generation to 0 so we know to invalidate the - * cache, but then we'd have to keep track of the block groups - * that fail this way so we know we _have_ to reset this cache - * before the next commit or risk reading stale cache. So to - * limit our exposure to horrible edge cases lets just abort the - * transaction, this only happens in really bad situations - * anyway. - */ - btrfs_abort_transaction(trans, ret); - goto out_put; - } - - /* We've already setup this transaction, go ahead and exit */ - if (block_group->cache_generation == trans->transid && - i_size_read(inode)) { - dcs = BTRFS_DC_SETUP; - goto out_put; - } - - if (i_size_read(inode) > 0) { - ret = btrfs_check_trunc_cache_free_space(fs_info, - &fs_info->global_block_rsv); - if (ret) - goto out_put; - - ret = btrfs_truncate_free_space_cache(trans, NULL, inode); - if (ret) - goto out_put; - } - - spin_lock(&block_group->lock); - if (block_group->cached != BTRFS_CACHE_FINISHED || - !btrfs_test_opt(fs_info, SPACE_CACHE)) { - /* - * don't bother trying to write stuff out _if_ - * a) we're not cached, - * b) we're with nospace_cache mount option, - * c) we're with v2 space_cache (FREE_SPACE_TREE). - */ - dcs = BTRFS_DC_WRITTEN; - spin_unlock(&block_group->lock); - goto out_put; - } - spin_unlock(&block_group->lock); - - /* - * We hit an ENOSPC when setting up the cache in this transaction, just - * skip doing the setup, we've already cleared the cache so we're safe. - */ - if (test_bit(BTRFS_TRANS_CACHE_ENOSPC, &trans->transaction->flags)) - goto out_put; - - /* - * Try to preallocate enough space based on how big the block group is. - * Keep in mind this has to include any pinned space which could end up - * taking up quite a bit since it's not folded into the other space - * cache. - */ - cache_size = div_u64(block_group->length, SZ_256M); - if (!cache_size) - cache_size = 1; - - cache_size *= 16; - cache_size *= fs_info->sectorsize; - - ret = btrfs_check_data_free_space(BTRFS_I(inode), &data_reserved, 0, - cache_size, false); - if (ret) - goto out_put; - - ret = btrfs_prealloc_file_range_trans(inode, trans, 0, 0, cache_size, - cache_size, cache_size, - &alloc_hint); - /* - * Our cache requires contiguous chunks so that we don't modify a bunch - * of metadata or split extents when writing the cache out, which means - * we can enospc if we are heavily fragmented in addition to just normal - * out of space conditions. So if we hit this just skip setting up any - * other block groups for this transaction, maybe we'll unpin enough - * space the next time around. - */ - if (!ret) - dcs = BTRFS_DC_SETUP; - else if (ret == -ENOSPC) - set_bit(BTRFS_TRANS_CACHE_ENOSPC, &trans->transaction->flags); - -out_put: - iput(inode); -out_free: - btrfs_release_path(path); -out: - spin_lock(&block_group->lock); - if (!ret && dcs == BTRFS_DC_SETUP) - block_group->cache_generation = trans->transid; - block_group->disk_cache_state = dcs; - spin_unlock(&block_group->lock); - - extent_changeset_free(data_reserved); -} - -int btrfs_setup_space_cache(struct btrfs_trans_handle *trans) -{ - struct btrfs_fs_info *fs_info = trans->fs_info; - struct btrfs_block_group *cache, *tmp; - struct btrfs_transaction *cur_trans = trans->transaction; - BTRFS_PATH_AUTO_FREE(path); - - if (list_empty(&cur_trans->dirty_bgs) || - !btrfs_test_opt(fs_info, SPACE_CACHE)) - return 0; - - path = btrfs_alloc_path(); - if (!path) - return -ENOMEM; - - /* Could add new block groups, use _safe just in case */ - list_for_each_entry_safe(cache, tmp, &cur_trans->dirty_bgs, - dirty_list) { - if (cache->disk_cache_state == BTRFS_DC_CLEAR) - cache_save_setup(cache, trans, path); - } - - return 0; -} - /* * Transaction commit does final block group cache writeback during a critical * section where nothing is allowed to change the FS. This is required in @@ -3622,10 +3412,8 @@ int btrfs_start_dirty_block_groups(struct btrfs_trans_handle *trans) struct btrfs_block_group *cache; struct btrfs_transaction *cur_trans = trans->transaction; int ret = 0; - int should_put; BTRFS_PATH_AUTO_FREE(path); LIST_HEAD(dirty); - struct list_head *io = &cur_trans->io_bgs; int loops = 0; spin_lock(&cur_trans->dirty_bgs_lock); @@ -3651,7 +3439,7 @@ int btrfs_start_dirty_block_groups(struct btrfs_trans_handle *trans) /* * cache_write_mutex is here only to save us from balance or automatic * removal of empty block groups deleting this block group while we are - * writing out the cache + * updating its item */ mutex_lock(&trans->transaction->cache_write_mutex); while (!list_empty(&dirty)) { @@ -3659,23 +3447,8 @@ int btrfs_start_dirty_block_groups(struct btrfs_trans_handle *trans) cache = list_first_entry(&dirty, struct btrfs_block_group, dirty_list); - /* - * This can happen if something re-dirties a block group that - * is already under IO. Just wait for it to finish and then do - * it all again - */ - if (!list_empty(&cache->io_list)) { - list_del_init(&cache->io_list); - btrfs_wait_cache_io(trans, cache, path); - btrfs_put_block_group(cache); - } - /* - * btrfs_wait_cache_io uses the cache->dirty_list to decide if - * it should update the cache_state. Don't delete until after - * we wait. - * * Since we're not running in the commit critical section * we need the dirty_bgs_lock to protect from update_block_group */ @@ -3683,66 +3456,33 @@ int btrfs_start_dirty_block_groups(struct btrfs_trans_handle *trans) list_del_init(&cache->dirty_list); spin_unlock(&cur_trans->dirty_bgs_lock); - should_put = 1; - - cache_save_setup(cache, trans, path); - - if (cache->disk_cache_state == BTRFS_DC_SETUP) { - cache->io_ctl.inode = NULL; - ret = btrfs_write_out_cache(trans, cache, path); - if (ret == 0 && cache->io_ctl.inode) { - should_put = 0; - - /* - * The cache_write_mutex is protecting the - * io_list, also refer to the definition of - * btrfs_transaction::io_bgs for more details - */ - list_add_tail(&cache->io_list, io); - } else { - /* - * If we failed to write the cache, the - * generation will be bad and life goes on - */ - ret = 0; - } - } - if (!ret) { - ret = update_block_group_item(trans, path, cache); - /* - * Our block group might still be attached to the list - * of new block groups in the transaction handle of some - * other task (struct btrfs_trans_handle->new_bgs). This - * means its block group item isn't yet in the extent - * tree. If this happens ignore the error, as we will - * try again later in the critical section of the - * transaction commit. - */ - if (ret == -ENOENT) { - ret = 0; - spin_lock(&cur_trans->dirty_bgs_lock); - if (list_empty(&cache->dirty_list)) { - list_add_tail(&cache->dirty_list, - &cur_trans->dirty_bgs); - btrfs_get_block_group(cache); - drop_reserve = false; - } - spin_unlock(&cur_trans->dirty_bgs_lock); - } else if (ret) { - btrfs_abort_transaction(trans, ret); + ret = update_block_group_item(trans, path, cache); + /* + * Our block group might still be attached to the list of new + * block groups in the transaction handle of some other task + * (struct btrfs_trans_handle->new_bgs). This means its block + * group item isn't yet in the extent tree. If this happens + * ignore the error, as we will try again later in the critical + * section of the transaction commit. + */ + if (ret == -ENOENT) { + ret = 0; + spin_lock(&cur_trans->dirty_bgs_lock); + if (list_empty(&cache->dirty_list)) { + list_add_tail(&cache->dirty_list, + &cur_trans->dirty_bgs); + btrfs_get_block_group(cache); + drop_reserve = false; } + spin_unlock(&cur_trans->dirty_bgs_lock); + } else if (ret) { + btrfs_abort_transaction(trans, ret); } - /* If it's not on the io list, we need to put the block group */ - if (should_put) - btrfs_put_block_group(cache); + btrfs_put_block_group(cache); if (drop_reserve) btrfs_dec_delayed_refs_rsv_bg_updates(fs_info); - /* - * Avoid blocking other tasks for too long. It might even save - * us from writing caches for block groups that are going to be - * removed. - */ + /* Avoid blocking other tasks for too long. */ mutex_unlock(&trans->transaction->cache_write_mutex); if (ret) goto out; @@ -3787,121 +3527,34 @@ int btrfs_write_dirty_block_groups(struct btrfs_trans_handle *trans) struct btrfs_block_group *cache; struct btrfs_transaction *cur_trans = trans->transaction; int ret = 0; - int should_put; BTRFS_PATH_AUTO_FREE(path); - struct list_head *io = &cur_trans->io_bgs; path = btrfs_alloc_path(); if (!path) return -ENOMEM; - /* - * Even though we are in the critical section of the transaction commit, - * we can still have concurrent tasks adding elements to this - * transaction's list of dirty block groups. These tasks correspond to - * endio free space workers started when writeback finishes for a - * space cache, which run inode.c:btrfs_finish_ordered_io(), and can - * allocate new block groups as a result of COWing nodes of the root - * tree when updating the free space inode. The writeback for the space - * caches is triggered by an earlier call to - * btrfs_start_dirty_block_groups() and iterations of the following - * loop. - * Also we want to do the cache_save_setup first and then run the - * delayed refs to make sure we have the best chance at doing this all - * in one shot. - */ spin_lock(&cur_trans->dirty_bgs_lock); while (!list_empty(&cur_trans->dirty_bgs)) { cache = list_first_entry(&cur_trans->dirty_bgs, struct btrfs_block_group, dirty_list); - - /* - * This can happen if cache_save_setup re-dirties a block group - * that is already under IO. Just wait for it to finish and - * then do it all again - */ - if (!list_empty(&cache->io_list)) { - spin_unlock(&cur_trans->dirty_bgs_lock); - list_del_init(&cache->io_list); - btrfs_wait_cache_io(trans, cache, path); - btrfs_put_block_group(cache); - spin_lock(&cur_trans->dirty_bgs_lock); - } - - /* - * Don't remove from the dirty list until after we've waited on - * any pending IO - */ list_del_init(&cache->dirty_list); spin_unlock(&cur_trans->dirty_bgs_lock); - should_put = 1; - - cache_save_setup(cache, trans, path); if (!ret) ret = btrfs_run_delayed_refs(trans, U64_MAX); - - if (!ret && cache->disk_cache_state == BTRFS_DC_SETUP) { - cache->io_ctl.inode = NULL; - ret = btrfs_write_out_cache(trans, cache, path); - if (ret == 0 && cache->io_ctl.inode) { - should_put = 0; - list_add_tail(&cache->io_list, io); - } else { - /* - * If we failed to write the cache, the - * generation will be bad and life goes on - */ - ret = 0; - } - } if (!ret) { ret = update_block_group_item(trans, path, cache); - /* - * One of the free space endio workers might have - * created a new block group while updating a free space - * cache's inode (at inode.c:btrfs_finish_ordered_io()) - * and hasn't released its transaction handle yet, in - * which case the new block group is still attached to - * its transaction handle and its creation has not - * finished yet (no block group item in the extent tree - * yet, etc). If this is the case, wait for all free - * space endio workers to finish and retry. This is a - * very rare case so no need for a more efficient and - * complex approach. - */ - if (ret == -ENOENT) { - wait_event(cur_trans->writer_wait, - atomic_read(&cur_trans->num_writers) == 1); - ret = update_block_group_item(trans, path, cache); - if (ret) - btrfs_abort_transaction(trans, ret); - } else if (ret) { + if (ret) btrfs_abort_transaction(trans, ret); - } } - /* If its not on the io list, we need to put the block group */ - if (should_put) - btrfs_put_block_group(cache); + btrfs_put_block_group(cache); btrfs_dec_delayed_refs_rsv_bg_updates(fs_info); spin_lock(&cur_trans->dirty_bgs_lock); } spin_unlock(&cur_trans->dirty_bgs_lock); - /* - * Refer to the definition of io_bgs member for details why it's safe - * to use it without any locking - */ - while (!list_empty(io)) { - cache = list_first_entry(io, struct btrfs_block_group, - io_list); - list_del_init(&cache->io_list); - btrfs_wait_cache_io(trans, cache, path); - btrfs_put_block_group(cache); - } - return ret; } diff --git a/fs/btrfs/block-group.h b/fs/btrfs/block-group.h index b349f94cf929ab..d69432b236ec02 100644 --- a/fs/btrfs/block-group.h +++ b/fs/btrfs/block-group.h @@ -368,7 +368,6 @@ int btrfs_inc_block_group_ro(struct btrfs_block_group *cache, void btrfs_dec_block_group_ro(struct btrfs_block_group *cache); int btrfs_start_dirty_block_groups(struct btrfs_trans_handle *trans); int btrfs_write_dirty_block_groups(struct btrfs_trans_handle *trans); -int btrfs_setup_space_cache(struct btrfs_trans_handle *trans); int btrfs_update_block_group(struct btrfs_trans_handle *trans, u64 bytenr, u64 num_bytes, bool alloc); int btrfs_add_reserved_bytes(struct btrfs_block_group *cache, diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index 90cf0647c47e36..841a28d234561e 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -4871,26 +4871,6 @@ static void btrfs_destroy_pinned_extent(struct btrfs_fs_info *fs_info, } } -static void btrfs_cleanup_bg_io(struct btrfs_block_group *cache) -{ - struct inode *inode; - - inode = cache->io_ctl.inode; - if (inode) { - unsigned int nofs_flag; - - nofs_flag = memalloc_nofs_save(); - invalidate_inode_pages2(inode->i_mapping); - memalloc_nofs_restore(nofs_flag); - - BTRFS_I(inode)->generation = 0; - cache->io_ctl.inode = NULL; - iput(inode); - } - ASSERT(cache->io_ctl.pages == NULL); - btrfs_put_block_group(cache); -} - void btrfs_cleanup_dirty_bgs(struct btrfs_transaction *cur_trans, struct btrfs_fs_info *fs_info) { @@ -4902,13 +4882,6 @@ void btrfs_cleanup_dirty_bgs(struct btrfs_transaction *cur_trans, struct btrfs_block_group, dirty_list); - if (!list_empty(&cache->io_list)) { - spin_unlock(&cur_trans->dirty_bgs_lock); - list_del_init(&cache->io_list); - btrfs_cleanup_bg_io(cache); - spin_lock(&cur_trans->dirty_bgs_lock); - } - list_del_init(&cache->dirty_list); spin_lock(&cache->lock); cache->disk_cache_state = BTRFS_DC_ERROR; @@ -4920,22 +4893,6 @@ void btrfs_cleanup_dirty_bgs(struct btrfs_transaction *cur_trans, spin_lock(&cur_trans->dirty_bgs_lock); } spin_unlock(&cur_trans->dirty_bgs_lock); - - /* - * Refer to the definition of io_bgs member for details why it's safe - * to use it without any locking - */ - while (!list_empty(&cur_trans->io_bgs)) { - cache = list_first_entry(&cur_trans->io_bgs, - struct btrfs_block_group, - io_list); - - list_del_init(&cache->io_list); - spin_lock(&cache->lock); - cache->disk_cache_state = BTRFS_DC_ERROR; - spin_unlock(&cache->lock); - btrfs_cleanup_bg_io(cache); - } } static void btrfs_free_all_qgroup_pertrans(struct btrfs_fs_info *fs_info) @@ -4971,7 +4928,6 @@ void btrfs_cleanup_one_transaction(struct btrfs_transaction *cur_trans) btrfs_cleanup_dirty_bgs(cur_trans, fs_info); ASSERT(list_empty(&cur_trans->dirty_bgs)); - ASSERT(list_empty(&cur_trans->io_bgs)); list_for_each_entry_safe(dev, tmp, &cur_trans->dev_update_list, post_commit_list) { diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c index cc71ad8f5e82eb..4c7f6b5a625c54 100644 --- a/fs/btrfs/transaction.c +++ b/fs/btrfs/transaction.c @@ -411,7 +411,6 @@ static noinline int join_transaction(struct btrfs_fs_info *fs_info, INIT_LIST_HEAD(&cur_trans->dev_update_list); INIT_LIST_HEAD(&cur_trans->switch_commits); INIT_LIST_HEAD(&cur_trans->dirty_bgs); - INIT_LIST_HEAD(&cur_trans->io_bgs); INIT_LIST_HEAD(&cur_trans->dropped_roots); mutex_init(&cur_trans->cache_write_mutex); spin_lock_init(&cur_trans->dirty_bgs_lock); @@ -1404,7 +1403,6 @@ static noinline int commit_cowonly_roots(struct btrfs_trans_handle *trans) { struct btrfs_fs_info *fs_info = trans->fs_info; struct list_head *dirty_bgs = &trans->transaction->dirty_bgs; - struct list_head *io_bgs = &trans->transaction->io_bgs; struct extent_buffer *eb; int ret; @@ -1434,10 +1432,6 @@ static noinline int commit_cowonly_roots(struct btrfs_trans_handle *trans) if (ret) return ret; - ret = btrfs_setup_space_cache(trans); - if (ret) - return ret; - again: while (!list_empty(&fs_info->dirty_cowonly_roots)) { struct btrfs_root *root; @@ -1458,7 +1452,7 @@ static noinline int commit_cowonly_roots(struct btrfs_trans_handle *trans) if (ret) return ret; - while (!list_empty(dirty_bgs) || !list_empty(io_bgs)) { + while (!list_empty(dirty_bgs)) { ret = btrfs_write_dirty_block_groups(trans); if (ret) return ret; @@ -2583,7 +2577,6 @@ int btrfs_commit_transaction(struct btrfs_trans_handle *trans) switch_commit_roots(trans); ASSERT(list_empty(&cur_trans->dirty_bgs)); - ASSERT(list_empty(&cur_trans->io_bgs)); update_super_roots(fs_info); btrfs_set_super_log_root(fs_info->super_copy, 0); diff --git a/fs/btrfs/transaction.h b/fs/btrfs/transaction.h index 0b2e69293b3b90..d4bbe44f955611 100644 --- a/fs/btrfs/transaction.h +++ b/fs/btrfs/transaction.h @@ -47,7 +47,6 @@ enum btrfs_trans_state { #define BTRFS_TRANS_HAVE_FREE_BGS 0 #define BTRFS_TRANS_DIRTY_BG_RUN 1 -#define BTRFS_TRANS_CACHE_ENOSPC 2 struct btrfs_transaction { u64 transid; @@ -78,23 +77,6 @@ struct btrfs_transaction { struct list_head dev_update_list; struct list_head switch_commits; struct list_head dirty_bgs; - - /* - * There is no explicit lock which protects io_bgs, rather its - * consistency is implied by the fact that all the sites which modify - * it do so under some form of transaction critical section, namely: - * - * - btrfs_start_dirty_block_groups - This function can only ever be - * run by one of the transaction committers. Refer to - * BTRFS_TRANS_DIRTY_BG_RUN usage in btrfs_commit_transaction - * - * - btrfs_write_dirty_blockgroups - this is called by - * commit_cowonly_roots from transaction critical section - * (TRANS_STATE_COMMIT_DOING) - * - * - btrfs_cleanup_dirty_bgs - called on transaction abort - */ - struct list_head io_bgs; struct list_head dropped_roots; struct extent_io_tree pinned_extents; From abdda43999227aea9a3cde8dd21e9ebc5d4bfe01 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Wed, 16 Sep 2026 23:59:59 -0400 Subject: [PATCH 0465/1352] btrfs: remove the free space cache endio workqueue Free space inodes are no longer written to, so nothing queues ordered extent completion on endio_freespace_worker. Remove it and always use endio_write_workers in btrfs_queue_ordered_fn(). Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/disk-io.c | 14 +++----------- fs/btrfs/fs.h | 1 - fs/btrfs/ordered-data.c | 7 ++----- fs/btrfs/super.c | 1 - 4 files changed, 5 insertions(+), 18 deletions(-) diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index 841a28d234561e..5d362fb5d4dd98 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -1780,7 +1780,6 @@ static void btrfs_stop_all_workers(struct btrfs_fs_info *fs_info) if (fs_info->rmw_workers) destroy_workqueue(fs_info->rmw_workers); btrfs_destroy_workqueue(fs_info->endio_write_workers); - btrfs_destroy_workqueue(fs_info->endio_freespace_worker); btrfs_destroy_workqueue(fs_info->delayed_workers); btrfs_destroy_workqueue(fs_info->caching_workers); btrfs_destroy_workqueue(fs_info->flush_workers); @@ -1991,9 +1990,6 @@ static int btrfs_init_workqueues(struct btrfs_fs_info *fs_info) fs_info->endio_write_workers = btrfs_alloc_workqueue(fs_info, "endio-write", flags, max_active, 2); - fs_info->endio_freespace_worker = - btrfs_alloc_workqueue(fs_info, "freespace-write", flags, - max_active, 0); fs_info->delayed_workers = btrfs_alloc_workqueue(fs_info, "delayed-meta", flags, max_active, 0); @@ -2006,8 +2002,7 @@ static int btrfs_init_workqueues(struct btrfs_fs_info *fs_info) if (!(fs_info->workers && fs_info->delalloc_workers && fs_info->flush_workers && fs_info->endio_workers && fs_info->endio_meta_workers && - fs_info->endio_write_workers && - fs_info->endio_freespace_worker && fs_info->rmw_workers && + fs_info->endio_write_workers && fs_info->rmw_workers && fs_info->caching_workers && fs_info->fixup_workers && fs_info->delayed_workers && fs_info->qgroup_rescan_workers && fs_info->discard_ctl.discard_workers)) { @@ -4453,9 +4448,8 @@ void __cold close_ctree(struct btrfs_fs_info *fs_info) * to finish an ordered extent - end_bbio_compressed_write() * calls btrfs_finish_ordered_extent() which in turns does a call to * btrfs_queue_ordered_fn(), and that queues the ordered extent - * completion either in the endio_write_workers work queue or in the - * fs_info->endio_freespace_worker work queue. We flush those queues - * below, so before we flush them we must flush this queue for the + * completion in the endio_write_workers work queue. We flush that + * queue below, so before we flush it we must flush this queue for the * workers of compressed writes. */ flush_workqueue(fs_info->endio_workers); @@ -4481,8 +4475,6 @@ void __cold close_ctree(struct btrfs_fs_info *fs_info) * btrfs_finish_ordered_io() when we are unmounting). */ btrfs_flush_workqueue(fs_info->endio_write_workers); - /* Ordered extents for free space inodes. */ - btrfs_flush_workqueue(fs_info->endio_freespace_worker); /* * Run delayed iputs in case an async reclaim worker is waiting for them * to be run as mentioned above. diff --git a/fs/btrfs/fs.h b/fs/btrfs/fs.h index 3eba8438593cde..441b315e8a9893 100644 --- a/fs/btrfs/fs.h +++ b/fs/btrfs/fs.h @@ -712,7 +712,6 @@ struct btrfs_fs_info { struct workqueue_struct *endio_meta_workers; struct workqueue_struct *rmw_workers; struct btrfs_workqueue *endio_write_workers; - struct btrfs_workqueue *endio_freespace_worker; struct btrfs_workqueue *caching_workers; struct workqueue_struct *fixup_workers; diff --git a/fs/btrfs/ordered-data.c b/fs/btrfs/ordered-data.c index b32d4eabe0abe4..e9f1cbeb555a4c 100644 --- a/fs/btrfs/ordered-data.c +++ b/fs/btrfs/ordered-data.c @@ -417,13 +417,10 @@ static bool can_finish_ordered_extent(struct btrfs_ordered_extent *ordered, static void btrfs_queue_ordered_fn(struct btrfs_ordered_extent *ordered) { - struct btrfs_inode *inode = ordered->inode; - struct btrfs_fs_info *fs_info = inode->root->fs_info; - struct btrfs_workqueue *wq = btrfs_is_free_space_inode(inode) ? - fs_info->endio_freespace_worker : fs_info->endio_write_workers; + struct btrfs_fs_info *fs_info = ordered->inode->root->fs_info; btrfs_init_work(&ordered->work, finish_ordered_fn, NULL); - btrfs_queue_work(wq, &ordered->work); + btrfs_queue_work(fs_info->endio_write_workers, &ordered->work); } void btrfs_finish_ordered_extent(struct btrfs_ordered_extent *ordered, diff --git a/fs/btrfs/super.c b/fs/btrfs/super.c index 213db18c6dfccf..c90cd377fcf316 100644 --- a/fs/btrfs/super.c +++ b/fs/btrfs/super.c @@ -1243,7 +1243,6 @@ static void btrfs_resize_thread_pool(struct btrfs_fs_info *fs_info, workqueue_set_max_active(fs_info->endio_workers, new_pool_size); workqueue_set_max_active(fs_info->endio_meta_workers, new_pool_size); btrfs_workqueue_set_max(fs_info->endio_write_workers, new_pool_size); - btrfs_workqueue_set_max(fs_info->endio_freespace_worker, new_pool_size); btrfs_workqueue_set_max(fs_info->delayed_workers, new_pool_size); } From 257d4b16d77bb2947b3a25030998bf6d49c84fcd Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:00 -0400 Subject: [PATCH 0466/1352] btrfs: remove the v1 space cache write path Nothing writes out a v1 space cache any more. Remove the writers and their io_ctl helpers, along with create_free_space_inode() and btrfs_prealloc_file_range_trans(), whose only user was the cache inode creation. The io_list and io_ctl block group fields were only used by the writers, so remove them too. btrfs_truncate_free_space_cache() only needed the block group to wait for and clear in-flight cache IO, so drop that parameter. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/block-group.c | 4 - fs/btrfs/block-group.h | 3 - fs/btrfs/btrfs_inode.h | 4 - fs/btrfs/free-space-cache.c | 688 ------------------------------------ fs/btrfs/free-space-cache.h | 10 - fs/btrfs/inode.c | 9 - fs/btrfs/relocation.c | 2 +- 7 files changed, 1 insertion(+), 719 deletions(-) diff --git a/fs/btrfs/block-group.c b/fs/btrfs/block-group.c index 059f984e053c72..4ac48e3650be4d 100644 --- a/fs/btrfs/block-group.c +++ b/fs/btrfs/block-group.c @@ -1268,7 +1268,6 @@ int btrfs_remove_block_group(struct btrfs_trans_handle *trans, spin_lock(&trans->transaction->dirty_bgs_lock); WARN_ON(!list_empty(&block_group->dirty_list)); - WARN_ON(!list_empty(&block_group->io_list)); spin_unlock(&trans->transaction->dirty_bgs_lock); btrfs_remove_free_space_cache(block_group); @@ -2452,7 +2451,6 @@ static struct btrfs_block_group *btrfs_create_block_group( INIT_LIST_HEAD(&cache->ro_list); INIT_LIST_HEAD(&cache->discard_list); INIT_LIST_HEAD(&cache->dirty_list); - INIT_LIST_HEAD(&cache->io_list); INIT_LIST_HEAD(&cache->active_bg_list); btrfs_init_free_space_ctl(cache, cache->free_space_ctl); atomic_set(&cache->frozen, 0); @@ -4337,7 +4335,6 @@ void btrfs_put_block_group_cache(struct btrfs_fs_info *info) block_group->inode = NULL; spin_unlock(&block_group->lock); - ASSERT(block_group->io_ctl.inode == NULL); iput(&inode->vfs_inode); } else { spin_unlock(&block_group->lock); @@ -4474,7 +4471,6 @@ int btrfs_free_block_groups(struct btrfs_fs_info *info) btrfs_remove_free_space_cache(block_group); ASSERT(block_group->cached != BTRFS_CACHE_STARTED); ASSERT(list_empty(&block_group->dirty_list)); - ASSERT(list_empty(&block_group->io_list)); ASSERT(list_empty(&block_group->bg_list)); ASSERT(refcount_read(&block_group->refs) == 1); ASSERT(block_group->swap_extents == 0); diff --git a/fs/btrfs/block-group.h b/fs/btrfs/block-group.h index d69432b236ec02..f7346a0170fc46 100644 --- a/fs/btrfs/block-group.h +++ b/fs/btrfs/block-group.h @@ -228,9 +228,6 @@ struct btrfs_block_group { /* For dirty block groups */ struct list_head dirty_list; - struct list_head io_list; - - struct btrfs_io_ctl io_ctl; /* * Incremented when doing extent allocations and holding a read lock diff --git a/fs/btrfs/btrfs_inode.h b/fs/btrfs/btrfs_inode.h index 03c00c99099d53..d1c014da52a35a 100644 --- a/fs/btrfs/btrfs_inode.h +++ b/fs/btrfs/btrfs_inode.h @@ -598,10 +598,6 @@ int btrfs_wait_on_delayed_iputs(struct btrfs_fs_info *fs_info); int btrfs_prealloc_file_range(struct inode *inode, int mode, u64 start, u64 num_bytes, u64 min_size, loff_t actual_len, u64 *alloc_hint); -int btrfs_prealloc_file_range_trans(struct inode *inode, - struct btrfs_trans_handle *trans, int mode, - u64 start, u64 num_bytes, u64 min_size, - loff_t actual_len, u64 *alloc_hint); int btrfs_run_delalloc_range(struct btrfs_inode *inode, struct folio *locked_folio, u64 start, u64 end, struct writeback_control *wbc); void btrfs_queue_writepage_fixup(struct btrfs_inode *inode, struct folio *folio); diff --git a/fs/btrfs/free-space-cache.c b/fs/btrfs/free-space-cache.c index e2af75a205ea53..336b546b0a9467 100644 --- a/fs/btrfs/free-space-cache.c +++ b/fs/btrfs/free-space-cache.c @@ -164,78 +164,6 @@ struct inode *lookup_free_space_inode(struct btrfs_block_group *block_group, return inode; } -static int __create_free_space_inode(struct btrfs_root *root, - struct btrfs_trans_handle *trans, - struct btrfs_path *path, - u64 ino, u64 offset) -{ - struct btrfs_key key; - struct btrfs_disk_key disk_key; - struct btrfs_free_space_header *header; - struct btrfs_inode_item *inode_item; - struct extent_buffer *leaf; - /* We inline CRCs for the free disk space cache */ - const u64 flags = BTRFS_INODE_NOCOMPRESS | BTRFS_INODE_PREALLOC | - BTRFS_INODE_NODATASUM | BTRFS_INODE_NODATACOW; - int ret; - - ret = btrfs_insert_empty_inode(trans, root, path, ino); - if (ret) - return ret; - - leaf = path->nodes[0]; - inode_item = btrfs_item_ptr(leaf, path->slots[0], - struct btrfs_inode_item); - btrfs_item_key(leaf, &disk_key, path->slots[0]); - memzero_extent_buffer(leaf, (unsigned long)inode_item, - sizeof(*inode_item)); - btrfs_set_inode_generation(leaf, inode_item, trans->transid); - btrfs_set_inode_size(leaf, inode_item, 0); - btrfs_set_inode_nbytes(leaf, inode_item, 0); - btrfs_set_inode_uid(leaf, inode_item, 0); - btrfs_set_inode_gid(leaf, inode_item, 0); - btrfs_set_inode_mode(leaf, inode_item, S_IFREG | 0600); - btrfs_set_inode_flags(leaf, inode_item, flags); - btrfs_set_inode_nlink(leaf, inode_item, 1); - btrfs_set_inode_transid(leaf, inode_item, trans->transid); - btrfs_set_inode_block_group(leaf, inode_item, offset); - btrfs_release_path(path); - - key.objectid = BTRFS_FREE_SPACE_OBJECTID; - key.type = 0; - key.offset = offset; - ret = btrfs_insert_empty_item(trans, root, path, &key, - sizeof(struct btrfs_free_space_header)); - if (ret < 0) { - btrfs_release_path(path); - return ret; - } - - leaf = path->nodes[0]; - header = btrfs_item_ptr(leaf, path->slots[0], - struct btrfs_free_space_header); - memzero_extent_buffer(leaf, (unsigned long)header, sizeof(*header)); - btrfs_set_free_space_key(leaf, header, &disk_key); - btrfs_release_path(path); - - return 0; -} - -int create_free_space_inode(struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, - struct btrfs_path *path) -{ - int ret; - u64 ino; - - ret = btrfs_get_free_objectid(trans->fs_info->tree_root, &ino); - if (ret < 0) - return ret; - - return __create_free_space_inode(trans->fs_info->tree_root, trans, path, - ino, block_group->start); -} - /* * inode is an optional sink: if it is NULL, btrfs_remove_free_space_inode * handles lookup, otherwise it takes ownership and iputs the inode. @@ -292,7 +220,6 @@ int btrfs_remove_free_space_inode(struct btrfs_trans_handle *trans, } int btrfs_truncate_free_space_cache(struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, struct inode *vfs_inode) { struct btrfs_truncate_control control = { @@ -306,33 +233,6 @@ int btrfs_truncate_free_space_cache(struct btrfs_trans_handle *trans, struct btrfs_root *root = inode->root; struct extent_state *cached_state = NULL; int ret = 0; - bool locked = false; - - if (block_group) { - BTRFS_PATH_AUTO_FREE(path); - - path = btrfs_alloc_path(); - if (!path) { - ret = -ENOMEM; - goto fail; - } - locked = true; - mutex_lock(&trans->transaction->cache_write_mutex); - if (!list_empty(&block_group->io_list)) { - list_del_init(&block_group->io_list); - - btrfs_wait_cache_io(trans, block_group, path); - btrfs_put_block_group(block_group); - } - - /* - * now that we've truncated the cache away, its no longer - * setup or written - */ - spin_lock(&block_group->lock); - block_group->disk_cache_state = BTRFS_DC_CLEAR; - spin_unlock(&block_group->lock); - } btrfs_i_size_write(inode, 0); truncate_pagecache(vfs_inode, 0); @@ -356,8 +256,6 @@ int btrfs_truncate_free_space_cache(struct btrfs_trans_handle *trans, ret = btrfs_update_inode(trans, inode); fail: - if (locked) - mutex_unlock(&trans->transaction->cache_write_mutex); if (ret) btrfs_abort_transaction(trans, ret); @@ -490,21 +388,6 @@ static int io_ctl_prepare_pages(struct btrfs_io_ctl *io_ctl, bool uptodate) return 0; } -static void io_ctl_set_generation(struct btrfs_io_ctl *io_ctl, u64 generation) -{ - io_ctl_map_page(io_ctl, 1); - - /* - * Skip the csum areas. If we don't check crcs then we just have a - * 64bit chunk at the front of the first page. - */ - io_ctl->cur += (sizeof(u32) * io_ctl->num_pages); - io_ctl->size -= sizeof(u64) + (sizeof(u32) * io_ctl->num_pages); - - put_unaligned_le64(generation, io_ctl->cur); - io_ctl->cur += sizeof(u64); -} - static int io_ctl_check_generation(struct btrfs_io_ctl *io_ctl, u64 generation) { u64 cache_gen; @@ -528,23 +411,6 @@ static int io_ctl_check_generation(struct btrfs_io_ctl *io_ctl, u64 generation) return 0; } -static void io_ctl_set_crc(struct btrfs_io_ctl *io_ctl, int index) -{ - u32 *tmp; - u32 crc = ~(u32)0; - unsigned offset = 0; - - if (index == 0) - offset = sizeof(u32) * io_ctl->num_pages; - - crc = crc32c(crc, io_ctl->orig + offset, PAGE_SIZE - offset); - btrfs_crc32c_final(crc, (u8 *)&crc); - io_ctl_unmap_page(io_ctl); - tmp = page_address(io_ctl->pages[0]); - tmp += index; - *tmp = crc; -} - static int io_ctl_check_crc(struct btrfs_io_ctl *io_ctl, int index) { u32 *tmp, val; @@ -574,76 +440,6 @@ static int io_ctl_check_crc(struct btrfs_io_ctl *io_ctl, int index) return 0; } -static int io_ctl_add_entry(struct btrfs_io_ctl *io_ctl, u64 offset, u64 bytes, - void *bitmap) -{ - struct btrfs_free_space_entry *entry; - - if (!io_ctl->cur) - return -ENOSPC; - - entry = io_ctl->cur; - put_unaligned_le64(offset, &entry->offset); - put_unaligned_le64(bytes, &entry->bytes); - entry->type = (bitmap) ? BTRFS_FREE_SPACE_BITMAP : - BTRFS_FREE_SPACE_EXTENT; - io_ctl->cur += sizeof(struct btrfs_free_space_entry); - io_ctl->size -= sizeof(struct btrfs_free_space_entry); - - if (io_ctl->size >= sizeof(struct btrfs_free_space_entry)) - return 0; - - io_ctl_set_crc(io_ctl, io_ctl->index - 1); - - /* No more pages to map */ - if (io_ctl->index >= io_ctl->num_pages) - return 0; - - /* map the next page */ - io_ctl_map_page(io_ctl, 1); - return 0; -} - -static int io_ctl_add_bitmap(struct btrfs_io_ctl *io_ctl, void *bitmap) -{ - if (!io_ctl->cur) - return -ENOSPC; - - /* - * If we aren't at the start of the current page, unmap this one and - * map the next one if there is any left. - */ - if (io_ctl->cur != io_ctl->orig) { - io_ctl_set_crc(io_ctl, io_ctl->index - 1); - if (io_ctl->index >= io_ctl->num_pages) - return -ENOSPC; - io_ctl_map_page(io_ctl, 0); - } - - copy_page(io_ctl->cur, bitmap); - io_ctl_set_crc(io_ctl, io_ctl->index - 1); - if (io_ctl->index < io_ctl->num_pages) - io_ctl_map_page(io_ctl, 0); - return 0; -} - -static void io_ctl_zero_remaining_pages(struct btrfs_io_ctl *io_ctl) -{ - /* - * If we're not on the boundary we know we've modified the page and we - * need to crc the page. - */ - if (io_ctl->cur != io_ctl->orig) - io_ctl_set_crc(io_ctl, io_ctl->index - 1); - else - io_ctl_unmap_page(io_ctl); - - while (io_ctl->index < io_ctl->num_pages) { - io_ctl_map_page(io_ctl, 1); - io_ctl_set_crc(io_ctl, io_ctl->index - 1); - } -} - static int io_ctl_read_entry(struct btrfs_io_ctl *io_ctl, struct btrfs_free_space *entry, u8 *type) { @@ -1065,490 +861,6 @@ int load_free_space_cache(struct btrfs_block_group *block_group) return ret; } -static noinline_for_stack -int write_cache_extent_entries(struct btrfs_io_ctl *io_ctl, - struct btrfs_block_group *block_group, - int *entries, int *bitmaps, - struct list_head *bitmap_list) -{ - int ret; - struct btrfs_free_space_ctl *ctl = block_group->free_space_ctl; - struct btrfs_free_cluster *cluster = NULL; - struct btrfs_free_cluster *cluster_locked = NULL; - struct rb_node *node = rb_first(&ctl->free_space_offset); - struct btrfs_trim_range *trim_entry; - - /* Get the cluster for this block_group if it exists */ - if (!list_empty(&block_group->cluster_list)) { - cluster = list_first_entry(&block_group->cluster_list, - struct btrfs_free_cluster, block_group_list); - } - - if (!node && cluster) { - cluster_locked = cluster; - spin_lock(&cluster_locked->lock); - node = rb_first(&cluster->root); - cluster = NULL; - } - - /* Write out the extent entries */ - while (node) { - struct btrfs_free_space *e; - - e = rb_entry(node, struct btrfs_free_space, offset_index); - *entries += 1; - - ret = io_ctl_add_entry(io_ctl, e->offset, e->bytes, - e->bitmap); - if (ret) - goto fail; - - if (e->bitmap) { - list_add_tail(&e->list, bitmap_list); - *bitmaps += 1; - } - node = rb_next(node); - if (!node && cluster) { - node = rb_first(&cluster->root); - cluster_locked = cluster; - spin_lock(&cluster_locked->lock); - cluster = NULL; - } - } - if (cluster_locked) { - spin_unlock(&cluster_locked->lock); - cluster_locked = NULL; - } - - /* - * Make sure we don't miss any range that was removed from our rbtree - * because trimming is running. Otherwise after a umount+mount (or crash - * after committing the transaction) we would leak free space and get - * an inconsistent free space cache report from fsck. - */ - list_for_each_entry(trim_entry, &ctl->trimming_ranges, list) { - ret = io_ctl_add_entry(io_ctl, trim_entry->start, - trim_entry->bytes, NULL); - if (ret) - goto fail; - *entries += 1; - } - - return 0; -fail: - if (cluster_locked) - spin_unlock(&cluster_locked->lock); - return -ENOSPC; -} - -static noinline_for_stack int -update_cache_item(struct btrfs_trans_handle *trans, - struct btrfs_root *root, - struct inode *inode, - struct btrfs_path *path, u64 offset, - int entries, int bitmaps) -{ - struct btrfs_key key; - struct btrfs_free_space_header *header; - struct extent_buffer *leaf; - int ret; - - key.objectid = BTRFS_FREE_SPACE_OBJECTID; - key.type = 0; - key.offset = offset; - - ret = btrfs_search_slot(trans, root, &key, path, 0, 1); - if (ret < 0) { - btrfs_clear_extent_bit(&BTRFS_I(inode)->io_tree, 0, inode->i_size - 1, - EXTENT_DELALLOC, NULL); - return ret; - } - leaf = path->nodes[0]; - if (ret > 0) { - struct btrfs_key found_key; - ASSERT(path->slots[0]); - path->slots[0]--; - btrfs_item_key_to_cpu(leaf, &found_key, path->slots[0]); - if (found_key.objectid != BTRFS_FREE_SPACE_OBJECTID || - found_key.offset != offset) { - btrfs_clear_extent_bit(&BTRFS_I(inode)->io_tree, 0, - inode->i_size - 1, EXTENT_DELALLOC, - NULL); - btrfs_release_path(path); - return -ENOENT; - } - } - - BTRFS_I(inode)->generation = trans->transid; - header = btrfs_item_ptr(leaf, path->slots[0], - struct btrfs_free_space_header); - btrfs_set_free_space_entries(leaf, header, entries); - btrfs_set_free_space_bitmaps(leaf, header, bitmaps); - btrfs_set_free_space_generation(leaf, header, trans->transid); - btrfs_release_path(path); - - return 0; -} - -static noinline_for_stack int write_pinned_extent_entries( - struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, - struct btrfs_io_ctl *io_ctl, - int *entries) -{ - u64 start, extent_start, extent_end, len; - const u64 block_group_end = btrfs_block_group_end(block_group); - struct extent_io_tree *unpin = NULL; - int ret; - - /* - * We want to add any pinned extents to our free space cache - * so we don't leak the space - * - * We shouldn't have switched the pinned extents yet so this is the - * right one - */ - unpin = &trans->transaction->pinned_extents; - - start = block_group->start; - - while (start < block_group_end) { - if (!btrfs_find_first_extent_bit(unpin, start, - &extent_start, &extent_end, - EXTENT_DIRTY, NULL)) - return 0; - - /* This pinned extent is out of our range */ - if (extent_start >= block_group_end) - return 0; - - extent_start = max(extent_start, start); - extent_end = min(block_group_end, extent_end + 1); - len = extent_end - extent_start; - - *entries += 1; - ret = io_ctl_add_entry(io_ctl, extent_start, len, NULL); - if (ret) - return -ENOSPC; - - start = extent_end; - } - - return 0; -} - -static noinline_for_stack int -write_bitmap_entries(struct btrfs_io_ctl *io_ctl, struct list_head *bitmap_list) -{ - struct btrfs_free_space *entry, *next; - int ret; - - /* Write out the bitmaps */ - list_for_each_entry_safe(entry, next, bitmap_list, list) { - ret = io_ctl_add_bitmap(io_ctl, entry->bitmap); - if (ret) - return -ENOSPC; - list_del_init(&entry->list); - } - - return 0; -} - -static int flush_dirty_cache(struct inode *inode) -{ - int ret; - - ret = btrfs_wait_ordered_range(BTRFS_I(inode), 0, (u64)-1); - if (ret) - btrfs_clear_extent_bit(&BTRFS_I(inode)->io_tree, 0, inode->i_size - 1, - EXTENT_DELALLOC, NULL); - - return ret; -} - -static void noinline_for_stack -cleanup_bitmap_list(struct list_head *bitmap_list) -{ - struct btrfs_free_space *entry, *next; - - list_for_each_entry_safe(entry, next, bitmap_list, list) - list_del_init(&entry->list); -} - -static void noinline_for_stack -cleanup_write_cache_enospc(struct inode *inode, - struct btrfs_io_ctl *io_ctl, - struct extent_state **cached_state) -{ - io_ctl_drop_pages(io_ctl); - btrfs_unlock_extent(&BTRFS_I(inode)->io_tree, 0, i_size_read(inode) - 1, - cached_state); -} - -static int __btrfs_wait_cache_io(struct btrfs_root *root, - struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, - struct btrfs_io_ctl *io_ctl, - struct btrfs_path *path, u64 offset) -{ - int ret; - struct inode *inode = io_ctl->inode; - - if (!inode) - return 0; - - /* Flush the dirty pages in the cache file. */ - ret = flush_dirty_cache(inode); - if (ret) - goto out; - - /* Update the cache item to tell everyone this cache file is valid. */ - ret = update_cache_item(trans, root, inode, path, offset, - io_ctl->entries, io_ctl->bitmaps); -out: - if (ret) { - invalidate_inode_pages2(inode->i_mapping); - BTRFS_I(inode)->generation = 0; - if (block_group) - btrfs_debug(root->fs_info, - "failed to write free space cache for block group %llu error %d", - block_group->start, ret); - } - btrfs_update_inode(trans, BTRFS_I(inode)); - - if (block_group) { - /* the dirty list is protected by the dirty_bgs_lock */ - spin_lock(&trans->transaction->dirty_bgs_lock); - - /* the disk_cache_state is protected by the block group lock */ - spin_lock(&block_group->lock); - - /* - * only mark this as written if we didn't get put back on - * the dirty list while waiting for IO. Otherwise our - * cache state won't be right, and we won't get written again - */ - if (!ret && list_empty(&block_group->dirty_list)) - block_group->disk_cache_state = BTRFS_DC_WRITTEN; - else if (ret) - block_group->disk_cache_state = BTRFS_DC_ERROR; - - spin_unlock(&block_group->lock); - spin_unlock(&trans->transaction->dirty_bgs_lock); - io_ctl->inode = NULL; - iput(inode); - } - - return ret; - -} - -int btrfs_wait_cache_io(struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, - struct btrfs_path *path) -{ - return __btrfs_wait_cache_io(block_group->fs_info->tree_root, trans, - block_group, &block_group->io_ctl, - path, block_group->start); -} - -/* - * Write out cached info to an inode. - * - * @inode: freespace inode we are writing out - * @ctl: free space cache we are going to write out - * @block_group: block_group for this cache if it belongs to a block_group - * @io_ctl: holds context for the io - * @trans: the trans handle - * - * This function writes out a free space cache struct to disk for quick recovery - * on mount. This will return 0 if it was successful in writing the cache out, - * or an errno if it was not. - */ -static int __btrfs_write_out_cache(struct inode *inode, - struct btrfs_block_group *block_group, - struct btrfs_trans_handle *trans) -{ - struct btrfs_free_space_ctl *ctl = block_group->free_space_ctl; - struct btrfs_io_ctl *io_ctl = &block_group->io_ctl; - struct extent_state *cached_state = NULL; - LIST_HEAD(bitmap_list); - int entries = 0; - int bitmaps = 0; - int ret; - bool must_iput = false; - int i_size; - - if (!i_size_read(inode)) - return -EIO; - - WARN_ON(io_ctl->pages); - ret = io_ctl_init(io_ctl, inode, 1); - if (ret) - return ret; - - if (block_group->flags & BTRFS_BLOCK_GROUP_DATA) { - down_write(&block_group->data_rwsem); - spin_lock(&block_group->lock); - if (block_group->delalloc_bytes) { - block_group->disk_cache_state = BTRFS_DC_WRITTEN; - spin_unlock(&block_group->lock); - up_write(&block_group->data_rwsem); - BTRFS_I(inode)->generation = 0; - ret = 0; - must_iput = true; - goto out; - } - spin_unlock(&block_group->lock); - } - - /* Lock all pages first so we can lock the extent safely. */ - ret = io_ctl_prepare_pages(io_ctl, false); - if (ret) - goto out_unlock; - - btrfs_lock_extent(&BTRFS_I(inode)->io_tree, 0, i_size_read(inode) - 1, - &cached_state); - - io_ctl_set_generation(io_ctl, trans->transid); - - mutex_lock(&ctl->cache_writeout_mutex); - /* Write out the extent entries in the free space cache */ - spin_lock(&ctl->tree_lock); - ret = write_cache_extent_entries(io_ctl, block_group, &entries, &bitmaps, - &bitmap_list); - if (ret) - goto out_nospc_locked; - - /* - * Some spaces that are freed in the current transaction are pinned, - * they will be added into free space cache after the transaction is - * committed, we shouldn't lose them. - * - * If this changes while we are working we'll get added back to - * the dirty list and redo it. No locking needed - */ - ret = write_pinned_extent_entries(trans, block_group, io_ctl, &entries); - if (ret) - goto out_nospc_locked; - - /* - * At last, we write out all the bitmaps and keep cache_writeout_mutex - * locked while doing it because a concurrent trim can be manipulating - * or freeing the bitmap. - */ - ret = write_bitmap_entries(io_ctl, &bitmap_list); - spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); - if (ret) - goto out_nospc; - - /* Zero out the rest of the pages just to make sure */ - io_ctl_zero_remaining_pages(io_ctl); - - /* Everything is written out, now we dirty the pages in the file. */ - i_size = i_size_read(inode); - for (int i = 0; i < round_up(i_size, PAGE_SIZE) / PAGE_SIZE; i++) { - u64 dirty_start = i * PAGE_SIZE; - u64 dirty_len = min_t(u64, dirty_start + PAGE_SIZE, i_size) - dirty_start; - - ret = btrfs_dirty_folio(BTRFS_I(inode), page_folio(io_ctl->pages[i]), - dirty_start, dirty_len, &cached_state, false); - if (ret < 0) - goto out_nospc; - } - - if (block_group->flags & BTRFS_BLOCK_GROUP_DATA) - up_write(&block_group->data_rwsem); - /* - * Release the pages and unlock the extent, we will flush - * them out later - */ - io_ctl_drop_pages(io_ctl); - io_ctl_free(io_ctl); - - btrfs_unlock_extent(&BTRFS_I(inode)->io_tree, 0, i_size_read(inode) - 1, - &cached_state); - - /* - * at this point the pages are under IO and we're happy, - * The caller is responsible for waiting on them and updating - * the cache and the inode - */ - io_ctl->entries = entries; - io_ctl->bitmaps = bitmaps; - - ret = btrfs_fdatawrite_range(BTRFS_I(inode), 0, (u64)-1); - if (ret) - goto out; - - return 0; - -out_nospc_locked: - cleanup_bitmap_list(&bitmap_list); - spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); - -out_nospc: - cleanup_write_cache_enospc(inode, io_ctl, &cached_state); - -out_unlock: - if (block_group->flags & BTRFS_BLOCK_GROUP_DATA) - up_write(&block_group->data_rwsem); - -out: - io_ctl->inode = NULL; - io_ctl_free(io_ctl); - if (ret) { - invalidate_inode_pages2(inode->i_mapping); - BTRFS_I(inode)->generation = 0; - } - btrfs_update_inode(trans, BTRFS_I(inode)); - if (must_iput) - iput(inode); - return ret; -} - -int btrfs_write_out_cache(struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, - struct btrfs_path *path) -{ - struct btrfs_fs_info *fs_info = trans->fs_info; - struct inode *inode; - int ret = 0; - - spin_lock(&block_group->lock); - if (block_group->disk_cache_state < BTRFS_DC_SETUP) { - spin_unlock(&block_group->lock); - return 0; - } - spin_unlock(&block_group->lock); - - inode = lookup_free_space_inode(block_group, path); - if (IS_ERR(inode)) - return 0; - - ret = __btrfs_write_out_cache(inode, block_group, trans); - if (ret) { - btrfs_debug(fs_info, - "failed to write free space cache for block group %llu error %d", - block_group->start, ret); - spin_lock(&block_group->lock); - block_group->disk_cache_state = BTRFS_DC_ERROR; - spin_unlock(&block_group->lock); - - block_group->io_ctl.inode = NULL; - iput(inode); - } - - /* - * if ret == 0 the caller is expected to call btrfs_wait_cache_io - * to wait for IO and put the inode - */ - - return ret; -} - static inline unsigned long offset_to_bit(u64 bitmap_start, u32 unit, u64 offset) { diff --git a/fs/btrfs/free-space-cache.h b/fs/btrfs/free-space-cache.h index 53fe8e293af1d5..2432f1783f47cd 100644 --- a/fs/btrfs/free-space-cache.h +++ b/fs/btrfs/free-space-cache.h @@ -105,23 +105,13 @@ int __init btrfs_free_space_init(void); void __cold btrfs_free_space_exit(void); struct inode *lookup_free_space_inode(struct btrfs_block_group *block_group, struct btrfs_path *path); -int create_free_space_inode(struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, - struct btrfs_path *path); int btrfs_remove_free_space_inode(struct btrfs_trans_handle *trans, struct inode *inode, struct btrfs_block_group *block_group); int btrfs_truncate_free_space_cache(struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, struct inode *inode); int load_free_space_cache(struct btrfs_block_group *block_group); -int btrfs_wait_cache_io(struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, - struct btrfs_path *path); -int btrfs_write_out_cache(struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, - struct btrfs_path *path); void btrfs_init_free_space_ctl(struct btrfs_block_group *block_group, struct btrfs_free_space_ctl *ctl); diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 2a32072849cbd4..6b57383b844160 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -9388,15 +9388,6 @@ int btrfs_prealloc_file_range(struct inode *inode, int mode, NULL); } -int btrfs_prealloc_file_range_trans(struct inode *inode, - struct btrfs_trans_handle *trans, int mode, - u64 start, u64 num_bytes, u64 min_size, - loff_t actual_len, u64 *alloc_hint) -{ - return __btrfs_prealloc_file_range(inode, mode, start, num_bytes, - min_size, actual_len, alloc_hint, trans); -} - /* * NOTE: in case you are adding MAY_EXEC check for directories: * we are marking them with IOP_FASTPERM_MAY_EXEC, allowing path lookup to diff --git a/fs/btrfs/relocation.c b/fs/btrfs/relocation.c index da54db75e7a9b9..630a7ad8f8e1ac 100644 --- a/fs/btrfs/relocation.c +++ b/fs/btrfs/relocation.c @@ -3357,7 +3357,7 @@ static int delete_block_group_cache(struct btrfs_block_group *block_group, goto out; } - ret = btrfs_truncate_free_space_cache(trans, block_group, inode); + ret = btrfs_truncate_free_space_cache(trans, inode); btrfs_end_transaction(trans); btrfs_btree_balance_dirty(fs_info); From a31da8d977cf9e3f9912701366cd8ac9e3aa1d66 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:01 -0400 Subject: [PATCH 0467/1352] btrfs: rename cache_write_mutex to dirty_bgs_update_mutex The v1 space cache writeout is gone, but the mutex is still needed. It keeps btrfs_remove_block_group() from deleting a block group item while btrfs_start_dirty_block_groups() is updating it outside the commit critical section. Rename it to reflect what it protects, and update the comments around the dirty block group writeout that still refer to the space cache. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/block-group.c | 46 +++++++++++++++++++++++------------------- fs/btrfs/transaction.c | 25 +++++++++-------------- fs/btrfs/transaction.h | 8 ++++---- 3 files changed, 39 insertions(+), 40 deletions(-) diff --git a/fs/btrfs/block-group.c b/fs/btrfs/block-group.c index 4ac48e3650be4d..2af84fcc3de76b 100644 --- a/fs/btrfs/block-group.c +++ b/fs/btrfs/block-group.c @@ -1196,7 +1196,11 @@ int btrfs_remove_block_group(struct btrfs_trans_handle *trans, inode = lookup_free_space_inode(block_group, path); - mutex_lock(&trans->transaction->cache_write_mutex); + /* + * Do not delete the block group item while + * btrfs_start_dirty_block_groups() is updating it. + */ + mutex_lock(&trans->transaction->dirty_bgs_update_mutex); spin_lock(&trans->transaction->dirty_bgs_lock); if (!list_empty(&block_group->dirty_list)) { list_del_init(&block_group->dirty_list); @@ -1204,7 +1208,7 @@ int btrfs_remove_block_group(struct btrfs_trans_handle *trans, btrfs_put_block_group(block_group); } spin_unlock(&trans->transaction->dirty_bgs_lock); - mutex_unlock(&trans->transaction->cache_write_mutex); + mutex_unlock(&trans->transaction->dirty_bgs_update_mutex); ret = btrfs_remove_free_space_inode(trans, inode, block_group); if (unlikely(ret)) { @@ -3393,15 +3397,15 @@ static int update_block_group_item(struct btrfs_trans_handle *trans, } /* - * Transaction commit does final block group cache writeback during a critical + * Transaction commit does the final block group item updates during a critical * section where nothing is allowed to change the FS. This is required in - * order for the cache to actually match the block group, but can introduce a + * order for the items to actually match the block groups, but can introduce a * lot of latency into the commit. * - * So, btrfs_start_dirty_block_groups is here to kick off block group cache IO. - * There's a chance we'll have to redo some of it if the block group changes - * again during the commit, but it greatly reduces the commit latency by - * getting rid of the easy block groups while we're still allowing others to + * So, btrfs_start_dirty_block_groups is here to update the block group items + * early. There's a chance we'll have to redo some of it if the block group + * changes again during the commit, but it greatly reduces the commit latency + * by getting rid of the easy block groups while we're still allowing others to * join the commit. */ int btrfs_start_dirty_block_groups(struct btrfs_trans_handle *trans) @@ -3435,11 +3439,11 @@ int btrfs_start_dirty_block_groups(struct btrfs_trans_handle *trans) } /* - * cache_write_mutex is here only to save us from balance or automatic - * removal of empty block groups deleting this block group while we are - * updating its item + * dirty_bgs_update_mutex is here only to save us from balance or + * automatic removal of empty block groups deleting this block group + * while we are updating its item */ - mutex_lock(&trans->transaction->cache_write_mutex); + mutex_lock(&trans->transaction->dirty_bgs_update_mutex); while (!list_empty(&dirty)) { bool drop_reserve = true; @@ -3481,12 +3485,12 @@ int btrfs_start_dirty_block_groups(struct btrfs_trans_handle *trans) if (drop_reserve) btrfs_dec_delayed_refs_rsv_bg_updates(fs_info); /* Avoid blocking other tasks for too long. */ - mutex_unlock(&trans->transaction->cache_write_mutex); + mutex_unlock(&trans->transaction->dirty_bgs_update_mutex); if (ret) goto out; - mutex_lock(&trans->transaction->cache_write_mutex); + mutex_lock(&trans->transaction->dirty_bgs_update_mutex); } - mutex_unlock(&trans->transaction->cache_write_mutex); + mutex_unlock(&trans->transaction->dirty_bgs_update_mutex); /* * Go through delayed refs for all the stuff we've just kicked off @@ -3500,7 +3504,7 @@ int btrfs_start_dirty_block_groups(struct btrfs_trans_handle *trans) list_splice_init(&cur_trans->dirty_bgs, &dirty); /* * dirty_bgs_lock protects us from concurrent block group - * deletes too (not just cache_write_mutex). + * deletes too (not just dirty_bgs_update_mutex). */ if (!list_empty(&dirty)) { spin_unlock(&cur_trans->dirty_bgs_lock); @@ -3596,10 +3600,10 @@ int btrfs_update_block_group(struct btrfs_trans_handle *trans, factor = btrfs_bg_type_to_factor(cache->flags); /* - * If this block group has free space cache written out, we need to make - * sure to load it if we are removing space. This is because we need - * the unpinning stage to actually add the space back to the block group, - * otherwise we will leak space. + * Make sure the free space of this block group is loaded if we are + * removing space. This is because we need the unpinning stage to + * actually add the space back to the block group, otherwise we will + * leak space. */ if (!alloc && !btrfs_block_group_done(cache)) btrfs_cache_block_group(cache, true); @@ -3655,7 +3659,7 @@ int btrfs_update_block_group(struct btrfs_trans_handle *trans, /* * No longer have used bytes in this block group, queue it for deletion. * We do this after adding the block group to the dirty list to avoid - * races between cleaner kthread and space cache writeout. + * races between the cleaner kthread and the dirty block group writeout. */ if (!alloc && old_val == 0) { if (!btrfs_test_opt(info, DISCARD_ASYNC)) diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c index 4c7f6b5a625c54..e4eff97a780a21 100644 --- a/fs/btrfs/transaction.c +++ b/fs/btrfs/transaction.c @@ -412,7 +412,7 @@ static noinline int join_transaction(struct btrfs_fs_info *fs_info, INIT_LIST_HEAD(&cur_trans->switch_commits); INIT_LIST_HEAD(&cur_trans->dirty_bgs); INIT_LIST_HEAD(&cur_trans->dropped_roots); - mutex_init(&cur_trans->cache_write_mutex); + mutex_init(&cur_trans->dirty_bgs_update_mutex); spin_lock_init(&cur_trans->dirty_bgs_lock); INIT_LIST_HEAD(&cur_trans->deleted_bgs); spin_lock_init(&cur_trans->dropped_roots_lock); @@ -2308,18 +2308,16 @@ int btrfs_commit_transaction(struct btrfs_trans_handle *trans) if (!test_bit(BTRFS_TRANS_DIRTY_BG_RUN, &cur_trans->flags)) { bool run_it = false; - /* this mutex is also taken before trying to set - * block groups readonly. We need to make sure - * that nobody has set a block group readonly - * after a extents from that block group have been - * allocated for cache files. btrfs_set_block_group_ro - * will wait for the transaction to commit if it - * finds BTRFS_TRANS_DIRTY_BG_RUN set. + /* + * This mutex is also taken before trying to set block groups + * readonly. btrfs_inc_block_group_ro() will wait for the + * transaction to commit if it finds BTRFS_TRANS_DIRTY_BG_RUN + * set. * * The BTRFS_TRANS_DIRTY_BG_RUN flag is also used to make sure - * only one process starts all the block group IO. It wouldn't - * hurt to have more than one go through, but there's no - * real advantage to it either. + * only one process starts all the block group item updates. It + * wouldn't hurt to have more than one go through, but there's + * no real advantage to it either. */ mutex_lock(&fs_info->ro_block_group_mutex); if (!test_and_set_bit(BTRFS_TRANS_DIRTY_BG_RUN, @@ -2553,10 +2551,7 @@ int btrfs_commit_transaction(struct btrfs_trans_handle *trans) if (unlikely(ret)) goto unlock_reloc; - /* - * The tasks which save the space cache and inode cache may also - * update ->aborted, check it. - */ + /* Other tasks may also have updated ->aborted, check it. */ if (TRANS_ABORTED(cur_trans)) { ret = cur_trans->aborted; goto unlock_reloc; diff --git a/fs/btrfs/transaction.h b/fs/btrfs/transaction.h index d4bbe44f955611..aa57c64a955dcf 100644 --- a/fs/btrfs/transaction.h +++ b/fs/btrfs/transaction.h @@ -81,11 +81,11 @@ struct btrfs_transaction { struct extent_io_tree pinned_extents; /* - * we need to make sure block group deletion doesn't race with - * free space cache writeout. This mutex keeps them from stomping - * on each other + * We need to make sure block group deletion doesn't race with the + * dirty block group item updates done outside the commit critical + * section. This mutex keeps them from stomping on each other. */ - struct mutex cache_write_mutex; + struct mutex dirty_bgs_update_mutex; spinlock_t dirty_bgs_lock; /* Protected by spin lock fs_info->unused_bgs_lock. */ struct list_head deleted_bgs; From da9a050baf01eff1075ffab8a3dfeb5de9feec98 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:02 -0400 Subject: [PATCH 0468/1352] btrfs: drop the transaction handle from the prealloc helpers The v1 space cache created its inode during the transaction commit, and btrfs_prealloc_file_range_trans() existed so that preallocation could reuse the open handle. It was the only caller passing a transaction, so __btrfs_prealloc_file_range() and insert_prealloc_file_extent() now always start their own. Fold the wrapper into btrfs_prealloc_file_range() and drop the parameter. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/inode.c | 47 ++++++++++------------------------------------- 1 file changed, 10 insertions(+), 37 deletions(-) diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 6b57383b844160..ba9053c2f24fb5 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -9156,14 +9156,13 @@ static int btrfs_symlink(struct mnt_idmap *idmap, struct inode *dir, } static struct btrfs_trans_handle *insert_prealloc_file_extent( - struct btrfs_trans_handle *trans_in, struct btrfs_inode *inode, struct btrfs_key *ins, u64 file_offset) { struct btrfs_file_extent_item stack_fi; struct btrfs_replace_extent_info extent_info; - struct btrfs_trans_handle *trans = trans_in; + struct btrfs_trans_handle *trans; struct btrfs_path *path; u64 start = ins->objectid; u64 len = ins->offset; @@ -9184,15 +9183,6 @@ static struct btrfs_trans_handle *insert_prealloc_file_extent( if (ret < 0) return ERR_PTR(ret); - if (trans) { - ret = insert_reserved_file_extent(trans, inode, - file_offset, &stack_fi, - true, qgroup_released); - if (ret) - goto free_qgroup; - return trans; - } - extent_info.disk_offset = start; extent_info.disk_len = len; extent_info.data_offset = 0; @@ -9232,12 +9222,12 @@ static struct btrfs_trans_handle *insert_prealloc_file_extent( return ERR_PTR(ret); } -static int __btrfs_prealloc_file_range(struct inode *inode, int mode, - u64 start, u64 num_bytes, u64 min_size, - loff_t actual_len, u64 *alloc_hint, - struct btrfs_trans_handle *trans) +int btrfs_prealloc_file_range(struct inode *inode, int mode, + u64 start, u64 num_bytes, u64 min_size, + loff_t actual_len, u64 *alloc_hint) { struct btrfs_fs_info *fs_info = inode_to_fs_info(inode); + struct btrfs_trans_handle *trans; struct extent_map *em; struct btrfs_root *root = BTRFS_I(inode)->root; struct btrfs_key ins; @@ -9247,11 +9237,8 @@ static int __btrfs_prealloc_file_range(struct inode *inode, int mode, u64 cur_bytes; u64 last_alloc = (u64)-1; int ret = 0; - bool own_trans = true; u64 end = start + num_bytes - 1; - if (trans) - own_trans = false; while (num_bytes > 0) { cur_bytes = min_t(u64, num_bytes, SZ_256M); cur_bytes = max(cur_bytes, min_size); @@ -9277,8 +9264,8 @@ static int __btrfs_prealloc_file_range(struct inode *inode, int mode, clear_offset += ins.offset; last_alloc = ins.offset; - trans = insert_prealloc_file_extent(trans, BTRFS_I(inode), - &ins, cur_offset); + trans = insert_prealloc_file_extent(BTRFS_I(inode), &ins, + cur_offset); /* * Now that we inserted the prealloc extent we can finally * decrement the number of reservations in the block group. @@ -9350,8 +9337,7 @@ static int __btrfs_prealloc_file_range(struct inode *inode, int mode, range_start, range_end - range_start); if (ret) { btrfs_abort_transaction(trans, ret); - if (own_trans) - btrfs_end_transaction(trans); + btrfs_end_transaction(trans); break; } @@ -9363,15 +9349,11 @@ static int __btrfs_prealloc_file_range(struct inode *inode, int mode, if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); - if (own_trans) - btrfs_end_transaction(trans); + btrfs_end_transaction(trans); break; } - if (own_trans) { - btrfs_end_transaction(trans); - trans = NULL; - } + btrfs_end_transaction(trans); } if (clear_offset < end) btrfs_free_reserved_data_space(BTRFS_I(inode), NULL, clear_offset, @@ -9379,15 +9361,6 @@ static int __btrfs_prealloc_file_range(struct inode *inode, int mode, return ret; } -int btrfs_prealloc_file_range(struct inode *inode, int mode, - u64 start, u64 num_bytes, u64 min_size, - loff_t actual_len, u64 *alloc_hint) -{ - return __btrfs_prealloc_file_range(inode, mode, start, num_bytes, - min_size, actual_len, alloc_hint, - NULL); -} - /* * NOTE: in case you are adding MAY_EXEC check for directories: * we are marking them with IOP_FASTPERM_MAY_EXEC, allowing path lookup to From 32c745fd0c5d6a3445446de8262c92db32774030 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:03 -0400 Subject: [PATCH 0469/1352] btrfs: remove the v1 space cache load path Nothing writes a v1 space cache any more, and since commit 545e560a5b0f ("btrfs: disable v1 space cache") the mount option can't be enabled to read one either. Remove load_free_space_cache(), its io_ctl helpers and struct btrfs_io_ctl. Drop the gfp constraint on the inode mapping as well, it only covered the cache's page cache allocations. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/block-group.c | 18 +- fs/btrfs/free-space-cache.c | 565 ------------------------------------ fs/btrfs/free-space-cache.h | 15 - 3 files changed, 1 insertion(+), 597 deletions(-) diff --git a/fs/btrfs/block-group.c b/fs/btrfs/block-group.c index 2af84fcc3de76b..dbe985db8e471f 100644 --- a/fs/btrfs/block-group.c +++ b/fs/btrfs/block-group.c @@ -904,22 +904,6 @@ static noinline void caching_thread(struct btrfs_work *work) down_read(&fs_info->commit_root_sem); load_block_group_size_class(caching_ctl); - if (btrfs_test_opt(fs_info, SPACE_CACHE)) { - ret = load_free_space_cache(block_group); - if (ret == 1) { - ret = 0; - goto done; - } - - /* - * We failed to load the space cache, set ourselves to - * CACHE_STARTED and carry on. - */ - spin_lock(&block_group->lock); - block_group->cached = BTRFS_CACHE_STARTED; - spin_unlock(&block_group->lock); - wake_up(&caching_ctl->wait); - } /* * If we are in the transaction that populated the free space tree we @@ -933,7 +917,7 @@ static noinline void caching_thread(struct btrfs_work *work) ret = btrfs_load_free_space_tree(caching_ctl); else ret = load_extent_tree_free(caching_ctl); -done: + spin_lock(&block_group->lock); block_group->caching_ctl = NULL; block_group->cached = ret ? BTRFS_CACHE_ERROR : BTRFS_CACHE_FINISHED; diff --git a/fs/btrfs/free-space-cache.c b/fs/btrfs/free-space-cache.c index 336b546b0a9467..a25c4db561b4bd 100644 --- a/fs/btrfs/free-space-cache.c +++ b/fs/btrfs/free-space-cache.c @@ -9,7 +9,6 @@ #include #include #include -#include #include #include #include "extent-tree.h" @@ -23,7 +22,6 @@ #include "space-info.h" #include "block-group.h" #include "discard.h" -#include "subpage.h" #include "inode-item.h" #include "accessors.h" #include "file-item.h" @@ -57,11 +55,6 @@ static void bitmap_clear_bits(struct btrfs_free_space_ctl *ctl, struct btrfs_free_space *info, u64 offset, u64 bytes, bool update_stats); -static void btrfs_crc32c_final(u32 crc, u8 *result) -{ - put_unaligned_le32(~crc, result); -} - static void __btrfs_remove_free_space_cache(struct btrfs_free_space_ctl *ctl) { struct btrfs_free_space *info; @@ -123,10 +116,6 @@ static struct inode *__lookup_free_space_inode(struct btrfs_root *root, if (IS_ERR(inode)) return ERR_CAST(inode); - mapping_set_gfp_mask(inode->vfs_inode.i_mapping, - mapping_gfp_constraint(inode->vfs_inode.i_mapping, - ~(__GFP_FS | __GFP_HIGHMEM))); - return &inode->vfs_inode; } @@ -262,226 +251,6 @@ int btrfs_truncate_free_space_cache(struct btrfs_trans_handle *trans, return ret; } -static void readahead_cache(struct inode *inode) -{ - struct file_ra_state ra; - pgoff_t last_index; - - file_ra_state_init(&ra, inode->i_mapping); - last_index = (i_size_read(inode) - 1) >> PAGE_SHIFT; - - page_cache_sync_readahead(inode->i_mapping, &ra, NULL, 0, last_index); -} - -static int io_ctl_init(struct btrfs_io_ctl *io_ctl, struct inode *inode, - int write) -{ - int num_pages; - - num_pages = DIV_ROUND_UP(i_size_read(inode), PAGE_SIZE); - - /* Make sure we can fit our crcs and generation into the first page */ - if (write && (num_pages * sizeof(u32) + sizeof(u64)) > PAGE_SIZE) - return -ENOSPC; - - memset(io_ctl, 0, sizeof(struct btrfs_io_ctl)); - - io_ctl->pages = kzalloc_objs(struct page *, num_pages, GFP_NOFS); - if (!io_ctl->pages) - return -ENOMEM; - - io_ctl->num_pages = num_pages; - io_ctl->fs_info = inode_to_fs_info(inode); - io_ctl->inode = inode; - - return 0; -} -ALLOW_ERROR_INJECTION(io_ctl_init, ERRNO); - -static void io_ctl_free(struct btrfs_io_ctl *io_ctl) -{ - kfree(io_ctl->pages); - io_ctl->pages = NULL; -} - -static void io_ctl_unmap_page(struct btrfs_io_ctl *io_ctl) -{ - if (io_ctl->cur) { - io_ctl->cur = NULL; - io_ctl->orig = NULL; - } -} - -static void io_ctl_map_page(struct btrfs_io_ctl *io_ctl, int clear) -{ - ASSERT(io_ctl->index < io_ctl->num_pages); - io_ctl->page = io_ctl->pages[io_ctl->index++]; - io_ctl->cur = page_address(io_ctl->page); - io_ctl->orig = io_ctl->cur; - io_ctl->size = PAGE_SIZE; - if (clear) - clear_page(io_ctl->cur); -} - -static void io_ctl_drop_pages(struct btrfs_io_ctl *io_ctl) -{ - int i; - - io_ctl_unmap_page(io_ctl); - - for (i = 0; i < io_ctl->num_pages; i++) { - if (io_ctl->pages[i]) { - unlock_page(io_ctl->pages[i]); - put_page(io_ctl->pages[i]); - } - } -} - -static int io_ctl_prepare_pages(struct btrfs_io_ctl *io_ctl, bool uptodate) -{ - struct folio *folio; - struct inode *inode = io_ctl->inode; - gfp_t mask = btrfs_alloc_write_mask(inode->i_mapping); - int i; - - for (i = 0; i < io_ctl->num_pages; i++) { - int ret; - - folio = __filemap_get_folio(inode->i_mapping, i, - FGP_LOCK | FGP_ACCESSED | FGP_CREAT, - mask); - if (IS_ERR(folio)) { - io_ctl_drop_pages(io_ctl); - return PTR_ERR(folio); - } - - ret = set_folio_extent_mapped(folio); - if (ret < 0) { - folio_unlock(folio); - folio_put(folio); - io_ctl_drop_pages(io_ctl); - return ret; - } - - io_ctl->pages[i] = &folio->page; - if (uptodate && !folio_test_uptodate(folio)) { - btrfs_read_folio(NULL, folio); - folio_lock(folio); - if (folio->mapping != inode->i_mapping) { - btrfs_err(BTRFS_I(inode)->root->fs_info, - "free space cache page truncated"); - io_ctl_drop_pages(io_ctl); - return -EIO; - } - if (!folio_test_uptodate(folio)) { - btrfs_err(BTRFS_I(inode)->root->fs_info, - "error reading free space cache"); - io_ctl_drop_pages(io_ctl); - return -EIO; - } - } - } - - for (i = 0; i < io_ctl->num_pages; i++) - clear_page_dirty_for_io(io_ctl->pages[i]); - - return 0; -} - -static int io_ctl_check_generation(struct btrfs_io_ctl *io_ctl, u64 generation) -{ - u64 cache_gen; - - /* - * Skip the crc area. If we don't check crcs then we just have a 64bit - * chunk at the front of the first page. - */ - io_ctl->cur += sizeof(u32) * io_ctl->num_pages; - io_ctl->size -= sizeof(u64) + (sizeof(u32) * io_ctl->num_pages); - - cache_gen = get_unaligned_le64(io_ctl->cur); - if (cache_gen != generation) { - btrfs_err_rl(io_ctl->fs_info, - "space cache generation (%llu) does not match inode (%llu)", - cache_gen, generation); - io_ctl_unmap_page(io_ctl); - return -EIO; - } - io_ctl->cur += sizeof(u64); - return 0; -} - -static int io_ctl_check_crc(struct btrfs_io_ctl *io_ctl, int index) -{ - u32 *tmp, val; - u32 crc = ~(u32)0; - unsigned offset = 0; - - if (index >= io_ctl->num_pages) - return -EIO; - - if (index == 0) - offset = sizeof(u32) * io_ctl->num_pages; - - tmp = page_address(io_ctl->pages[0]); - tmp += index; - val = *tmp; - - io_ctl_map_page(io_ctl, 0); - crc = crc32c(crc, io_ctl->orig + offset, PAGE_SIZE - offset); - btrfs_crc32c_final(crc, (u8 *)&crc); - if (val != crc) { - btrfs_err_rl(io_ctl->fs_info, - "csum mismatch on free space cache"); - io_ctl_unmap_page(io_ctl); - return -EIO; - } - - return 0; -} - -static int io_ctl_read_entry(struct btrfs_io_ctl *io_ctl, - struct btrfs_free_space *entry, u8 *type) -{ - struct btrfs_free_space_entry *e; - int ret; - - if (!io_ctl->cur) { - ret = io_ctl_check_crc(io_ctl, io_ctl->index); - if (ret) - return ret; - } - - e = io_ctl->cur; - entry->offset = get_unaligned_le64(&e->offset); - entry->bytes = get_unaligned_le64(&e->bytes); - *type = e->type; - io_ctl->cur += sizeof(struct btrfs_free_space_entry); - io_ctl->size -= sizeof(struct btrfs_free_space_entry); - - if (io_ctl->size >= sizeof(struct btrfs_free_space_entry)) - return 0; - - io_ctl_unmap_page(io_ctl); - - return 0; -} - -static int io_ctl_read_bitmap(struct btrfs_io_ctl *io_ctl, - struct btrfs_free_space *entry) -{ - int ret; - - ret = io_ctl_check_crc(io_ctl, io_ctl->index); - if (ret) - return ret; - - copy_page(entry->bitmap, io_ctl->cur); - io_ctl_unmap_page(io_ctl); - - return 0; -} - static void recalculate_thresholds(struct btrfs_free_space_ctl *ctl) { struct btrfs_block_group *block_group = ctl->block_group; @@ -527,340 +296,6 @@ static void recalculate_thresholds(struct btrfs_free_space_ctl *ctl) div_u64(extent_bytes, sizeof(struct btrfs_free_space)); } -static int __load_free_space_cache(struct btrfs_root *root, struct inode *inode, - struct btrfs_free_space_ctl *ctl, - struct btrfs_path *path, u64 offset) -{ - struct btrfs_fs_info *fs_info = root->fs_info; - struct btrfs_free_space_header *header; - struct extent_buffer *leaf; - struct btrfs_io_ctl io_ctl; - struct btrfs_key key; - struct btrfs_free_space *e, *n; - LIST_HEAD(bitmaps); - u64 num_entries; - u64 num_bitmaps; - u64 generation; - u8 type; - int ret = 0; - - /* Nothing in the space cache, goodbye */ - if (!i_size_read(inode)) - return 0; - - key.objectid = BTRFS_FREE_SPACE_OBJECTID; - key.type = 0; - key.offset = offset; - - ret = btrfs_search_slot(NULL, root, &key, path, 0, 0); - if (ret < 0) - return 0; - else if (ret > 0) { - btrfs_release_path(path); - return 0; - } - - ret = -1; - - leaf = path->nodes[0]; - header = btrfs_item_ptr(leaf, path->slots[0], - struct btrfs_free_space_header); - num_entries = btrfs_free_space_entries(leaf, header); - num_bitmaps = btrfs_free_space_bitmaps(leaf, header); - generation = btrfs_free_space_generation(leaf, header); - btrfs_release_path(path); - - if (!BTRFS_I(inode)->generation) { - btrfs_info(fs_info, - "the free space cache file (%llu) is invalid, skip it", - offset); - return 0; - } - - if (BTRFS_I(inode)->generation != generation) { - btrfs_err(fs_info, - "free space inode generation (%llu) did not match free space cache generation (%llu)", - BTRFS_I(inode)->generation, generation); - return 0; - } - - if (!num_entries) - return 0; - - ret = io_ctl_init(&io_ctl, inode, 0); - if (ret) - return ret; - - readahead_cache(inode); - - ret = io_ctl_prepare_pages(&io_ctl, true); - if (ret) - goto out; - - ret = io_ctl_check_crc(&io_ctl, 0); - if (ret) - goto free_cache; - - ret = io_ctl_check_generation(&io_ctl, generation); - if (ret) - goto free_cache; - - while (num_entries) { - e = kmem_cache_zalloc(btrfs_free_space_cachep, - GFP_NOFS); - if (!e) { - ret = -ENOMEM; - goto free_cache; - } - - ret = io_ctl_read_entry(&io_ctl, e, &type); - if (ret) { - kmem_cache_free(btrfs_free_space_cachep, e); - goto free_cache; - } - - if (!e->bytes) { - ret = -1; - kmem_cache_free(btrfs_free_space_cachep, e); - goto free_cache; - } - - if (type == BTRFS_FREE_SPACE_EXTENT) { - spin_lock(&ctl->tree_lock); - ret = link_free_space(ctl, e); - spin_unlock(&ctl->tree_lock); - if (ret) { - btrfs_err(fs_info, - "Duplicate entries in free space cache, dumping"); - kmem_cache_free(btrfs_free_space_cachep, e); - goto free_cache; - } - } else { - ASSERT(num_bitmaps); - num_bitmaps--; - e->bitmap = kmem_cache_zalloc( - btrfs_free_space_bitmap_cachep, GFP_NOFS); - if (!e->bitmap) { - ret = -ENOMEM; - kmem_cache_free( - btrfs_free_space_cachep, e); - goto free_cache; - } - spin_lock(&ctl->tree_lock); - ret = link_free_space(ctl, e); - if (ret) { - spin_unlock(&ctl->tree_lock); - btrfs_err(fs_info, - "Duplicate entries in free space cache, dumping"); - kmem_cache_free(btrfs_free_space_bitmap_cachep, e->bitmap); - kmem_cache_free(btrfs_free_space_cachep, e); - goto free_cache; - } - ctl->total_bitmaps++; - recalculate_thresholds(ctl); - spin_unlock(&ctl->tree_lock); - list_add_tail(&e->list, &bitmaps); - } - - num_entries--; - } - - io_ctl_unmap_page(&io_ctl); - - /* - * We add the bitmaps at the end of the entries in order that - * the bitmap entries are added to the cache. - */ - list_for_each_entry_safe(e, n, &bitmaps, list) { - list_del_init(&e->list); - ret = io_ctl_read_bitmap(&io_ctl, e); - if (ret) - goto free_cache; - } - - io_ctl_drop_pages(&io_ctl); - ret = 1; -out: - io_ctl_free(&io_ctl); - return ret; -free_cache: - io_ctl_drop_pages(&io_ctl); - - spin_lock(&ctl->tree_lock); - __btrfs_remove_free_space_cache(ctl); - spin_unlock(&ctl->tree_lock); - goto out; -} - -static int copy_free_space_cache(struct btrfs_free_space_ctl *ctl) -{ - struct btrfs_free_space *info; - struct rb_node *n; - int ret = 0; - - while (!ret && (n = rb_first(&ctl->free_space_offset)) != NULL) { - info = rb_entry(n, struct btrfs_free_space, offset_index); - if (!info->bitmap) { - const u64 offset = info->offset; - const u64 bytes = info->bytes; - - unlink_free_space(ctl, info, true); - spin_unlock(&ctl->tree_lock); - kmem_cache_free(btrfs_free_space_cachep, info); - ret = btrfs_add_free_space(ctl->block_group, offset, bytes); - spin_lock(&ctl->tree_lock); - } else { - u64 offset = info->offset; - u64 bytes = ctl->block_group->fs_info->sectorsize; - - ret = search_bitmap(ctl, info, &offset, &bytes, false); - if (ret == 0) { - bitmap_clear_bits(ctl, info, offset, bytes, true); - spin_unlock(&ctl->tree_lock); - ret = btrfs_add_free_space(ctl->block_group, offset, - bytes); - spin_lock(&ctl->tree_lock); - } else { - free_bitmap(ctl, info); - ret = 0; - } - } - cond_resched_lock(&ctl->tree_lock); - } - return ret; -} - -static struct lock_class_key btrfs_free_space_inode_key; - -int load_free_space_cache(struct btrfs_block_group *block_group) -{ - struct btrfs_fs_info *fs_info = block_group->fs_info; - struct btrfs_free_space_ctl *ctl = block_group->free_space_ctl; - struct btrfs_free_space_ctl tmp_ctl = {}; - struct inode *inode; - struct btrfs_path *path; - int ret = 0; - bool matched; - u64 used = block_group->used; - - /* - * Because we could potentially discard our loaded free space, we want - * to load everything into a temporary structure first, and then if it's - * valid copy it all into the actual free space ctl. - */ - btrfs_init_free_space_ctl(block_group, &tmp_ctl); - - /* - * If this block group has been marked to be cleared for one reason or - * another then we can't trust the on disk cache, so just return. - */ - spin_lock(&block_group->lock); - if (block_group->disk_cache_state != BTRFS_DC_WRITTEN) { - spin_unlock(&block_group->lock); - return 0; - } - spin_unlock(&block_group->lock); - - path = btrfs_alloc_path(); - if (!path) - return 0; - path->search_commit_root = true; - path->skip_locking = true; - - /* - * We must pass a path with search_commit_root set to btrfs_iget in - * order to avoid a deadlock when allocating extents for the tree root. - * - * When we are COWing an extent buffer from the tree root, when looking - * for a free extent, at extent-tree.c:find_free_extent(), we can find - * block group without its free space cache loaded. When we find one - * we must load its space cache which requires reading its free space - * cache's inode item from the root tree. If this inode item is located - * in the same leaf that we started COWing before, then we end up in - * deadlock on the extent buffer (trying to read lock it when we - * previously write locked it). - * - * It's safe to read the inode item using the commit root because - * block groups, once loaded, stay in memory forever (until they are - * removed) as well as their space caches once loaded. New block groups - * once created get their ->cached field set to BTRFS_CACHE_FINISHED so - * we will never try to read their inode item while the fs is mounted. - */ - inode = lookup_free_space_inode(block_group, path); - if (IS_ERR(inode)) { - btrfs_free_path(path); - return 0; - } - - /* We may have converted the inode and made the cache invalid. */ - spin_lock(&block_group->lock); - if (block_group->disk_cache_state != BTRFS_DC_WRITTEN) { - spin_unlock(&block_group->lock); - btrfs_free_path(path); - goto out; - } - spin_unlock(&block_group->lock); - - /* - * Reinitialize the class of struct inode's mapping->invalidate_lock for - * free space inodes to prevent false positives related to locks for normal - * inodes. - */ - lockdep_set_class(&(&inode->i_data)->invalidate_lock, - &btrfs_free_space_inode_key); - - ret = __load_free_space_cache(fs_info->tree_root, inode, &tmp_ctl, - path, block_group->start); - btrfs_free_path(path); - if (ret <= 0) - goto out; - - matched = (tmp_ctl.free_space == (block_group->length - used - - block_group->bytes_super)); - - if (matched) { - spin_lock(&tmp_ctl.tree_lock); - ret = copy_free_space_cache(&tmp_ctl); - spin_unlock(&tmp_ctl.tree_lock); - /* - * ret == 1 means we successfully loaded the free space cache, - * so we need to re-set it here. - */ - if (ret == 0) - ret = 1; - } else { - /* - * We need to call the _locked variant so we don't try to update - * the discard counters. - */ - spin_lock(&tmp_ctl.tree_lock); - __btrfs_remove_free_space_cache(&tmp_ctl); - spin_unlock(&tmp_ctl.tree_lock); - btrfs_warn(fs_info, - "block group %llu has wrong amount of free space", - block_group->start); - ret = -1; - } -out: - if (ret < 0) { - /* This cache is bogus, make sure it gets cleared */ - spin_lock(&block_group->lock); - block_group->disk_cache_state = BTRFS_DC_CLEAR; - spin_unlock(&block_group->lock); - ret = 0; - - btrfs_warn(fs_info, - "failed to load free space cache for block group %llu, rebuilding it now", - block_group->start); - } - - spin_lock(&ctl->tree_lock); - btrfs_discard_update_discardable(block_group); - spin_unlock(&ctl->tree_lock); - iput(inode); - return ret; -} - static inline unsigned long offset_to_bit(u64 bitmap_start, u32 unit, u64 offset) { diff --git a/fs/btrfs/free-space-cache.h b/fs/btrfs/free-space-cache.h index 2432f1783f47cd..29166cc09b9012 100644 --- a/fs/btrfs/free-space-cache.h +++ b/fs/btrfs/free-space-cache.h @@ -14,7 +14,6 @@ #include "fs.h" struct inode; -struct page; struct btrfs_fs_info; struct btrfs_path; struct btrfs_trans_handle; @@ -88,19 +87,6 @@ struct btrfs_free_space_ctl { struct list_head trimming_ranges; }; -struct btrfs_io_ctl { - void *cur, *orig; - struct page *page; - struct page **pages; - struct btrfs_fs_info *fs_info; - struct inode *inode; - unsigned long size; - int index; - int num_pages; - int entries; - int bitmaps; -}; - int __init btrfs_free_space_init(void); void __cold btrfs_free_space_exit(void); struct inode *lookup_free_space_inode(struct btrfs_block_group *block_group, @@ -111,7 +97,6 @@ int btrfs_remove_free_space_inode(struct btrfs_trans_handle *trans, int btrfs_truncate_free_space_cache(struct btrfs_trans_handle *trans, struct inode *inode); -int load_free_space_cache(struct btrfs_block_group *block_group); void btrfs_init_free_space_ctl(struct btrfs_block_group *block_group, struct btrfs_free_space_ctl *ctl); From ca4f022682f7252c984159853d1e006cd6de70ce Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:04 -0400 Subject: [PATCH 0470/1352] btrfs: remove btrfs_disk_cache_state With neither the writer nor the loader left, nothing acts on disk_cache_state. Remove it, the need_clear handling when reading block groups, and the enum. While at it, drop the unused cache_generation field from struct btrfs_block_group. lookup_free_space_inode() converted old style space inodes by clearing disk_cache_state so the cache would be rewritten with the new inode flags. Without that it only sets flags on the in-memory inode, which every remaining caller truncates or deletes right after, so drop the conversion too. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/block-group.c | 32 ++------------------------------ fs/btrfs/block-group.h | 10 ---------- fs/btrfs/disk-io.c | 4 ---- fs/btrfs/free-space-cache.c | 8 -------- 4 files changed, 2 insertions(+), 52 deletions(-) diff --git a/fs/btrfs/block-group.c b/fs/btrfs/block-group.c index dbe985db8e471f..8a0b7e2a374b64 100644 --- a/fs/btrfs/block-group.c +++ b/fs/btrfs/block-group.c @@ -2494,8 +2494,7 @@ static int check_chunk_block_group_mappings(struct btrfs_fs_info *fs_info) static int read_one_block_group(struct btrfs_fs_info *info, struct btrfs_block_group_item_v2 *bgi, - const struct btrfs_key *key, - bool need_clear) + const struct btrfs_key *key) { struct btrfs_block_group *cache; const bool mixed = btrfs_fs_incompat(info, MIXED_GROUPS); @@ -2521,20 +2520,6 @@ static int read_one_block_group(struct btrfs_fs_info *info, btrfs_set_free_space_tree_thresholds(cache); - if (need_clear) { - /* - * When we mount with old space cache, we need to - * set BTRFS_DC_CLEAR and set dirty flag. - * - * a) Setting 'BTRFS_DC_CLEAR' makes sure that we - * truncate the old free space cache inode and - * setup a new one. - * b) Setting 'dirty flag' makes sure that we flush - * the new space cache info onto disk. - */ - if (btrfs_test_opt(info, SPACE_CACHE)) - cache->disk_cache_state = BTRFS_DC_CLEAR; - } if (!mixed && ((cache->flags & BTRFS_BLOCK_GROUP_METADATA) && (cache->flags & BTRFS_BLOCK_GROUP_DATA))) { btrfs_err(info, @@ -2675,8 +2660,6 @@ int btrfs_read_block_groups(struct btrfs_fs_info *info) struct btrfs_block_group *cache; struct btrfs_space_info *space_info; struct btrfs_key key; - bool need_clear = false; - u64 cache_gen; /* * Either no extent root (with ibadroots rescue option) or we have @@ -2697,13 +2680,6 @@ int btrfs_read_block_groups(struct btrfs_fs_info *info) if (!path) return -ENOMEM; - cache_gen = btrfs_super_cache_generation(info->super_copy); - if (btrfs_test_opt(info, SPACE_CACHE) && - btrfs_super_generation(info->super_copy) != cache_gen) - need_clear = true; - if (btrfs_test_opt(info, CLEAR_CACHE)) - need_clear = true; - while (1) { struct btrfs_block_group_item_v2 bgi; struct extent_buffer *leaf; @@ -2732,7 +2708,7 @@ int btrfs_read_block_groups(struct btrfs_fs_info *info) btrfs_item_key_to_cpu(leaf, &key, slot); btrfs_release_path(path); - ret = read_one_block_group(info, &bgi, &key, need_clear); + ret = read_one_block_group(info, &bgi, &key); if (ret < 0) goto error; key.objectid += key.offset; @@ -3595,10 +3571,6 @@ int btrfs_update_block_group(struct btrfs_trans_handle *trans, spin_lock(&space_info->lock); spin_lock(&cache->lock); - if (btrfs_test_opt(info, SPACE_CACHE) && - cache->disk_cache_state < BTRFS_DC_CLEAR) - cache->disk_cache_state = BTRFS_DC_CLEAR; - old_val = cache->used; if (alloc) { old_val += num_bytes; diff --git a/fs/btrfs/block-group.h b/fs/btrfs/block-group.h index f7346a0170fc46..d567ed822e55da 100644 --- a/fs/btrfs/block-group.h +++ b/fs/btrfs/block-group.h @@ -20,13 +20,6 @@ struct btrfs_fs_info; struct btrfs_inode; struct btrfs_trans_handle; -enum btrfs_disk_cache_state { - BTRFS_DC_WRITTEN, - BTRFS_DC_ERROR, - BTRFS_DC_CLEAR, - BTRFS_DC_SETUP, -}; - enum btrfs_block_group_size_class { /* Unset */ BTRFS_BG_SZ_NONE, @@ -131,7 +124,6 @@ struct btrfs_block_group { u64 delalloc_bytes; u64 bytes_super; u64 flags; - u64 cache_generation; u64 global_root_id; u64 remap_bytes; u32 identity_remap_count; @@ -171,8 +163,6 @@ struct btrfs_block_group { unsigned long full_stripe_len; unsigned long runtime_flags; - enum btrfs_disk_cache_state disk_cache_state; - /* Cache tracking stuff */ enum btrfs_caching_type cached; struct btrfs_caching_control *caching_ctl; diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index 5d362fb5d4dd98..06eb2587e61f50 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -4875,10 +4875,6 @@ void btrfs_cleanup_dirty_bgs(struct btrfs_transaction *cur_trans, dirty_list); list_del_init(&cache->dirty_list); - spin_lock(&cache->lock); - cache->disk_cache_state = BTRFS_DC_ERROR; - spin_unlock(&cache->lock); - spin_unlock(&cur_trans->dirty_bgs_lock); btrfs_put_block_group(cache); btrfs_dec_delayed_refs_rsv_bg_updates(fs_info); diff --git a/fs/btrfs/free-space-cache.c b/fs/btrfs/free-space-cache.c index a25c4db561b4bd..3ba9ed4a39d0da 100644 --- a/fs/btrfs/free-space-cache.c +++ b/fs/btrfs/free-space-cache.c @@ -124,7 +124,6 @@ struct inode *lookup_free_space_inode(struct btrfs_block_group *block_group, { struct btrfs_fs_info *fs_info = block_group->fs_info; struct inode *inode = NULL; - u32 flags = BTRFS_INODE_NODATASUM | BTRFS_INODE_NODATACOW; spin_lock(&block_group->lock); if (block_group->inode) @@ -139,13 +138,6 @@ struct inode *lookup_free_space_inode(struct btrfs_block_group *block_group, return inode; spin_lock(&block_group->lock); - if (!((BTRFS_I(inode)->flags & flags) == flags)) { - btrfs_info(fs_info, "Old style space inode found, converting."); - BTRFS_I(inode)->flags |= BTRFS_INODE_NODATASUM | - BTRFS_INODE_NODATACOW; - block_group->disk_cache_state = BTRFS_DC_CLEAR; - } - if (!test_and_set_bit(BLOCK_GROUP_FLAG_IREF, &block_group->runtime_flags)) block_group->inode = BTRFS_I(igrab(inode)); spin_unlock(&block_group->lock); From 71f6afa2c289d24b9a56f0e0269019648ade2e9d Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:05 -0400 Subject: [PATCH 0471/1352] btrfs: remove the SPACE_CACHE mount option flag Nothing sets BTRFS_MOUNT_SPACE_CACHE anymore, so every test of it is false. Remove the flag, the checks rejecting the v1 cache on zoned filesystems and for sector sizes other than the page size, and the deprecation warning. Show a read-only filesystem that still has an old cache as nospace_cache, since that's what's in effect. space_cache and space_cache=v1 keep falling back to no space cache with a warning. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/disk-io.c | 19 ++----------------- fs/btrfs/fs.h | 1 - fs/btrfs/super.c | 31 ++----------------------------- fs/btrfs/transaction.c | 4 +--- fs/btrfs/zoned.c | 9 --------- 5 files changed, 5 insertions(+), 59 deletions(-) diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index 06eb2587e61f50..29976f5d02f8e5 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -3069,7 +3069,6 @@ static int btrfs_cleanup_fs_roots(struct btrfs_fs_info *fs_info) int btrfs_start_pre_rw_mount(struct btrfs_fs_info *fs_info) { int ret; - const bool cache_opt = btrfs_test_opt(fs_info, SPACE_CACHE); bool rebuild_free_space_tree = false; if (btrfs_test_opt(fs_info, CLEAR_CACHE) && @@ -3164,8 +3163,8 @@ int btrfs_start_pre_rw_mount(struct btrfs_fs_info *fs_info) } } - if (cache_opt != btrfs_free_space_cache_v1_active(fs_info)) { - ret = btrfs_set_free_space_cache_v1_active(fs_info, cache_opt); + if (btrfs_free_space_cache_v1_active(fs_info)) { + ret = btrfs_set_free_space_cache_v1_active(fs_info, false); if (ret) return ret; } @@ -3281,20 +3280,6 @@ int btrfs_check_features(struct btrfs_fs_info *fs_info, bool is_rw_mount) return -EINVAL; } - /* - * Subpage/bs > ps runtime limitation on v1 cache. - * - * V1 space cache still has some hard coded PAGE_SIZE usage, while - * we're already defaulting to v2 cache, no need to bother v1 as it's - * going to be deprecated anyway. - */ - if (fs_info->sectorsize != PAGE_SIZE && btrfs_test_opt(fs_info, SPACE_CACHE)) { - btrfs_warn(fs_info, - "v1 space cache is not supported for page size %lu with sectorsize %u", - PAGE_SIZE, fs_info->sectorsize); - return -EINVAL; - } - /* This can be called by remount, we need to protect the super block. */ spin_lock(&fs_info->super_lock); btrfs_set_super_incompat_flags(disk_super, incompat); diff --git a/fs/btrfs/fs.h b/fs/btrfs/fs.h index 441b315e8a9893..79d0828c51c7d7 100644 --- a/fs/btrfs/fs.h +++ b/fs/btrfs/fs.h @@ -259,7 +259,6 @@ enum { BTRFS_MOUNT_NOSSD = (1ULL << 9), BTRFS_MOUNT_DISCARD_SYNC = (1ULL << 10), BTRFS_MOUNT_FORCE_COMPRESS = (1ULL << 11), - BTRFS_MOUNT_SPACE_CACHE = (1ULL << 12), BTRFS_MOUNT_CLEAR_CACHE = (1ULL << 13), BTRFS_MOUNT_USER_SUBVOL_RM_ALLOWED = (1ULL << 14), BTRFS_MOUNT_ENOSPC_DEBUG = (1ULL << 15), diff --git a/fs/btrfs/super.c b/fs/btrfs/super.c index c90cd377fcf316..14ed0ed823a3c6 100644 --- a/fs/btrfs/super.c +++ b/fs/btrfs/super.c @@ -515,7 +515,6 @@ static int btrfs_parse_param(struct fs_context *fc, struct fs_parameter *param) btrfs_warn(NULL, "v1 space cache is deprecated, falling back to no space cache"); btrfs_set_opt(ctx->mount_opt, NOSPACECACHE); - btrfs_clear_opt(ctx->mount_opt, SPACE_CACHE); btrfs_clear_opt(ctx->mount_opt, FREE_SPACE_TREE); break; case Opt_space_cache_version: @@ -524,11 +523,9 @@ static int btrfs_parse_param(struct fs_context *fc, struct fs_parameter *param) btrfs_warn(NULL, "v1 space cache is deprecated, falling back to no space cache"); btrfs_set_opt(ctx->mount_opt, NOSPACECACHE); - btrfs_clear_opt(ctx->mount_opt, SPACE_CACHE); btrfs_clear_opt(ctx->mount_opt, FREE_SPACE_TREE); break; case Opt_space_cache_v2: - btrfs_clear_opt(ctx->mount_opt, SPACE_CACHE); btrfs_set_opt(ctx->mount_opt, FREE_SPACE_TREE); break; default: @@ -705,13 +702,6 @@ bool btrfs_check_options(const struct btrfs_fs_info *info, if (btrfs_check_mountopts_zoned(info, mount_opt)) ret = false; - if (!test_bit(BTRFS_FS_STATE_REMOUNTING, &info->fs_state)) { - if (btrfs_raw_test_opt(*mount_opt, SPACE_CACHE)) { - btrfs_warn(info, -"space cache v1 is being deprecated and will be removed in a future release, please use -o space_cache=v2"); - } - } - return ret; } @@ -729,14 +719,6 @@ bool btrfs_check_options(const struct btrfs_fs_info *info, */ void btrfs_set_free_space_cache_settings(struct btrfs_fs_info *fs_info) { - if (fs_info->sectorsize != PAGE_SIZE && btrfs_test_opt(fs_info, SPACE_CACHE)) { - btrfs_info(fs_info, - "forcing free space tree for sector size %u with page size %lu", - fs_info->sectorsize, PAGE_SIZE); - btrfs_clear_opt(fs_info->mount_opt, SPACE_CACHE); - btrfs_set_opt(fs_info->mount_opt, FREE_SPACE_TREE); - } - /* * At this point our mount options are populated, so we only mess with * these settings if we don't have any settings already. @@ -751,9 +733,6 @@ void btrfs_set_free_space_cache_settings(struct btrfs_fs_info *fs_info) return; } - if (btrfs_test_opt(fs_info, SPACE_CACHE)) - return; - if (btrfs_test_opt(fs_info, NOSPACECACHE)) return; @@ -1107,9 +1086,7 @@ static int btrfs_show_options(struct seq_file *seq, struct dentry *dentry) seq_puts(seq, ",discard=async"); if (!(info->sb->s_flags & SB_POSIXACL)) seq_puts(seq, ",noacl"); - if (btrfs_free_space_cache_v1_active(info)) - seq_puts(seq, ",space_cache"); - else if (btrfs_fs_compat_ro(info, FREE_SPACE_TREE)) + if (btrfs_fs_compat_ro(info, FREE_SPACE_TREE)) seq_puts(seq, ",space_cache=v2"); else seq_puts(seq, ",nospace_cache"); @@ -1441,7 +1418,6 @@ static void btrfs_emit_options(struct btrfs_fs_info *info, btrfs_info_if_set(info, old, DISCARD_SYNC, "turning on sync discard"); btrfs_info_if_set(info, old, DISCARD_ASYNC, "turning on async discard"); btrfs_info_if_set(info, old, FREE_SPACE_TREE, "enabling free space tree"); - btrfs_info_if_set(info, old, SPACE_CACHE, "enabling disk space caching"); btrfs_info_if_set(info, old, CLEAR_CACHE, "force clearing of disk cache"); btrfs_info_if_set(info, old, AUTO_DEFRAG, "enabling auto defrag"); btrfs_info_if_set(info, old, FRAGMENT_DATA, "fragmenting data"); @@ -1459,7 +1435,6 @@ static void btrfs_emit_options(struct btrfs_fs_info *info, btrfs_info_if_unset(info, old, SSD_SPREAD, "not using spread ssd allocation scheme"); btrfs_info_if_unset(info, old, NOBARRIER, "turning on barriers"); btrfs_info_if_unset(info, old, NOTREELOG, "enabling tree log"); - btrfs_info_if_unset(info, old, SPACE_CACHE, "disabling disk space caching"); btrfs_info_if_unset(info, old, FREE_SPACE_TREE, "disabling free space tree"); btrfs_info_if_unset(info, old, AUTO_DEFRAG, "disabling auto defrag"); btrfs_info_if_unset(info, old, COMPRESS, "use no compression"); @@ -1524,10 +1499,8 @@ static int btrfs_reconfigure(struct fs_context *fc) btrfs_warn(fs_info, "remount supports changing free space tree only from RO to RW"); /* Make sure free space cache options match the state on disk. */ - if (btrfs_fs_compat_ro(fs_info, FREE_SPACE_TREE)) { + if (btrfs_fs_compat_ro(fs_info, FREE_SPACE_TREE)) btrfs_set_opt(fs_info->mount_opt, FREE_SPACE_TREE); - btrfs_clear_opt(fs_info->mount_opt, SPACE_CACHE); - } } ret = 0; diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c index e4eff97a780a21..260088f30f0068 100644 --- a/fs/btrfs/transaction.c +++ b/fs/btrfs/transaction.c @@ -2024,9 +2024,7 @@ static void update_super_roots(struct btrfs_fs_info *fs_info) super->root = root_item->bytenr; super->generation = root_item->generation; super->root_level = root_item->level; - if (btrfs_test_opt(fs_info, SPACE_CACHE)) - super->cache_generation = root_item->generation; - else if (test_bit(BTRFS_FS_CLEANUP_SPACE_CACHE_V1, &fs_info->flags)) + if (test_bit(BTRFS_FS_CLEANUP_SPACE_CACHE_V1, &fs_info->flags)) super->cache_generation = 0; if (test_bit(BTRFS_FS_UPDATE_UUID_TREE_GEN, &fs_info->flags)) super->uuid_tree_generation = root_item->generation; diff --git a/fs/btrfs/zoned.c b/fs/btrfs/zoned.c index 9f562cc34e3836..1cda88332284e8 100644 --- a/fs/btrfs/zoned.c +++ b/fs/btrfs/zoned.c @@ -804,15 +804,6 @@ int btrfs_check_mountopts_zoned(const struct btrfs_fs_info *info, if (!btrfs_is_zoned(info)) return 0; - /* - * Space cache writing is not COWed. Disable that to avoid write errors - * in sequential zones. - */ - if (btrfs_raw_test_opt(*mount_opt, SPACE_CACHE)) { - btrfs_err(info, "zoned: space cache v1 is not supported"); - return -EINVAL; - } - if (btrfs_raw_test_opt(*mount_opt, NODATACOW)) { btrfs_err(info, "zoned: NODATACOW not supported"); return -EINVAL; From f84e1c65542ddde195561960fab02389642f6760 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:06 -0400 Subject: [PATCH 0472/1352] btrfs: replace btrfs_set_free_space_cache_v1_active() with a cleanup helper The only caller passes active = false. Turn it into btrfs_cleanup_free_space_cache_v1() and fold the block group loop into it. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/disk-io.c | 2 +- fs/btrfs/free-space-cache.c | 42 +++++++++++-------------------------- fs/btrfs/free-space-cache.h | 2 +- 3 files changed, 14 insertions(+), 32 deletions(-) diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index 29976f5d02f8e5..340673d63e484d 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -3164,7 +3164,7 @@ int btrfs_start_pre_rw_mount(struct btrfs_fs_info *fs_info) } if (btrfs_free_space_cache_v1_active(fs_info)) { - ret = btrfs_set_free_space_cache_v1_active(fs_info, false); + ret = btrfs_cleanup_free_space_cache_v1(fs_info); if (ret) return ret; } diff --git a/fs/btrfs/free-space-cache.c b/fs/btrfs/free-space-cache.c index 3ba9ed4a39d0da..fb6ff3db1241d8 100644 --- a/fs/btrfs/free-space-cache.c +++ b/fs/btrfs/free-space-cache.c @@ -2891,47 +2891,29 @@ bool btrfs_free_space_cache_v1_active(struct btrfs_fs_info *fs_info) return btrfs_super_cache_generation(fs_info->super_copy); } -static int cleanup_free_space_cache_v1(struct btrfs_fs_info *fs_info, - struct btrfs_trans_handle *trans) +int btrfs_cleanup_free_space_cache_v1(struct btrfs_fs_info *fs_info) { - struct btrfs_block_group *block_group; + struct btrfs_trans_handle *trans; struct rb_node *node; + int ret; btrfs_info(fs_info, "cleaning free space cache v1"); - node = rb_first_cached(&fs_info->block_group_cache_tree); - while (node) { - int ret; - - block_group = rb_entry(node, struct btrfs_block_group, cache_node); - ret = btrfs_remove_free_space_inode(trans, NULL, block_group); - if (ret) - return ret; - node = rb_next(node); - } - return 0; -} - -int btrfs_set_free_space_cache_v1_active(struct btrfs_fs_info *fs_info, bool active) -{ - struct btrfs_trans_handle *trans; - int ret; - /* - * update_super_roots will appropriately set or unset - * super_copy->cache_generation based on SPACE_CACHE and - * BTRFS_FS_CLEANUP_SPACE_CACHE_V1. For this reason, we need a - * transaction commit whether we are enabling space cache v1 and don't - * have any other work to do, or are disabling it and removing free - * space inodes. + * update_super_roots() zeroes super_copy->cache_generation while + * BTRFS_FS_CLEANUP_SPACE_CACHE_V1 is set, so this needs a commit. */ trans = btrfs_start_transaction(fs_info->tree_root, 0); if (IS_ERR(trans)) return PTR_ERR(trans); - if (!active) { - set_bit(BTRFS_FS_CLEANUP_SPACE_CACHE_V1, &fs_info->flags); - ret = cleanup_free_space_cache_v1(fs_info, trans); + set_bit(BTRFS_FS_CLEANUP_SPACE_CACHE_V1, &fs_info->flags); + for (node = rb_first_cached(&fs_info->block_group_cache_tree); node; + node = rb_next(node)) { + struct btrfs_block_group *block_group; + + block_group = rb_entry(node, struct btrfs_block_group, cache_node); + ret = btrfs_remove_free_space_inode(trans, NULL, block_group); if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); btrfs_end_transaction(trans); diff --git a/fs/btrfs/free-space-cache.h b/fs/btrfs/free-space-cache.h index 29166cc09b9012..f5f18e397b130e 100644 --- a/fs/btrfs/free-space-cache.h +++ b/fs/btrfs/free-space-cache.h @@ -136,7 +136,7 @@ int btrfs_trim_block_group_bitmaps(struct btrfs_block_group *block_group, void btrfs_trim_fully_remapped_block_group(struct btrfs_block_group *bg); bool btrfs_free_space_cache_v1_active(struct btrfs_fs_info *fs_info); -int btrfs_set_free_space_cache_v1_active(struct btrfs_fs_info *fs_info, bool active); +int btrfs_cleanup_free_space_cache_v1(struct btrfs_fs_info *fs_info); /* Support functions for running our sanity tests */ #ifdef CONFIG_BTRFS_FS_RUN_SANITY_TESTS bool btrfs_use_bitmap(struct btrfs_free_space_ctl *ctl, From f6ea070a748ed3199b1096ba412514f06b6fc20e Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:07 -0400 Subject: [PATCH 0473/1352] btrfs: remove the free space cache trimming ranges cache_writeout_mutex and trimming_ranges let the v1 cache writer see ranges that were unlinked from the free space tree while being discarded. Nothing consumes the list anymore, and the tree itself is protected by tree_lock, so remove them. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/free-space-cache.c | 41 +++---------------------------------- fs/btrfs/free-space-cache.h | 2 -- 2 files changed, 3 insertions(+), 40 deletions(-) diff --git a/fs/btrfs/free-space-cache.c b/fs/btrfs/free-space-cache.c index fb6ff3db1241d8..2a40167c3fb657 100644 --- a/fs/btrfs/free-space-cache.c +++ b/fs/btrfs/free-space-cache.c @@ -36,12 +36,6 @@ static struct kmem_cache *btrfs_free_space_cachep; static struct kmem_cache *btrfs_free_space_bitmap_cachep; -struct btrfs_trim_range { - u64 start; - u64 bytes; - struct list_head list; -}; - static int link_free_space(struct btrfs_free_space_ctl *ctl, struct btrfs_free_space *info); static void unlink_free_space(struct btrfs_free_space_ctl *ctl, @@ -1692,8 +1686,6 @@ void btrfs_init_free_space_ctl(struct btrfs_block_group *block_group, spin_lock_init(&ctl->tree_lock); ctl->block_group = block_group; ctl->free_space_bytes = RB_ROOT_CACHED; - INIT_LIST_HEAD(&ctl->trimming_ranges); - mutex_init(&ctl->cache_writeout_mutex); /* * we only want to have 32k of ram per block group for keeping @@ -2389,12 +2381,10 @@ void btrfs_init_free_cluster(struct btrfs_free_cluster *cluster) static int do_trimming(struct btrfs_block_group *block_group, u64 *total_trimmed, u64 start, u64 bytes, u64 reserved_start, u64 reserved_bytes, - enum btrfs_trim_state reserved_trim_state, - struct btrfs_trim_range *trim_entry) + enum btrfs_trim_state reserved_trim_state) { struct btrfs_space_info *space_info = block_group->space_info; struct btrfs_fs_info *fs_info = block_group->fs_info; - struct btrfs_free_space_ctl *ctl = block_group->free_space_ctl; int ret; bool bg_ro; const u64 end = start + bytes; @@ -2420,7 +2410,6 @@ static int do_trimming(struct btrfs_block_group *block_group, trim_state = BTRFS_TRIM_STATE_TRIMMED; } - mutex_lock(&ctl->cache_writeout_mutex); if (reserved_start < start) __btrfs_add_free_space(block_group, reserved_start, start - reserved_start, @@ -2429,8 +2418,6 @@ static int do_trimming(struct btrfs_block_group *block_group, __btrfs_add_free_space(block_group, end, reserved_end - end, reserved_trim_state); __btrfs_add_free_space(block_group, start, bytes, trim_state); - list_del(&trim_entry->list); - mutex_unlock(&ctl->cache_writeout_mutex); if (!bg_ro) { spin_lock(&space_info->lock); @@ -2468,9 +2455,6 @@ static int trim_no_bitmap(struct btrfs_block_group *block_group, const u64 max_discard_size = READ_ONCE(discard_ctl->max_discard_size); while (start < end) { - struct btrfs_trim_range trim_entry; - - mutex_lock(&ctl->cache_writeout_mutex); spin_lock(&ctl->tree_lock); if (ctl->free_space < minlen) @@ -2501,7 +2485,6 @@ static int trim_no_bitmap(struct btrfs_block_group *block_group, bytes = entry->bytes; if (bytes < minlen) { spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); goto next; } unlink_free_space(ctl, entry, true); @@ -2526,7 +2509,6 @@ static int trim_no_bitmap(struct btrfs_block_group *block_group, bytes = min(extent_start + extent_bytes, end) - start; if (bytes < minlen) { spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); goto next; } @@ -2535,14 +2517,9 @@ static int trim_no_bitmap(struct btrfs_block_group *block_group, } spin_unlock(&ctl->tree_lock); - trim_entry.start = extent_start; - trim_entry.bytes = extent_bytes; - list_add_tail(&trim_entry.list, &ctl->trimming_ranges); - mutex_unlock(&ctl->cache_writeout_mutex); ret = do_trimming(block_group, total_trimmed, start, bytes, - extent_start, extent_bytes, extent_trim_state, - &trim_entry); + extent_start, extent_bytes, extent_trim_state); if (ret) { block_group->discard_cursor = start + bytes; break; @@ -2566,7 +2543,6 @@ static int trim_no_bitmap(struct btrfs_block_group *block_group, out_unlock: block_group->discard_cursor = btrfs_block_group_end(block_group); spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); return ret; } @@ -2677,16 +2653,13 @@ static int trim_bitmaps(struct btrfs_block_group *block_group, while (offset < end) { bool next_bitmap = false; - struct btrfs_trim_range trim_entry; - mutex_lock(&ctl->cache_writeout_mutex); spin_lock(&ctl->tree_lock); if (ctl->free_space < minlen) { block_group->discard_cursor = btrfs_block_group_end(block_group); spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); break; } @@ -2702,7 +2675,6 @@ static int trim_bitmaps(struct btrfs_block_group *block_group, if (!entry || (async && minlen && start == offset && btrfs_free_space_trimmed(entry))) { spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); next_bitmap = true; goto next; } @@ -2728,7 +2700,6 @@ static int trim_bitmaps(struct btrfs_block_group *block_group, else entry->trim_state = BTRFS_TRIM_STATE_UNTRIMMED; spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); next_bitmap = true; goto next; } @@ -2739,14 +2710,12 @@ static int trim_bitmaps(struct btrfs_block_group *block_group, */ if (async && *total_trimmed) { spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); return ret; } bytes = min(bytes, end - start); if (bytes < minlen || (async && maxlen && bytes > maxlen)) { spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); goto next; } @@ -2766,13 +2735,9 @@ static int trim_bitmaps(struct btrfs_block_group *block_group, free_bitmap(ctl, entry); spin_unlock(&ctl->tree_lock); - trim_entry.start = start; - trim_entry.bytes = bytes; - list_add_tail(&trim_entry.list, &ctl->trimming_ranges); - mutex_unlock(&ctl->cache_writeout_mutex); ret = do_trimming(block_group, total_trimmed, start, bytes, - start, bytes, 0, &trim_entry); + start, bytes, 0); if (ret) { reset_trimming_bitmap(ctl, offset); block_group->discard_cursor = diff --git a/fs/btrfs/free-space-cache.h b/fs/btrfs/free-space-cache.h index f5f18e397b130e..e22443598b8efe 100644 --- a/fs/btrfs/free-space-cache.h +++ b/fs/btrfs/free-space-cache.h @@ -83,8 +83,6 @@ struct btrfs_free_space_ctl { s32 discardable_extents[BTRFS_STAT_NR_ENTRIES]; s64 discardable_bytes[BTRFS_STAT_NR_ENTRIES]; struct btrfs_block_group *block_group; - struct mutex cache_writeout_mutex; - struct list_head trimming_ranges; }; int __init btrfs_free_space_init(void); From a1a7d608c84d8d15e45c853dceddaa5eeaf20041 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:08 -0400 Subject: [PATCH 0474/1352] btrfs: remove BTRFS_RESERVE_FLUSH_FREE_SPACE_INODE Free space inodes no longer reserve data or delalloc space, as nothing writes to them. Remove the flush mode and the special cases that selected it. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/delalloc-space.c | 13 ++----------- fs/btrfs/space-info.c | 2 -- fs/btrfs/space-info.h | 4 ---- 3 files changed, 2 insertions(+), 17 deletions(-) diff --git a/fs/btrfs/delalloc-space.c b/fs/btrfs/delalloc-space.c index d357ed7efd99bd..77781852e41782 100644 --- a/fs/btrfs/delalloc-space.c +++ b/fs/btrfs/delalloc-space.c @@ -132,9 +132,7 @@ int btrfs_alloc_data_chunk_ondemand(const struct btrfs_inode *inode, u64 bytes) /* Make sure bytes are sectorsize aligned */ bytes = ALIGN(bytes, fs_info->sectorsize); - if (btrfs_is_free_space_inode(inode)) - flush = BTRFS_RESERVE_FLUSH_FREE_SPACE_INODE; - else if (btrfs_is_zoned(fs_info) && btrfs_is_data_reloc_root(root)) + if (btrfs_is_zoned(fs_info) && btrfs_is_data_reloc_root(root)) flush = BTRFS_RESERVE_FLUSH_ZONED_RELOCATION; return btrfs_reserve_data_bytes(data_sinfo_for_inode(inode), bytes, flush); @@ -155,8 +153,6 @@ int btrfs_check_data_free_space(struct btrfs_inode *inode, if (noflush) flush = BTRFS_RESERVE_NO_FLUSH; - else if (btrfs_is_free_space_inode(inode)) - flush = BTRFS_RESERVE_FLUSH_FREE_SPACE_INODE; ret = btrfs_reserve_data_bytes(data_sinfo_for_inode(inode), len, flush); if (ret < 0) @@ -326,15 +322,10 @@ int btrfs_delalloc_reserve_metadata(struct btrfs_inode *inode, u64 num_bytes, int ret = 0; /* - * If we are a free space inode we need to not flush since we will be in - * the middle of a transaction commit. We also don't need the delalloc - * mutex since we won't race with anybody. We need this mostly to make - * lockdep shut its filthy mouth. - * * If we have a transaction open (can happen if we call truncate_block * from truncate), then we need FLUSH_LIMIT so we don't deadlock. */ - if (noflush || btrfs_is_free_space_inode(inode)) { + if (noflush) { flush = BTRFS_RESERVE_NO_FLUSH; } else { if (current->journal_info) diff --git a/fs/btrfs/space-info.c b/fs/btrfs/space-info.c index 39a28e1bec8ad8..01018152c054b0 100644 --- a/fs/btrfs/space-info.c +++ b/fs/btrfs/space-info.c @@ -1704,7 +1704,6 @@ static int handle_reserve_ticket(struct btrfs_space_info *space_info, evict_flush_states, ARRAY_SIZE(evict_flush_states)); break; - case BTRFS_RESERVE_FLUSH_FREE_SPACE_INODE: case BTRFS_RESERVE_FLUSH_ZONED_RELOCATION: priority_reclaim_data_space(space_info, ticket); break; @@ -1968,7 +1967,6 @@ int btrfs_reserve_data_bytes(struct btrfs_space_info *space_info, u64 bytes, int ret; ASSERT(flush == BTRFS_RESERVE_FLUSH_DATA || - flush == BTRFS_RESERVE_FLUSH_FREE_SPACE_INODE || flush == BTRFS_RESERVE_FLUSH_ZONED_RELOCATION || flush == BTRFS_RESERVE_NO_FLUSH, "flush=%d", flush); ASSERT(!current->journal_info || flush != BTRFS_RESERVE_FLUSH_DATA, diff --git a/fs/btrfs/space-info.h b/fs/btrfs/space-info.h index aa836e8a9d4a6f..d0130c8ba3ddaf 100644 --- a/fs/btrfs/space-info.h +++ b/fs/btrfs/space-info.h @@ -66,7 +66,6 @@ enum btrfs_reserve_flush_enum { * Can be interrupted by a fatal signal. */ BTRFS_RESERVE_FLUSH_DATA, - BTRFS_RESERVE_FLUSH_FREE_SPACE_INODE, BTRFS_RESERVE_FLUSH_ALL, /* @@ -82,9 +81,6 @@ enum btrfs_reserve_flush_enum { * priority flushing for this, because otherwise we can deadlock on * waiting for a ticket, that cannot be granted, because we cannot do * any allocations. - * - * Apart from being specific to zoned relocation, it is equal to - * BTRFS_FLUSH_FREE_SPACE_INODE. */ BTRFS_RESERVE_FLUSH_ZONED_RELOCATION, From 9a9ea6d4dab9a000c990e4f6f9c856e41265bdbb Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:09 -0400 Subject: [PATCH 0475/1352] btrfs: remove the free space inode ordered extent special cases Free space inodes never have ordered extents anymore. Drop the lockdep exceptions for them and btrfs_join_transaction_spacecache(), which was only used to finish their ordered extents during a commit. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/inode.c | 20 +++----------------- fs/btrfs/ordered-data.c | 20 ++------------------ fs/btrfs/transaction.c | 6 ------ fs/btrfs/transaction.h | 1 - 4 files changed, 5 insertions(+), 42 deletions(-) diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index ba9053c2f24fb5..840bb85f7ffca2 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -3221,7 +3221,6 @@ int btrfs_finish_one_ordered(struct btrfs_ordered_extent *ordered_extent) int compress_type = 0; int ret = 0; u64 logical_len = ordered_extent->num_bytes; - bool freespace_inode; bool truncated = false; bool clear_reserved_extent = true; unsigned int clear_bits = 0; @@ -3238,9 +3237,7 @@ int btrfs_finish_one_ordered(struct btrfs_ordered_extent *ordered_extent) if (!test_bit(BTRFS_ORDERED_NOCOW, &ordered_extent->flags)) clear_bits |= EXTENT_DEFRAG; - freespace_inode = btrfs_is_free_space_inode(inode); - if (!freespace_inode) - btrfs_lockdep_acquire(fs_info, btrfs_ordered_extent); + btrfs_lockdep_acquire(fs_info, btrfs_ordered_extent); if (unlikely(test_bit(BTRFS_ORDERED_IOERR, &ordered_extent->flags))) { ret = -EIO; @@ -3275,10 +3272,7 @@ int btrfs_finish_one_ordered(struct btrfs_ordered_extent *ordered_extent) &cached_state); } - if (freespace_inode) - trans = btrfs_join_transaction_spacecache(root); - else - trans = btrfs_join_transaction(root); + trans = btrfs_join_transaction(root); if (IS_ERR(trans)) { ret = PTR_ERR(trans); trans = NULL; @@ -8135,7 +8129,6 @@ void btrfs_destroy_inode(struct inode *vfs_inode) struct btrfs_ordered_extent *ordered; struct btrfs_inode *inode = BTRFS_I(vfs_inode); struct btrfs_root *root = inode->root; - bool freespace_inode; WARN_ON(!hlist_empty(&vfs_inode->i_dentry)); WARN_ON(vfs_inode->i_data.nrpages); @@ -8158,12 +8151,6 @@ void btrfs_destroy_inode(struct inode *vfs_inode) if (!root) return; - /* - * If this is a free space inode do not take the ordered extents lockdep - * map. - */ - freespace_inode = btrfs_is_free_space_inode(inode); - while (1) { ordered = btrfs_lookup_first_ordered_extent(inode, (u64)-1); if (!ordered) @@ -8173,8 +8160,7 @@ void btrfs_destroy_inode(struct inode *vfs_inode) "found ordered extent %llu %llu on inode cleanup", ordered->file_offset, ordered->num_bytes); - if (!freespace_inode) - btrfs_lockdep_acquire(root->fs_info, btrfs_ordered_extent); + btrfs_lockdep_acquire(root->fs_info, btrfs_ordered_extent); btrfs_remove_ordered_extent(ordered); btrfs_put_ordered_extent(ordered); diff --git a/fs/btrfs/ordered-data.c b/fs/btrfs/ordered-data.c index e9f1cbeb555a4c..df74c75d6c2991 100644 --- a/fs/btrfs/ordered-data.c +++ b/fs/btrfs/ordered-data.c @@ -654,13 +654,6 @@ void btrfs_remove_ordered_extent(struct btrfs_ordered_extent *entry) struct btrfs_fs_info *fs_info = root->fs_info; struct rb_node *node; bool pending; - bool freespace_inode; - - /* - * If this is a free space inode the thread has not acquired the ordered - * extents lockdep map. - */ - freespace_inode = btrfs_is_free_space_inode(btrfs_inode); btrfs_lockdep_acquire(fs_info, btrfs_trans_pending_ordered); /* This is paired with alloc_ordered_extent(). */ @@ -735,8 +728,7 @@ void btrfs_remove_ordered_extent(struct btrfs_ordered_extent *entry) } spin_unlock(&root->ordered_extent_lock); wake_up(&entry->wait); - if (!freespace_inode) - btrfs_lockdep_release(fs_info, btrfs_ordered_extent); + btrfs_lockdep_release(fs_info, btrfs_ordered_extent); } static void btrfs_run_ordered_extent_work(struct btrfs_work *work) @@ -867,16 +859,9 @@ void btrfs_start_ordered_extent_nowriteback(struct btrfs_ordered_extent *entry, u64 start = entry->file_offset; u64 end = start + entry->num_bytes - 1; struct btrfs_inode *inode = entry->inode; - bool freespace_inode; trace_btrfs_ordered_extent_start(inode, entry); - /* - * If this is a free space inode do not take the ordered extents lockdep - * map. - */ - freespace_inode = btrfs_is_free_space_inode(inode); - /* * pages in the range can be dirty, clean or writeback. We * start IO on any dirty ones so the wait doesn't stall waiting @@ -896,8 +881,7 @@ void btrfs_start_ordered_extent_nowriteback(struct btrfs_ordered_extent *entry, } } - if (!freespace_inode) - btrfs_might_wait_for_event(inode->root->fs_info, btrfs_ordered_extent); + btrfs_might_wait_for_event(inode->root->fs_info, btrfs_ordered_extent); wait_event(entry->wait, test_bit(BTRFS_ORDERED_COMPLETE, &entry->flags)); } diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c index 260088f30f0068..87194f3e897991 100644 --- a/fs/btrfs/transaction.c +++ b/fs/btrfs/transaction.c @@ -895,12 +895,6 @@ struct btrfs_trans_handle *btrfs_join_transaction(struct btrfs_root *root) true); } -struct btrfs_trans_handle *btrfs_join_transaction_spacecache(struct btrfs_root *root) -{ - return start_transaction(root, 0, TRANS_JOIN_NOLOCK, - BTRFS_RESERVE_NO_FLUSH, true); -} - /* * Similar to regular join but it never starts a transaction when none is * running or when there's a running one at a state >= TRANS_STATE_UNBLOCKED. diff --git a/fs/btrfs/transaction.h b/fs/btrfs/transaction.h index aa57c64a955dcf..684b46e5f2e9a4 100644 --- a/fs/btrfs/transaction.h +++ b/fs/btrfs/transaction.h @@ -287,7 +287,6 @@ struct btrfs_trans_handle *btrfs_start_transaction_fallback_global_rsv( struct btrfs_root *root, unsigned int num_items); struct btrfs_trans_handle *btrfs_join_transaction(struct btrfs_root *root); -struct btrfs_trans_handle *btrfs_join_transaction_spacecache(struct btrfs_root *root); struct btrfs_trans_handle *btrfs_join_transaction_nostart(struct btrfs_root *root); struct btrfs_trans_handle *btrfs_attach_transaction(struct btrfs_root *root); struct btrfs_trans_handle *btrfs_attach_transaction_barrier( From ae2b863aaf06b4bb0dff01a90da8e9e3d3f62738 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:10 -0400 Subject: [PATCH 0476/1352] btrfs: remove the free space inode special cases from the COW paths Free space inodes are never written anymore, so drop the special cases for them in cow_file_range(), fallback_to_cow(), can_nocow_file_extent() and btrfs_finish_one_ordered(). Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/inode.c | 58 +++++++++++++----------------------------------- 1 file changed, 15 insertions(+), 43 deletions(-) diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 840bb85f7ffca2..87eb7bcb74fce5 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -1369,11 +1369,6 @@ static noinline int cow_file_range(struct btrfs_inode *inode, goto out_unlock; } - if (btrfs_is_free_space_inode(inode)) { - ret = -EINVAL; - goto out_unlock; - } - num_bytes = ALIGN(end - start + 1, blocksize); num_bytes = max(blocksize, num_bytes); ASSERT(num_bytes <= btrfs_super_total_bytes(fs_info->super_copy)); @@ -1683,7 +1678,6 @@ static int fallback_to_cow(struct btrfs_inode *inode, struct folio *locked_folio, const u64 start, const u64 end) { - const bool is_space_ino = btrfs_is_free_space_inode(inode); const bool is_reloc_ino = btrfs_is_data_reloc_root(inode->root); const u64 range_bytes = end + 1 - start; struct extent_io_tree *io_tree = &inode->io_tree; @@ -1716,23 +1710,22 @@ static int fallback_to_cow(struct btrfs_inode *inode, * extent_clear_unlock_delalloc()) the bytes_may_use counter of the * data space info, which we incremented in the step above. * - * If we need to fallback to cow and the inode corresponds to a free - * space cache inode or an inode of the data relocation tree, we must - * also increment bytes_may_use of the data space_info for the same - * reason. Space caches and relocated data extents always get a prealloc - * extent for them, however scrub or balance may have set the block - * group that contains that extent to RO mode and therefore force COW - * when starting writeback. + * If we need to fallback to cow and the inode is in the data relocation + * tree, we must also increment bytes_may_use of the data space_info for + * the same reason. Relocated data extents always get a prealloc extent, + * however scrub or balance may have set the block group that contains + * that extent to RO mode and therefore force COW when starting + * writeback. */ btrfs_lock_extent(io_tree, start, end, &cached_state); count = btrfs_count_range_bits(io_tree, &range_start, end, range_bytes, EXTENT_NORESERVE, false, NULL); - if (count > 0 || is_space_ino || is_reloc_ino) { + if (count > 0 || is_reloc_ino) { u64 bytes = count; struct btrfs_fs_info *fs_info = inode->root->fs_info; struct btrfs_space_info *sinfo = fs_info->data_sinfo; - if (is_space_ino || is_reloc_ino) + if (is_reloc_ino) bytes = range_bytes; spin_lock(&sinfo->lock); @@ -1797,7 +1790,6 @@ static int can_nocow_file_extent(struct btrfs_path *path, struct btrfs_inode *inode, struct can_nocow_file_extent_args *args) { - const bool is_freespace_inode = btrfs_is_free_space_inode(inode); struct extent_buffer *leaf = path->nodes[0]; struct btrfs_root *root = inode->root; struct btrfs_file_extent_item *fi; @@ -1810,8 +1802,7 @@ static int can_nocow_file_extent(struct btrfs_path *path, bool nowait = path->nowait; /* If there are pending snapshots for this root, we must do COW. */ - if (args->writeback_path && !is_freespace_inode && - atomic_read(&root->snapshot_force_cow)) + if (args->writeback_path && atomic_read(&root->snapshot_force_cow)) goto out; fi = btrfs_item_ptr(leaf, path->slots[0], struct btrfs_file_extent_item); @@ -1860,7 +1851,6 @@ static int can_nocow_file_extent(struct btrfs_path *path, ret = btrfs_cross_ref_exist(inode, key->offset - args->file_extent.offset, args->file_extent.disk_bytenr, path); - WARN_ON_ONCE(ret > 0 && is_freespace_inode); if (ret != 0) goto out; @@ -1895,7 +1885,6 @@ static int can_nocow_file_extent(struct btrfs_path *path, ret = btrfs_lookup_csums_list(csum_root, io_start, io_start + args->file_extent.num_bytes - 1, NULL, nowait); - WARN_ON_ONCE(ret > 0 && is_freespace_inode); if (ret != 0) goto out; @@ -3221,6 +3210,7 @@ int btrfs_finish_one_ordered(struct btrfs_ordered_extent *ordered_extent) int compress_type = 0; int ret = 0; u64 logical_len = ordered_extent->num_bytes; + u64 unwritten_start; bool truncated = false; bool clear_reserved_extent = true; unsigned int clear_bits = 0; @@ -3382,29 +3372,11 @@ int btrfs_finish_one_ordered(struct btrfs_ordered_extent *ordered_extent) if (ret) btrfs_mark_ordered_extent_error(ordered_extent); - /* - * Drop extent maps for the part of the extent we didn't write. - * - * We have an exception here for the free_space_inode, this is - * because when we do btrfs_get_extent() on the free space inode - * we will search the commit root. If this is a new block group - * we won't find anything, and we will trip over the assert in - * writepage where we do ASSERT(em->block_start != - * EXTENT_MAP_HOLE). - * - * Theoretically we could also skip this for any NOCOW extent as - * we don't mess with the extent map tree in the NOCOW case, but - * for now simply skip this if we are the free space inode. - */ - if (!btrfs_is_free_space_inode(inode)) { - u64 unwritten_start = start; - - if (truncated) - unwritten_start += logical_len; - - btrfs_drop_extent_map_range(inode, unwritten_start, - end, false); - } + /* Drop extent maps for the part of the extent we didn't write. */ + unwritten_start = start; + if (truncated) + unwritten_start += logical_len; + btrfs_drop_extent_map_range(inode, unwritten_start, end, false); /* * If the ordered extent had an IOERR or something else went From 1f7f2df27c958a47bfa833fbac0db16c45cd4143 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:11 -0400 Subject: [PATCH 0477/1352] btrfs: stop special-casing free space inodes in the delalloc accounting Free space inodes never have delalloc or outstanding extents any more, so they don't need to be kept off the root's delalloc inode list or out of the outstanding extents tracepoint. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/btrfs_inode.h | 2 -- fs/btrfs/inode.c | 5 ++--- 2 files changed, 2 insertions(+), 5 deletions(-) diff --git a/fs/btrfs/btrfs_inode.h b/fs/btrfs/btrfs_inode.h index d1c014da52a35a..f7aa0d1451851d 100644 --- a/fs/btrfs/btrfs_inode.h +++ b/fs/btrfs/btrfs_inode.h @@ -409,8 +409,6 @@ static inline void btrfs_mod_outstanding_extents(struct btrfs_inode *inode, { lockdep_assert_held(&inode->lock); inode->outstanding_extents += mod; - if (btrfs_is_free_space_inode(inode)) - return; trace_btrfs_inode_mod_outstanding_extents(inode->root, btrfs_ino(inode), mod, inode->outstanding_extents); } diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 87eb7bcb74fce5..845ad8cdfa41fa 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -2632,7 +2632,7 @@ void btrfs_set_delalloc_extent(struct btrfs_inode *inode, struct extent_state *s * and are therefore protected against concurrent calls of this * function and btrfs_clear_delalloc_extent(). */ - if (!btrfs_is_free_space_inode(inode) && prev_delalloc_bytes == 0) + if (prev_delalloc_bytes == 0) btrfs_add_delalloc_inode(inode); } @@ -2690,7 +2690,6 @@ void btrfs_clear_delalloc_extent(struct btrfs_inode *inode, return; if (!btrfs_is_data_reloc_root(root) && - !btrfs_is_free_space_inode(inode) && !(state->state & EXTENT_NORESERVE) && (bits & EXTENT_CLEAR_DATA_RESV)) btrfs_free_reserved_data_space_noquota(inode, len); @@ -2708,7 +2707,7 @@ void btrfs_clear_delalloc_extent(struct btrfs_inode *inode, * and are therefore protected against concurrent calls of this * function and btrfs_set_delalloc_extent(). */ - if (!btrfs_is_free_space_inode(inode) && new_delalloc_bytes == 0) { + if (new_delalloc_bytes == 0) { spin_lock(&root->delalloc_lock); btrfs_del_delalloc_inode(inode); spin_unlock(&root->delalloc_lock); From 8e56d0dd65313ac4cfb4e1533ff53e6e9c33691a Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:12 -0400 Subject: [PATCH 0478/1352] btrfs: stop reading free space inodes from the commit root Free space inode data was only read when loading the v1 cache, which is gone, so btrfs_get_extent() and btrfs_lookup_bio_sums() no longer need to search the commit root for them. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/file-item.c | 11 ----------- fs/btrfs/inode.c | 10 ---------- 2 files changed, 21 deletions(-) diff --git a/fs/btrfs/file-item.c b/fs/btrfs/file-item.c index ae1fd4da38d31b..ff8f8cad00fc0d 100644 --- a/fs/btrfs/file-item.c +++ b/fs/btrfs/file-item.c @@ -396,17 +396,6 @@ int btrfs_lookup_bio_sums(struct btrfs_bio *bbio) if (nblocks > fs_info->csums_per_leaf) path->reada = READA_FORWARD; - /* - * the free space stuff is only read when it hasn't been - * updated in the current transaction. So, we can safely - * read from the commit root and sidestep a nasty deadlock - * between reading the free space cache and updating the csum tree. - */ - if (btrfs_is_free_space_inode(inode)) { - path->search_commit_root = true; - path->skip_locking = true; - } - /* * If we are searching for a csum of an extent from a past * transaction, we can search in the commit root and reduce diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 845ad8cdfa41fa..568467dcd20ef5 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -7234,16 +7234,6 @@ struct extent_map *btrfs_get_extent(struct btrfs_inode *inode, /* Chances are we'll be called again, so go ahead and do readahead */ path->reada = READA_FORWARD; - /* - * The same explanation in load_free_space_cache applies here as well, - * we only read when we're loading the free space cache, and at that - * point the commit_root has everything we need. - */ - if (btrfs_is_free_space_inode(inode)) { - path->search_commit_root = true; - path->skip_locking = true; - } - ret = btrfs_lookup_file_extent(NULL, root, path, objectid, start, 0); if (ret < 0) { goto out; From f61fb1feb9ce3bf1055c45d3150fcdf9e87d09e0 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Tue, 8 Sep 2026 07:47:39 +0930 Subject: [PATCH 0479/1352] btrfs: use kvmalloc() for b-tree split_item() [BUG] There is a bug report that the kmalloc() call inside split_item() failed with the following call trace, and triggered a transaction abort: kworker/u69:8: page allocation failure: order:4, mode:0x40c40(GFP_NOFS|__GFP_COMP), nodemask=(null) CPU: 3 UID: 0 PID: 1154528 Comm: kworker/u69:8 Not tainted 7.0.2 #1 PREEMPTLAZY Workqueue: events_unbound btrfs_async_reclaim_metadata_space Call Trace: dump_stack_lvl+0x47/0x60 warn_alloc.cold+0x67/0xec __alloc_pages_slowpath.constprop.0+0x9bf/0xed0 __alloc_frozen_pages_noprof+0x1ac/0x1c0 ___kmalloc_large_node+0x9d/0xc0 __kmalloc_noprof+0x17b/0x1f0 split_item+0x9e/0x2e0 btrfs_del_csums+0x285/0x400 __btrfs_free_extent.isra.0+0x6de/0x12b0 __btrfs_run_delayed_refs+0x522/0x10c0 btrfs_run_delayed_refs+0x4d/0x1d0 flush_space+0x34d/0x4e0 do_async_reclaim_metadata_space+0x89/0x1d0 btrfs_async_reclaim_metadata_space+0x44/0x60 process_one_work+0x145/0x230 worker_thread+0x185/0x2e0 kthread+0xca/0x100 ret_from_fork+0x14e/0x200 ret_from_fork_asm+0x11/0x20 BTRFS error (device dm-3 state A): Transaction aborted (error -12) BTRFS: error (device dm-3 state A) in btrfs_del_csums:1053: errno=-12 Out of memory BTRFS info (device dm-3 state EA): forced readonly BTRFS: error (device dm-3 state EA) in do_free_extent_accounting:3168: errno=-12 Out of memory BTRFS error (device dm-3 state EA): failed to run delayed ref for logical 1202913873920 num_bytes 274432 type 184 action 2 ref_mod 1: -12 BTRFS: error (device dm-3 state EA) in btrfs_run_delayed_refs:2247: errno=-12 Out of memory [CAUSE] The kmalloc() call is to allocate a buffer to store the full item. However as shown in the above call trace, the order can be high (4), and since we're using GFP_NOFS, it's impossible to reclaim memory by writing back dirty pages. When there is no physically contiguous memory left, such high order allocation can easily fail, and if such kmalloc() happens in a critical path we can trigger a transaction abort. [FIX] Instead of kmalloc(), which requires physically contiguous pages, use kvmalloc(). There is no special requirement for physically contiguous pages here, we just want virtually contiguous memory as a buffer. Reported-by: xavierbachmeyer182 Link: https://lore.kernel.org/linux-btrfs/250decb0-d940-4fe6-9b54-d06e1b293a1b@suse.com/ Reviewed-by: Johannes Thumshirn Reviewed-by: Daniel Vacek Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/ctree.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/ctree.c b/fs/btrfs/ctree.c index 8fe330d81b8ff3..fef0e49dd91882 100644 --- a/fs/btrfs/ctree.c +++ b/fs/btrfs/ctree.c @@ -3943,7 +3943,7 @@ static noinline int split_item(struct btrfs_trans_handle *trans, orig_offset = btrfs_item_offset(leaf, path->slots[0]); item_size = btrfs_item_size(leaf, path->slots[0]); - buf = kmalloc(item_size, GFP_NOFS); + buf = kvmalloc(item_size, GFP_NOFS); if (!buf) return -ENOMEM; @@ -3981,7 +3981,7 @@ static noinline int split_item(struct btrfs_trans_handle *trans, btrfs_mark_buffer_dirty(trans, leaf); BUG_ON(btrfs_leaf_free_space(leaf) < 0); - kfree(buf); + kvfree(buf); return 0; } From 9693a85a134b7e71deb273f1b9d10c1289e9cad7 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Tue, 8 Sep 2026 16:45:39 +0930 Subject: [PATCH 0480/1352] btrfs: tree-log: use kvmalloc() for overwrite_item() The @src_copy buffer utilized inside overwrite_item() can be as large as the nodesize. For an existing btrfs with 64KiB nodesize, it means there is a high chance to fail the kmalloc() call if there is not enough physically contiguous pages. Meanwhile there is really no need for such physically contiguous pages, as we only use that buffer to compare the content of the item. Use kvmalloc() to replace the kmalloc() call. For most cases that kvmalloc() call will be easily fulfilled by regular kmalloc(), but for really large items and large nodes, kvmalloc() will have a much higher chance to get memory allocated. Reviewed-by: Daniel Vacek Reviewed-by: Johannes Thumshirn Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/tree-log.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/tree-log.c b/fs/btrfs/tree-log.c index a00094604e544e..b4c1fb90d9d0dd 100644 --- a/fs/btrfs/tree-log.c +++ b/fs/btrfs/tree-log.c @@ -503,7 +503,7 @@ static int overwrite_item(struct walk_control *wc) btrfs_release_path(wc->subvol_path); return 0; } - src_copy = kmalloc(item_size, GFP_NOFS); + src_copy = kvmalloc(item_size, GFP_NOFS); if (!src_copy) { btrfs_abort_log_replay(wc, -ENOMEM, "failed to allocate memory for log leaf item"); @@ -514,7 +514,7 @@ static int overwrite_item(struct walk_control *wc) dst_ptr = btrfs_item_ptr_offset(dst_eb, dst_slot); ret = memcmp_extent_buffer(dst_eb, src_copy, dst_ptr, item_size); - kfree(src_copy); + kvfree(src_copy); /* * they have the same contents, just return, this saves * us from cowing blocks in the destination tree and doing From a8b2f051020777081f4a65243dcdc9ca0fad5f62 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Tue, 8 Sep 2026 16:45:40 +0930 Subject: [PATCH 0481/1352] btrfs: use kvmalloc() for uncompress_inline() Although btrfs doesn't support inlined extents larger than PAGE_SIZE for bs > ps cases, it's still possible for the experimental bs > ps support to mount a btrfs created on a system with a much larger page size, thus can still hit an inlined extent that is way larger than the current page size. E.g. a compressed inline extent which has 32K compressed size, is created on 64K page sized ARM64 with 64K sectorsize, then mounted on x86_64 with the experimental bs > ps support. In that case, when reading the compressed inline extent, we need to allocate a buffer that is the same size as the compressed inline extent (32K). That kmalloc() call will request physically contiguous memory for that 32K allocation, and if the system has a very fragmented memory space, such allocation can fail. But there is really no reason that we require such buffer to be physically contiguous, so change it to kvmalloc() to reduce the chance of allocation failure for bs > ps cases. And for all bs <= ps cases, the kvmalloc() call will just be fulfilled by kmalloc() so this will not bring any change to the most common cases. Only bs > ps will get the benefit of less memory allocation failure. Reviewed-by: Daniel Vacek Reviewed-by: Johannes Thumshirn Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/inode.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index e97446fc9e1063..04c45ea9b9c9bf 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -7164,7 +7164,7 @@ static noinline int uncompress_inline(struct btrfs_path *path, compress_type = btrfs_file_extent_compression(leaf, item); max_size = btrfs_file_extent_ram_bytes(leaf, item); inline_size = btrfs_file_extent_inline_item_len(leaf, path->slots[0]); - tmp = kmalloc(inline_size, GFP_NOFS); + tmp = kvmalloc(inline_size, GFP_NOFS); if (!tmp) return -ENOMEM; ptr = btrfs_file_extent_inline_start(item); @@ -7185,7 +7185,7 @@ static noinline int uncompress_inline(struct btrfs_path *path, if (max_size < blocksize) folio_zero_range(folio, max_size, blocksize - max_size); - kfree(tmp); + kvfree(tmp); return ret; } From 9eb9f4e5a184aa3d19d6e842f4fc1479cd0def6a Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Fri, 18 Sep 2026 11:58:00 +0100 Subject: [PATCH 0482/1352] btrfs: clear BTRFS_ROOT_IN_TRANS_SETUP on early exit from record_root_in_trans() If we exit early because the transaction that last used the root already matches the current transaction, we leave the BTRFS_ROOT_IN_TRANS_SETUP bit set in the root (which we just set right before the exit). While this does not cause any functional issue, it makes callers of btrfs_record_root_in_trans() lock fs_info->reloc_mutex and call record_root_in_trans() for nothing, causing unnecessary lock contention, until one of them clears the bit in record_root_in_trans(). One caller of btrfs_record_root_in_trans() is start_transaction(), used to start new transaction or joining an existing one, which is a hot path. So clear BTRFS_ROOT_IN_TRANS_SETUP on early exit. Assisted-by: LLM Reviewed-by: Boris Burkov Reviewed-by: Qu Wenruo Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/transaction.c | 1 + 1 file changed, 1 insertion(+) diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c index 6802b94ed76ff4..11b38133f737e6 100644 --- a/fs/btrfs/transaction.c +++ b/fs/btrfs/transaction.c @@ -432,6 +432,7 @@ static int record_root_in_trans(struct btrfs_trans_handle *trans, spin_lock(&fs_info->fs_roots_radix_lock); if (btrfs_get_root_last_trans(root) == trans->transid && !force) { spin_unlock(&fs_info->fs_roots_radix_lock); + clear_bit(BTRFS_ROOT_IN_TRANS_SETUP, &root->state); return 0; } radix_tree_tag_set(&fs_info->fs_roots_radix, From cb66fd17b1e758aca1c1c076ad1251c8eb88d17e Mon Sep 17 00:00:00 2001 From: Wentao Liang Date: Wed, 16 Sep 2026 17:16:10 +0000 Subject: [PATCH 0483/1352] btrfs: scrub: fix local_root reference leak in scrub_print_warning_inode() When paths_from_inode() fails, scrub_print_warning_inode() jumps to err without dropping the reference taken by btrfs_get_fs_root(), leaking a reference to the root every time path resolution fails while printing scrub warnings. Every other error and success path of the function drops the reference. Drop the reference on the paths_from_inode() failure path too. Fixes: 558540c17771 ("btrfs scrub: print paths of corrupted files") CC: stable@vger.kernel.org Reviewed-by: Qu Wenruo Signed-off-by: Wentao Liang Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/scrub.c | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/fs/btrfs/scrub.c b/fs/btrfs/scrub.c index c09d4213ad8915..479c0e037015ab 100644 --- a/fs/btrfs/scrub.c +++ b/fs/btrfs/scrub.c @@ -536,8 +536,10 @@ static int scrub_print_warning_inode(u64 inum, u64 offset, u64 num_bytes, } ret = paths_from_inode(inum, ipath); - if (ret < 0) + if (ret < 0) { + btrfs_put_root(local_root); goto err; + } /* * we deliberately ignore the bit ipath might have been too small to From e389442d40dcb1b50cbe9f141df46831ae74b867 Mon Sep 17 00:00:00 2001 From: Yang Xiuwei Date: Wed, 19 Aug 2026 10:54:32 +0800 Subject: [PATCH 0484/1352] btrfs: always return -EIOCBQUEUED after btrfs_uring_read_extent_endio If all bios finish before btrfs_encoded_read_regular_fill_pages() returns, it calls btrfs_uring_read_extent_endio() and previously returned the I/O status. A negative errno then made btrfs_uring_read_extent() unlock and free while btrfs_uring_read_finished() did the same again. Return -EIOCBQUEUED so only the deferred path cleans up. Reported-by: Yue Sun Closes: https://lore.kernel.org/linux-btrfs/20260630091609.3414-1-samsun1006219@gmail.com/ Suggested-by: Jens Axboe Fixes: 34310c442e17 ("btrfs: add io_uring command for encoded reads (ENCODED_READ ioctl)") Signed-off-by: Yang Xiuwei Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/inode.c | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 04c45ea9b9c9bf..33ed0bf676564f 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -9623,7 +9623,6 @@ int btrfs_encoded_read_regular_fill_pages(struct btrfs_inode *inode, struct completion sync_reads; unsigned long i = 0; struct btrfs_bio *bbio; - int ret; /* * Fast path for synchronous reads which completes in this call, io_uring @@ -9670,10 +9669,10 @@ int btrfs_encoded_read_regular_fill_pages(struct btrfs_inode *inode, if (uring_ctx) { if (refcount_dec_and_test(&priv->pending_refs)) { - ret = blk_status_to_errno(READ_ONCE(priv->status)); - btrfs_uring_read_extent_endio(uring_ctx, ret); + int error = blk_status_to_errno(READ_ONCE(priv->status)); + + btrfs_uring_read_extent_endio(uring_ctx, error); kfree(priv); - return ret; } return -EIOCBQUEUED; From 882102868b0922e8218176a1b5b6c4bebb46f7ae Mon Sep 17 00:00:00 2001 From: Yang Xiuwei Date: Wed, 19 Aug 2026 10:54:33 +0800 Subject: [PATCH 0485/1352] btrfs: free iov when btrfs_uring_read_extent() fails After btrfs_uring_read_extent(), the caller always jumped to out_acct. That skips kfree(data->iov), which is only correct for -EIOCBQUEUED where the deferred path owns the iov. On failure, fall through to out_free instead. Fixes: 34310c442e17 ("btrfs: add io_uring command for encoded reads (ENCODED_READ ioctl)") Signed-off-by: Yang Xiuwei Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/ioctl.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/ioctl.c b/fs/btrfs/ioctl.c index 21c4e755f9c5db..58eba94a73b89b 100644 --- a/fs/btrfs/ioctl.c +++ b/fs/btrfs/ioctl.c @@ -4865,8 +4865,8 @@ static int btrfs_uring_encoded_read(struct io_uring_cmd *cmd, unsigned int issue cached_state, disk_bytenr, disk_io_size, count, data->args.compression, data->iov, cmd); - - goto out_acct; + if (ret == -EIOCBQUEUED) + goto out_acct; } out_free: From a04b077d0bb1b422128eb2d4a3c8f275f5d2e84c Mon Sep 17 00:00:00 2001 From: Yang Xiuwei Date: Wed, 19 Aug 2026 10:54:34 +0800 Subject: [PATCH 0486/1352] btrfs: unlock inode and extent in caller when io_uring read extent fails btrfs_uring_read_extent() runs only after btrfs_encoded_read() has taken the inode shared lock and the extent lock. On failure it used to unlock in out_fail, and a pages-array allocation failure returned -ENOMEM without unlocking at all. Unlock in the caller instead on all failure returns, matching the copy_to_user() error path. The deferred -EIOCBQUEUED path still unlocks in btrfs_uring_read_finished(). Fixes: 34310c442e17 ("btrfs: add io_uring command for encoded reads (ENCODED_READ ioctl)") Suggested-by: Qu Wenruo Reviewed-by: Qu Wenruo Signed-off-by: Yang Xiuwei Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/ioctl.c | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/fs/btrfs/ioctl.c b/fs/btrfs/ioctl.c index 58eba94a73b89b..131373bf1e30a9 100644 --- a/fs/btrfs/ioctl.c +++ b/fs/btrfs/ioctl.c @@ -4601,7 +4601,7 @@ static void btrfs_uring_read_finished(struct io_tw_req tw_req, io_tw_token_t tw) size_t page_offset; ssize_t ret; - /* The inode lock has already been acquired in btrfs_uring_read_extent. */ + /* The inode lock has already been acquired in btrfs_encoded_read(). */ btrfs_lockdep_inode_acquire(inode, i_rwsem); if (priv->err) { @@ -4667,7 +4667,6 @@ static int btrfs_uring_read_extent(struct kiocb *iocb, struct iov_iter *iter, struct iovec *iov, struct io_uring_cmd *cmd) { struct btrfs_inode *inode = BTRFS_I(file_inode(iocb->ki_filp)); - struct extent_io_tree *io_tree = &inode->io_tree; struct page **pages = NULL; struct btrfs_uring_priv *priv = NULL; unsigned long nr_pages; @@ -4723,8 +4722,6 @@ static int btrfs_uring_read_extent(struct kiocb *iocb, struct iov_iter *iter, return -EIOCBQUEUED; out_fail: - btrfs_unlock_extent(io_tree, start, lockend, &cached_state); - btrfs_inode_unlock(inode, BTRFS_ILOCK_SHARED); kfree(priv); for (int i = 0; i < nr_pages; i++) { if (pages[i]) @@ -4867,6 +4864,8 @@ static int btrfs_uring_encoded_read(struct io_uring_cmd *cmd, unsigned int issue data->iov, cmd); if (ret == -EIOCBQUEUED) goto out_acct; + btrfs_unlock_extent(io_tree, start, lockend, &cached_state); + btrfs_inode_unlock(inode, BTRFS_ILOCK_SHARED); } out_free: From f2ef04a338a4bb85445d22e192e1f396004d4118 Mon Sep 17 00:00:00 2001 From: Yang Xiuwei Date: Wed, 19 Aug 2026 10:54:35 +0800 Subject: [PATCH 0487/1352] btrfs: don't stash io_uring encoded data across -EAGAIN Returning -EAGAIN while leaving btrfs_uring_encoded_data in the cmd PDU leaks if the request is cancelled or the ring exits before reissue. io_uring does not free driver PDU allocations on cleanup. Write: io_queue_sqe() always issues with IO_URING_F_NONBLOCK first, so return -EAGAIN before allocating and free data on every exit. Read: free on nowait -EAGAIN too; only -EIOCBQUEUED keeps the allocation for btrfs_uring_read_finished(). Fixes: 34310c442e17 ("btrfs: add io_uring command for encoded reads (ENCODED_READ ioctl)") Fixes: e32dcdb0af9f ("btrfs: add io_uring interface for encoded writes") Signed-off-by: Yang Xiuwei Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/ioctl.c | 20 +++++++++++--------- 1 file changed, 11 insertions(+), 9 deletions(-) diff --git a/fs/btrfs/ioctl.c b/fs/btrfs/ioctl.c index 131373bf1e30a9..44d582d8cf3f09 100644 --- a/fs/btrfs/ioctl.c +++ b/fs/btrfs/ioctl.c @@ -4834,7 +4834,7 @@ static int btrfs_uring_encoded_read(struct io_uring_cmd *cmd, unsigned int issue ret = btrfs_encoded_read(&kiocb, &data->iter, &data->args, &cached_state, &disk_bytenr, &disk_io_size); if (ret == -EAGAIN) - goto out_acct; + goto out_free; if (ret < 0 && ret != -EIOCBQUEUED) goto out_free; @@ -4876,8 +4876,10 @@ static int btrfs_uring_encoded_read(struct io_uring_cmd *cmd, unsigned int issue add_rchar(current, ret); inc_syscr(current); - if (ret != -EIOCBQUEUED && ret != -EAGAIN) + if (ret != -EIOCBQUEUED) { kfree(data); + bc->data = NULL; + } return ret; } @@ -4906,6 +4908,11 @@ static int btrfs_uring_encoded_write(struct io_uring_cmd *cmd, unsigned int issu goto out_acct; } + if (issue_flags & IO_URING_F_NONBLOCK) { + ret = -EAGAIN; + goto out_acct; + } + if (!data) { data = kzalloc_obj(*data, GFP_NOFS); if (!data) { @@ -4974,11 +4981,6 @@ static int btrfs_uring_encoded_write(struct io_uring_cmd *cmd, unsigned int issu } } - if (issue_flags & IO_URING_F_NONBLOCK) { - ret = -EAGAIN; - goto out_acct; - } - pos = data->args.offset; ret = rw_verify_area(WRITE, file, &pos, data->args.len); if (ret < 0) @@ -5004,8 +5006,8 @@ static int btrfs_uring_encoded_write(struct io_uring_cmd *cmd, unsigned int issu add_wchar(current, ret); inc_syscw(current); - if (ret != -EAGAIN) - kfree(data); + kfree(data); + bc->data = NULL; return ret; } From 823b8041984fa4964dc6f895b9c3fcca30a3ca97 Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Fri, 25 Sep 2026 13:47:21 +0100 Subject: [PATCH 0488/1352] btrfs: fix xattr replace when multiple xattrs are packed in the same item If we have a btrfs_dir_item item that packs multiple xattrs and then we replace the value of one of them (with the setxattr(2) family of syscalls) with another value of a different size, we end up not having a fully initialized btrfs_dir_item, resulting in a corruption that the tree checker will detect at extent buffer writeback time. This is because in btrfs_setxattr() when we find a btrfs_dir_item with multiple xattrs (due to the crc32c hash of their name being the same) we delete one of the xattr items (btrfs_dir_item) and then insert a new one, but the deletion and insertion results in shifting existing data in the leaf and therefore when the new value of a xattr has a different size, the new btrfs_dir_item is placed in a leaf section that was not initialized and we only copy the value's data and set the value's length in the new btrfs_dir_item, without setting the name, the name's length, the key (which must be all zeroes for xattrs), flags (BTRFS_FT_XATTR) and transaction ID. The following script reproduces the issue: $ cat test.sh #!/bin/bash DEV=/dev/sdi MNT=/mnt/sdi mkfs.btrfs -f $DEV mount $DEV $MNT touch $MNT/testfile # Add two xattrs that, on btrfs, have the same hash (crc32c) for their # name and therefore are packed into the same btrfs_dir_item. setfattr -n user.foobar -v 123 $MNT/testfile setfattr -n user.WvG1c1Td -v qwerty $MNT/testfile # Verify the xattrs are present. echo "xattrs before:" getfattr --absolute-names --dump $MNT/testfile # Now replace the value of the foobar xattr with a significantly larger # value. setfattr -n user.foobar -v abcdefghijklmnopqrstuvwxyz $MNT/testfile # Check the xattrs have the expected values. echo "xattrs after:" getfattr --absolute-names --dump $MNT/testfile umount $MNT Running it: $ ./test.sh (...) xattrs before: # file: /mnt/sdi/testfile user.WvG1c1Td="qwerty" user.foobar="123" xattrs after: # file: /mnt/sdi/testfile user.WvG1c1Td="qwerty" So the "user.foobar" xattr is missing and there was a transaction abort when unmounting the fs with the following traces in dmesg: $ dmesg [869800.159271] BTRFS warning (device sdi): access to eb bytenr 30474240 len 16384 out of range start 16015 len 25964 [869800.159293] ------------[ cut here ]------------ [869800.159296] WARNING: fs/btrfs/extent_io.c:4408 at report_eb_range+0x44/0x60 [btrfs], CPU#8: getfattr/2605179 [869800.168961] Modules linked in: btrfs dm_thin_pool (...) [869800.190527] CPU: 8 UID: 0 PID: 2605179 Comm: getfattr Tainted: G W 7.3.0-rc3-btrfs-next-244+ #1 PREEMPT(full) [869800.193726] Tainted: [W]=WARN [869800.194502] Hardware name: QEMU Standard PC (i440FX + PIIX, 1996), BIOS rel-1.16.2-0-gea1b7a073390-prebuilt.qemu.org 04/01/2014 [869800.197576] RIP: 0010:report_eb_range+0x44/0x60 [btrfs] [869800.198985] Code: 48 8b 7b 18 (...) [869800.203531] RSP: 0018:ffffce4541937d58 EFLAGS: 00010246 [869800.204609] RAX: 0000000000000000 RBX: ffff8dde054b8738 RCX: 0000000000000000 [869800.206137] RDX: 0000000000000000 RSI: 0000000000000001 RDI: ffffffffc04c92a0 [869800.207645] RBP: 0000000000003e8f R08: 0000000000000000 R09: 3fffffffffefffff [869800.226175] R10: ffffce4541937a88 R11: 0000000000000003 R12: 000000000000656c [869800.227594] R13: ffff8dde14be800f R14: 000000000000656c R15: 0000000000000069 [869800.229097] FS: 00007f59023bb780(0000) GS:ffff8de5788ed000(0000) knlGS:0000000000000000 [869800.231137] CS: 0010 DS: 0000 ES: 0000 CR0: 0000000080050033 [869800.232321] CR2: 0000558dc45e6a78 CR3: 0000000765b16004 CR4: 0000000000370ef0 [869800.233783] Call Trace: [869800.234330] [869800.234791] read_extent_buffer+0x4d/0x100 [btrfs] [869800.235898] btrfs_listxattr+0x199/0x240 [btrfs] [869800.236912] vfs_listxattr+0x51/0xa0 [869800.237683] listxattr+0x7e/0x100 [869800.238398] path_listxattrat+0x9e/0x190 [869800.239117] do_syscall_64+0x89/0x470 [869800.239877] entry_SYSCALL_64_after_hwframe+0x76/0x7e [869800.240914] RIP: 0033:0x7f59024cdcb7 [869800.241686] Code: f0 ff ff 73 (...) [869800.245353] RSP: 002b:00007ffd056dcdb8 EFLAGS: 00000246 ORIG_RAX: 00000000000000c2 [869800.246892] RAX: ffffffffffffffda RBX: 00007ffd056df2e2 RCX: 00007f59024cdcb7 [869800.248853] RDX: 0000000000006600 RSI: 0000558dc45e0470 RDI: 00007ffd056df2e2 [869800.250514] RBP: 00007ffd056df2e2 R08: 0000000000006600 R09: 0000000000006600 [869800.252295] R10: 0000000000000004 R11: 0000000000000246 R12: 00000000ffffff9c [869800.254105] R13: 0000558dc45e0470 R14: 0000000000006600 R15: 0000000000000000 [869800.255931] [869800.256529] ---[ end trace 0000000000000000 ]--- [869800.260071] page: refcount:2 mapcount:0 mapping:000000007ccfc77f index:0x1d10 pfn:0x608c69 [869800.260075] memcg:ffff8dde00344d40 [869800.260076] aops:btree_aops [btrfs] ino:1 [869800.260151] flags: 0x17fffc00000402a(uptodate|lru|private|writeback|node=0|zone=2|lastcpupid=0x1ffff) [869800.260154] raw: 017fffc00000402a fffff4adc773d4c8 fffff4adc48f3d88 ffff8de36504ba30 [869800.260155] raw: 0000000000001d10 ffff8dde054b8738 00000002ffffffff ffff8dde00344d40 [869800.260156] page dumped because: eb page dump [869800.260157] BTRFS critical (device sdi): corrupt leaf: root=5 block=30474240 slot=5 ino=257, invalid location key type, have 46, expect 132 or 1 [869800.260161] BTRFS info (device sdi): leaf 30474240 gen 9 total ptrs 6 free space 15629 owner 5 [869800.260163] BTRFS info (device sdi): refs 3 lock_owner 0 current 2550294 [869800.260164] item 0 key (256 INODE_ITEM 0) itemoff 16123 itemsize 160 [869800.260165] inode generation 3 transid 0 size 0 nbytes 16384 [869800.260166] block group 0 mode 40755 links 1 uid 0 gid 0 [869800.260167] rdev 0 sequence 0 flags 0x0 [869800.260168] atime 1790340637.0 [869800.260169] ctime 1790340637.0 [869800.260169] mtime 1790340637.0 [869800.260170] otime 1790340637.0 [869800.260170] item 1 key (256 INODE_REF 256) itemoff 16111 itemsize 12 [869800.260172] index 0 name_len 2 [869800.260172] item 2 key (256 DIR_ITEM 982728850) itemoff 16073 itemsize 38 [869800.260173] location key (257 1 0) type 1 [869800.260174] transid 9 data_len 0 name_len 8 [869800.260175] item 3 key (257 INODE_ITEM 0) itemoff 15913 itemsize 160 [869800.260176] inode generation 9 transid 9 size 0 nbytes 0 [869800.260177] block group 0 mode 100664 links 1 uid 0 gid 0 [869800.260177] rdev 0 sequence 0 flags 0x0 [869800.260178] atime 1790340638.38778502 [869800.260179] ctime 1790340638.38778502 [869800.260179] mtime 1790340638.38778502 [869800.260180] otime 1790340638.38778502 [869800.260180] item 4 key (257 INODE_REF 256) itemoff 15895 itemsize 18 [869800.260181] index 2 name_len 8 [869800.260182] item 5 key (257 XATTR_ITEM 751495445) itemoff 15779 itemsize 116 [869800.260183] location key (0 0 0) type 8 [869800.266052] transid 9 data_len 6 name_len 13 [869800.266053] location key (8243121639454149888 46 7229457603934778967) type 0 [869800.266055] transid 113 data_len 26 name_len 0 [869800.266056] location key (8608196880778817904 120 162425) type 9 [869800.266057] transid 8391162079612502016 data_len 26982 name_len 25964 [869800.266058] BTRFS error (device sdi): block=30474240 write time tree block corruption detected [869800.266091] ------------[ cut here ]------------ [869800.266092] WARNING: fs/btrfs/disk-io.c:336 at btree_csum_one_bio+0x20b/0x220 [btrfs], CPU#7: kworker/u50:7/2550294 [869800.268392] Modules linked in: btrfs dm_thin_pool (...) [869800.365851] CPU: 7 UID: 0 PID: 2550294 Comm: kworker/u50:7 Tainted: G W 7.3.0-rc3-btrfs-next-244+ #1 PREEMPT(full) [869800.368937] Tainted: [W]=WARN [869800.369834] Hardware name: QEMU Standard PC (i440FX + PIIX, 1996), BIOS rel-1.16.2-0-gea1b7a073390-prebuilt.qemu.org 04/01/2014 [869800.372778] Workqueue: writeback wb_workfn (flush-btrfs-3821) [869800.374314] RIP: 0010:btree_csum_one_bio+0x20b/0x220 [btrfs] [869800.375915] Code: 89 44 24 04 (...) [869800.380639] RSP: 0018:ffffce4548e3f7d0 EFLAGS: 00010246 [869800.382008] RAX: 0000000000000000 RBX: ffff8dde054b8738 RCX: 0000000000000000 [869800.383850] RDX: 0000000000000000 RSI: 0000000000000001 RDI: ffff8de091c2ddc0 [869800.385710] RBP: ffff8dde196a2000 R08: 0000000000000000 R09: 3fffffffffefffff [869800.387388] R10: ffffce4548e3f500 R11: 0000000000000003 R12: ffffce4548e3f7d8 [869800.388973] R13: ffff8dde196a2000 R14: ffff8de36504b750 R15: ffff8dde4c497b00 [869800.390414] FS: 0000000000000000(0000) GS:ffff8de5788ad000(0000) knlGS:0000000000000000 [869800.392002] CS: 0010 DS: 0000 ES: 0000 CR0: 0000000080050033 [869800.393160] CR2: 000055cd6e92ad5c CR3: 00000007cb264001 CR4: 0000000000370ef0 [869800.394595] Call Trace: [869800.395113] [869800.395570] btrfs_submit_bbio+0x872/0x890 [btrfs] [869800.397284] write_meta_extent_buffer+0x70/0x80 [btrfs] [869800.398940] btree_writepages+0x141/0x4f0 [btrfs] [869800.400426] ? get_random_u32+0x8a/0xf0 [869800.401417] ? build_slab_freelist+0x47/0x130 [869800.402574] ? preempt_count_add+0x6b/0xa0 [869800.403633] ? _raw_spin_lock_irqsave+0x23/0x50 [869800.404807] ? _raw_spin_unlock_irqrestore+0x22/0x40 [869800.406085] ? alloc_from_new_slab+0x18f/0x330 [869800.407223] do_writepages+0xc6/0x160 [869800.408191] ? refill_objects+0xd8/0x300 [869800.409211] __writeback_single_inode+0x42/0x350 [869800.410408] writeback_sb_inodes+0x231/0x560 [869800.411511] wb_writeback+0x8a/0x300 [869800.412440] wb_workfn+0xbf/0x460 [869800.413291] ? _raw_spin_unlock+0x14/0x30 [869800.414328] ? finish_task_switch.isra.0+0xb9/0x380 [869800.415105] process_one_work+0x1d1/0x3d0 [869800.416633] worker_thread+0x1c4/0x330 [869800.417467] ? __pfx_worker_thread+0x10/0x10 [869800.418452] kthread+0xfc/0x130 [869800.419257] ? __pfx_kthread+0x10/0x10 [869800.420089] ret_from_fork+0x1f7/0x2c0 [869800.420863] ? __pfx_kthread+0x10/0x10 [869800.421654] ret_from_fork_asm+0x1a/0x30 [869800.422484] [869800.422953] ---[ end trace 0000000000000000 ]--- [869800.424005] BTRFS error (device sdi state A): Transaction 9 aborted (-EIO) [869800.424010] BTRFS: error (device sdi state A) in __btrfs_run_delayed_items:1162: errno=-5 IO failure [869800.424011] BTRFS info (device sdi state EA): forced readonly [869800.424013] BTRFS warning (device sdi state EA): Skipping commit of aborted transaction. [869800.424014] BTRFS: error (device sdi state EA) in cleanup_transaction:2076: errno=-5 IO failure Fix this by always setting all fields in the new btrfs_dir_item when we replace an existing xattr. Fixes: 5f5bc6b1e2d5 ("Btrfs: make xattr replace operations atomic") Reviewed-by: Qu Wenruo Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/xattr.c | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/fs/btrfs/xattr.c b/fs/btrfs/xattr.c index ab55d10bd71fdb..e8762e12f3e1e5 100644 --- a/fs/btrfs/xattr.c +++ b/fs/btrfs/xattr.c @@ -161,6 +161,7 @@ int btrfs_setxattr(struct btrfs_trans_handle *trans, struct inode *inode, const u16 old_data_len = btrfs_dir_data_len(leaf, di); const u32 item_size = btrfs_item_size(leaf, slot); const u32 data_size = sizeof(*di) + name_len + size; + unsigned long name_ptr; unsigned long data_ptr; char *ptr; @@ -189,8 +190,16 @@ int btrfs_setxattr(struct btrfs_trans_handle *trans, struct inode *inode, ptr = btrfs_item_ptr(leaf, slot, char); ptr += btrfs_item_size(leaf, slot) - data_size; di = (struct btrfs_dir_item *)ptr; + memzero_extent_buffer(leaf, (unsigned long)ptr + + offsetof(struct btrfs_dir_item, location), + sizeof(struct btrfs_disk_key)); + btrfs_set_dir_flags(leaf, di, BTRFS_FT_XATTR); + btrfs_set_dir_transid(leaf, di, trans->transid); + btrfs_set_dir_name_len(leaf, di, name_len); btrfs_set_dir_data_len(leaf, di, size); + name_ptr = (unsigned long)(di + 1); data_ptr = ((unsigned long)(di + 1)) + name_len; + write_extent_buffer(leaf, name, name_ptr, name_len); write_extent_buffer(leaf, value, data_ptr, size); } else { /* From 1281d7e2f6fc82aa3b5b4fbc8b7d4d2f56471109 Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Fri, 25 Sep 2026 17:12:45 +0100 Subject: [PATCH 0489/1352] btrfs: fix lost error return value in btrfs_listxattr() If the input buffer does not have enough space to store the current xattr, we set 'iter_ret' to -ERANGE and then do "break", but that only exits the while loop over the xattrs in the current btrfs_dir_item, and then we continue the btrfs_for_each_slot() iteration, which overwrites the value of 'iter_ret' causing us to lose the error return value and proceed as if the buffer has enough space. Fix this by returning -ERANGE directly (the path is automatically freed) instead of breaking from the while loop. Fixes: 184b3d190087 ("btrfs: use btrfs_for_each_slot in btrfs_listxattr") Assisted-by: LLM (found the bug) Reviewed-by: Qu Wenruo Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/xattr.c | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/fs/btrfs/xattr.c b/fs/btrfs/xattr.c index e8762e12f3e1e5..cb001a211f5575 100644 --- a/fs/btrfs/xattr.c +++ b/fs/btrfs/xattr.c @@ -329,10 +329,8 @@ ssize_t btrfs_listxattr(struct dentry *dentry, char *buffer, size_t size) if (!size) goto next; - if (!buffer || (name_len + 1) > size_left) { - iter_ret = -ERANGE; - break; - } + if (!buffer || (name_len + 1) > size_left) + return -ERANGE; read_extent_buffer(leaf, buffer, name_ptr, name_len); buffer[name_len] = '\0'; From 8ef8e9839a23ca7f5dda6a69b670633de5a23716 Mon Sep 17 00:00:00 2001 From: Liu Dalin Date: Thu, 27 Aug 2026 16:03:20 +0800 Subject: [PATCH 0490/1352] rtc: ftrtc010: fix integer overflow in time calculation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit In ftrtc010_rtc_read_time() and ftrtc010_rtc_set_time(), all variables (offset, days, hour, min, sec) are u32. The expression "days * 86400 + hour * 3600 + min * 60 + sec" is evaluated in 32-bit arithmetic, which silently wraps around for dates beyond approximately year 2106 (U32_MAX / 86400 ≈ 49710 days). In ftrtc010_rtc_read_time(), perform the addition in 32-bit arithmetic first to preserve two's complement wrapping for negative offsets, then cast the result to timeu64_t. In ftrtc010_rtc_set_time(), cast individual u32 operands to timeu64_t before multiplication to prevent overflow in the right-hand side of the offset calculation. Fixes: 1d61d2592c1f ("rtc: ftrtc010: Rename to Faraday FTRTC010") Assisted-by: Sashiko AI [static analysis] Signed-off-by: Liu Dalin Reviewed-by: Linus Walleij Tested-by: Liu Dalin Link: https://patch.msgid.link/42C8FBD59AA34EA9+20260827080320.3351155-4-liudalin@kylinsec.com.cn Signed-off-by: Alexandre Belloni --- drivers/rtc/rtc-ftrtc010.c | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/drivers/rtc/rtc-ftrtc010.c b/drivers/rtc/rtc-ftrtc010.c index 19e6333bfbab8f..b29c96be40f4a2 100644 --- a/drivers/rtc/rtc-ftrtc010.c +++ b/drivers/rtc/rtc-ftrtc010.c @@ -72,7 +72,7 @@ static int ftrtc010_rtc_read_time(struct device *dev, struct rtc_time *tm) days = readl(rtc->rtc_base + FTRTC010_RTC_DAYS); offset = readl(rtc->rtc_base + FTRTC010_RTC_RECORD); - time = offset + days * 86400 + hour * 3600 + min * 60 + sec; + time = (timeu64_t)(offset + days * 86400 + hour * 3600 + min * 60 + sec); rtc_time64_to_tm(time, tm); @@ -92,7 +92,8 @@ static int ftrtc010_rtc_set_time(struct device *dev, struct rtc_time *tm) hour = readl(rtc->rtc_base + FTRTC010_RTC_HOUR); day = readl(rtc->rtc_base + FTRTC010_RTC_DAYS); - offset = time - (day * 86400 + hour * 3600 + min * 60 + sec); + offset = time - ((timeu64_t)day * 86400 + (timeu64_t)hour * 3600 + + (timeu64_t)min * 60 + (timeu64_t)sec); writel(offset, rtc->rtc_base + FTRTC010_RTC_RECORD); writel(0x01, rtc->rtc_base + FTRTC010_RTC_CR); From 042b855e19f336c8526c4cf2a107c5279735a4a0 Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Wed, 30 Sep 2026 17:59:57 +0300 Subject: [PATCH 0491/1352] drm/intel: move i915_gtt_view_types.h to include/drm/intel MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The i915 and xe drivers share i915_gtt_view_types.h from i915 source. Move it to include/drm/intel/gtt_view_types.h. Remove the i915 compat header. v2: Keep comment for header guard #endif (Ville) Reviewed-by: Maarten Lankhorst Reviewed-by: Ville Syrjälä Link: https://patch.msgid.link/174f7e26be755b29a1286bf8b8ef0342dcc083a6.1790780340.git.jani.nikula@intel.com Signed-off-by: Jani Nikula --- drivers/gpu/drm/i915/display/intel_display_types.h | 2 +- drivers/gpu/drm/i915/i915_vma_types.h | 3 +-- .../gpu/drm/xe/compat-i915-headers/i915_gtt_view_types.h | 7 ------- drivers/gpu/drm/xe/display/xe_fb_pin.c | 4 +--- .../drm/intel/gtt_view_types.h | 6 +++--- 5 files changed, 6 insertions(+), 16 deletions(-) delete mode 100644 drivers/gpu/drm/xe/compat-i915-headers/i915_gtt_view_types.h rename drivers/gpu/drm/i915/i915_gtt_view_types.h => include/drm/intel/gtt_view_types.h (92%) diff --git a/drivers/gpu/drm/i915/display/intel_display_types.h b/drivers/gpu/drm/i915/display/intel_display_types.h index 79f30660c2b69b..fe3b6bec4817d8 100644 --- a/drivers/gpu/drm/i915/display/intel_display_types.h +++ b/drivers/gpu/drm/i915/display/intel_display_types.h @@ -41,10 +41,10 @@ #include #include #include +#include #include #include -#include "i915_gtt_view_types.h" #include "intel_bios.h" #include "intel_display.h" #include "intel_display_conversion.h" diff --git a/drivers/gpu/drm/i915/i915_vma_types.h b/drivers/gpu/drm/i915/i915_vma_types.h index a499a3bea87402..83fe02833b5e5c 100644 --- a/drivers/gpu/drm/i915/i915_vma_types.h +++ b/drivers/gpu/drm/i915/i915_vma_types.h @@ -29,11 +29,10 @@ #include #include +#include #include "gem/i915_gem_object_types.h" -#include "i915_gtt_view_types.h" - /** * DOC: Global GTT views * diff --git a/drivers/gpu/drm/xe/compat-i915-headers/i915_gtt_view_types.h b/drivers/gpu/drm/xe/compat-i915-headers/i915_gtt_view_types.h deleted file mode 100644 index b261910cd6f94c..00000000000000 --- a/drivers/gpu/drm/xe/compat-i915-headers/i915_gtt_view_types.h +++ /dev/null @@ -1,7 +0,0 @@ -/* SPDX-License-Identifier: MIT */ -/* Copyright © 2025 Intel Corporation */ - -#include "../../i915/i915_gtt_view_types.h" - -/* Partial view not supported in xe, fail build if used. */ -#define I915_GTT_VIEW_PARTIAL diff --git a/drivers/gpu/drm/xe/display/xe_fb_pin.c b/drivers/gpu/drm/xe/display/xe_fb_pin.c index 73469ea5f333d3..ce2601069e7d05 100644 --- a/drivers/gpu/drm/xe/display/xe_fb_pin.c +++ b/drivers/gpu/drm/xe/display/xe_fb_pin.c @@ -4,11 +4,9 @@ */ #include +#include #include -/* FIXME move the types to parent interface? */ -#include "i915_gtt_view_types.h" - /* FIXME move intel_remapped_info_size() & co. to parent interface? */ #include "intel_fb.h" diff --git a/drivers/gpu/drm/i915/i915_gtt_view_types.h b/include/drm/intel/gtt_view_types.h similarity index 92% rename from drivers/gpu/drm/i915/i915_gtt_view_types.h rename to include/drm/intel/gtt_view_types.h index 9c4f38db32ffaa..5770fc53b8236f 100644 --- a/drivers/gpu/drm/i915/i915_gtt_view_types.h +++ b/include/drm/intel/gtt_view_types.h @@ -1,8 +1,8 @@ /* SPDX-License-Identifier: MIT */ /* Copyright © 2025 Intel Corporation */ -#ifndef __I915_GTT_VIEW_TYPES_H__ -#define __I915_GTT_VIEW_TYPES_H__ +#ifndef __DRM_INTEL_GTT_VIEW_TYPES_H__ +#define __DRM_INTEL_GTT_VIEW_TYPES_H__ #include @@ -71,4 +71,4 @@ static inline bool i915_gtt_view_is_rotated(const struct i915_gtt_view *view) return view->type == I915_GTT_VIEW_ROTATED; } -#endif /* __I915_GTT_VIEW_TYPES_H__ */ +#endif /* __DRM_INTEL_GTT_VIEW_TYPES_H__ */ From fe05df7a540aaf940d294cad137b0ae8bb578387 Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Wed, 30 Sep 2026 17:59:58 +0300 Subject: [PATCH 0492/1352] drm/intel: rename i915_gtt_view_is_*() helpers to intel_gtt_view_is_*() MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Make the i915_gtt_view_is_*() helpers less i915 specific, and rename them intel_gtt_view_is_*(). $ sed -i 's/i915_gtt_view_is_/intel_gtt_view_is_/g' -- $(git grep -l i915_gtt_view_is_) Reviewed-by: Maarten Lankhorst Reviewed-by: Ville Syrjälä Link: https://patch.msgid.link/8230339be1738a1f7c87047f2516dfa66b03d9aa.1790780340.git.jani.nikula@intel.com Signed-off-by: Jani Nikula --- drivers/gpu/drm/i915/display/intel_fb.c | 8 ++++---- include/drm/intel/gtt_view_types.h | 6 +++--- 2 files changed, 7 insertions(+), 7 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_fb.c b/drivers/gpu/drm/i915/display/intel_fb.c index c3e4c3f6d8cf01..63b7644c565907 100644 --- a/drivers/gpu/drm/i915/display/intel_fb.c +++ b/drivers/gpu/drm/i915/display/intel_fb.c @@ -1289,7 +1289,7 @@ bool intel_plane_uses_fence(const struct intel_plane_state *plane_state) return intel_plane_needs_fence(display) || (plane->fbc && !plane_state->no_fbc_reason && - i915_gtt_view_is_normal(&plane_state->view.gtt)); + intel_gtt_view_is_normal(&plane_state->view.gtt)); } static int intel_fb_pitch(const struct intel_framebuffer *fb, int color_plane, unsigned int rotation) @@ -1511,7 +1511,7 @@ static u32 calc_plane_remap_info(const struct intel_framebuffer *fb, int color_p plane_view_height_tiles(fb, color_plane, dims, y)); } - if (i915_gtt_view_is_rotated(&view->gtt)) { + if (intel_gtt_view_is_rotated(&view->gtt)) { drm_WARN_ON(display->drm, remap_info->linear); check_array_bounds(display, view->gtt.rotated.plane, color_plane); @@ -1536,7 +1536,7 @@ static u32 calc_plane_remap_info(const struct intel_framebuffer *fb, int color_p /* rotate the tile dimensions to match the GTT view */ swap(tile_width, tile_height); } else { - drm_WARN_ON(display->drm, !i915_gtt_view_is_remapped(&view->gtt)); + drm_WARN_ON(display->drm, !intel_gtt_view_is_remapped(&view->gtt)); check_array_bounds(display, view->gtt.remapped.plane, color_plane); @@ -1638,7 +1638,7 @@ static void intel_fb_view_init(struct intel_display *display, memset(view, 0, sizeof(*view)); view->gtt.type = view_type; - if (i915_gtt_view_is_remapped(&view->gtt) && + if (intel_gtt_view_is_remapped(&view->gtt) && intel_fb_needs_pot_stride_remap(fb)) view->gtt.remapped.plane_alignment = SZ_2M / PAGE_SIZE; } diff --git a/include/drm/intel/gtt_view_types.h b/include/drm/intel/gtt_view_types.h index 5770fc53b8236f..61d2289a8921e6 100644 --- a/include/drm/intel/gtt_view_types.h +++ b/include/drm/intel/gtt_view_types.h @@ -56,17 +56,17 @@ struct i915_gtt_view { }; }; -static inline bool i915_gtt_view_is_normal(const struct i915_gtt_view *view) +static inline bool intel_gtt_view_is_normal(const struct i915_gtt_view *view) { return view->type == I915_GTT_VIEW_NORMAL; } -static inline bool i915_gtt_view_is_remapped(const struct i915_gtt_view *view) +static inline bool intel_gtt_view_is_remapped(const struct i915_gtt_view *view) { return view->type == I915_GTT_VIEW_REMAPPED; } -static inline bool i915_gtt_view_is_rotated(const struct i915_gtt_view *view) +static inline bool intel_gtt_view_is_rotated(const struct i915_gtt_view *view) { return view->type == I915_GTT_VIEW_ROTATED; } From 468a755f203ea2f364452f7f1e009ffe8484a6a3 Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Wed, 30 Sep 2026 17:59:59 +0300 Subject: [PATCH 0493/1352] drm/intel: rename i915_gtt_view* struct/enum to intel_gtt_view* MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Make enum i915_gtt_view_type and struct i915_gtt_view less i915 specific, and rename them enum intel_gtt_view_type and struct intel_gtt_view, respectively. $ sed -i 's/i915_gtt_view/intel_gtt_view/g' -- $(git grep -l i915_gtt_view) Reviewed-by: Maarten Lankhorst Reviewed-by: Ville Syrjälä Link: https://patch.msgid.link/4ca4ad3d11b0f8167a54d69d6eef0c8a78218c53.1790780340.git.jani.nikula@intel.com Signed-off-by: Jani Nikula --- .../gpu/drm/i915/display/intel_display_types.h | 2 +- drivers/gpu/drm/i915/display/intel_fb.c | 2 +- drivers/gpu/drm/i915/display/intel_parent.c | 4 ++-- drivers/gpu/drm/i915/display/intel_parent.h | 6 +++--- drivers/gpu/drm/i915/gem/i915_gem_domain.c | 2 +- drivers/gpu/drm/i915/gem/i915_gem_mman.c | 6 +++--- drivers/gpu/drm/i915/gem/i915_gem_object.h | 2 +- .../gpu/drm/i915/gem/selftests/i915_gem_mman.c | 4 ++-- drivers/gpu/drm/i915/i915_gem.c | 4 ++-- drivers/gpu/drm/i915/i915_gem.h | 6 +++--- drivers/gpu/drm/i915/i915_vma.c | 8 ++++---- drivers/gpu/drm/i915/i915_vma.h | 4 ++-- drivers/gpu/drm/i915/i915_vma_types.h | 16 ++++++++-------- drivers/gpu/drm/i915/selftests/i915_vma.c | 16 ++++++++-------- drivers/gpu/drm/xe/display/xe_fb_pin.c | 10 +++++----- include/drm/intel/display_parent_interface.h | 8 ++++---- include/drm/intel/gtt_view_types.h | 12 ++++++------ 17 files changed, 56 insertions(+), 56 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_display_types.h b/drivers/gpu/drm/i915/display/intel_display_types.h index fe3b6bec4817d8..0e3083e852d5fc 100644 --- a/drivers/gpu/drm/i915/display/intel_display_types.h +++ b/drivers/gpu/drm/i915/display/intel_display_types.h @@ -111,7 +111,7 @@ struct intel_fb_view { * In the normal view the FB object's backing store sg list is used * directly and hence the remap information here is not used. */ - struct i915_gtt_view gtt; + struct intel_gtt_view gtt; /* * The GTT view (gtt.type) specific information for each FB color diff --git a/drivers/gpu/drm/i915/display/intel_fb.c b/drivers/gpu/drm/i915/display/intel_fb.c index 63b7644c565907..d1b4efa08bcc35 100644 --- a/drivers/gpu/drm/i915/display/intel_fb.c +++ b/drivers/gpu/drm/i915/display/intel_fb.c @@ -1632,7 +1632,7 @@ calc_plane_normal_size(const struct intel_framebuffer *fb, int color_plane, static void intel_fb_view_init(struct intel_display *display, struct intel_fb_view *view, - enum i915_gtt_view_type view_type, + enum intel_gtt_view_type view_type, const struct intel_framebuffer *fb) { memset(view, 0, sizeof(*view)); diff --git a/drivers/gpu/drm/i915/display/intel_parent.c b/drivers/gpu/drm/i915/display/intel_parent.c index 99cebb17763ac1..ed3e21b0a9273a 100644 --- a/drivers/gpu/drm/i915/display/intel_parent.c +++ b/drivers/gpu/drm/i915/display/intel_parent.c @@ -94,9 +94,9 @@ void intel_parent_fb_pin_dpt_unpin(struct intel_display *display, struct i915_vma *intel_parent_fb_pin_reuse_vma(struct intel_display *display, struct i915_vma *old_ggtt_vma, struct drm_gem_object *old_obj, - const struct i915_gtt_view *old_view, + const struct intel_gtt_view *old_view, struct drm_gem_object *new_obj, - const struct i915_gtt_view *new_view, + const struct intel_gtt_view *new_view, u32 *out_offset) { if (!display->parent->fb_pin->reuse_vma) diff --git a/drivers/gpu/drm/i915/display/intel_parent.h b/drivers/gpu/drm/i915/display/intel_parent.h index cc4a58f6316637..923cb6464a9878 100644 --- a/drivers/gpu/drm/i915/display/intel_parent.h +++ b/drivers/gpu/drm/i915/display/intel_parent.h @@ -11,12 +11,12 @@ struct dma_fence; struct drm_file; struct drm_gem_object; struct drm_scanout_buffer; -struct i915_gtt_view; struct i915_vma; struct intel_display; struct intel_dpt; struct intel_fb_pin_params; struct intel_frontbuffer; +struct intel_gtt_view; struct intel_hdcp_gsc_context; struct intel_panic; struct intel_stolen_node; @@ -53,9 +53,9 @@ void intel_parent_fb_pin_dpt_unpin(struct intel_display *display, struct i915_vma *intel_parent_fb_pin_reuse_vma(struct intel_display *display, struct i915_vma *old_ggtt_vma, struct drm_gem_object *old_obj, - const struct i915_gtt_view *old_view, + const struct intel_gtt_view *old_view, struct drm_gem_object *new_obj, - const struct i915_gtt_view *new_view, + const struct intel_gtt_view *new_view, u32 *out_offset); void intel_parent_fb_pin_get_map(struct intel_display *display, struct i915_vma *vma, struct iosys_map *map); diff --git a/drivers/gpu/drm/i915/gem/i915_gem_domain.c b/drivers/gpu/drm/i915/gem/i915_gem_domain.c index b7297c0a3a48e8..0cb41704e54e74 100644 --- a/drivers/gpu/drm/i915/gem/i915_gem_domain.c +++ b/drivers/gpu/drm/i915/gem/i915_gem_domain.c @@ -421,7 +421,7 @@ struct i915_vma * i915_gem_object_pin_to_display_plane(struct drm_i915_gem_object *obj, struct i915_gem_ww_ctx *ww, u32 alignment, unsigned int guard, - const struct i915_gtt_view *view, + const struct intel_gtt_view *view, unsigned int flags) { struct drm_i915_private *i915 = to_i915(obj->base.dev); diff --git a/drivers/gpu/drm/i915/gem/i915_gem_mman.c b/drivers/gpu/drm/i915/gem/i915_gem_mman.c index 9ca90c1bb5b422..ee8fbbcbb5fefb 100644 --- a/drivers/gpu/drm/i915/gem/i915_gem_mman.c +++ b/drivers/gpu/drm/i915/gem/i915_gem_mman.c @@ -196,12 +196,12 @@ int i915_gem_mmap_gtt_version(void) return 5; } -static inline struct i915_gtt_view +static inline struct intel_gtt_view compute_partial_view(const struct drm_i915_gem_object *obj, pgoff_t page_offset, unsigned int chunk) { - struct i915_gtt_view view; + struct intel_gtt_view view; if (i915_gem_object_is_tiled(obj)) chunk = roundup(chunk, tile_row_pages(obj) ?: 1); @@ -391,7 +391,7 @@ static vm_fault_t vm_fault_gtt(struct vm_fault *vmf) PIN_NOEVICT); if (IS_ERR(vma) && vma != ERR_PTR(-EDEADLK)) { /* Use a partial view if it is bigger than available space */ - struct i915_gtt_view view = + struct intel_gtt_view view = compute_partial_view(obj, page_offset, MIN_CHUNK_PAGES); unsigned int flags; diff --git a/drivers/gpu/drm/i915/gem/i915_gem_object.h b/drivers/gpu/drm/i915/gem/i915_gem_object.h index 2c5d20e4dbafcc..a2f4b6758a65b0 100644 --- a/drivers/gpu/drm/i915/gem/i915_gem_object.h +++ b/drivers/gpu/drm/i915/gem/i915_gem_object.h @@ -776,7 +776,7 @@ struct i915_vma * __must_check i915_gem_object_pin_to_display_plane(struct drm_i915_gem_object *obj, struct i915_gem_ww_ctx *ww, u32 alignment, unsigned int guard, - const struct i915_gtt_view *view, + const struct intel_gtt_view *view, unsigned int flags); void i915_gem_object_make_unshrinkable(struct drm_i915_gem_object *obj); diff --git a/drivers/gpu/drm/i915/gem/selftests/i915_gem_mman.c b/drivers/gpu/drm/i915/gem/selftests/i915_gem_mman.c index d01acfb7d93d09..4d49fd6c15e3cb 100644 --- a/drivers/gpu/drm/i915/gem/selftests/i915_gem_mman.c +++ b/drivers/gpu/drm/i915/gem/selftests/i915_gem_mman.c @@ -97,7 +97,7 @@ static int check_partial_mapping(struct drm_i915_gem_object *obj, { const unsigned long npages = obj->base.size / PAGE_SIZE; struct drm_i915_private *i915 = to_i915(obj->base.dev); - struct i915_gtt_view view; + struct intel_gtt_view view; struct i915_vma *vma; unsigned long offset; unsigned long page; @@ -214,7 +214,7 @@ static int check_partial_mappings(struct drm_i915_gem_object *obj, } for_each_prime_number_from(page, 1, npages) { - struct i915_gtt_view view = + struct intel_gtt_view view = compute_partial_view(obj, page, MIN_CHUNK_PAGES); unsigned long offset; u32 __iomem *io; diff --git a/drivers/gpu/drm/i915/i915_gem.c b/drivers/gpu/drm/i915/i915_gem.c index a432daf8038a5c..62987ae59a5084 100644 --- a/drivers/gpu/drm/i915/i915_gem.c +++ b/drivers/gpu/drm/i915/i915_gem.c @@ -902,7 +902,7 @@ static void discard_ggtt_vma(struct i915_vma *vma) struct i915_vma * i915_gem_object_ggtt_pin_ww(struct drm_i915_gem_object *obj, struct i915_gem_ww_ctx *ww, - const struct i915_gtt_view *view, + const struct intel_gtt_view *view, u64 size, u64 alignment, u64 flags) { struct drm_i915_private *i915 = to_i915(obj->base.dev); @@ -1004,7 +1004,7 @@ i915_gem_object_ggtt_pin_ww(struct drm_i915_gem_object *obj, struct i915_vma * __must_check i915_gem_object_ggtt_pin(struct drm_i915_gem_object *obj, - const struct i915_gtt_view *view, + const struct intel_gtt_view *view, u64 size, u64 alignment, u64 flags) { struct i915_gem_ww_ctx ww; diff --git a/drivers/gpu/drm/i915/i915_gem.h b/drivers/gpu/drm/i915/i915_gem.h index 20b3cb29cfffa2..80d9758fe68446 100644 --- a/drivers/gpu/drm/i915/i915_gem.h +++ b/drivers/gpu/drm/i915/i915_gem.h @@ -36,8 +36,8 @@ struct drm_file; struct drm_i915_gem_object; struct drm_i915_private; struct i915_gem_ww_ctx; -struct i915_gtt_view; struct i915_vma; +struct intel_gtt_view; #define I915_GEM_GPU_DOMAINS \ (I915_GEM_DOMAIN_RENDER | \ @@ -55,12 +55,12 @@ void i915_gem_drain_workqueue(struct drm_i915_private *i915); struct i915_vma * __must_check i915_gem_object_ggtt_pin_ww(struct drm_i915_gem_object *obj, struct i915_gem_ww_ctx *ww, - const struct i915_gtt_view *view, + const struct intel_gtt_view *view, u64 size, u64 alignment, u64 flags); struct i915_vma * __must_check i915_gem_object_ggtt_pin(struct drm_i915_gem_object *obj, - const struct i915_gtt_view *view, + const struct intel_gtt_view *view, u64 size, u64 alignment, u64 flags); int i915_gem_object_unbind(struct drm_i915_gem_object *obj, diff --git a/drivers/gpu/drm/i915/i915_vma.c b/drivers/gpu/drm/i915/i915_vma.c index afc192d9931b88..61ddfa5d289550 100644 --- a/drivers/gpu/drm/i915/i915_vma.c +++ b/drivers/gpu/drm/i915/i915_vma.c @@ -147,7 +147,7 @@ static void __i915_vma_retire(struct i915_active *ref) static struct i915_vma * vma_create(struct drm_i915_gem_object *obj, struct i915_address_space *vm, - const struct i915_gtt_view *view) + const struct intel_gtt_view *view) { struct i915_vma *pos = ERR_PTR(-E2BIG); struct i915_vma *vma; @@ -286,7 +286,7 @@ vma_create(struct drm_i915_gem_object *obj, static struct i915_vma * i915_vma_lookup(struct drm_i915_gem_object *obj, struct i915_address_space *vm, - const struct i915_gtt_view *view) + const struct intel_gtt_view *view) { struct rb_node *rb; @@ -324,7 +324,7 @@ i915_vma_lookup(struct drm_i915_gem_object *obj, struct i915_vma * i915_vma_instance(struct drm_i915_gem_object *obj, struct i915_address_space *vm, - const struct i915_gtt_view *view) + const struct intel_gtt_view *view) { struct i915_vma *vma; @@ -1267,7 +1267,7 @@ intel_remap_pages(struct intel_remapped_info *rem_info, } static noinline struct sg_table * -intel_partial_pages(const struct i915_gtt_view *view, +intel_partial_pages(const struct intel_gtt_view *view, struct drm_i915_gem_object *obj) { struct sg_table *st; diff --git a/drivers/gpu/drm/i915/i915_vma.h b/drivers/gpu/drm/i915/i915_vma.h index 892306ab935dcc..a8a89bb0270cd0 100644 --- a/drivers/gpu/drm/i915/i915_vma.h +++ b/drivers/gpu/drm/i915/i915_vma.h @@ -43,7 +43,7 @@ struct i915_vma * i915_vma_instance(struct drm_i915_gem_object *obj, struct i915_address_space *vm, - const struct i915_gtt_view *view); + const struct intel_gtt_view *view); void i915_vma_unpin_and_release(struct i915_vma **p_vma, unsigned int flags); #define I915_VMA_RELEASE_MAP BIT(0) @@ -207,7 +207,7 @@ static inline void i915_vma_put(struct i915_vma *vma) static inline long i915_vma_compare(struct i915_vma *vma, struct i915_address_space *vm, - const struct i915_gtt_view *view) + const struct intel_gtt_view *view) { ptrdiff_t cmp; diff --git a/drivers/gpu/drm/i915/i915_vma_types.h b/drivers/gpu/drm/i915/i915_vma_types.h index 83fe02833b5e5c..8815421eaefba9 100644 --- a/drivers/gpu/drm/i915/i915_vma_types.h +++ b/drivers/gpu/drm/i915/i915_vma_types.h @@ -66,22 +66,22 @@ * Implementation and usage * * GGTT views are implemented using VMAs and are distinguished via enum - * i915_gtt_view_type and struct i915_gtt_view. + * intel_gtt_view_type and struct intel_gtt_view. * * A new flavour of core GEM functions which work with GGTT bound objects were * added with the _ggtt_ infix, and sometimes with _view postfix to avoid - * renaming in large amounts of code. They take the struct i915_gtt_view + * renaming in large amounts of code. They take the struct intel_gtt_view * parameter encapsulating all metadata required to implement a view. * * As a helper for callers which are only interested in the normal view, - * globally const i915_gtt_view_normal singleton instance exists. All old core + * globally const intel_gtt_view_normal singleton instance exists. All old core * GEM API functions, the ones not taking the view parameter, are operating on, * or with the normal GGTT view. * * Code wanting to add or use a new GGTT view needs to: * * 1. Add a new enum with a suitable name. - * 2. Extend the metadata in the i915_gtt_view structure if required. + * 2. Extend the metadata in the intel_gtt_view structure if required. * 3. Add support to i915_get_vma_pages(). * * New views are required to build a scatter-gather table from within the @@ -89,7 +89,7 @@ * exists for the lifetime of an VMA. * * Core API is designed to have copy semantics which means that passed in - * struct i915_gtt_view does not need to be persistent (left around after + * struct intel_gtt_view does not need to be persistent (left around after * calling the core API functions). * */ @@ -111,7 +111,7 @@ static inline void assert_i915_gem_gtt_types(void) /* As we encode the size of each branch inside the union into its type, * we have to be careful that each branch has a unique size. */ - switch ((enum i915_gtt_view_type)0) { + switch ((enum intel_gtt_view_type)0) { case I915_GTT_VIEW_NORMAL: case I915_GTT_VIEW_PARTIAL: case I915_GTT_VIEW_ROTATED: @@ -230,11 +230,11 @@ struct i915_vma { /** * Support different GGTT views into the same object. * This means there can be multiple VMA mappings per object and per VM. - * i915_gtt_view_type is used to distinguish between those entries. + * intel_gtt_view_type is used to distinguish between those entries. * The default one of zero (I915_GTT_VIEW_NORMAL) is default and also * assumed in GEM functions which take no ggtt view parameter. */ - struct i915_gtt_view gtt_view; + struct intel_gtt_view gtt_view; /** This object's place on the active/inactive lists */ struct list_head vm_link; diff --git a/drivers/gpu/drm/i915/selftests/i915_vma.c b/drivers/gpu/drm/i915/selftests/i915_vma.c index 7c4111e60f2ec6..b16297a6a4e8de 100644 --- a/drivers/gpu/drm/i915/selftests/i915_vma.c +++ b/drivers/gpu/drm/i915/selftests/i915_vma.c @@ -63,7 +63,7 @@ static bool assert_vma(struct i915_vma *vma, static struct i915_vma * checked_vma_instance(struct drm_i915_gem_object *obj, struct i915_address_space *vm, - const struct i915_gtt_view *view) + const struct intel_gtt_view *view) { struct i915_vma *vma; bool ok = true; @@ -533,7 +533,7 @@ assert_remapped(struct drm_i915_gem_object *obj, return sg; } -static unsigned int remapped_size(enum i915_gtt_view_type view_type, +static unsigned int remapped_size(enum intel_gtt_view_type view_type, const struct intel_remapped_plane_info *a, const struct intel_remapped_plane_info *b) { @@ -572,7 +572,7 @@ static int igt_vma_rotate_remap(void *arg) { } }, *a, *b; - enum i915_gtt_view_type types[] = { + enum intel_gtt_view_type types[] = { I915_GTT_VIEW_ROTATED, I915_GTT_VIEW_REMAPPED, 0, @@ -592,7 +592,7 @@ static int igt_vma_rotate_remap(void *arg) for (t = types; *t; t++) { for (a = planes; a->width; a++) { for (b = planes + ARRAY_SIZE(planes); b-- != planes; ) { - struct i915_gtt_view view = { + struct intel_gtt_view view = { .type = *t, .remapped.plane[0] = *a, .remapped.plane[1] = *b, @@ -745,7 +745,7 @@ static bool assert_partial(struct drm_i915_gem_object *obj, } static bool assert_pin(struct i915_vma *vma, - struct i915_gtt_view *view, + struct intel_gtt_view *view, u64 size, const char *name) { @@ -823,7 +823,7 @@ static int igt_vma_partial(void *arg) nvma = 0; for_each_prime_number_from(sz, 1, npages) { for_each_prime_number_from(offset, 0, npages - sz) { - struct i915_gtt_view view; + struct intel_gtt_view view; view.type = I915_GTT_VIEW_PARTIAL; view.partial.offset = offset; @@ -981,7 +981,7 @@ static int igt_vma_remapped_gtt(void *arg) { } }, *p; - enum i915_gtt_view_type types[] = { + enum intel_gtt_view_type types[] = { I915_GTT_VIEW_ROTATED, I915_GTT_VIEW_REMAPPED, 0, @@ -1001,7 +1001,7 @@ static int igt_vma_remapped_gtt(void *arg) for (t = types; *t; t++) { for (p = planes; p->width; p++) { - struct i915_gtt_view view = { + struct intel_gtt_view view = { .type = *t, .rotated.plane[0] = *p, }; diff --git a/drivers/gpu/drm/xe/display/xe_fb_pin.c b/drivers/gpu/drm/xe/display/xe_fb_pin.c index ce2601069e7d05..1ba2dfc2f5748b 100644 --- a/drivers/gpu/drm/xe/display/xe_fb_pin.c +++ b/drivers/gpu/drm/xe/display/xe_fb_pin.c @@ -148,7 +148,7 @@ static int __xe_pin_fb_vma_dpt(struct drm_gem_object *obj, struct xe_device *xe = to_xe_device(obj->dev); struct xe_tile *tile0 = xe_device_get_root_tile(xe); struct xe_ggtt *ggtt = tile0->mem.ggtt; - const struct i915_gtt_view *view = pin_params->view; + const struct intel_gtt_view *view = pin_params->view; struct xe_bo *bo = gem_to_xe_bo(obj), *dpt; u32 dpt_size, size = bo->ttm.base.size; @@ -231,7 +231,7 @@ write_ggtt_rotated(struct xe_ggtt *ggtt, u32 *ggtt_ofs, } struct fb_rotate_args { - const struct i915_gtt_view *view; + const struct intel_gtt_view *view; struct xe_bo *bo; }; @@ -256,7 +256,7 @@ static int __xe_pin_fb_vma_ggtt(struct drm_gem_object *obj, const struct intel_fb_pin_params *pin_params, struct i915_vma *vma) { - const struct i915_gtt_view *view = pin_params->view; + const struct intel_gtt_view *view = pin_params->view; struct xe_bo *bo = gem_to_xe_bo(obj); struct xe_device *xe = to_xe_device(obj->dev); struct xe_tile *tile0 = xe_device_get_root_tile(xe); @@ -457,9 +457,9 @@ static void xe_fb_pin_dpt_unpin(struct intel_dpt *dpt, static struct i915_vma * xe_fb_pin_reuse_vma(struct i915_vma *old_ggtt_vma, struct drm_gem_object *old_obj, - const struct i915_gtt_view *old_view, + const struct intel_gtt_view *old_view, struct drm_gem_object *new_obj, - const struct i915_gtt_view *new_view, + const struct intel_gtt_view *new_view, u32 *out_offset) { if (old_ggtt_vma && old_obj == new_obj && diff --git a/include/drm/intel/display_parent_interface.h b/include/drm/intel/display_parent_interface.h index f36134e89cb6d1..464fb588cb60fd 100644 --- a/include/drm/intel/display_parent_interface.h +++ b/include/drm/intel/display_parent_interface.h @@ -16,11 +16,11 @@ struct drm_mode_fb_cmd2; struct drm_plane_state; struct drm_scanout_buffer; struct fb_info; -struct i915_gtt_view; struct i915_vma; struct intel_dpt; struct intel_dsb_buffer; struct intel_frontbuffer; +struct intel_gtt_view; struct intel_hdcp_gsc_context; struct intel_initial_plane_config; struct intel_panic; @@ -31,7 +31,7 @@ struct seq_file; struct vm_area_struct; struct intel_fb_pin_params { - const struct i915_gtt_view *view; + const struct intel_gtt_view *view; unsigned int alignment; unsigned int phys_alignment; unsigned int vtd_guard; @@ -101,9 +101,9 @@ struct intel_display_fb_pin_interface { struct i915_vma *ggtt_vma); struct i915_vma *(*reuse_vma)(struct i915_vma *old_ggtt_vma, struct drm_gem_object *old_obj, - const struct i915_gtt_view *old_view, + const struct intel_gtt_view *old_view, struct drm_gem_object *new_obj, - const struct i915_gtt_view *new_view, + const struct intel_gtt_view *new_view, u32 *out_offset); void (*get_map)(struct i915_vma *vma, struct iosys_map *map); }; diff --git a/include/drm/intel/gtt_view_types.h b/include/drm/intel/gtt_view_types.h index 61d2289a8921e6..3ad2a1026db134 100644 --- a/include/drm/intel/gtt_view_types.h +++ b/include/drm/intel/gtt_view_types.h @@ -39,15 +39,15 @@ struct intel_remapped_info { u32 plane_alignment; } __packed; -enum i915_gtt_view_type { +enum intel_gtt_view_type { I915_GTT_VIEW_NORMAL = 0, I915_GTT_VIEW_ROTATED = sizeof(struct intel_rotation_info), I915_GTT_VIEW_PARTIAL = sizeof(struct intel_partial_info), I915_GTT_VIEW_REMAPPED = sizeof(struct intel_remapped_info), }; -struct i915_gtt_view { - enum i915_gtt_view_type type; +struct intel_gtt_view { + enum intel_gtt_view_type type; union { /* Members need to contain no holes/padding */ struct intel_partial_info partial; @@ -56,17 +56,17 @@ struct i915_gtt_view { }; }; -static inline bool intel_gtt_view_is_normal(const struct i915_gtt_view *view) +static inline bool intel_gtt_view_is_normal(const struct intel_gtt_view *view) { return view->type == I915_GTT_VIEW_NORMAL; } -static inline bool intel_gtt_view_is_remapped(const struct i915_gtt_view *view) +static inline bool intel_gtt_view_is_remapped(const struct intel_gtt_view *view) { return view->type == I915_GTT_VIEW_REMAPPED; } -static inline bool intel_gtt_view_is_rotated(const struct i915_gtt_view *view) +static inline bool intel_gtt_view_is_rotated(const struct intel_gtt_view *view) { return view->type == I915_GTT_VIEW_ROTATED; } From 4a197e47221236f39abe072df065dcab2ce31b69 Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Wed, 30 Sep 2026 18:00:00 +0300 Subject: [PATCH 0494/1352] drm/intel: add intel_gtt_view_is_partial() for completeness MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit There's a helper for all other view types, add one for intel_gtt_view_is_partial() too. Reviewed-by: Maarten Lankhorst Reviewed-by: Ville Syrjälä Link: https://patch.msgid.link/be522421df45be071ec8f298f4d5b7a49fa6bab8.1790780340.git.jani.nikula@intel.com Signed-off-by: Jani Nikula --- include/drm/intel/gtt_view_types.h | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/include/drm/intel/gtt_view_types.h b/include/drm/intel/gtt_view_types.h index 3ad2a1026db134..d77292897d6448 100644 --- a/include/drm/intel/gtt_view_types.h +++ b/include/drm/intel/gtt_view_types.h @@ -71,4 +71,9 @@ static inline bool intel_gtt_view_is_rotated(const struct intel_gtt_view *view) return view->type == I915_GTT_VIEW_ROTATED; } +static inline bool intel_gtt_view_is_partial(const struct intel_gtt_view *view) +{ + return view->type == I915_GTT_VIEW_PARTIAL; +} + #endif /* __DRM_INTEL_GTT_VIEW_TYPES_H__ */ From 0586e5f135513c4fc7656769b9c15974ca2be59c Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Wed, 30 Sep 2026 18:00:01 +0300 Subject: [PATCH 0495/1352] drm/xe/display: use intel_gtt_view_is_*() helpers more MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Prefer using the intel_gtt_view_is_*() helpers instead of comparing the view type directly. Reviewed-by: Maarten Lankhorst Reviewed-by: Ville Syrjälä Link: https://patch.msgid.link/e05bf916d46007f8e655beb6c31748a8f69e09f3.1790780340.git.jani.nikula@intel.com Signed-off-by: Jani Nikula --- drivers/gpu/drm/xe/display/xe_fb_pin.c | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/drivers/gpu/drm/xe/display/xe_fb_pin.c b/drivers/gpu/drm/xe/display/xe_fb_pin.c index 1ba2dfc2f5748b..ab2be62a7fb49d 100644 --- a/drivers/gpu/drm/xe/display/xe_fb_pin.c +++ b/drivers/gpu/drm/xe/display/xe_fb_pin.c @@ -152,9 +152,9 @@ static int __xe_pin_fb_vma_dpt(struct drm_gem_object *obj, struct xe_bo *bo = gem_to_xe_bo(obj), *dpt; u32 dpt_size, size = bo->ttm.base.size; - if (view->type == I915_GTT_VIEW_NORMAL) + if (intel_gtt_view_is_normal(view)) dpt_size = ALIGN(size / XE_PAGE_SIZE * 8, XE_PAGE_SIZE); - else if (view->type == I915_GTT_VIEW_REMAPPED) + else if (intel_gtt_view_is_remapped(view)) dpt_size = ALIGN(intel_remapped_info_size(&view->remapped) * 8, XE_PAGE_SIZE); else @@ -173,7 +173,7 @@ static int __xe_pin_fb_vma_dpt(struct drm_gem_object *obj, if (IS_ERR(dpt)) return PTR_ERR(dpt); - if (view->type == I915_GTT_VIEW_NORMAL) { + if (intel_gtt_view_is_normal(view)) { u64 pte = xe_ggtt_encode_pte_flags(ggtt, bo, xe_cache_pat_idx(xe, XE_CACHE_NONE)); u32 x; @@ -182,7 +182,7 @@ static int __xe_pin_fb_vma_dpt(struct drm_gem_object *obj, iosys_map_wr(&dpt->vmap, x * 8, u64, pte | addr); } - } else if (view->type == I915_GTT_VIEW_REMAPPED) { + } else if (intel_gtt_view_is_remapped(view)) { write_dpt_remapped(bo, &view->remapped, &dpt->vmap); } else { const struct intel_rotation_info *rot_info = &view->rotated; @@ -275,7 +275,7 @@ static int __xe_pin_fb_vma_ggtt(struct drm_gem_object *obj, align = max(align, SZ_64K); /* Fast case, preallocated GGTT view? */ - if (bo->ggtt_node[tile0->id] && view->type == I915_GTT_VIEW_NORMAL) { + if (bo->ggtt_node[tile0->id] && intel_gtt_view_is_normal(view)) { vma->node = bo->ggtt_node[tile0->id]; return 0; } @@ -283,7 +283,7 @@ static int __xe_pin_fb_vma_ggtt(struct drm_gem_object *obj, /* TODO: Consider sharing framebuffer mapping? * embed i915_vma inside intel_framebuffer */ - if (view->type == I915_GTT_VIEW_NORMAL) + if (intel_gtt_view_is_normal(view)) size = xe_bo_size(bo); else /* display uses tiles instead of bytes here, so convert it back.. */ @@ -292,7 +292,7 @@ static int __xe_pin_fb_vma_ggtt(struct drm_gem_object *obj, pte = xe_ggtt_encode_pte_flags(ggtt, bo, xe_cache_pat_idx(xe, XE_CACHE_NONE)); vma->node = xe_ggtt_insert_node_transform(ggtt, bo, pte, ALIGN(size, align), align, - view->type == I915_GTT_VIEW_NORMAL ? + intel_gtt_view_is_normal(view) ? NULL : write_ggtt_rotated_node, &(struct fb_rotate_args){view, bo}); if (IS_ERR(vma->node)) From 1a4aab52c6ea890d0cef2dfc21900715c086da83 Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Wed, 30 Sep 2026 18:00:02 +0300 Subject: [PATCH 0496/1352] drm/i915/gem: use intel_gtt_view_is_*() helpers more MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Prefer using the intel_gtt_view_is_*() helpers instead of comparing the view type directly. Reviewed-by: Maarten Lankhorst Reviewed-by: Ville Syrjälä Link: https://patch.msgid.link/1b7f309d0c5c5891caa2ec84c04ce8a4438f1207.1790780340.git.jani.nikula@intel.com Signed-off-by: Jani Nikula --- drivers/gpu/drm/i915/gem/i915_gem_domain.c | 2 +- drivers/gpu/drm/i915/gem/i915_gem_mman.c | 2 +- drivers/gpu/drm/i915/i915_gem.c | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/drivers/gpu/drm/i915/gem/i915_gem_domain.c b/drivers/gpu/drm/i915/gem/i915_gem_domain.c index 0cb41704e54e74..9a82ed77ccaf9e 100644 --- a/drivers/gpu/drm/i915/gem/i915_gem_domain.c +++ b/drivers/gpu/drm/i915/gem/i915_gem_domain.c @@ -462,7 +462,7 @@ i915_gem_object_pin_to_display_plane(struct drm_i915_gem_object *obj, */ vma = ERR_PTR(-ENOSPC); if ((flags & PIN_MAPPABLE) == 0 && - (!view || view->type == I915_GTT_VIEW_NORMAL)) + (!view || intel_gtt_view_is_normal(view))) vma = i915_gem_object_ggtt_pin_ww(obj, ww, view, 0, alignment, flags | PIN_MAPPABLE | PIN_NONBLOCK); diff --git a/drivers/gpu/drm/i915/gem/i915_gem_mman.c b/drivers/gpu/drm/i915/gem/i915_gem_mman.c index ee8fbbcbb5fefb..055f8d3161dabc 100644 --- a/drivers/gpu/drm/i915/gem/i915_gem_mman.c +++ b/drivers/gpu/drm/i915/gem/i915_gem_mman.c @@ -396,7 +396,7 @@ static vm_fault_t vm_fault_gtt(struct vm_fault *vmf) unsigned int flags; flags = PIN_MAPPABLE | PIN_NOSEARCH; - if (view.type == I915_GTT_VIEW_NORMAL) + if (intel_gtt_view_is_normal(&view)) flags |= PIN_NONBLOCK; /* avoid warnings for pinned */ /* diff --git a/drivers/gpu/drm/i915/i915_gem.c b/drivers/gpu/drm/i915/i915_gem.c index 62987ae59a5084..1209d5f1fc8d4b 100644 --- a/drivers/gpu/drm/i915/i915_gem.c +++ b/drivers/gpu/drm/i915/i915_gem.c @@ -913,7 +913,7 @@ i915_gem_object_ggtt_pin_ww(struct drm_i915_gem_object *obj, GEM_WARN_ON(!ww); if (flags & PIN_MAPPABLE && - (!view || view->type == I915_GTT_VIEW_NORMAL)) { + (!view || intel_gtt_view_is_normal(view))) { /* * If the required space is larger than the available * aperture, we will not able to find a slot for the From d37667551cfe2828b2530740ebaccce70cec3377 Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Wed, 30 Sep 2026 18:00:03 +0300 Subject: [PATCH 0497/1352] drm/i915/vma: use intel_gtt_view_is_*() helpers more MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Prefer using the intel_gtt_view_is_*() helpers instead of comparing the view type directly. Reviewed-by: Maarten Lankhorst Reviewed-by: Ville Syrjälä Link: https://patch.msgid.link/5bc8331cab1d15855b378bd01c76fb0fc4be5c7f.1790780340.git.jani.nikula@intel.com Signed-off-by: Jani Nikula --- drivers/gpu/drm/i915/i915_vma.c | 36 +++++++++++---------------------- 1 file changed, 12 insertions(+), 24 deletions(-) diff --git a/drivers/gpu/drm/i915/i915_vma.c b/drivers/gpu/drm/i915/i915_vma.c index 61ddfa5d289550..16782d8579a850 100644 --- a/drivers/gpu/drm/i915/i915_vma.c +++ b/drivers/gpu/drm/i915/i915_vma.c @@ -179,9 +179,9 @@ vma_create(struct drm_i915_gem_object *obj, INIT_LIST_HEAD(&vma->obj_link); RB_CLEAR_NODE(&vma->obj_node); - if (view && view->type != I915_GTT_VIEW_NORMAL) { + if (view && !intel_gtt_view_is_normal(view)) { vma->gtt_view = *view; - if (view->type == I915_GTT_VIEW_PARTIAL) { + if (intel_gtt_view_is_partial(view)) { GEM_BUG_ON(range_overflows_t(u64, view->partial.offset, view->partial.size, @@ -189,10 +189,10 @@ vma_create(struct drm_i915_gem_object *obj, vma->size = view->partial.size; vma->size <<= PAGE_SHIFT; GEM_BUG_ON(vma->size > obj->base.size); - } else if (view->type == I915_GTT_VIEW_ROTATED) { + } else if (intel_gtt_view_is_rotated(view)) { vma->size = intel_rotation_info_size(&view->rotated); vma->size <<= PAGE_SHIFT; - } else if (view->type == I915_GTT_VIEW_REMAPPED) { + } else if (intel_gtt_view_is_remapped(view)) { vma->size = intel_remapped_info_size(&view->remapped); vma->size <<= PAGE_SHIFT; } @@ -1311,27 +1311,15 @@ __i915_vma_get_pages(struct i915_vma *vma) */ GEM_BUG_ON(!i915_gem_object_has_pinned_pages(vma->obj)); - switch (vma->gtt_view.type) { - default: - GEM_BUG_ON(vma->gtt_view.type); - fallthrough; - case I915_GTT_VIEW_NORMAL: - pages = vma->obj->mm.pages; - break; - - case I915_GTT_VIEW_ROTATED: - pages = - intel_rotate_pages(&vma->gtt_view.rotated, vma->obj); - break; - - case I915_GTT_VIEW_REMAPPED: - pages = - intel_remap_pages(&vma->gtt_view.remapped, vma->obj); - break; - - case I915_GTT_VIEW_PARTIAL: + if (intel_gtt_view_is_rotated(&vma->gtt_view)) { + pages = intel_rotate_pages(&vma->gtt_view.rotated, vma->obj); + } else if (intel_gtt_view_is_remapped(&vma->gtt_view)) { + pages = intel_remap_pages(&vma->gtt_view.remapped, vma->obj); + } else if (intel_gtt_view_is_partial(&vma->gtt_view)) { pages = intel_partial_pages(&vma->gtt_view, vma->obj); - break; + } else { + GEM_BUG_ON(!intel_gtt_view_is_normal(&vma->gtt_view)); + pages = vma->obj->mm.pages; } if (IS_ERR(pages)) { From 88d9d2bab07325c6649f1e227466c8974af64fb7 Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Wed, 30 Sep 2026 18:00:04 +0300 Subject: [PATCH 0498/1352] drm/i915/selftests: use intel_gtt_view_is_*() helpers more MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Prefer using the intel_gtt_view_is_*() helpers instead of comparing the view type directly. v3: Use view, not vma->gtt_view even more (Ville) v2: Use view, not vma->gtt_view at the end (Sashiko) Reviewed-by: Maarten Lankhorst Reviewed-by: Ville Syrjälä Link: https://patch.msgid.link/350f0be2c56ae4ac7b7bb97186f4d501faa7bb4e.1790780340.git.jani.nikula@intel.com Signed-off-by: Jani Nikula --- drivers/gpu/drm/i915/selftests/i915_vma.c | 34 +++++++++++------------ 1 file changed, 17 insertions(+), 17 deletions(-) diff --git a/drivers/gpu/drm/i915/selftests/i915_vma.c b/drivers/gpu/drm/i915/selftests/i915_vma.c index b16297a6a4e8de..182c9f8606281b 100644 --- a/drivers/gpu/drm/i915/selftests/i915_vma.c +++ b/drivers/gpu/drm/i915/selftests/i915_vma.c @@ -51,7 +51,7 @@ static bool assert_vma(struct i915_vma *vma, ok = false; } - if (vma->gtt_view.type != I915_GTT_VIEW_NORMAL) { + if (!intel_gtt_view_is_normal(&vma->gtt_view)) { pr_err("VMA created with wrong type [%d]\n", vma->gtt_view.type); ok = false; @@ -533,12 +533,12 @@ assert_remapped(struct drm_i915_gem_object *obj, return sg; } -static unsigned int remapped_size(enum intel_gtt_view_type view_type, +static unsigned int remapped_size(const struct intel_gtt_view *view, const struct intel_remapped_plane_info *a, const struct intel_remapped_plane_info *b) { - if (view_type == I915_GTT_VIEW_ROTATED) + if (intel_gtt_view_is_rotated(view)) return a->dst_stride * a->width + b->dst_stride * b->width; else return a->dst_stride * a->height + b->dst_stride * b->height; @@ -606,11 +606,11 @@ static int igt_vma_rotate_remap(void *arg) max_offset = max_pages - max_offset; if (!plane_info[0].dst_stride) - plane_info[0].dst_stride = view.type == I915_GTT_VIEW_ROTATED ? + plane_info[0].dst_stride = intel_gtt_view_is_rotated(&view) ? plane_info[0].height : plane_info[0].width; if (!plane_info[1].dst_stride) - plane_info[1].dst_stride = view.type == I915_GTT_VIEW_ROTATED ? + plane_info[1].dst_stride = intel_gtt_view_is_rotated(&view) ? plane_info[1].height : plane_info[1].width; @@ -632,9 +632,9 @@ static int igt_vma_rotate_remap(void *arg) goto out_object; } - expected_pages = remapped_size(view.type, &plane_info[0], &plane_info[1]); + expected_pages = remapped_size(&view, &plane_info[0], &plane_info[1]); - if (view.type == I915_GTT_VIEW_ROTATED && + if (intel_gtt_view_is_rotated(&view) && vma->size != expected_pages * PAGE_SIZE) { pr_err("VMA is wrong size, expected %lu, found %llu\n", PAGE_SIZE * expected_pages, vma->size); @@ -642,7 +642,7 @@ static int igt_vma_rotate_remap(void *arg) goto out_object; } - if (view.type == I915_GTT_VIEW_REMAPPED && + if (intel_gtt_view_is_remapped(&view) && vma->size > expected_pages * PAGE_SIZE) { pr_err("VMA is wrong size, expected %lu, found %llu\n", PAGE_SIZE * expected_pages, vma->size); @@ -672,13 +672,13 @@ static int igt_vma_rotate_remap(void *arg) sg = vma->pages->sgl; for (n = 0; n < ARRAY_SIZE(view.rotated.plane); n++) { - if (view.type == I915_GTT_VIEW_ROTATED) + if (intel_gtt_view_is_rotated(&view)) sg = assert_rotated(obj, &view.rotated, n, sg); else sg = assert_remapped(obj, &view.remapped, n, sg); if (IS_ERR(sg)) { pr_err("Inconsistent %s VMA pages for plane %d: [(%d, %d, %d, %d, %d), (%d, %d, %d, %d, %d)]\n", - view.type == I915_GTT_VIEW_ROTATED ? + intel_gtt_view_is_rotated(&view) ? "rotated" : "remapped", n, plane_info[0].width, plane_info[0].height, @@ -763,7 +763,7 @@ static bool assert_pin(struct i915_vma *vma, ok = false; } - if (view && view->type != I915_GTT_VIEW_NORMAL) { + if (view && !intel_gtt_view_is_normal(view)) { if (memcmp(&vma->gtt_view, view, sizeof(*view))) { pr_err("(%s) VMA mismatch upon creation!\n", name); @@ -776,7 +776,7 @@ static bool assert_pin(struct i915_vma *vma, ok = false; } } else { - if (vma->gtt_view.type != I915_GTT_VIEW_NORMAL) { + if (!intel_gtt_view_is_normal(&vma->gtt_view)) { pr_err("Not the normal ggtt view! Found %d\n", vma->gtt_view.type); ok = false; @@ -1017,7 +1017,7 @@ static int igt_vma_remapped_gtt(void *arg) goto out; if (!plane_info[0].dst_stride) - plane_info[0].dst_stride = *t == I915_GTT_VIEW_ROTATED ? + plane_info[0].dst_stride = intel_gtt_view_is_rotated(&view) ? p->height : p->width; vma = i915_gem_object_ggtt_pin(obj, &view, 0, 0, PIN_MAPPABLE); @@ -1040,7 +1040,7 @@ static int igt_vma_remapped_gtt(void *arg) unsigned int offset; u32 val = y << 16 | x; - if (*t == I915_GTT_VIEW_ROTATED) + if (intel_gtt_view_is_rotated(&view)) offset = (x * plane_info[0].dst_stride + y) * PAGE_SIZE; else offset = (y * plane_info[0].dst_stride + x) * PAGE_SIZE; @@ -1057,7 +1057,7 @@ static int igt_vma_remapped_gtt(void *arg) goto out; } - GEM_BUG_ON(vma->gtt_view.type != I915_GTT_VIEW_NORMAL); + GEM_BUG_ON(!intel_gtt_view_is_normal(&vma->gtt_view)); map = i915_vma_pin_iomap(vma); i915_vma_unpin(vma); @@ -1072,7 +1072,7 @@ static int igt_vma_remapped_gtt(void *arg) u32 exp = y << 16 | x; u32 val; - if (*t == I915_GTT_VIEW_ROTATED) + if (intel_gtt_view_is_rotated(&view)) src_idx = rotated_index(&view.rotated, 0, x, y); else src_idx = remapped_index(&view.remapped, 0, x, y); @@ -1081,7 +1081,7 @@ static int igt_vma_remapped_gtt(void *arg) val = ioread32(&map[offset / sizeof(*map)]); if (val != exp) { pr_err("%s VMA write test failed, expected 0x%x, found 0x%x\n", - *t == I915_GTT_VIEW_ROTATED ? "Rotated" : "Remapped", + intel_gtt_view_is_rotated(&view) ? "Rotated" : "Remapped", exp, val); i915_vma_unpin_iomap(vma); err = -EINVAL; From 105852cba40e500775207144e22c72e4ca2347a7 Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Wed, 30 Sep 2026 18:00:05 +0300 Subject: [PATCH 0499/1352] drm/i915/debugfs: use the intel_gtt_view_is_*() helpers more MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Prefer using the intel_gtt_view_is_*() helpers instead of comparing the view type directly. Reviewed-by: Maarten Lankhorst Reviewed-by: Ville Syrjälä Link: https://patch.msgid.link/6d9279ce0522f6768f55fb6c760862fdb92ffc03.1790780340.git.jani.nikula@intel.com Signed-off-by: Jani Nikula --- drivers/gpu/drm/i915/i915_debugfs.c | 20 +++++--------------- 1 file changed, 5 insertions(+), 15 deletions(-) diff --git a/drivers/gpu/drm/i915/i915_debugfs.c b/drivers/gpu/drm/i915/i915_debugfs.c index 93ba7369ba6e2a..04bc3d27b836e1 100644 --- a/drivers/gpu/drm/i915/i915_debugfs.c +++ b/drivers/gpu/drm/i915/i915_debugfs.c @@ -209,18 +209,13 @@ i915_debugfs_describe_obj(struct seq_file *m, struct drm_i915_gem_object *obj) stringify_page_sizes(vma->resource->page_sizes_gtt, NULL, 0)); if (i915_vma_is_ggtt(vma) || i915_vma_is_dpt(vma)) { - switch (vma->gtt_view.type) { - case I915_GTT_VIEW_NORMAL: + if (intel_gtt_view_is_normal(&vma->gtt_view)) { seq_puts(m, ", normal"); - break; - - case I915_GTT_VIEW_PARTIAL: + } else if (intel_gtt_view_is_partial(&vma->gtt_view)) { seq_printf(m, ", partial [%08llx+%x]", vma->gtt_view.partial.offset << PAGE_SHIFT, vma->gtt_view.partial.size << PAGE_SHIFT); - break; - - case I915_GTT_VIEW_ROTATED: + } else if (intel_gtt_view_is_rotated(&vma->gtt_view)) { seq_printf(m, ", rotated [(%ux%u, src_stride=%u, dst_stride=%u, offset=%u), (%ux%u, src_stride=%u, dst_stride=%u, offset=%u)]", vma->gtt_view.rotated.plane[0].width, vma->gtt_view.rotated.plane[0].height, @@ -232,9 +227,7 @@ i915_debugfs_describe_obj(struct seq_file *m, struct drm_i915_gem_object *obj) vma->gtt_view.rotated.plane[1].src_stride, vma->gtt_view.rotated.plane[1].dst_stride, vma->gtt_view.rotated.plane[1].offset); - break; - - case I915_GTT_VIEW_REMAPPED: + } else if (intel_gtt_view_is_remapped(&vma->gtt_view)) { seq_printf(m, ", remapped [(%ux%u, src_stride=%u, dst_stride=%u, offset=%u), (%ux%u, src_stride=%u, dst_stride=%u, offset=%u)]", vma->gtt_view.remapped.plane[0].width, vma->gtt_view.remapped.plane[0].height, @@ -246,11 +239,8 @@ i915_debugfs_describe_obj(struct seq_file *m, struct drm_i915_gem_object *obj) vma->gtt_view.remapped.plane[1].src_stride, vma->gtt_view.remapped.plane[1].dst_stride, vma->gtt_view.remapped.plane[1].offset); - break; - - default: + } else { MISSING_CASE(vma->gtt_view.type); - break; } } if (vma->fence) From 2c1bb96681bea17c2ffbd134816961615292a159 Mon Sep 17 00:00:00 2001 From: Jani Nikula Date: Wed, 30 Sep 2026 18:00:06 +0300 Subject: [PATCH 0500/1352] drm/intel: rename I915_GTT_VIEW_* enumerations to INTEL_GTT_VIEW_* MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Make the I915_GTT_VIEW_* enumerators less i915 specific, and rename them INTEL_GTT_VIEW_*. $ sed -i 's/I915_GTT_VIEW_/INTEL_GTT_VIEW_/g' -- $(git grep -l I915_GTT_VIEW_) Reviewed-by: Maarten Lankhorst Reviewed-by: Ville Syrjälä Link: https://patch.msgid.link/6e66fd6b61e361ffde316e5399baf9432383ced5.1790780340.git.jani.nikula@intel.com Signed-off-by: Jani Nikula --- drivers/gpu/drm/i915/display/intel_fb.c | 8 ++++---- drivers/gpu/drm/i915/gem/i915_gem_mman.c | 6 +++--- drivers/gpu/drm/i915/i915_vma.h | 8 ++++---- drivers/gpu/drm/i915/i915_vma_types.h | 10 +++++----- drivers/gpu/drm/i915/selftests/i915_vma.c | 12 ++++++------ include/drm/intel/gtt_view_types.h | 16 ++++++++-------- 6 files changed, 30 insertions(+), 30 deletions(-) diff --git a/drivers/gpu/drm/i915/display/intel_fb.c b/drivers/gpu/drm/i915/display/intel_fb.c index d1b4efa08bcc35..9d71f594537f2c 100644 --- a/drivers/gpu/drm/i915/display/intel_fb.c +++ b/drivers/gpu/drm/i915/display/intel_fb.c @@ -1706,7 +1706,7 @@ int intel_fill_fb_info(struct intel_display *display, struct intel_framebuffer * unsigned int tile_size = intel_tile_size(display); intel_fb_view_init(display, &fb->normal_view, - I915_GTT_VIEW_NORMAL, fb); + INTEL_GTT_VIEW_NORMAL, fb); drm_WARN_ON(display->drm, intel_fb_supports_90_270_rotation(fb) && @@ -1714,10 +1714,10 @@ int intel_fill_fb_info(struct intel_display *display, struct intel_framebuffer * if (intel_fb_supports_90_270_rotation(fb)) intel_fb_view_init(display, &fb->rotated_view, - I915_GTT_VIEW_ROTATED, fb); + INTEL_GTT_VIEW_ROTATED, fb); if (intel_fb_needs_pot_stride_remap(fb)) intel_fb_view_init(display, &fb->remapped_view, - I915_GTT_VIEW_REMAPPED, fb); + INTEL_GTT_VIEW_REMAPPED, fb); for (i = 0; i < num_planes; i++) { struct fb_plane_view_dims view_dims; @@ -1845,7 +1845,7 @@ static void intel_plane_remap_gtt(struct intel_plane_state *plane_state) intel_fb_view_init(display, &plane_state->view, drm_rotation_90_or_270(rotation) ? - I915_GTT_VIEW_ROTATED : I915_GTT_VIEW_REMAPPED, + INTEL_GTT_VIEW_ROTATED : INTEL_GTT_VIEW_REMAPPED, intel_fb); src_x = plane_state->uapi.src.x1 >> 16; diff --git a/drivers/gpu/drm/i915/gem/i915_gem_mman.c b/drivers/gpu/drm/i915/gem/i915_gem_mman.c index 055f8d3161dabc..6b337684dbfe88 100644 --- a/drivers/gpu/drm/i915/gem/i915_gem_mman.c +++ b/drivers/gpu/drm/i915/gem/i915_gem_mman.c @@ -206,7 +206,7 @@ compute_partial_view(const struct drm_i915_gem_object *obj, if (i915_gem_object_is_tiled(obj)) chunk = roundup(chunk, tile_row_pages(obj) ?: 1); - view.type = I915_GTT_VIEW_PARTIAL; + view.type = INTEL_GTT_VIEW_PARTIAL; view.partial.offset = rounddown(page_offset, chunk); view.partial.size = min_t(unsigned int, chunk, @@ -214,7 +214,7 @@ compute_partial_view(const struct drm_i915_gem_object *obj, /* If the partial covers the entire object, just create a normal VMA. */ if (chunk >= obj->base.size >> PAGE_SHIFT) - view.type = I915_GTT_VIEW_NORMAL; + view.type = INTEL_GTT_VIEW_NORMAL; return view; } @@ -407,7 +407,7 @@ static vm_fault_t vm_fault_gtt(struct vm_fault *vmf) vma = i915_gem_object_ggtt_pin_ww(obj, &ww, &view, 0, 0, flags); if (IS_ERR(vma) && vma != ERR_PTR(-EDEADLK)) { flags = PIN_MAPPABLE; - view.type = I915_GTT_VIEW_PARTIAL; + view.type = INTEL_GTT_VIEW_PARTIAL; vma = i915_gem_object_ggtt_pin_ww(obj, &ww, &view, 0, 0, flags); } diff --git a/drivers/gpu/drm/i915/i915_vma.h b/drivers/gpu/drm/i915/i915_vma.h index a8a89bb0270cd0..b0aca10bf8e7a9 100644 --- a/drivers/gpu/drm/i915/i915_vma.h +++ b/drivers/gpu/drm/i915/i915_vma.h @@ -217,7 +217,7 @@ i915_vma_compare(struct i915_vma *vma, if (cmp) return cmp; - BUILD_BUG_ON(I915_GTT_VIEW_NORMAL != 0); + BUILD_BUG_ON(INTEL_GTT_VIEW_NORMAL != 0); cmp = vma->gtt_view.type; if (!view) return cmp; @@ -238,9 +238,9 @@ i915_vma_compare(struct i915_vma *vma, * we assert above that all branches have the same address, and that * each branch has a unique type/size. */ - BUILD_BUG_ON(I915_GTT_VIEW_NORMAL >= I915_GTT_VIEW_PARTIAL); - BUILD_BUG_ON(I915_GTT_VIEW_PARTIAL >= I915_GTT_VIEW_ROTATED); - BUILD_BUG_ON(I915_GTT_VIEW_ROTATED >= I915_GTT_VIEW_REMAPPED); + BUILD_BUG_ON(INTEL_GTT_VIEW_NORMAL >= INTEL_GTT_VIEW_PARTIAL); + BUILD_BUG_ON(INTEL_GTT_VIEW_PARTIAL >= INTEL_GTT_VIEW_ROTATED); + BUILD_BUG_ON(INTEL_GTT_VIEW_ROTATED >= INTEL_GTT_VIEW_REMAPPED); BUILD_BUG_ON(offsetof(typeof(*view), rotated) != offsetof(typeof(*view), partial)); BUILD_BUG_ON(offsetof(typeof(*view), rotated) != diff --git a/drivers/gpu/drm/i915/i915_vma_types.h b/drivers/gpu/drm/i915/i915_vma_types.h index 8815421eaefba9..95eaf68fdfae0c 100644 --- a/drivers/gpu/drm/i915/i915_vma_types.h +++ b/drivers/gpu/drm/i915/i915_vma_types.h @@ -112,10 +112,10 @@ static inline void assert_i915_gem_gtt_types(void) * we have to be careful that each branch has a unique size. */ switch ((enum intel_gtt_view_type)0) { - case I915_GTT_VIEW_NORMAL: - case I915_GTT_VIEW_PARTIAL: - case I915_GTT_VIEW_ROTATED: - case I915_GTT_VIEW_REMAPPED: + case INTEL_GTT_VIEW_NORMAL: + case INTEL_GTT_VIEW_PARTIAL: + case INTEL_GTT_VIEW_ROTATED: + case INTEL_GTT_VIEW_REMAPPED: /* gcc complains if these are identical cases */ break; } @@ -231,7 +231,7 @@ struct i915_vma { * Support different GGTT views into the same object. * This means there can be multiple VMA mappings per object and per VM. * intel_gtt_view_type is used to distinguish between those entries. - * The default one of zero (I915_GTT_VIEW_NORMAL) is default and also + * The default one of zero (INTEL_GTT_VIEW_NORMAL) is default and also * assumed in GEM functions which take no ggtt view parameter. */ struct intel_gtt_view gtt_view; diff --git a/drivers/gpu/drm/i915/selftests/i915_vma.c b/drivers/gpu/drm/i915/selftests/i915_vma.c index 182c9f8606281b..fc068dab27f440 100644 --- a/drivers/gpu/drm/i915/selftests/i915_vma.c +++ b/drivers/gpu/drm/i915/selftests/i915_vma.c @@ -573,8 +573,8 @@ static int igt_vma_rotate_remap(void *arg) { } }, *a, *b; enum intel_gtt_view_type types[] = { - I915_GTT_VIEW_ROTATED, - I915_GTT_VIEW_REMAPPED, + INTEL_GTT_VIEW_ROTATED, + INTEL_GTT_VIEW_REMAPPED, 0, }, *t; const unsigned int max_pages = 64; @@ -825,12 +825,12 @@ static int igt_vma_partial(void *arg) for_each_prime_number_from(offset, 0, npages - sz) { struct intel_gtt_view view; - view.type = I915_GTT_VIEW_PARTIAL; + view.type = INTEL_GTT_VIEW_PARTIAL; view.partial.offset = offset; view.partial.size = sz; if (sz == npages) - view.type = I915_GTT_VIEW_NORMAL; + view.type = INTEL_GTT_VIEW_NORMAL; vma = checked_vma_instance(obj, vm, &view); if (IS_ERR(vma)) { @@ -982,8 +982,8 @@ static int igt_vma_remapped_gtt(void *arg) { } }, *p; enum intel_gtt_view_type types[] = { - I915_GTT_VIEW_ROTATED, - I915_GTT_VIEW_REMAPPED, + INTEL_GTT_VIEW_ROTATED, + INTEL_GTT_VIEW_REMAPPED, 0, }, *t; struct drm_i915_gem_object *obj; diff --git a/include/drm/intel/gtt_view_types.h b/include/drm/intel/gtt_view_types.h index d77292897d6448..128392d98f3a57 100644 --- a/include/drm/intel/gtt_view_types.h +++ b/include/drm/intel/gtt_view_types.h @@ -40,10 +40,10 @@ struct intel_remapped_info { } __packed; enum intel_gtt_view_type { - I915_GTT_VIEW_NORMAL = 0, - I915_GTT_VIEW_ROTATED = sizeof(struct intel_rotation_info), - I915_GTT_VIEW_PARTIAL = sizeof(struct intel_partial_info), - I915_GTT_VIEW_REMAPPED = sizeof(struct intel_remapped_info), + INTEL_GTT_VIEW_NORMAL = 0, + INTEL_GTT_VIEW_ROTATED = sizeof(struct intel_rotation_info), + INTEL_GTT_VIEW_PARTIAL = sizeof(struct intel_partial_info), + INTEL_GTT_VIEW_REMAPPED = sizeof(struct intel_remapped_info), }; struct intel_gtt_view { @@ -58,22 +58,22 @@ struct intel_gtt_view { static inline bool intel_gtt_view_is_normal(const struct intel_gtt_view *view) { - return view->type == I915_GTT_VIEW_NORMAL; + return view->type == INTEL_GTT_VIEW_NORMAL; } static inline bool intel_gtt_view_is_remapped(const struct intel_gtt_view *view) { - return view->type == I915_GTT_VIEW_REMAPPED; + return view->type == INTEL_GTT_VIEW_REMAPPED; } static inline bool intel_gtt_view_is_rotated(const struct intel_gtt_view *view) { - return view->type == I915_GTT_VIEW_ROTATED; + return view->type == INTEL_GTT_VIEW_ROTATED; } static inline bool intel_gtt_view_is_partial(const struct intel_gtt_view *view) { - return view->type == I915_GTT_VIEW_PARTIAL; + return view->type == INTEL_GTT_VIEW_PARTIAL; } #endif /* __DRM_INTEL_GTT_VIEW_TYPES_H__ */ From 712b27836b59885bd976028fbda24d05f8f91487 Mon Sep 17 00:00:00 2001 From: Krzysztof Kozlowski Date: Thu, 1 Oct 2026 14:26:21 +0200 Subject: [PATCH 0501/1352] soc: document merges Signed-off-by: Krzysztof Kozlowski --- arch/arm/arm-soc-for-next-contents.txt | 24 ++++++++++++++++++++++++ 1 file changed, 24 insertions(+) diff --git a/arch/arm/arm-soc-for-next-contents.txt b/arch/arm/arm-soc-for-next-contents.txt index 2f1c655335b588..c3856c535cfcf6 100644 --- a/arch/arm/arm-soc-for-next-contents.txt +++ b/arch/arm/arm-soc-for-next-contents.txt @@ -1,10 +1,34 @@ soc/arm soc/dt + renesas/dt + https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel tags/renesas-dts-for-v7.4-tag1 + rockchip/dt64 + ssh://gitolite.kernel.org/pub/scm/linux/kernel/git/mmind/linux-rockchip tags/v7.4-rockchip-dts64-1 + qcom/dt64 + ssh://gitolite.kernel.org/pub/scm/linux/kernel/git/qcom/linux tags/qcom-arm64-for-7.4 + mediatek/dt64 + ssh://gitolite.kernel.org/pub/scm/linux/kernel/git/mediatek/linux tags/mtk-dts64-for-v7.4 + mediatek/dt32 + ssh://gitolite.kernel.org/pub/scm/linux/kernel/git/mediatek/linux tags/mtk-dts32-for-v7.4 + arm-juno/dt + https://git.kernel.org/pub/scm/linux/kernel/git/sudeep.holla/linux tags/juno-updates-7.4 soc/drivers + renesas/drivers + https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel tags/renesas-drivers-for-v7.4-tag1 + patch + platform: cznic: turris-omnia-mcu: Add missing MODULE_DEVICE_TABLE() + mediatek/drivers + ssh://gitolite.kernel.org/pub/scm/linux/kernel/git/mediatek/linux tags/mtk-soc-for-v7.4 + scmi/drivers + https://git.kernel.org/pub/scm/linux/kernel/git/sudeep.holla/linux tags/scmi-updates-7.4 soc/defconfig + renesas/defconfig + https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel tags/renesas-arm-defconfig-for-v7.4-tag1 + qcom/defconfig + ssh://gitolite.kernel.org/pub/scm/linux/kernel/git/qcom/linux tags/qcom-arm64-defconfig-for-7.4 soc/late From e25be1ba74f2251b4badc1402c392f0abd62cb0a Mon Sep 17 00:00:00 2001 From: Krzysztof Kozlowski Date: Thu, 1 Oct 2026 14:39:58 +0200 Subject: [PATCH 0502/1352] soc: document merges Signed-off-by: Krzysztof Kozlowski --- arch/arm/arm-soc-for-next-contents.txt | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/arch/arm/arm-soc-for-next-contents.txt b/arch/arm/arm-soc-for-next-contents.txt index c3856c535cfcf6..381938ff5b74bd 100644 --- a/arch/arm/arm-soc-for-next-contents.txt +++ b/arch/arm/arm-soc-for-next-contents.txt @@ -1,4 +1,19 @@ soc/arm + patch + ARM: remove sa1100 platform + ARM: remove footbridge + ARM: remove legacy pxa board files + ARM: orion/dove/mv78xx0: remove all board files + ARM: orion5x: fold plat-orion/pcie.c and hw_pci into pci.c + ARM: omap2: remove omap24xx support + ARM: imx: remove i.MX31 SoC support + ARM: versatile: remove Integrator/CM1136JF-S option + ARM: imx: remove nommu support + ARM: lpc18xx: remove entire platform + ARM: stm32: remove stm32f4/f7/h7 MCU support + ARM: versatile: remove mps2 support + ARM: at91: remove samv7 support + ARM: axxia: remove entire platform soc/dt renesas/dt From 97958feb6560b4f5eb57addeb9d3a214d55b154b Mon Sep 17 00:00:00 2001 From: Nikola Prica Date: Mon, 21 Sep 2026 13:19:03 +0200 Subject: [PATCH 0503/1352] PCI: Accept AtomicOps already enabled by the hypervisor MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit pci_enable_atomic_ops_to_root() currently fails when no Root Port is visible. That is common in passthrough guests (ESXi, Hyper-V): the Endpoint is assigned to the VM, but the Root Port above it is not visible in the guest topology. In those setups the hypervisor may already have enabled AtomicOp Requester Enable on the device. If PCI_EXP_DEVCTL2_ATOMIC_REQ is set, treat AtomicOps as already enabled and return success instead of failing the Root Port walk. After 1ae8c4ce1570 ("PCI: Enable AtomicOps only if Root Port supports them"), pci_enable_atomic_ops_to_root() always returns failure if the Root Port is not visible, so drivers don't use atomics when they could. On systems where the Root Port is not visible but *does* support AtomicOps, this is a regression: prior to 1ae8c4ce1570, it enabled AtomicOps in the endpoint and returned success. Fixes: 1ae8c4ce1570 ("PCI: Enable AtomicOps only if Root Port supports them") Signed-off-by: Nikola Prica [bhelgaas: commit log] Signed-off-by: Bjorn Helgaas Tested-by: Gerd Bayer Reviewed-by: Christian König Reviewed-by: Gerd Bayer Link: https://patch.msgid.link/20260921111903.978687-1-nikprica@amd.com --- drivers/pci/pci.c | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/drivers/pci/pci.c b/drivers/pci/pci.c index b2879a6be5f808..62729ade496fca 100644 --- a/drivers/pci/pci.c +++ b/drivers/pci/pci.c @@ -3769,8 +3769,18 @@ int pci_enable_atomic_ops_to_root(struct pci_dev *dev, u32 cap_mask) } root = pcie_find_root_port(dev); - if (!root) + if (!root) { + /* + * A hypervisor may expose a headless topology with no + * visible root port. If it has already set AtomicOp + * Requester Enable, there is nothing more to do. + */ + pcie_capability_read_dword(dev, PCI_EXP_DEVCTL2, &ctl2); + if (ctl2 & PCI_EXP_DEVCTL2_ATOMIC_REQ) + return 0; + return -EINVAL; + } pcie_capability_read_dword(root, PCI_EXP_DEVCAP2, &cap); if ((cap & cap_mask) != cap_mask) From d7a50931398d6ac35f2a2cbd2f944410b75d7e6f Mon Sep 17 00:00:00 2001 From: Yong-Xuan Wang Date: Mon, 17 Aug 2026 00:37:04 -0700 Subject: [PATCH 0504/1352] RISC-V: errata: Add SiFive MAL-9092 workaround SiFive cores with MAL-9092 require fence.rw instructions around HLV/HLVX/HSV hypervisor instructions to ensure proper memory ordering when accessing guest memory. Note that HSTATUS.HU is cleared during KVM initialization, restricting hypervisor instructions to HS-mode only. If future changes enable HSTATUS.HU for U-mode access, additional errata considerations will be needed. Signed-off-by: Yong-Xuan Wang Reviewed-by: Samuel Holland Link: https://patch.msgid.link/20260817-sifive-errata-v1-1-6a80f1a2743d@sifive.com Signed-off-by: Paul Walmsley --- arch/riscv/Kconfig.errata | 14 ++++++++++++++ arch/riscv/errata/sifive/errata.c | 20 ++++++++++++++++++++ arch/riscv/include/asm/errata_list_vendors.h | 3 ++- arch/riscv/include/asm/insn-def.h | 16 +++++++++++++--- 4 files changed, 49 insertions(+), 4 deletions(-) diff --git a/arch/riscv/Kconfig.errata b/arch/riscv/Kconfig.errata index 38ade0ea4e9cb7..a6aac8774f250a 100644 --- a/arch/riscv/Kconfig.errata +++ b/arch/riscv/Kconfig.errata @@ -96,6 +96,20 @@ config ERRATA_STARFIVE_JH7100 Say "Y" if you want to support the BeagleV Starlight and/or StarFive VisionFive V1 boards. +config ERRATA_SIFIVE_MAL_9092 + bool "Apply SiFive HLV/HLVX/HSV fence errata" + depends on ERRATA_SIFIVE && 64BIT + default y + help + This will apply the SiFive MAL-9092 errata to add fence.rw + instructions around all HLV/HLVX/HSV instructions. + + Note: HSTATUS.HU is cleared during KVM initialization. If future + changes enable HSTATUS.HU to allow hypervisor instructions in + U-mode, additional errata handling may be required. + + If you don't know what to do here, say "Y". + config ERRATA_THEAD bool "T-HEAD errata" depends on RISCV_ALTERNATIVE diff --git a/arch/riscv/errata/sifive/errata.c b/arch/riscv/errata/sifive/errata.c index df80c9614df1a0..1a5d223ebf35b8 100644 --- a/arch/riscv/errata/sifive/errata.c +++ b/arch/riscv/errata/sifive/errata.c @@ -51,6 +51,22 @@ static bool errata_cip_1200_check_func(unsigned long arch_id, unsigned long imp return true; } +static bool errata_mal_9092_check_func(unsigned long arch_id, unsigned long impid) +{ + /* + * Affected cores: + * Architecture ID: 0x8000000000000109 + * Implementation ID: 0x19251031, 0x19251217, 0x19260320 + */ + + if (arch_id != 0x8000000000000109) + return false; + if (impid != 0x19251031 && impid != 0x19251217 && impid != 0x19260320) + return false; + + return true; +} + static struct errata_info_t errata_list[ERRATA_SIFIVE_NUMBER] = { { .name = "cip-453", @@ -60,6 +76,10 @@ static struct errata_info_t errata_list[ERRATA_SIFIVE_NUMBER] = { .name = "cip-1200", .check_func = errata_cip_1200_check_func }, + { + .name = "mal-9092", + .check_func = errata_mal_9092_check_func + }, }; static u32 __init_or_module sifive_errata_probe(unsigned long archid, diff --git a/arch/riscv/include/asm/errata_list_vendors.h b/arch/riscv/include/asm/errata_list_vendors.h index ec7eba3734371a..c62a82f60468a3 100644 --- a/arch/riscv/include/asm/errata_list_vendors.h +++ b/arch/riscv/include/asm/errata_list_vendors.h @@ -11,7 +11,8 @@ #ifdef CONFIG_ERRATA_SIFIVE #define ERRATA_SIFIVE_CIP_453 0 #define ERRATA_SIFIVE_CIP_1200 1 -#define ERRATA_SIFIVE_NUMBER 2 +#define ERRATA_SIFIVE_MAL_9092 2 +#define ERRATA_SIFIVE_NUMBER 3 #endif #ifdef CONFIG_ERRATA_THEAD diff --git a/arch/riscv/include/asm/insn-def.h b/arch/riscv/include/asm/insn-def.h index 7c6daf11675675..8fcc848c196e00 100644 --- a/arch/riscv/include/asm/insn-def.h +++ b/arch/riscv/include/asm/insn-def.h @@ -192,18 +192,28 @@ INSN_R(OPCODE_SYSTEM, FUNC3(0), FUNC7(49), \ __RD(0), RS1(gaddr), RS2(vmid)) +#define ALT_SIFIVE_MAL_9092_FENCE \ +ALTERNATIVE("nop", "fence rw, rw", SIFIVE_VENDOR_ID, \ + ERRATA_SIFIVE_MAL_9092, CONFIG_ERRATA_SIFIVE_MAL_9092) + #define HLVX_HU(dest, addr) \ + ALT_SIFIVE_MAL_9092_FENCE "\n" \ INSN_R(OPCODE_SYSTEM, FUNC3(4), FUNC7(50), \ - RD(dest), RS1(addr), __RS2(3)) + RD(dest), RS1(addr), __RS2(3)) "\n" \ + ALT_SIFIVE_MAL_9092_FENCE #define HLV_W(dest, addr) \ + ALT_SIFIVE_MAL_9092_FENCE "\n" \ INSN_R(OPCODE_SYSTEM, FUNC3(4), FUNC7(52), \ - RD(dest), RS1(addr), __RS2(0)) + RD(dest), RS1(addr), __RS2(0)) "\n" \ + ALT_SIFIVE_MAL_9092_FENCE #ifdef CONFIG_64BIT #define HLV_D(dest, addr) \ + ALT_SIFIVE_MAL_9092_FENCE "\n" \ INSN_R(OPCODE_SYSTEM, FUNC3(4), FUNC7(54), \ - RD(dest), RS1(addr), __RS2(0)) + RD(dest), RS1(addr), __RS2(0)) "\n" \ + ALT_SIFIVE_MAL_9092_FENCE #else #define HLV_D(dest, addr) \ __ASM_STR(.error "hlv.d requires 64-bit support") From 247ac82ac82d82cfae97da2a76cd30a071950d30 Mon Sep 17 00:00:00 2001 From: Yong-Xuan Wang Date: Thu, 24 Sep 2026 03:16:13 -0700 Subject: [PATCH 0505/1352] RISC-V: Clear HSTATUS.HU on CPU initialization The RISC-V privileged specification does not mandate HSTATUS reset values, leaving the HU bit potentially set after hardware reset. When HU=1, hypervisor instructions (HLV/HLVX/HSV) can execute in U-mode to access guest memory, which may cause unintended behavior if not explicitly controlled. Clear HSTATUS.HU during CPU initialization to ensure hypervisor instructions are only available in HS-mode, preventing unexpected guest memory access from U-mode code. Signed-off-by: Yong-Xuan Wang Reviewed-by: Samuel Holland Link: https://patch.msgid.link/20260924-hstatus_hu-v2-1-7e970f5f1d8d@sifive.com [pjw@kernel.org: capitalized RISCV_ISA_EXT_h to fix build] Signed-off-by: Paul Walmsley --- arch/riscv/include/asm/cpufeature.h | 2 ++ arch/riscv/kernel/cpufeature.c | 12 ++++++++++++ arch/riscv/kernel/setup.c | 2 ++ arch/riscv/kernel/smpboot.c | 2 ++ arch/riscv/kernel/suspend.c | 2 ++ 5 files changed, 20 insertions(+) diff --git a/arch/riscv/include/asm/cpufeature.h b/arch/riscv/include/asm/cpufeature.h index 739fcc84bf7b28..5efa72823475fd 100644 --- a/arch/riscv/include/asm/cpufeature.h +++ b/arch/riscv/include/asm/cpufeature.h @@ -40,6 +40,8 @@ extern u32 thead_vlenb_of; void __init riscv_user_isa_enable(void); +void riscv_clear_hypervisor_csr(void); + #define _RISCV_ISA_EXT_DATA(_name, _id, _subset_exts, _subset_exts_size, _validate) { \ .name = #_name, \ .property = #_name, \ diff --git a/arch/riscv/kernel/cpufeature.c b/arch/riscv/kernel/cpufeature.c index 61d21f7148305c..7329792978cc0f 100644 --- a/arch/riscv/kernel/cpufeature.c +++ b/arch/riscv/kernel/cpufeature.c @@ -1233,6 +1233,18 @@ void __init riscv_user_isa_enable(void) pr_warn("Zicbop disabled as it is unavailable on some harts\n"); } +void riscv_clear_hypervisor_csr(void) +{ + if (!riscv_has_extension_unlikely(RISCV_ISA_EXT_H)) + return; + + /* + * Clear HSTATUS.HU to restrict hypervisor instructions to HS-mode. + * This prevents user-mode from executing HLV/HSV instructions. + */ + csr_clear(CSR_HSTATUS, HSTATUS_HU); +} + #ifdef CONFIG_RISCV_ALTERNATIVE /* * Alternative patch sites consider 48 bits when determining when to patch diff --git a/arch/riscv/kernel/setup.c b/arch/riscv/kernel/setup.c index a32344bb220dff..d79f99b8603342 100644 --- a/arch/riscv/kernel/setup.c +++ b/arch/riscv/kernel/setup.c @@ -366,6 +366,8 @@ void __init setup_arch(char **cmdline_p) if (!IS_ENABLED(CONFIG_RISCV_ISA_ZBB) || !riscv_isa_extension_available(NULL, ZBB)) static_branch_disable(&efficient_ffs_key); + + riscv_clear_hypervisor_csr(); } bool arch_cpu_is_hotpluggable(int cpu) diff --git a/arch/riscv/kernel/smpboot.c b/arch/riscv/kernel/smpboot.c index f6ef57930b50a8..a9de2dae804c61 100644 --- a/arch/riscv/kernel/smpboot.c +++ b/arch/riscv/kernel/smpboot.c @@ -244,6 +244,8 @@ asmlinkage __visible void smp_callin(void) numa_add_cpu(curr_cpuid); + riscv_clear_hypervisor_csr(); + pr_debug("CPU%u: Booted secondary hartid %lu\n", curr_cpuid, cpuid_to_hartid_map(curr_cpuid)); diff --git a/arch/riscv/kernel/suspend.c b/arch/riscv/kernel/suspend.c index 3efbf7874f3ba2..db220966f78272 100644 --- a/arch/riscv/kernel/suspend.c +++ b/arch/riscv/kernel/suspend.c @@ -43,6 +43,8 @@ void suspend_save_csrs(struct suspend_context *context) void suspend_restore_csrs(struct suspend_context *context) { + riscv_clear_hypervisor_csr(); + csr_write(CSR_SCRATCH, 0); if (riscv_has_extension_unlikely(RISCV_ISA_EXT_XLINUXENVCFG)) csr_write(CSR_ENVCFG, context->envcfg); From 4ae9ed2068014eca91ea5b8dfa5626aadbe44911 Mon Sep 17 00:00:00 2001 From: "Paul E. McKenney" Date: Thu, 3 Sep 2026 15:26:24 -0700 Subject: [PATCH 0506/1352] rcu: Add running and boosted indications to RCU task stall dump Currently, rcu_print_task_stall() will dump out the PID, RCU reader nesting level, the rcu_special structure's flags, and whether or not that reader is on the ->blkd_tasks list. When debugging RCU priority boosting, it is also good to know whether the stalled RCU reader is currently running and whether it is currently being RCU priority boosted. This commit therefore adds this information to the output. [ paulmck: Apply kernel test robot feedback. ] [ paulmck: Apply feedback from Arnd Bergmann and kernel test robot. ] Link: https://lore.kernel.org/all/20260925133922.1356404-1-arnd@kernel.org/ Link: https://lore.kernel.org/all/202609261005.GrnaXBXT-lkp@intel.com/ Signed-off-by: Paul E. McKenney Reviewed-by: Bradley Morgan --- kernel/rcu/tree_plugin.h | 2 -- kernel/rcu/tree_stall.h | 24 ++++++++++++++++++++++-- 2 files changed, 22 insertions(+), 4 deletions(-) diff --git a/kernel/rcu/tree_plugin.h b/kernel/rcu/tree_plugin.h index 743c16247fc057..ae5d21bae7bc0f 100644 --- a/kernel/rcu/tree_plugin.h +++ b/kernel/rcu/tree_plugin.h @@ -11,8 +11,6 @@ * Paul E. McKenney */ -#include "../locking/rtmutex_common.h" - static bool rcu_rdp_is_offloaded(struct rcu_data *rdp) { /* diff --git a/kernel/rcu/tree_stall.h b/kernel/rcu/tree_stall.h index 93ba31a619b670..803a56af3259f6 100644 --- a/kernel/rcu/tree_stall.h +++ b/kernel/rcu/tree_stall.h @@ -11,6 +11,8 @@ #include #include #include +#include +#include "../locking/rtmutex_common.h" ////////////////////////////////////////////////////////////////////////////// // @@ -305,6 +307,8 @@ struct rcu_stall_chk_rdr { int nesting; union rcu_special rs; bool on_blkd_list; + bool rcu_rdr_running; + int rcu_rdr_boosted; }; /* @@ -313,6 +317,7 @@ struct rcu_stall_chk_rdr { */ static int check_slow_task(struct task_struct *t, void *arg) { + struct rcu_node *rnp; struct rcu_stall_chk_rdr *rscrp = arg; if (task_curr(t)) @@ -320,6 +325,19 @@ static int check_slow_task(struct task_struct *t, void *arg) rscrp->nesting = t->rcu_read_lock_nesting; rscrp->rs = t->rcu_read_unlock_special; rscrp->on_blkd_list = !list_empty(&t->rcu_node_entry); + rscrp->rcu_rdr_running = task_curr(t); + rscrp->rcu_rdr_boosted = 0; + if (rscrp->on_blkd_list) { + rnp = READ_ONCE(t->rcu_blocked_node); + raw_spin_lock_rcu_node(rnp); /* irqs already disabled. */ + if (rnp == READ_ONCE(t->rcu_blocked_node)) { + if (rt_mutex_owner(&rnp->boost_mtx.rtmutex) == t) + rscrp->rcu_rdr_boosted = 1; + } else { + rscrp->rcu_rdr_boosted = 2; + } + raw_spin_unlock_rcu_node(rnp); /* irqs remain disabled. */ + } return 0; } @@ -357,12 +375,14 @@ static int rcu_print_task_stall(struct rcu_node *rnp, unsigned long flags) if (task_call_func(t, check_slow_task, &rscr)) pr_cont(" P%d", t->pid); else - pr_cont(" P%d/%d:%c%c%c%c", + pr_cont(" P%d/%d:%c%c%c%c%c%c", t->pid, rscr.nesting, ".b"[rscr.rs.b.blocked], ".q"[rscr.rs.b.need_qs], ".e"[rscr.rs.b.exp_hint], - ".l"[rscr.on_blkd_list]); + ".l"[rscr.on_blkd_list], + ".R"[rscr.rcu_rdr_running], + ".B?"[rscr.rcu_rdr_boosted]); lockdep_assert_irqs_disabled(); put_task_struct(t); ndetected++; From d493512fb4ac017178f8ace0be6004be50089e5f Mon Sep 17 00:00:00 2001 From: Hemanth Selam Date: Mon, 7 Sep 2026 12:06:22 +0530 Subject: [PATCH 0507/1352] rcu: Fix typo "upto" in comment Correct "upto" to "up to", reported by scripts/checkpatch.pl using the misspelling list in scripts/spelling.txt. Only touches comments, no code changes. Assisted-by: Cursor:claude-opus-5 Signed-off-by: Hemanth Selam Signed-off-by: Paul E. McKenney --- kernel/rcu/srcutree.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/kernel/rcu/srcutree.c b/kernel/rcu/srcutree.c index ed204b3f4b8440..01f19bb17dd6d6 100644 --- a/kernel/rcu/srcutree.c +++ b/kernel/rcu/srcutree.c @@ -624,7 +624,7 @@ module_param(srcu_retry_check_delay, ulong, 0444); #define SRCU_UL_CLAMP_LO(val, low) ((val) > (low) ? (val) : (low)) #define SRCU_UL_CLAMP_HI(val, high) ((val) < (high) ? (val) : (high)) #define SRCU_UL_CLAMP(val, low, high) SRCU_UL_CLAMP_HI(SRCU_UL_CLAMP_LO((val), (low)), (high)) -// per-GP-phase no-delay instances adjusted to allow non-sleeping poll upto +// per-GP-phase no-delay instances adjusted to allow non-sleeping poll up to // one jiffies time duration. Mult by 2 is done to factor in the srcu_get_delay() // called from process_srcu(). #define SRCU_DEFAULT_MAX_NODELAY_PHASE_ADJUSTED \ From 22ff74ca7664c90bb8e9eece00ec9470d14a0d1e Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Thu, 10 Sep 2026 17:25:37 +0000 Subject: [PATCH 0508/1352] rcu: Drop the private tick-internal.h include from tree.c Nothing in the tree.c translation unit uses anything from the private timekeeping header ../time/tick-internal.h, included since commit 48d07c04b4cc ("rcu: Enable elimination of Tree-RCU softirq processing"). The tick symbols RCU uses, tick_dep_set(), tick_dep_clear(), their _cpu and _task variants, tick_nohz_full_cpu() and TICK_DEP_BIT_RCU, are all declared in the public linux/tick.h, which tree.c already includes. Drop the include. Signed-off-by: Bradley Morgan Signed-off-by: Paul E. McKenney --- kernel/rcu/tree.c | 1 - 1 file changed, 1 deletion(-) diff --git a/kernel/rcu/tree.c b/kernel/rcu/tree.c index 96848fc1f02b8f..338737b9781cba 100644 --- a/kernel/rcu/tree.c +++ b/kernel/rcu/tree.c @@ -64,7 +64,6 @@ #include #include #include -#include "../time/tick-internal.h" #include "tree.h" #include "rcu.h" From a18b96518d370940ce5daa829bcee84f9d6fda76 Mon Sep 17 00:00:00 2001 From: Matthias Goergens Date: Fri, 11 Sep 2026 11:40:59 +0800 Subject: [PATCH 0509/1352] rcu: Make userspace barrier hook drain kvfree_rcu work The bcachefs ktest allocation-leak check writes rcutree.do_rcu_barrier before reading /proc/allocinfo. While testing bcachefs performance changes, small objects released with kfree_rcu() remained visible after repeated writes to the hook and 20 seconds of waiting, causing otherwise clean tests to fail their leak check. The test assumes a stronger contract than the hook currently documents: rcu_barrier() waits for ordinary callbacks, but does not flush objects still held in kfree_rcu() batching or per-CPU SLUB sheaves. The retained population eventually fell as a sheaf filled; there is no evidence here of unbounded growth or OOM. Changing the hook to drain kvfree_rcu() work let the same unmodified bcachefs workload pass its allocation check. All eight checkpoints in one VM, after 50 through 400 option changes, reported zero retained reconcile_scan objects. This motivated the separate private-cache test used to isolate the incomplete drain from bcachefs. Calling kvfree_rcu_barrier() from rcu_barrier_throttled() was proposed when kvfree_rcu_barrier() was added in 2024, to restore a clean baseline between userspace benchmark runs. The discussion concluded that keeping the existing hook name, adding the second operation and documenting both was the safest compatibility choice, but the follow-up was not added. Add that drain and document the stronger test interface. Always retain the existing start-rate limit and perform the kvfree_rcu() drain: an unrelated ordinary barrier does not establish that this work completed. Retain the entry ordinary-barrier sequence snapshot. After draining, skip the final ordinary barrier only if that snapshot is complete, preserving the memory barrier on the completion path. Otherwise, invoke rcu_barrier() explicitly. This keeps the ordinary-callback guarantee independent of whether kvfree_rcu_barrier() embeds an ordinary barrier. Clarify that the documented completion guarantee covers work queued before the request, without preventing new work from being queued. Earlier validation of the unconditional-drain version used four fresh VM pairs with a private-cache fixture: controls retained the queued object (60 to 60 active objects), and treatments drained it (60 to 59). An ordinary-callback test passed on both kernels. Those runs predated the guarded skip and do not validate that change. No elapsed-time improvement is claimed. Link: https://lore.kernel.org/all/20240820155935.1167988-1-urezki@gmail.com/ Signed-off-by: Matthias Goergens Signed-off-by: Paul E. McKenney --- .../admin-guide/kernel-parameters.txt | 9 ++++-- kernel/rcu/tree.c | 30 ++++++++++++------- 2 files changed, 26 insertions(+), 13 deletions(-) diff --git a/Documentation/admin-guide/kernel-parameters.txt b/Documentation/admin-guide/kernel-parameters.txt index 68647ff4bdd24b..914b65ae941346 100644 --- a/Documentation/admin-guide/kernel-parameters.txt +++ b/Documentation/admin-guide/kernel-parameters.txt @@ -5699,9 +5699,12 @@ Kernel parameters there is an ongoing too-long CSD-lock wait. rcutree.do_rcu_barrier= [KNL] - Request a call to rcu_barrier(). This is - throttled so that userspace tests can safely - hammer on the sysfs variable if they so choose. + Wait for deferred kfree_rcu() frees and ordinary + call_rcu() callbacks queued before this request to + complete. This does not prevent new work from being + queued concurrently. Requests are throttled so that + userspace tests can safely hammer on the sysfs + variable if they so choose. If triggered before the RCU grace-period machinery is fully active, this will error out with EAGAIN. diff --git a/kernel/rcu/tree.c b/kernel/rcu/tree.c index 338737b9781cba..f60252390d5dea 100644 --- a/kernel/rcu/tree.c +++ b/kernel/rcu/tree.c @@ -3988,12 +3988,12 @@ EXPORT_SYMBOL_GPL(rcu_barrier); static unsigned long rcu_barrier_last_throttle; /** - * rcu_barrier_throttled - Do rcu_barrier(), but limit to one per second + * rcu_barrier_throttled - Drain deferred RCU frees, but rate-limit starts * - * This can be thought of as guard rails around rcu_barrier() that - * permits unrestricted userspace use, at least assuming the hardware's - * try_cmpxchg() is robust. There will be at most one call per second to - * rcu_barrier() system-wide from use of this function, which means that + * This can be thought of as guard rails around the deferred-free barriers + * that permit unrestricted userspace use, at least assuming the hardware's + * try_cmpxchg() is robust. There will be at most one drain operation started + * per sixteenth of a second from use of this function, which means that * callers might needlessly wait a second or three. * * This is intended for use by test suites to avoid OOM by flushing RCU @@ -4015,14 +4015,24 @@ static void rcu_barrier_throttled(void) while (time_in_range(j, old, old + HZ / 16) || !try_cmpxchg(&rcu_barrier_last_throttle, &old, j)) { schedule_timeout_idle(HZ / 16); - if (rcu_seq_done(&rcu_state.barrier_sequence, s)) { - smp_mb(); /* caller's subsequent code after above check. */ - return; - } j = jiffies; old = READ_ONCE(rcu_barrier_last_throttle); } - rcu_barrier(); + /* + * kfree_rcu() can retain objects outside the ordinary callback lists in + * per-CPU SLUB sheaves and kvfree_rcu batches. Always drain those queues: + * an ordinary barrier does not establish that this work was drained. + */ + kvfree_rcu_barrier(); + /* + * A completed barrier can still cover ordinary callbacks queued before + * our entry snapshot. Otherwise, retain an explicit ordinary barrier + * without depending on the implementation of kvfree_rcu_barrier(). + */ + if (rcu_seq_done(&rcu_state.barrier_sequence, s)) + smp_mb(); /* caller's subsequent code after above check. */ + else + rcu_barrier(); } /* From 0c70f3fa57cbad8baf58b417f114600289e5674e Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Fri, 11 Sep 2026 23:36:43 +0200 Subject: [PATCH 0510/1352] rcuref: Fix the rcuread_is_dead reference in rcuref_read() kernel-doc The kernel-doc comment of rcuref_read() refers to rcuread_is_dead, which does not exist. The name is rcuref_is_dead. Say rcuref_is_dead. Fixes: 3efa66ce6ee1 ("rcuref: Provide rcuref_is_dead()") Assisted-by: LLM Signed-off-by: Karl Mehltretter Reviewed-by: Sebastian Andrzej Siewior Signed-off-by: Paul E. McKenney --- include/linux/rcuref.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/include/linux/rcuref.h b/include/linux/rcuref.h index 2fb2af6d982497..01fee161b67c3e 100644 --- a/include/linux/rcuref.h +++ b/include/linux/rcuref.h @@ -34,7 +34,7 @@ static inline void rcuref_init(rcuref_t *ref, unsigned int cnt) * indicate that it is safe to schedule the object, protected by this reference * counter, for deconstruction. * If you want to know if the reference counter has been marked DEAD (as - * signaled by rcuref_put()) please use rcuread_is_dead(). + * signaled by rcuref_put()) please use rcuref_is_dead(). */ static inline unsigned int rcuref_read(rcuref_t *ref) { From c070b83bbb8213a834077e2970fc6298e2154f50 Mon Sep 17 00:00:00 2001 From: "Paul E. McKenney" Date: Tue, 11 Aug 2026 11:44:00 -0700 Subject: [PATCH 0511/1352] rcu-tasks: Disable callback contend/collapse messages by default New workloads can do large bursts of call_rcu_tasks() invocations in a short time period, followed by a quiet time period long enough to drain all of the callbacks, followed by another burst of call_rcu_tasks() invocations. This can cause RCU Tasks to switch back and forth between queuing callbacks only on CPU 0 (during quiet periods) and on all CPUs (during bursts). Which is fine. Except for the fact that each cycle from CPU-0-only to all-CPUs queuing and back generates three console messages, one announcing the shift to all-CPUs queuing, another announcing the start of the shift back to CPU-0-only queuing, and the third announcing completion of this shift after an RCU grace period. And these console messages can overrun console-log communications channels and obscure other console-message-based debugging information. And the only known use for these console messages is debugging RCU Tasks itself. This commit therefore adds a rcupdate.rcu_task_collapse_debug module parameter that defaults to false (suppressing these console messages). Those debugging or otherwise playing with RCU Tasks callback queuing auto-adjustment can set this parameter to the value true. [ paulmck: Apply Breno Leitao feedback. ] Reported-by: Breno Leitao Reported-by: David Dai Signed-off-by: Paul E. McKenney Reviewed-by: Breno Leitao --- Documentation/admin-guide/kernel-parameters.txt | 7 +++++++ kernel/rcu/tasks.h | 15 ++++++++++++--- 2 files changed, 19 insertions(+), 3 deletions(-) diff --git a/Documentation/admin-guide/kernel-parameters.txt b/Documentation/admin-guide/kernel-parameters.txt index 914b65ae941346..6cc6d45b59d68a 100644 --- a/Documentation/admin-guide/kernel-parameters.txt +++ b/Documentation/admin-guide/kernel-parameters.txt @@ -6405,6 +6405,13 @@ Kernel parameters period to instead use normal non-expedited grace-period processing. + rcupdate.rcu_task_collapse_debug= [KNL] + Enable debugging prints that record when RCU Tasks + and RCU Tasks Trace expand to per-CPU callback + queuing and collapse back to CPU-0 queuing. + This is default-disabled due to the fact that + some workloads can make it quite noisy. + rcupdate.rcu_task_collapse_lim= [KNL] Set the maximum number of callbacks present at the beginning of a grace period that allows diff --git a/kernel/rcu/tasks.h b/kernel/rcu/tasks.h index 627295396cd91d..fcac7361ec51ec 100644 --- a/kernel/rcu/tasks.h +++ b/kernel/rcu/tasks.h @@ -178,6 +178,8 @@ static int rcu_task_contend_lim __read_mostly = 100; module_param(rcu_task_contend_lim, int, 0444); static int rcu_task_collapse_lim __read_mostly = 10; module_param(rcu_task_collapse_lim, int, 0444); +static bool rcu_task_collapse_debug __read_mostly; +module_param(rcu_task_collapse_debug, bool, 0644); static int rcu_task_lazy_lim __read_mostly = 32; module_param(rcu_task_lazy_lim, int, 0444); @@ -390,7 +392,8 @@ static void call_rcu_tasks_generic(struct rcu_head *rhp, rcu_callback_t func, WRITE_ONCE(rtp->percpu_enqueue_shift, 0); WRITE_ONCE(rtp->percpu_dequeue_lim, rcu_task_cpu_ids); smp_store_release(&rtp->percpu_enqueue_lim, rcu_task_cpu_ids); - pr_info("Switching %s to per-CPU callback queuing.\n", rtp->name); + if (data_race(rcu_task_collapse_debug)) + pr_info("Switching %s to per-CPU callback queuing.\n", rtp->name); } raw_spin_unlock_irqrestore(&rtp->cbs_gbl_lock, flags); } @@ -511,7 +514,9 @@ static int rcu_tasks_need_gpcb(struct rcu_tasks *rtp) smp_store_release(&rtp->percpu_enqueue_lim, 1); rtp->percpu_dequeue_gpseq = get_state_synchronize_rcu(); gpdone = false; - pr_info("Starting switch %s to CPU-0 callback queuing.\n", rtp->name); + if (data_race(rcu_task_collapse_debug)) + pr_info("Starting switch %s to CPU-0 callback queuing.\n", + rtp->name); } raw_spin_unlock_irqrestore(&rtp->cbs_gbl_lock, flags); } @@ -519,7 +524,9 @@ static int rcu_tasks_need_gpcb(struct rcu_tasks *rtp) raw_spin_lock_irqsave(&rtp->cbs_gbl_lock, flags); if (rtp->percpu_enqueue_lim < rtp->percpu_dequeue_lim) { WRITE_ONCE(rtp->percpu_dequeue_lim, 1); - pr_info("Completing switch %s to CPU-0 callback queuing.\n", rtp->name); + if (data_race(rcu_task_collapse_debug)) + pr_info("Completing switch %s to CPU-0 callback queuing.\n", + rtp->name); } if (rtp->percpu_dequeue_lim == 1) { for (cpu = rtp->percpu_dequeue_lim; cpu < rcu_task_cpu_ids; cpu++) { @@ -704,6 +711,8 @@ static void __init rcu_tasks_bootup_oddness(void) pr_info("\tTasks-RCU CPU stall info multiplier clamped to %d (rcu_task_stall_info_mult).\n", rtsimc); rcu_task_stall_info_mult = rtsimc; } + if (rcu_task_collapse_debug) + pr_info("\tTasks-RCU callback contend/collapse debug enabled.\n"); #endif /* #ifdef CONFIG_TASKS_RCU */ #ifdef CONFIG_TASKS_RCU pr_info("\tTrampoline variant of Tasks RCU enabled.\n"); From 8d7916fdfdf8209e51e35a4a6626f325df8c5a9b Mon Sep 17 00:00:00 2001 From: Ricardo Ribalda Date: Mon, 14 Sep 2026 13:53:31 +0000 Subject: [PATCH 0512/1352] rcu: Drop the address space qualifier from the dereference macros The RCU dereference macros end with a cast that is meant to hand back a plain kernel pointer from a __rcu pointer. For that it uses: ((typeof(*p) __force __kernel *)(local)) The problem is that typeof preserves every qualifier, including the address space qualifiers (__rcu). Recent versions of smatch[1] care about this and throw tens of warnings like this one: ./include/trace/events/vb2.h:46:1: warning: incorrect type in assignment (different address spaces) ./include/trace/events/vb2.h:46:1: expected struct tracepoint_func *it_func_ptr ./include/trace/events/vb2.h:46:1: got struct tracepoint_func __rcu * Use a new macro TYPEOF_NO_ADDRESS_SPACE() for the result type. This new macro strips all the qualifiers when running with sparse (so const and volatile are gone). But keeps all the qualifiers when running with the compiler, caring about const/volatile mismatch. [1] https://github.com/error27/smatch/commit/e53027a4e816a772403baafa83c09e4a94c1cb8f Suggested-by: Dan Carpenter Link: https://lore.kernel.org/r/20260825-unqual-v1-1-7024fb81b4f9@chromium.org Signed-off-by: Ricardo Ribalda Reviewed-by: Joel Fernandes Signed-off-by: Paul E. McKenney --- include/linux/compiler.h | 17 +++++++++++++++++ include/linux/rcupdate.h | 10 +++++----- 2 files changed, 22 insertions(+), 5 deletions(-) diff --git a/include/linux/compiler.h b/include/linux/compiler.h index cb2f6050bdf7dc..70cb31d6053846 100644 --- a/include/linux/compiler.h +++ b/include/linux/compiler.h @@ -239,6 +239,23 @@ void ftrace_likely_update(struct ftrace_likely_data *f, int val, # define TYPEOF_UNQUAL(exp) __typeof__(exp) #endif +/* + * TYPEOF_NO_ADDRESS_SPACE() - typeof() without the address space qualifiers + * + * No operator strips only the address space qualifiers: typeof() keeps every + * qualifier and TYPEOF_UNQUAL() drops every qualifier, const and volatile + * included. + * + * Approximate one by dropping the qualifiers for sparse only, as it is the + * only one that knows about address spaces. The compiler keeps seeing the fully + * qualified type, so a missing const or volatile will still throw a warning. + */ +#ifdef __CHECKER__ +# define TYPEOF_NO_ADDRESS_SPACE(exp) TYPEOF_UNQUAL(exp) +#else +# define TYPEOF_NO_ADDRESS_SPACE(exp) __typeof__(exp) +#endif + #endif /* __KERNEL__ */ #if defined(CONFIG_CFI) && !defined(__DISABLE_EXPORTS) && !defined(BUILD_VDSO) diff --git a/include/linux/rcupdate.h b/include/linux/rcupdate.h index 44c07a66edfff2..3f74ae6d6e1f20 100644 --- a/include/linux/rcupdate.h +++ b/include/linux/rcupdate.h @@ -488,7 +488,7 @@ static __always_inline bool lockdep_assert_rcu_helper(bool c, const struct __ctx context_unsafe( \ typeof(*p) *local = (typeof(*p) *__force)(p); \ rcu_check_sparse(p, __rcu); \ - ((typeof(*p) __force __kernel *)(local)) \ + ((TYPEOF_NO_ADDRESS_SPACE(*p) __force __kernel *)(local)) \ ) /** * unrcu_pointer - mark a pointer as not being RCU protected @@ -503,7 +503,7 @@ context_unsafe( \ ({ \ typeof(*p) *local = (typeof(*p) *__force)READ_ONCE(p); \ rcu_check_sparse(p, space); \ - ((typeof(*p) __force __kernel *)(local)); \ + ((TYPEOF_NO_ADDRESS_SPACE(*p) __force __kernel *)(local)); \ }) ) #define __rcu_dereference_check(p, local, c, space) \ ({ \ @@ -511,19 +511,19 @@ context_unsafe( \ typeof(*p) *local = (typeof(*p) *__force)READ_ONCE(p); \ RCU_LOCKDEP_WARN(!(c), "suspicious rcu_dereference_check() usage"); \ rcu_check_sparse(p, space); \ - ((typeof(*p) __force __kernel *)(local)); \ + ((TYPEOF_NO_ADDRESS_SPACE(*p) __force __kernel *)(local)); \ }) #define __rcu_dereference_protected(p, local, c, space) \ ({ \ RCU_LOCKDEP_WARN(!(c), "suspicious rcu_dereference_protected() usage"); \ rcu_check_sparse(p, space); \ - ((typeof(*p) __force __kernel *)(p)); \ + ((TYPEOF_NO_ADDRESS_SPACE(*p) __force __kernel *)(p)); \ }) #define __rcu_dereference_raw(p, local) \ ({ \ /* Dependency order vs. p above. */ \ typeof(p) local = READ_ONCE(p); \ - ((typeof(*p) __force __kernel *)(local)); \ + ((TYPEOF_NO_ADDRESS_SPACE(*p) __force __kernel *)(local)); \ }) #define rcu_dereference_raw(p) __rcu_dereference_raw(p, __UNIQUE_ID(rcu)) From 96ac38e9c8cb1674783f5eed8adc2b533d80c269 Mon Sep 17 00:00:00 2001 From: "Uladzislau Rezki (Sony)" Date: Sat, 5 Sep 2026 17:27:17 +0200 Subject: [PATCH 0513/1352] mm/vmalloc: use dedicated unbound workqueues for vmap drain drain_vmap_area_work() function can take >10ms to complete when there are many accumulated vmap areas in a system with high CPU count, causing workqueue watchdog warnings when run via schedule_work(): workqueue: drain_vmap_area_work hogged CPU for >10000us Move the top-level drain work to a dedicated WQ_UNBOUND workqueue so the scheduler can run this background work on any available CPU, improving responsiveness. Use the WQ_MEM_RECLAIM to ensure forward progress under memory pressure. Move purge helpers to separate WQ_UNBOUND | WQ_MEM_RECLAIM workqueue. This allows drain_vmap_work to wait for helpers completion without creating dependency on the same rescuer thread and avoid a potential parent/child deadlock. Simplify purge helper scheduling by removing cpumask-based iteration to iterating directly over vmap nodes checking work_queued state. Link: https://lore.kernel.org/20260905152717.11711-1-urezki@gmail.com Fixes: 72210662c5a2 ("mm: vmalloc: offload free_vmap_area_lock lock") Signed-off-by: Uladzislau Rezki (Sony) Signed-off-by: Andrew Morton Reported-by: Li RongQing Closes: https://lore.kernel.org/all/20260319074307.2325-1-lirongqing@baidu.com/ Reviewed-by: Baoquan He Reviewed-by: Ye Liu Cc: Dev Jain Cc: --- mm/vmalloc.c | 79 ++++++++++++++++++++++++++++++++++------------------ 1 file changed, 52 insertions(+), 27 deletions(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index bea9f76ed7e742..89c327a6ce7d9f 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -972,6 +972,7 @@ static struct vmap_node { struct list_head purge_list; struct work_struct purge_work; unsigned long nr_purged; + bool work_queued; } single; /* @@ -1090,6 +1091,8 @@ static void reclaim_and_purge_vmap_areas(void); static BLOCKING_NOTIFIER_HEAD(vmap_notify_list); static void drain_vmap_area_work(struct work_struct *work); static DECLARE_WORK(drain_vmap_work, drain_vmap_area_work); +static struct workqueue_struct *drain_vmap_helpers_wq; +static struct workqueue_struct *drain_vmap_wq; static __cacheline_aligned_in_smp atomic_long_t vmap_lazy_nr; @@ -2351,6 +2354,16 @@ static void purge_vmap_node(struct work_struct *work) reclaim_list_global(&local_list); } +static bool +schedule_drain_vmap_work(struct workqueue_struct *wq, + struct work_struct *work) +{ + if (wq) + return queue_work(wq, work); + + return false; +} + /* * Purges all lazily-freed vmap areas. */ @@ -2358,19 +2371,12 @@ static bool __purge_vmap_area_lazy(unsigned long start, unsigned long end, bool full_pool_decay) { unsigned long nr_purged_areas = 0; + unsigned int nr_purge_nodes = 0; unsigned int nr_purge_helpers; - static cpumask_t purge_nodes; - unsigned int nr_purge_nodes; struct vmap_node *vn; - int i; lockdep_assert_held(&vmap_purge_lock); - /* - * Use cpumask to mark which node has to be processed. - */ - purge_nodes = CPU_MASK_NONE; - for_each_vmap_node(vn) { INIT_LIST_HEAD(&vn->purge_list); vn->skip_populate = full_pool_decay; @@ -2390,10 +2396,9 @@ static bool __purge_vmap_area_lazy(unsigned long start, unsigned long end, end = max(end, list_last_entry(&vn->purge_list, struct vmap_area, list)->va_end); - cpumask_set_cpu(node_to_id(vn), &purge_nodes); + nr_purge_nodes++; } - nr_purge_nodes = cpumask_weight(&purge_nodes); if (nr_purge_nodes > 0) { flush_tlb_kernel_range(start, end); @@ -2401,29 +2406,31 @@ static bool __purge_vmap_area_lazy(unsigned long start, unsigned long end, nr_purge_helpers = atomic_long_read(&vmap_lazy_nr) / lazy_max_pages(); nr_purge_helpers = clamp(nr_purge_helpers, 1U, nr_purge_nodes) - 1; - for_each_cpu(i, &purge_nodes) { - vn = &vmap_nodes[i]; + for_each_vmap_node(vn) { + vn->work_queued = false; + + if (list_empty(&vn->purge_list)) + continue; if (nr_purge_helpers > 0) { INIT_WORK(&vn->purge_work, purge_vmap_node); + vn->work_queued = schedule_drain_vmap_work( + READ_ONCE(drain_vmap_helpers_wq), &vn->purge_work); - if (cpumask_test_cpu(i, cpu_online_mask)) - schedule_work_on(i, &vn->purge_work); - else - schedule_work(&vn->purge_work); - - nr_purge_helpers--; - } else { - vn->purge_work.func = NULL; - purge_vmap_node(&vn->purge_work); - nr_purged_areas += vn->nr_purged; + if (vn->work_queued) { + nr_purge_helpers--; + continue; + } } - } - for_each_cpu(i, &purge_nodes) { - vn = &vmap_nodes[i]; + /* Sync path. Process locally. */ + purge_vmap_node(&vn->purge_work); + nr_purged_areas += vn->nr_purged; + } - if (vn->purge_work.func) { + /* Wait for completion if queued any. */ + for_each_vmap_node(vn) { + if (vn->work_queued) { flush_work(&vn->purge_work); nr_purged_areas += vn->nr_purged; } @@ -2487,7 +2494,8 @@ static void free_vmap_area_noflush(struct vmap_area *va) /* After this point, we may free va at any time */ if (unlikely(nr_lazy > nr_lazy_max)) - schedule_work(&drain_vmap_work); + schedule_drain_vmap_work(READ_ONCE(drain_vmap_wq), + &drain_vmap_work); } /* @@ -5587,3 +5595,20 @@ void __init vmalloc_init(void) vmap_node_shrinker->scan_objects = vmap_node_shrink_scan; shrinker_register(vmap_node_shrinker); } + +static int __init vmalloc_init_workqueue(void) +{ + struct workqueue_struct *drain_wq, *helpers_wq; + unsigned int flags = WQ_UNBOUND | WQ_MEM_RECLAIM; + + drain_wq = alloc_workqueue("vmap_drain", flags, 0); + WARN_ON_ONCE(drain_wq == NULL); + WRITE_ONCE(drain_vmap_wq, drain_wq); + + helpers_wq = alloc_workqueue("vmap_drain_helpers", flags, 0); + WARN_ON_ONCE(helpers_wq == NULL); + WRITE_ONCE(drain_vmap_helpers_wq, helpers_wq); + + return 0; +} +early_initcall(vmalloc_init_workqueue); From 8e1b0e4fc7140073381fb3d140872f1aecbd9e98 Mon Sep 17 00:00:00 2001 From: Krystian Kaniewski Date: Fri, 4 Sep 2026 14:12:59 +0200 Subject: [PATCH 0514/1352] xarray: fix index jumping backwards in xas_find() A bug in the XArray iterator xas_find() causes the iterator's index (xas->xa_index) to jump backwards when iterating over a multi-index entry (like a THP) that resides in a non-leaf node and is concurrently split. When iterating over a multi-index entry in a non-leaf node, xas_load() sets xas->xa_offset to the base offset of the entry, but leaves xas->xa_index at the requested index. When the caller subsequently wants to advance to the next entry, xas_find() is called. xas_find() attempts to synchronize xas->xa_offset with xas->xa_index before advancing. However, the fixup logic was incorrectly restricted to leaf nodes (!xas->xa_node->shift). Because the THP resides in a non-leaf node, the fixup is skipped. As a result, xas_find() simply increments xas->xa_offset and recalculates xas->xa_index based on this new offset. This causes xas->xa_index to jump backwards. If the THP was concurrently split, the entry at the new offset is a node pointer, so xas_find() descends into it and returns the folio at the backwards index. The caller (filemap_map_pages()) then calculates the PTE pointer based on this backwards index, resulting in an invalid memory access such as an out-of-bounds read or use-after-free on a page-table page freed via tlb_remove_table_rcu(). A userspace access that faults in a file-backed mapping can trigger this path. When the index moves backwards, filemap_map_pages() can calculate a PTE outside the page locked for fault-around and dereference a freed page-table page, resulting in a KASAN-detected use-after-free read. To fix this, check if xas->xa_offset matches get_offset(xas->xa_index, xas->xa_node). If it does not and the node is a non-leaf node, set xas->xa_offset to get_offset(xas->xa_index, xas->xa_node) before advancing. Also add test cases in test_xarray to verify xas_find() behavior when iterating over and splitting multi-index entries. Link: https://lore.kernel.org/20260904121301.200049-1-krystianmkaniewski@gmail.com Fixes: b803b42823d0 ("xarray: Add XArray iterators") Reported-by: syzbot+b72767277f29b6407083@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=b72767277f29b6407083 Closes: https://syzkaller.appspot.com/ai_job?id=a01c56bd-74d0-411c-afb4-ee6f0cb6cb61 Signed-off-by: Krystian Kaniewski Signed-off-by: Andrew Morton Assisted-by: Gemini:gemini-3.7-flash syzbot Cc: Matthew Wilcox (Oracle) Cc: Mike Rapoport Cc: --- lib/test_xarray.c | 70 +++++++++++++++++++++++++++++++++++++++++++++++ lib/xarray.c | 8 ++++-- 2 files changed, 75 insertions(+), 3 deletions(-) diff --git a/lib/test_xarray.c b/lib/test_xarray.c index 5ca0aefee9aa5f..615f7730a5bd4a 100644 --- a/lib/test_xarray.c +++ b/lib/test_xarray.c @@ -1247,6 +1247,75 @@ static noinline void check_multi_find_3(struct xarray *xa) } } +static noinline void check_multi_find_4(struct xarray *xa) +{ +#ifdef CONFIG_XARRAY_MULTI + unsigned int order = XA_CHUNK_SHIFT + 1; + unsigned long next = 1UL << order; + unsigned long start = next - 1; + XA_STATE(xas, xa, start); + XA_STATE_ORDER(split, xa, 0, 0); + void *entry; + unsigned long i; + + /* Multi-index entry in two slots of a non-leaf node. */ + xa_store_order(xa, 0, order, xa_mk_index(0), GFP_KERNEL); + XA_BUG_ON(xa, xa_store_index(xa, next, GFP_KERNEL) != NULL); + + rcu_read_lock(); + entry = xas_find(&xas, ULONG_MAX); + XA_BUG_ON(xa, entry != xa_mk_index(0)); + XA_BUG_ON(xa, xas.xa_index != start); + + entry = xas_find(&xas, ULONG_MAX); + XA_BUG_ON(xa, entry != xa_mk_index(next)); + XA_BUG_ON(xa, xas.xa_index != next); + + entry = xas_find(&xas, ULONG_MAX); + XA_BUG_ON(xa, entry != NULL); + rcu_read_unlock(); + + xa_erase_index(xa, next); + xa_erase_index(xa, 0); + XA_BUG_ON(xa, !xa_empty(xa)); + + /* Split the multi-index entry after a lookup begins inside it. */ + xa_store_order(xa, 0, order, xa_mk_index(0), GFP_KERNEL); + XA_BUG_ON(xa, xa_store_index(xa, next, GFP_KERNEL) != NULL); + + xas_set(&xas, start); + rcu_read_lock(); + entry = xas_find(&xas, ULONG_MAX); + XA_BUG_ON(xa, entry != xa_mk_index(0)); + XA_BUG_ON(xa, xas.xa_index != start); + rcu_read_unlock(); + + xas_split_alloc(&split, xa_mk_index(0), order, GFP_KERNEL); + if (xas_error(&split)) { + XA_BUG_ON(xa, true); + goto out; + } + xas_lock(&split); + xas_split(&split, xa_mk_index(0), order); + for (i = 0; i < next; i++) + __xa_store(xa, i, xa_mk_index(i), 0); + xas_unlock(&split); + + rcu_read_lock(); + entry = xas_find(&xas, ULONG_MAX); + XA_BUG_ON(xa, entry != xa_mk_index(next)); + XA_BUG_ON(xa, xas.xa_index != next); + + entry = xas_find(&xas, ULONG_MAX); + XA_BUG_ON(xa, entry != NULL); + rcu_read_unlock(); + +out: + xa_destroy(xa); + XA_BUG_ON(xa, !xa_empty(xa)); +#endif +} + static noinline void check_find_1(struct xarray *xa) { unsigned long i, j, k; @@ -1370,6 +1439,7 @@ static noinline void check_find(struct xarray *xa) check_multi_find_1(xa, i); check_multi_find_2(xa); check_multi_find_3(xa); + check_multi_find_4(xa); } /* See find_swap_entry() in mm/shmem.c */ diff --git a/lib/xarray.c b/lib/xarray.c index bfe7bef80f34eb..3913c8d6486bb1 100644 --- a/lib/xarray.c +++ b/lib/xarray.c @@ -1409,9 +1409,11 @@ void *xas_find(struct xa_state *xas, unsigned long max) entry = xas_load(xas); if (entry || xas_not_node(xas->xa_node)) return entry; - } else if (!xas->xa_node->shift && - xas->xa_offset != (xas->xa_index & XA_CHUNK_MASK)) { - xas->xa_offset = ((xas->xa_index - 1) & XA_CHUNK_MASK) + 1; + } else if (xas->xa_offset != get_offset(xas->xa_index, xas->xa_node)) { + if (!xas->xa_node->shift) + xas->xa_offset = ((xas->xa_index - 1) & XA_CHUNK_MASK) + 1; + else + xas->xa_offset = get_offset(xas->xa_index, xas->xa_node); } xas_next_offset(xas); From 33dbff70402ccb1297e913de642f6436f8b60ead Mon Sep 17 00:00:00 2001 From: Zhao Li Date: Tue, 28 Apr 2026 19:30:38 +0800 Subject: [PATCH 0515/1352] mm/hugetlb: fix max-only subpool accounting on alloc_hugetlb_folio failure Failed hugetlbfs page allocations can permanently consume the mount's size= quota without allocating a huge page. Repeated failures can make the filesystem appear full and cause later huge-page faults or allocations to fail with SIGBUS/allocation failure despite available huge pages and unused real filesystem capacity. alloc_hugetlb_folio() calls hugepage_subpool_get_pages() when map_chg is set. For a subpool with max_hpages != -1, that bumps used_hpages regardless of whether it returns gbl_chg = 0 (rsv slot consumed) or gbl_chg > 0 (used_hpages slot only). If the allocation later fails before a folio is returned, the unwind must undo the used_hpages bump. The old cleanup only ran for !gbl_chg, leaking used_hpages on the gbl_chg > 0 path. For gbl_chg > 0 on max-only subpools (max_hpages != -1, min_hpages == -1), hugepage_subpool_get_pages() took only a speculative used_hpages slot. Drop that slot directly under spool->lock. In that configuration hugepage_subpool_put_pages() cannot restore rsv_hpages, so the direct decrement is the exact inverse and is race-free against concurrent puts. This matches the used_hpages-only part of hugetlb_reserve_pages()'s out_put_pages cleanup, but restricts it to the max-only case where no rsv_hpages restoration is possible. Mounts with min_hpages != -1 are left unchanged for now. v2's approach (hugepage_subpool_put_pages() + h->resv_huge_pages++ to back a restored rsv_hpages slot) double-counts global backing under concurrent free_huge_folio() and creates phantom reservations under concurrent hugetlb_unreserve_pages(). Safe cleanup of that quadrant needs a coordinated fix across multiple call sites. Reproduced on size=20M hugetlbfs with the faulting task in a hugetlb cgroup whose limit is exceeded. Vanilla leaks 6/8 hugepages of subpool quota; this patch leaks 0/8. Verified under QEMU. Link: https://lore.kernel.org/20260428113037.88766-2-enderaoelyther@gmail.com Fixes: a833a693a490 ("mm: hugetlb: fix incorrect fallback for subpool") Signed-off-by: Zhao Li Signed-off-by: Andrew Morton Tested-by: Ackerley Tng Cc: David Hildenbrand Cc: Ma Wupeng Cc: Muchun Song Cc: Oscar Salvador Cc: # v6.15+ --- mm/hugetlb.c | 25 ++++++++++++++++++------- 1 file changed, 18 insertions(+), 7 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index cea25773a6c953..49ffbcb54f8c08 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -3070,13 +3070,24 @@ struct folio *alloc_hugetlb_folio(struct vm_area_struct *vma, return folio; out_subpool_put: - /* - * put page to subpool iff the quota of subpool's rsv_hpages is used - * during hugepage_subpool_get_pages. - */ - if (map_chg && !gbl_chg) { - gbl_reserve = hugepage_subpool_put_pages(spool, 1); - hugetlb_acct_memory(h, -gbl_reserve); + if (map_chg) { + if (!gbl_chg) { + /* Full inverse when subpool_get_pages() consumed rsv_hpages. */ + gbl_reserve = hugepage_subpool_put_pages(spool, 1); + hugetlb_acct_memory(h, -gbl_reserve); + } else if (gbl_chg > 0 && spool && spool->min_hpages == -1 && + spool->max_hpages != -1) { + unsigned long flags; + + /* + * For max-only subpools, subpool_get_pages() took only a + * speculative used_hpages slot. Drop that slot directly. + */ + spin_lock_irqsave(&spool->lock, flags); + if (spool->used_hpages > 0) + spool->used_hpages--; + unlock_or_release_subpool(spool, flags); + } } out_end_reservation: From 43f32cc34b1510149202cb2981d742db1482a421 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Wed, 30 Sep 2026 19:47:09 +0100 Subject: [PATCH 0516/1352] mm/mremap: fix locked_vm leak from MREMAP_DONTUNMAP self-merge Patch series "mm/mremap: fix two issues with MREMAP_DONTUNMAP", v2. The MREMAP_DONTUNMAP feature is highly unusual in that it permits mremap() operations that keep the original VMA in place. Historically this has led to a lot of bugs where non-obvious interactions occur between existing mremap() operations and the original VMA. Commit 397432cab17b ("mm/mremap: account mm->locked_vm correctly for MREMAP_DONTUNMAP") fixed an accidentally introduced bug around mm->locked_vm accounting, but this wasn't the only issue. And thus history repeats itself, as it turns out that mm->locked_vm accounting is broken by MREMAP_DONTUNMAP yet again by two further cases, and has been broken ever since the feature was introduced. Both relate to the fact that VMA_LOCKED_BIT is cleared on the source VMA (it has to be as all page tables are moved): 1. If an unfaulted VMA_LOCKONFAULT_BIT anonymous VMA self-merges it clears the VMA_LOCKED_BIT flag and permanently leaks mm->locked_vm pages. 2. If a partial mremap() is performed on a locked VMA there is a leak equal to the number of pages not copied. (Both for MREMAP_DONTUNMAP operations only) Both issues can be fixed by treating the source range as distinct from the destination range, which is the definition of what MREMAP_DONTUNMAP does so is appropriate. In case 1, simply disallow the self-merge, keeping adjacent source and destination VMAs distinct. In case 2, split the source range ahead of time if the VMA is mlock()'d, so accounting is always correct. Both changes were tested locally and confirmed to fix the issues. For the purposes of a backport, the fixes are kept distinct, a follow-up series can add self-tests. This patch (of 2): The MREMAP_DONTUNMAP feature is highly unusual in that it permits mremap() operations that keep the original VMA in place. Historically this has led to a lot of bugs where non-obvious interactions occur between existing mremap() operations and the original VMA. Fix another of these - self-merge. Self-merge occurs when a VMA is moved in front of or behind itself and the attributes of the VMA permit such a merge. Practically this can only happen for unfaulted anonymous VMAs due to the page offset equality requirement for merge: |------------| | | | v |...........||-----------||...........| | || unfaulted || | |...........||-----------||...........| ^ | | | |------------| This becomes problematic if the VMA is configured by the user to mlock-on-fault, i.e. the VMA_LOCKED_BIT, VMA_LOCKONFAULT_BIT VMA flags are set. MREMAP_DONTUNMAP clears mlock flags for the source VMA and maintains them for the destination VMA. Self-merge makes this impossible (there is only one VMA) and incorrectly clears the destination VMA's mlock flags. This causes a leak in mm->locked_vm as clearing this flag does not decrement the counter and the VMA no longer has VMA_LOCKED_BIT set so it is not decremented on unmap. Resolve this by simply disallowing a self-merge in this case - the source and destination VMAs are kept distinct and then are able to have distinct mlock() flags. Update dontunmap_complete() to make the now-redundant self-merge check a VM_WARN_ON_ONCE() instead to guard against future regressions. Also update the VMA userland tests to reflect the change. Link: https://lore.kernel.org/20260930-fix-dontunmap-partial-self-merge-v2-0-f388985a0f0a@kernel.org Link: https://lore.kernel.org/20260930-fix-dontunmap-partial-self-merge-v2-1-f388985a0f0a@kernel.org Fixes: e346b3813067 ("mm/mremap: add MREMAP_DONTUNMAP to mremap()") Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Pedro Falcato Acked-by: Kiryl Shutsemau (Meta) Reviewed-by: Jose A. Perez de Azpillaga Tested-by: Anirudh Srinivasan Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Jann Horn Cc: Brian Geffon Cc: Minchan Kim Cc: --- mm/mremap.c | 8 ++++++-- mm/vma.c | 17 ++++++++++++++++- mm/vma.h | 2 +- tools/testing/vma/tests/vma.c | 10 +++++----- 4 files changed, 28 insertions(+), 9 deletions(-) diff --git a/mm/mremap.c b/mm/mremap.c index 7c368440fafe24..058dd66fe1d762 100644 --- a/mm/mremap.c +++ b/mm/mremap.c @@ -1275,7 +1275,8 @@ static int copy_vma_and_data(struct vma_remap_struct *vrm, PAGETABLE_MOVE(pmc, NULL, NULL, vrm->addr, vrm->new_addr, vrm->old_len); new_vma = copy_vma(&vma, vrm->new_addr, vrm->new_len, new_pgoff, - new_anon_pgoff, &pmc.need_rmap_locks); + new_anon_pgoff, &pmc.need_rmap_locks, + vrm->flags & MREMAP_DONTUNMAP); if (!new_vma) { vrm_uncharge(vrm); *new_vma_ptr = NULL; @@ -1335,6 +1336,9 @@ static void dontunmap_complete(struct vma_remap_struct *vrm, unsigned long old_start = vma->vm_start; unsigned long old_end = vma->vm_end; + /* Self-merge is disallowed. */ + VM_WARN_ON_ONCE(new_vma == vma); + /* We always clear VMA_LOCKED[ONFAULT]_BIT on the old VMA. */ vma_clear_flags_mask(vma, VMA_LOCKED_MASK); @@ -1342,7 +1346,7 @@ static void dontunmap_complete(struct vma_remap_struct *vrm, * anon_vma links of the old vma is no longer needed after its page * table has been moved. */ - if (new_vma != vma && start == old_start && end == old_end) { + if (start == old_start && end == old_end) { const pgoff_t pgoff_unfaulted = vma->vm_start >> PAGE_SHIFT; unlink_anon_vmas(vma); diff --git a/mm/vma.c b/mm/vma.c index 6cde67883fb04a..d8308c319daa7e 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -1943,7 +1943,7 @@ static int vma_link(struct mm_struct *mm, struct vm_area_struct *vma) */ struct vm_area_struct *copy_vma(struct vm_area_struct **vmap, unsigned long addr, unsigned long len, pgoff_t pgoff, - pgoff_t anon_pgoff, bool *need_rmap_locks) + pgoff_t anon_pgoff, bool *need_rmap_locks, bool keep_source) { struct vm_area_struct *vma = *vmap; unsigned long old_vma_start = vma->vm_start; @@ -1981,6 +1981,21 @@ struct vm_area_struct *copy_vma(struct vm_area_struct **vmap, vmg.pgoff = pgoff; vmg.anon_pgoff = anon_pgoff; vmg.next = vma_iter_next_rewind(&vmi, NULL); + + /* + * If the original VMA is kept (MREMAP_DONTUNMAP), the source and + * destination VMA must be treated distinctly. + * + * A merge violates this, so in this case disallow a self-merge. + */ + if (can_self_merge && keep_source) { + if (vmg.prev == vma) + vmg.prev = NULL; + if (vmg.next == vma) + vmg.next = NULL; + can_self_merge = false; + } + new_vma = vma_merge_copied_range(&vmg); if (new_vma) { diff --git a/mm/vma.h b/mm/vma.h index 024fabe63560a0..39514fac72fc70 100644 --- a/mm/vma.h +++ b/mm/vma.h @@ -535,7 +535,7 @@ void unlink_file_vma_batch_add(struct unlink_vma_file_batch *vb, struct vm_area_struct *copy_vma(struct vm_area_struct **vmap, unsigned long addr, unsigned long len, pgoff_t pgoff, - pgoff_t anon_pgoff, bool *need_rmap_locks); + pgoff_t anon_pgoff, bool *need_rmap_locks, bool keep_source); struct anon_vma *find_mergeable_anon_vma(struct vm_area_struct *vma); diff --git a/tools/testing/vma/tests/vma.c b/tools/testing/vma/tests/vma.c index c8ef7b8cd46be7..e973e0a6d1a829 100644 --- a/tools/testing/vma/tests/vma.c +++ b/tools/testing/vma/tests/vma.c @@ -40,7 +40,7 @@ static bool test_copy_vma(void) vma = alloc_and_link_vma(&mm, 0x1000, 0x2000, 1, vma_flags); vma_set_anonymous(vma); vma_orig = vma; - vma_new = copy_vma(&vma, 0x2000, 0x1000, 1, 1, &need_locks); + vma_new = copy_vma(&vma, 0x2000, 0x1000, 1, 1, &need_locks, false); ASSERT_EQ(vma_new, vma_orig); ASSERT_EQ(vma, vma_orig); ASSERT_EQ(vma_new->vm_start, 0x1000); @@ -53,7 +53,7 @@ static bool test_copy_vma(void) vma = alloc_and_link_vma(&mm, 0x2000, 0x3000, 2, vma_flags); vma_set_anonymous(vma); vma_orig = vma; - vma_new = copy_vma(&vma, 0x1000, 0x1000, 2, 2, &need_locks); + vma_new = copy_vma(&vma, 0x1000, 0x1000, 2, 2, &need_locks, false); ASSERT_EQ(vma_new, vma_orig); ASSERT_EQ(vma, vma_orig); ASSERT_EQ(vma_new->vm_start, 0x1000); @@ -71,7 +71,7 @@ static bool test_copy_vma(void) vma = alloc_and_link_vma(&mm, 0x3000, 0x4000, 3, vma_flags); vma_set_anonymous(vma); vma_orig = vma; - vma_new = copy_vma(&vma, 0x2000, 0x1000, 3, 3, &need_locks); + vma_new = copy_vma(&vma, 0x2000, 0x1000, 3, 3, &need_locks, false); ASSERT_NE(vma_new, vma_orig); ASSERT_EQ(vma_new, vma); ASSERT_EQ(vma_new->vm_start, 0x1000); @@ -82,7 +82,7 @@ static bool test_copy_vma(void) /* Move backwards and do not merge. */ vma = alloc_and_link_vma(&mm, 0x3000, 0x5000, 3, vma_flags); - vma_new = copy_vma(&vma, 0, 0x2000, 0, 3, &need_locks); + vma_new = copy_vma(&vma, 0, 0x2000, 0, 3, &need_locks, false); ASSERT_NE(vma_new, vma); ASSERT_EQ(vma_new->vm_start, 0); ASSERT_EQ(vma_new->vm_end, 0x2000); @@ -95,7 +95,7 @@ static bool test_copy_vma(void) vma = alloc_and_link_vma(&mm, 0, 0x2000, 0, vma_flags); vma_next = alloc_and_link_vma(&mm, 0x6000, 0x8000, 6, vma_flags); - vma_new = copy_vma(&vma, 0x4000, 0x2000, 4, 4, &need_locks); + vma_new = copy_vma(&vma, 0x4000, 0x2000, 4, 4, &need_locks, false); vma_assert_attached(vma_new); ASSERT_EQ(vma_new, vma_next); From 50d2653d2583404b952dafa1b48fdb964e9667a1 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Wed, 30 Sep 2026 19:47:10 +0100 Subject: [PATCH 0517/1352] mm/mremap: fix locked_vm leak by splitting VMA for MREMAP_DONTUNMAP The MREMAP_DONTUNMAP feature is highly unusual in that it permits mremap() operations that keep the original VMA in place. Historically this has led to a lot of bugs where non-obvious interactions occur between existing mremap() operations and the original VMA. Fix another of these - partial copies. The long-standing mremap() partial VMA logic has the baked-in assumption that the originating VMA is unmapped and thus moved. However MREMAP_DONTUNMAP defeats this by performing a partial copy instead since it keeps the source VMA around. An mremap(..., MREMAP_DONTUNMAP) operation disallows resizing of the VMA, but the operation can be performed partially: |-----------------| | | | v <------> <------> .new_sz. new_sz |--.------.--| |------| | .source. | | dest | |--.------.--| |------| <------------> old_sz The page tables in the specified range are moved, but the original VMA is kept intact. This interacts poorly with mlock()'d VMAs, as the VMA_LOCKED_BIT flag is cleared for the entire source VMA and set for the entire destination VMA. This results in an mm->locked_vm leak as the change is therefore not accounted correctly. The clear solution here is to make the portion of the source VMA which is mremap()'d distinct from the rest of it, a.k.a. split it. Therefore resolve this issue by splitting it ahead of the rest of the mremap() operation if the VMA is mlock()'d. This is valid, as the source VMA will lose its VMA_LOCKED_BIT flag, so if a partial remap it will become distinct from the rest of the VMA. In order to make this change re-expose split_vma() in vma.h for CONFIG_MMU (nommu doesn't compile mremap.c and uses a static helper instead). Finally, update the sys_map_count check to account for this case. Note that the early check does not use needs_pre_split() - this is because the VMA has not been looked up by this point, so be conservative and assume that the VMA is mlock()'d in this case for the purposes of the sys_map_count check. Link: https://lore.kernel.org/20260930-fix-dontunmap-partial-self-merge-v2-2-f388985a0f0a@kernel.org Fixes: e346b3813067 ("mm/mremap: add MREMAP_DONTUNMAP to mremap()") Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Pedro Falcato Acked-by: Kiryl Shutsemau (Meta) Reviewed-by: Jose A. Perez de Azpillaga Tested-by: Anirudh Srinivasan Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Jann Horn Cc: Brian Geffon Cc: Minchan Kim Cc: --- mm/mremap.c | 62 +++++++++++++++++++++++++++++++++++++++-------------- mm/vma.c | 4 ++-- mm/vma.h | 5 +++++ 3 files changed, 53 insertions(+), 18 deletions(-) diff --git a/mm/mremap.c b/mm/mremap.c index 058dd66fe1d762..444f37d7cd28ec 100644 --- a/mm/mremap.c +++ b/mm/mremap.c @@ -1037,6 +1037,7 @@ static void vrm_stat_account(struct vma_remap_struct *vrm, } static bool __check_map_count_against_split(struct mm_struct *mm, + bool pre_split, bool before_unmaps) { const int sys_map_count = get_sysctl_max_map_count(); @@ -1088,26 +1089,42 @@ static bool __check_map_count_against_split(struct mm_struct *mm, * Therefore we must check to ensure we have headroom of 2 additional * VMAs. */ - return map_count + 2 <= sys_map_count; + map_count += 2; + + /* If pre-split, the -1 observed above doesn't apply. */ + if (pre_split) + map_count++; + + return map_count <= sys_map_count; +} + +static bool needs_pre_split(struct vma_remap_struct *vrm) +{ + /* + * An MREMAP_DONTUNMAP of a mlock()'d VMA needs to unlock the + * source VMA, so split in this case. + */ + return (vrm->flags & MREMAP_DONTUNMAP) && + vma_test(vrm->vma, VMA_LOCKED_BIT); } /* Do we violate the map count limit if we split VMAs when moving the VMA? */ -static bool check_map_count_against_split(void) +static bool check_map_count_against_split(struct vma_remap_struct *vrm) { return __check_map_count_against_split(current->mm, - /*before_unmaps=*/false); + needs_pre_split(vrm), /*before_unmaps=*/false); } /* Do we violate the map count limit if we split VMAs prior to early unmaps? */ -static bool check_map_count_against_split_early(void) +static bool check_map_count_against_split_early(struct vma_remap_struct *vrm) { return __check_map_count_against_split(current->mm, - /*before_unmaps=*/true); + vrm->flags & MREMAP_DONTUNMAP, /*before_unmaps=*/true); } /* - * Perform checks before attempting to write a VMA prior to it being - * moved. + * Perform checks and preparation before attempting to write a VMA prior to it + * being moved. */ static unsigned long prep_move_vma(struct vma_remap_struct *vrm) { @@ -1116,19 +1133,17 @@ static unsigned long prep_move_vma(struct vma_remap_struct *vrm) unsigned long old_addr = vrm->addr; unsigned long old_len = vrm->old_len; vm_flags_t dummy = vma->vm_flags; + const bool split_before = vma->vm_start != old_addr; + const bool split_after = vma->vm_end != old_addr + old_len; - /* - * We'd prefer to avoid failure later on in do_munmap: we copy a VMA, - * which may not merge, then (if MREMAP_DONTUNMAP is not set) unmap the - * source, which may split, causing a net increase of 2 mappings. - */ - if (!check_map_count_against_split()) + /* Avoid failure later on. */ + if (!check_map_count_against_split(vrm)) return -ENOMEM; if (vma->vm_ops && vma->vm_ops->may_split) { - if (vma->vm_start != old_addr) + if (split_before) err = vma->vm_ops->may_split(vma, old_addr); - if (!err && vma->vm_end != old_addr + old_len) + if (!err && split_after) err = vma->vm_ops->may_split(vma, old_addr + old_len); if (err) return err; @@ -1146,6 +1161,21 @@ static unsigned long prep_move_vma(struct vma_remap_struct *vrm) if (err) return err; + /* + * To account mlock()'d pages correctly in the MREMAP_DONTUNMAP + * case perform any split ahead of time for an mlock()'d VMA. + */ + if (needs_pre_split(vrm)) { + VMA_ITERATOR(vmi, vma->vm_mm, old_addr); + + if (split_before) + err = split_vma(&vmi, vma, old_addr, 1); + if (!err && split_after) + err = split_vma(&vmi, vma, old_addr + old_len, 0); + vrm->vmi_needs_invalidate = true; + return err; + } + return 0; } @@ -2006,7 +2036,7 @@ static unsigned long do_mremap(struct vma_remap_struct *vrm) return -EINTR; vrm->mmap_locked = true; - if (!check_map_count_against_split_early()) { + if (!check_map_count_against_split_early(vrm)) { mmap_write_unlock(mm); return -ENOMEM; } diff --git a/mm/vma.c b/mm/vma.c index d8308c319daa7e..6c35f0d775ab6e 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -634,8 +634,8 @@ __split_vma(struct vma_iterator *vmi, struct vm_area_struct *vma, * Split a vma into two pieces at address 'addr', a new vma is allocated * either for the first part or the tail. */ -static int split_vma(struct vma_iterator *vmi, struct vm_area_struct *vma, - unsigned long addr, int new_below) +int split_vma(struct vma_iterator *vmi, struct vm_area_struct *vma, + unsigned long addr, int new_below) { if (vma->vm_mm->map_count >= get_sysctl_max_map_count()) return -ENOMEM; diff --git a/mm/vma.h b/mm/vma.h index 39514fac72fc70..36973abaa015a8 100644 --- a/mm/vma.h +++ b/mm/vma.h @@ -556,6 +556,11 @@ int do_brk_flags(struct vma_iterator *vmi, struct vm_area_struct *brkvma, unsigned long unmapped_area(struct vm_unmapped_area_info *info); unsigned long unmapped_area_topdown(struct vm_unmapped_area_info *info); +#ifdef CONFIG_MMU +int split_vma(struct vma_iterator *vmi, struct vm_area_struct *vma, + unsigned long addr, int new_below); +#endif + static inline bool vma_wants_manual_pte_write_upgrade(struct vm_area_struct *vma) { /* From 221d5bfc957c85de55de6e869553b9ea490ca60d Mon Sep 17 00:00:00 2001 From: John Garry Date: Wed, 23 Sep 2026 08:53:24 +0100 Subject: [PATCH 0518/1352] mailmap: update addresses for John Garry Point any employment addresses at my personal dev address. Link: https://lore.kernel.org/20260923075324.1382927-1-john.garry@linux.dev Signed-off-by: John Garry Signed-off-by: Andrew Morton --- .mailmap | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/.mailmap b/.mailmap index 389b94a0124e31..38b72611b3e995 100644 --- a/.mailmap +++ b/.mailmap @@ -459,7 +459,8 @@ Johan Hovold Johan Hovold John Crispin John Fastabend -John Garry +John Garry +John Garry John Keeping John Moon John Paul Adrian Glaubitz From 15629d8d32f790e0b6e0bf4619f9ab860c33f2a9 Mon Sep 17 00:00:00 2001 From: Mikhail Gavrilov Date: Fri, 25 Sep 2026 10:06:47 +0500 Subject: [PATCH 0519/1352] mm: don't schedule deferred kernel page table freeing while booting Booting with a boot-time function tracer and a filter, for example ftrace=function ftrace_filter=pud_free_pmd_page panics on 7.3-rc4 as soon as the tracer starts: [ 23.531178] Starting tracer 'function' [ 23.675800] Oops: general protection fault, probably for non-canonical address 0xdffffc0000000038: 0000 [#1] SMP KASAN NOPTI [ 23.819917] KASAN: null-ptr-deref in range [0x00000000000001c0-0x00000000000001c7] [ 23.964025] CPU: 0 UID: 0 PID: 0 Comm: swapper Not tainted 7.3.0-rc4-fe2ec83746e5-with-fixes-v2+ #195 PREEMPT(undef) [ 24.252248] RIP: 0010:__queue_work+0xab/0xf00 [ 25.981629] Call Trace: [ 26.125727] [ 26.413912] ? pagetable_free_kernel+0x20/0x120 [ 26.990283] queue_work_on+0x97/0xf0 [ 27.134382] __cpa_collapse_large_pages+0x501/0x6f0 [ 27.566662] cpa_flush+0x394/0x620 [ 27.998953] change_page_attr_set_clr+0x321/0x4a0 [ 29.151729] set_memory_rox+0xa2/0xf0 [ 29.584018] create_trampoline+0x431/0x6f0 ... [ 44.343347] Kernel panic - not syncing: Attempted to kill the idle task! The boot-time tracer is started from early_trace_init(), which runs before workqueue_init_early(). Making its trampoline read-only splits a large page, and CPA collapses it again right away. The split table has been a kernel page table since commit 9e4a3ec3411b ("x86/mm/pat: Allocate split page tables as kernel page tables"), so the collapse frees it through pagetable_free_kernel(), which queues work on system_percpu_wq - still NULL at that point. That commit is correct in itself; it only lets CPA reach pagetable_free_kernel() before the workqueue that function relies on exists. Keep putting the table on the list, but don't schedule the work while the system is still booting. A core_initcall schedules it once to free whatever was queued by then. Link: https://lore.kernel.org/20260925050647.86913-1-mikhail.v.gavrilov@gmail.com Link: https://lore.kernel.org/20260924064321.23787-1-mikhail.v.gavrilov@gmail.com Fixes: 9e4a3ec3411b ("x86/mm/pat: Allocate split page tables as kernel page tables") Signed-off-by: Mikhail Gavrilov Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand (Arm) Suggested-by: Lorenzo Stoakes (ARM) Reviewed-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Jose A. Perez de Azpillaga Tested-by: Jose A. Perez de Azpillaga Cc: Dave Hansen Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Vishal Moola (Oracle) Cc: Ingo Molnar Cc: Lu Baolu Cc: Jason Gunthorpe Cc: Steven Rostedt Cc: --- mm/pgtable-generic.c | 20 +++++++++++++++++++- 1 file changed, 19 insertions(+), 1 deletion(-) diff --git a/mm/pgtable-generic.c b/mm/pgtable-generic.c index b91b1a98029c7f..cd227fc05d2d8f 100644 --- a/mm/pgtable-generic.c +++ b/mm/pgtable-generic.c @@ -438,12 +438,30 @@ static void kernel_pgtable_work_func(struct work_struct *work) __pagetable_free(pt); } +static void schedule_kernel_pgtable_free(void) +{ + schedule_work(&kernel_pgtable_work.work); +} + void pagetable_free_kernel(struct ptdesc *pt) { spin_lock(&kernel_pgtable_work.lock); list_add(&pt->pt_list, &kernel_pgtable_work.list); spin_unlock(&kernel_pgtable_work.lock); - schedule_work(&kernel_pgtable_work.work); + /* + * The workqueue may not exist yet while the system is booting. + * kernel_pgtable_drain_early() schedules the work once it does. + */ + if (system_state != SYSTEM_BOOTING) + schedule_kernel_pgtable_free(); +} + +static int __init kernel_pgtable_drain_early(void) +{ + /* Free the kernel page tables queued while booting. */ + schedule_kernel_pgtable_free(); + return 0; } +core_initcall(kernel_pgtable_drain_early); #endif From fe13d58cba8f49dd2206a8cfd47ccbe1918d9bdb Mon Sep 17 00:00:00 2001 From: Carlos Llamas Date: Sun, 27 Sep 2026 16:24:18 +0000 Subject: [PATCH 0520/1352] selftests/mm: cleanup -Wformat issues in hugetlb-mmap Commit ae571cd6015c ("selftests/mm: hugetlb-mmap: add setup of HugeTLB pages") and commit 9c5a65f374f8 ("selftests/mm: merge map_hugetlb into hugepage-mmap") added logs of 'hugepage_size' which has a size_t type. However, the incorrect format specifier '%lu' was used which triggers -Wformat warnings when building for 32-bit: hugetlb-mmap.c:125:55: warning: format specifies type 'unsigned long' but the argument has type 'size_t' (aka 'unsigned int') [-Wformat] 125 | ksft_print_msg("Default size hugepages (%lu kB)\n", hugepage_size >> 10); | ~~~ ^~~~~~~~~~~~~~~~~~~ | %zu hugetlb-mmap.c:134:47: warning: format specifies type 'unsigned long' but the argument has type 'size_t' (aka 'unsigned int') [-Wformat] 134 | ksft_exit_skip("Not enough %lu Kb pages\n", hugepage_size >> 10); | ~~~ ^~~~~~~~~~~~~~~~~~~ | %zu Fix this by switching to the expected '%zu' format specifier. Link: https://lore.kernel.org/20260927162419.820609-1-cmllamas@google.com Fixes: ae571cd6015c ("selftests/mm: hugetlb-mmap: add setup of HugeTLB pages") Fixes: 9c5a65f374f8 ("selftests/mm: merge map_hugetlb into hugepage-mmap") Signed-off-by: Carlos Llamas Signed-off-by: Andrew Morton Reviewed-by: Sarthak Sharma Reviewed-by: SJ Park Acked-by: Lorenzo Stoakes (ARM) Cc: Mike Rapoport Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Shuah Khan Cc: --- tools/testing/selftests/mm/hugetlb-mmap.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tools/testing/selftests/mm/hugetlb-mmap.c b/tools/testing/selftests/mm/hugetlb-mmap.c index 0f2aad1b7dbd6f..2edbc992e6bcf0 100644 --- a/tools/testing/selftests/mm/hugetlb-mmap.c +++ b/tools/testing/selftests/mm/hugetlb-mmap.c @@ -122,7 +122,7 @@ int main(int argc, char **argv) hugepage_size = default_huge_page_size(); if (!hugepage_size) ksft_exit_skip("Could not detect default hugetlb page size."); - ksft_print_msg("Default size hugepages (%lu kB)\n", hugepage_size >> 10); + ksft_print_msg("Default size hugepages (%zu kB)\n", hugepage_size >> 10); } /* munmap will fail if the length is not page aligned */ @@ -131,7 +131,7 @@ int main(int argc, char **argv) hugetlb_set_nr_pages(hugepage_size, nr); if (hugetlb_free_pages(hugepage_size) < nr) - ksft_exit_skip("Not enough %lu Kb pages\n", hugepage_size >> 10); + ksft_exit_skip("Not enough %zu Kb pages\n", hugepage_size >> 10); ksft_set_plan(2); ksft_print_msg("Mapping %lu Mbytes\n", (unsigned long)length >> 20); From 5533ed6bff81db05315893984c94a9e4ee1e9ebc Mon Sep 17 00:00:00 2001 From: Donggeun Yoo Date: Sat, 26 Sep 2026 21:41:44 +0900 Subject: [PATCH 0521/1352] userfaultfd: clear the inherited uffd bit in move_swap_pte() Patch series "userfaultfd: clear the inherited uffd bit in move_swap_pte()", v3. UFFDIO_MOVE on a swapped-out page installs the source PTE at the destination unchanged, so a uffd bit set on a write-protected or RWP-protected source lands in a destination VMA that was never registered for either, and nothing clears it afterwards. Patch 1 clears the bit, then re-arms it if the destination is RWP-registered, which is what the present-page and zeropage move paths already do. Patch 2 adds the tests that catch it. This patch (of 2): UFFDIO_MOVE on a swapped-out page installs the source PTE at the destination unchanged, so a uffd bit set on a write-protected or RWP-protected source lands in a destination VMA that was never registered for either. Nothing clears it there, and userspace sees: - /proc//pagemap reports the page as uffd-tracked (bit 57), both while it is swapped out and after it is faulted back in. - MADV_COLLAPSE fails with EINVAL on a range containing it, because the collapse scan bails on a swap entry with the uffd bit set. - With CONFIG_PAGE_TABLE_CHECK, faulting the page in warns. do_swap_page() carries the bit into the present PTE and, since the destination isn't WP-registered, also makes it writable: WARNING: mm/page_table_check.c:202 at __page_table_check_ptes_set+0x185/0x1e0 Call Trace: set_ptes+0x67/0xc0 do_swap_page+0x990/0xfe0 __handle_mm_fault+0x7d0/0xeb0 handle_mm_fault+0x9c/0x250 do_user_addr_fault+0x207/0x650 exc_page_fault+0x65/0x150 asm_exc_page_fault+0x26/0x30 Reaching it takes UFFDIO_MOVE out of a WP- or RWP-protected area into one that isn't, on a page that is swapped out at the time. It hasn't been seen in practice. A resident page doesn't carry the bit, because move_present_ptes() builds the destination PTE from dst_vma->vm_page_prot. Clear it on the moved swap entry as well, then re-arm it if the destination is RWP-registered. The WP case has been there since v6.8, where UFFDIO_MOVE was added. Link: https://lore.kernel.org/20260926124145.2878520-1-donggeunyoo.kernel@gmail.com Link: https://lore.kernel.org/20260926124145.2878520-2-donggeunyoo.kernel@gmail.com Fixes: adef440691ba ("userfaultfd: UFFDIO_MOVE uABI") Signed-off-by: Donggeun Yoo Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Mike Rapoport Cc: Peter Xu Cc: Suren Baghdasaryan Cc: Andrea Arcangeli Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Michal Hocko Cc: Shuah Khan Cc: Kiryl Shutsemau Cc: --- mm/userfaultfd.c | 1 + 1 file changed, 1 insertion(+) diff --git a/mm/userfaultfd.c b/mm/userfaultfd.c index 74f04c323c50fb..f39f109f17989e 100644 --- a/mm/userfaultfd.c +++ b/mm/userfaultfd.c @@ -1449,6 +1449,7 @@ static int move_swap_pte(struct mm_struct *mm, struct vm_area_struct *dst_vma, orig_src_pte = ptep_get_and_clear(mm, src_addr, src_pte); if (pgtable_supports_soft_dirty()) orig_src_pte = pte_swp_mksoft_dirty(orig_src_pte); + orig_src_pte = pte_swp_clear_uffd(orig_src_pte); /* Re-arm RWP on the moved swap entry if dst_vma is RWP-registered. */ if (userfaultfd_rwp(dst_vma)) orig_src_pte = pte_swp_mkuffd(orig_src_pte); From 22bee0ec35d1fe264a819a9b2aaec442b7b37609 Mon Sep 17 00:00:00 2001 From: Donggeun Yoo Date: Sat, 26 Sep 2026 21:41:45 +0900 Subject: [PATCH 0522/1352] selftests/mm: add tests for UFFDIO_MOVE of a uffd-protected swap entry Move a swapped-out page out of a write-protected or RWP-protected area into a destination registered for missing faults only, and read pagemap bit 57 at the destination. The destination was never protected, so the bit must be clear. Link: https://lore.kernel.org/20260926124145.2878520-3-donggeunyoo.kernel@gmail.com Signed-off-by: Donggeun Yoo Signed-off-by: Andrew Morton Assisted-by: LLM Cc: Mike Rapoport Cc: Peter Xu Cc: Suren Baghdasaryan Cc: Andrea Arcangeli Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Michal Hocko Cc: Shuah Khan Cc: Kiryl Shutsemau --- tools/testing/selftests/mm/uffd-unit-tests.c | 84 ++++++++++++++++++++ 1 file changed, 84 insertions(+) diff --git a/tools/testing/selftests/mm/uffd-unit-tests.c b/tools/testing/selftests/mm/uffd-unit-tests.c index ef9b3956bdcfd9..580178630eded8 100644 --- a/tools/testing/selftests/mm/uffd-unit-tests.c +++ b/tools/testing/selftests/mm/uffd-unit-tests.c @@ -2037,6 +2037,75 @@ static void uffd_move_pmd_split_test(uffd_global_test_opts_t *gopts, uffd_test_a uffd_move_pmd_handle_fault); } +/* + * Moving a swapped-out page out of a write-protected or RWP-protected area + * must not carry the uffd bit into a destination registered for missing + * faults only: such a bit is never cleared afterwards, so pagemap keeps + * reporting the page as uffd-tracked. + * + * Needs a swap device; skipped if MADV_PAGEOUT cannot evict the page. + */ +static void uffd_move_swap_test_common(uffd_global_test_opts_t *gopts, + bool rwp) +{ + unsigned long page_size = gopts->page_size; + struct uffdio_move move = { }; + int pagemap_fd; + + if (rwp) { + if (uffd_register_rwp(gopts->uffd, gopts->area_src, page_size)) + err("register src failure"); + } else if (uffd_register(gopts->uffd, gopts->area_src, page_size, + false, true, false)) { + err("register src failure"); + } + if (uffd_register(gopts->uffd, gopts->area_dst, page_size, + true, false, false)) + err("register dst failure"); + + if (rwp) + rwprotect_range(gopts->uffd, (unsigned long)gopts->area_src, + page_size, true); + else + wp_range(gopts->uffd, (unsigned long)gopts->area_src, + page_size, true); + + pagemap_fd = pagemap_open(); + if (madvise(gopts->area_src, page_size, MADV_PAGEOUT)) + err("MADV_PAGEOUT"); + if (!pagemap_is_swapped(pagemap_fd, gopts->area_src)) { + uffd_test_skip("MADV_PAGEOUT did not swap the page; is swap enabled?"); + goto out; + } + + move.dst = (unsigned long)gopts->area_dst; + move.src = (unsigned long)gopts->area_src; + move.len = page_size; + if (ioctl(gopts->uffd, UFFDIO_MOVE, &move)) + err("UFFDIO_MOVE"); + + if (pagemap_get_entry(pagemap_fd, gopts->area_dst) & PM_UFFD_WP) + uffd_test_fail("uffd bit moved into an area registered for missing faults only"); + else + uffd_test_pass(); +out: + close(pagemap_fd); + uffd_unregister(gopts->uffd, gopts->area_src, page_size); + uffd_unregister(gopts->uffd, gopts->area_dst, page_size); +} + +static void uffd_move_swap_wp_test(uffd_global_test_opts_t *gopts, + uffd_test_args_t *targs) +{ + uffd_move_swap_test_common(gopts, false); +} + +static void uffd_move_swap_rwp_test(uffd_global_test_opts_t *gopts, + uffd_test_args_t *targs) +{ + uffd_move_swap_test_common(gopts, true); +} + static bool uffdio_verify_results(const char *name, int ret, int error, long result) { @@ -2359,6 +2428,21 @@ uffd_test_case_t uffd_tests[] = { .uffd_feature_required = UFFD_FEATURE_MOVE, .test_case_ops = &uffd_move_test_pmd_case_ops, }, + { + .name = "move-swap-wp", + .uffd_fn = uffd_move_swap_wp_test, + .mem_targets = MEM_ANON, + .uffd_feature_required = UFFD_FEATURE_MOVE | + UFFD_FEATURE_PAGEFAULT_FLAG_WP, + .test_case_ops = &uffd_move_test_case_ops, + }, + { + .name = "move-swap-rwp", + .uffd_fn = uffd_move_swap_rwp_test, + .mem_targets = MEM_ANON, + .uffd_feature_required = UFFD_FEATURE_MOVE | UFFD_FEATURE_RWP, + .test_case_ops = &uffd_move_test_case_ops, + }, { .name = "wp-fork", .uffd_fn = uffd_wp_fork_test, From cfc0ca8317a6367669ae76e9c38fed356bfb67e6 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 24 Sep 2026 15:48:24 +0100 Subject: [PATCH 0523/1352] drivers/char/mem: mmap readonly MAP_SHARED-/dev/zero correctly Rather surprisingly, opening /dev/zero read-only then mmap()'ing it MAP_SHARED gets you true anonymous memory (albeit in a VMA with non-NULL vma->vm_file). This is a by-product of MAP_PRIVATE-/dev/zero being how anonymous memory was mapped in Linux's distant past. It happens because mmap_zero_prepare() gates on VMA_SHARED_BIT and when mapping a read-only file MAP_SHARED, do_mmap() clears VMA_SHARED_BIT and VMA_MAYWRITE_BIT. The gating is incorrect - the (poorly named) VMA_MAYSHARE_BIT flag exists explicitly to tell you if something was originally mapped MAP_SHARED. So the fix is simple - gate on this instead. This isn't exactly a common use case, but it's unexpected behaviour which now causes an assert if CONFIG_DEBUG_VM is set. While this bug has existed since the dawn of time for linux (or at least since 2.6.12), it hasn't caused issues in the past, so while it's incorrect behaviour, it doesn't seem necessary to backport that far. The mapping is now accounted at mmap time and can fail with -ENOMEM under strict overcommit, and read faults allocate folios. However this is normal behaviour for a read-only shmem mapping. Commit 93c0c8dc87f6 ("mm/rmap: use anon pgoff to track MAP_PRIVATE file-backed anon folios") is the first patch at which the debug assert fires, so target that instead. Link: https://lore.kernel.org/20260924-fix-dev-zero-readonly-shared-v1-1-153c2111e323@kernel.org Fixes: 93c0c8dc87f6 ("mm/rmap: use anon pgoff to track MAP_PRIVATE file-backed anon folios") Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reported-by: Closes: https://lore.kernel.org/linux-mm/6ab4ae75.80e1c6cc.1e8e5f.000d.GAE@google.com/ Acked-by: David Hildenbrand (Arm) Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Jann Horn Cc: Pedro Falcato Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Arnd Bergmann Cc: Greg Kroah-Hartman Cc: Lance Yang --- drivers/char/mem.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/char/mem.c b/drivers/char/mem.c index 63253d1de5d70b..5b93c92c2cf194 100644 --- a/drivers/char/mem.c +++ b/drivers/char/mem.c @@ -503,7 +503,7 @@ static int mmap_zero_prepare(struct vm_area_desc *desc) #ifndef CONFIG_MMU return -ENOSYS; #endif - if (vma_desc_test(desc, VMA_SHARED_BIT)) + if (vma_desc_test(desc, VMA_MAYSHARE_BIT)) return shmem_zero_setup_desc(desc); /* From f7bd18b173f11d7bfee502defc487f349fc024bd Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Tue, 29 Sep 2026 18:45:51 +0100 Subject: [PATCH 0524/1352] mm: page_alloc: make defrag_mode retries follow the promoted order Since commit 7e8756d7ad22 ("mm: page_alloc: fix non-movable reclaim storm in defrag_mode"), direct reclaim and compaction for non-movable requests under defrag_mode run at pageblock_order, to produce the whole blocks that ALLOC_NOFRAGMENT needs. The retry decisions that follow still use the request order. An order-0 request can therefore retry indefinitely without ever reaching the ALLOC_NOFRAGMENT fallback: - Reclaim at pageblock_order gives up after one pass as soon as a zone looks compaction_ready(), and do_try_to_free_pages() then returns 1 even though nothing was reclaimed. It returns before the retry that would reclaim memory.low-protected cgroups, so when most memory is protected, the pass that did run finds next to nothing. - Compaction at pageblock_order fails or is deferred. - should_reclaim_retry() takes the reported progress as progress for the order-0 request and resets no_progress_loops. The request retries. Order 1-3 requests loop the same way, and should_compact_retry() also checks their pageblock_order compaction result against the request order. On a production host (64G, defrag_mode, memory.low covering most of the workload), 95% of direct reclaim runs were order-9 runs that returned 1 with nothing reclaimed, at up to 60k runs per second. Across ~200M should_reclaim_retry() calls in a day, no_progress_loops never left 0. The spinning allocations were SLUB slab refills for inode and dentry caches. The time spent registers as memory pressure, and pressure-based OOM killing takes down both workloads and system services. Treat promoted requests like costly orders: - Reclaim progress does not reset no_progress_loops for them. - should_compact_retry() checks the compaction result at the promoted order. It does not retry COMPACT_SKIPPED, since the request can fall back, and it does not escalate compaction to COMPACT_PRIO_SYNC_FULL. When the fallback is taken, reset the retry counters, so that the fallback attempt gets a full retry budget before the OOM killer is considered. In a VM reproducer (32G, defrag_mode, inode churn under memory.low): before after should_reclaim_retry() calls 63M 293k peak memory pressure (PSI some avg10) 99% 12% File creation runs 5.7x faster. Link: https://lore.kernel.org/20260929174553.175333-1-kirill@shutemov.name Fixes: 7e8756d7ad22 ("mm: page_alloc: fix non-movable reclaim storm in defrag_mode") Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Assisted-by: LLM Cc: Vlastimil Babka Cc: Johannes Weiner Cc: David Hildenbrand Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Brendan Jackman Cc: Zi Yan Cc: Shakeel Butt Cc: Usama Arif Cc: Harry Yoo Cc: --- mm/page_alloc.c | 85 ++++++++++++++++++++++++++++++++----------------- 1 file changed, 56 insertions(+), 29 deletions(-) diff --git a/mm/page_alloc.c b/mm/page_alloc.c index 12fac9084c483d..608487672d933e 100644 --- a/mm/page_alloc.c +++ b/mm/page_alloc.c @@ -4127,6 +4127,31 @@ __alloc_pages_may_oom(gfp_t gfp_mask, unsigned int order, return page; } +/* + * If fallbacks are not permitted (defrag_mode), we either need to + * reclaim space in a block of matching type, or clear out an entire + * block to allow __rmqueue_claim() to convert. + * + * Reclaim by itself is primarily freeing space in movable blocks, + * since that's where the LRU pages live. So this works for movable + * requests, but not for others. + * + * For those, promote the order of reclaim and compaction to help make + * blocks, instead of spinning in reclaim alone unproductively. Retry + * decisions based on the outcome of that work - reclaim progress and + * compaction results - must account for the promotion as well, see + * should_reclaim_retry() and should_compact_retry(). + */ +static inline unsigned int nofrag_promote_order(unsigned int order, + unsigned int alloc_flags, + const struct alloc_context *ac) +{ + if ((alloc_flags & ALLOC_NOFRAGMENT) && ac->migratetype != MIGRATE_MOVABLE) + return max(order, pageblock_order); + + return order; +} + /* * Maximum number of compaction retries with a progress before OOM * killer is consider as the only way to move forward. @@ -4149,22 +4174,7 @@ __alloc_pages_direct_compact(gfp_t gfp_mask, unsigned int order, .order = order, .page = NULL, }; - int compact_order = order; - - /* - * If fallbacks are not permitted (defrag_mode), we either - * need to reclaim space in a block of matching type, or clear - * out an entire block to allow __rmqueue_claim() to convert. - * - * Reclaim by itself is primarily freeing space in movable - * blocks, since that's where the LRU pages live. So this - * works for movable requests, but not for others. - * - * For those, promote the order to help make blocks, instead - * of spinning in reclaim alone unproductively. - */ - if ((alloc_flags & ALLOC_NOFRAGMENT) && ac->migratetype != MIGRATE_MOVABLE) - compact_order = max(order, pageblock_order); + unsigned int compact_order = nofrag_promote_order(order, alloc_flags, ac); if (!compact_order) return NULL; @@ -4256,8 +4266,11 @@ should_compact_retry(gfp_t gfp_mask, struct alloc_context *ac, int order, bool ret = false; int retries = *compaction_retries; enum compact_priority priority = *compact_priority; + unsigned int compact_order; - if (!order) + /* Check the compaction result at the order compaction ran at */ + compact_order = nofrag_promote_order(order, alloc_flags, ac); + if (!compact_order) return false; if (fatal_signal_pending(current)) @@ -4266,10 +4279,14 @@ should_compact_retry(gfp_t gfp_mask, struct alloc_context *ac, int order, /* * Compaction was skipped due to a lack of free order-0 * migration targets. Continue if reclaim can help. + * + * Promoted requests have exhausted their reclaim retries at + * this point, and they can fall back instead. */ if (compact_result == COMPACT_SKIPPED) { - ret = compaction_zonelist_suitable(ac, order, alloc_flags, - gfp_mask); + if (compact_order == order) + ret = compaction_zonelist_suitable(ac, order, alloc_flags, + gfp_mask); goto out; } @@ -4287,7 +4304,7 @@ should_compact_retry(gfp_t gfp_mask, struct alloc_context *ac, int order, * need much more detailed feedback from compaction to * make a better decision. */ - if (order > PAGE_ALLOC_COSTLY_ORDER) + if (compact_order > PAGE_ALLOC_COSTLY_ORDER) max_retries /= 4; if (++(*compaction_retries) <= max_retries) { @@ -4299,7 +4316,7 @@ should_compact_retry(gfp_t gfp_mask, struct alloc_context *ac, int order, /* * Compaction failed. Retry with increasing priority. */ - min_priority = (order > PAGE_ALLOC_COSTLY_ORDER) ? + min_priority = (compact_order > PAGE_ALLOC_COSTLY_ORDER) ? MIN_COMPACT_COSTLY_PRIORITY : MIN_COMPACT_PRIORITY; if (*compact_priority > min_priority) { @@ -4468,11 +4485,7 @@ __alloc_pages_direct_reclaim(gfp_t gfp_mask, unsigned int order, struct page *page = NULL; unsigned long pflags; bool drained = false; - int reclaim_order = order; - - /* Match the slowpath compaction promotion in __alloc_pages_direct_compact */ - if ((alloc_flags & ALLOC_NOFRAGMENT) && ac->migratetype != MIGRATE_MOVABLE) - reclaim_order = max(order, pageblock_order); + unsigned int reclaim_order = nofrag_promote_order(order, alloc_flags, ac); psi_memstall_enter(&pflags); *did_some_progress = __perform_reclaim(gfp_mask, reclaim_order, ac); @@ -4648,9 +4661,17 @@ should_reclaim_retry(gfp_t gfp_mask, unsigned order, /* * Costly allocations might have made a progress but this doesn't mean * their order will become available due to high fragmentation so - * always increment the no progress counter for them + * always increment the no progress counter for them. + * + * The same goes for requests whose reclaim is promoted to make whole + * blocks. At that order, reclaim also reports progress when it backs + * off for compaction without freeing anything. + * + * The watermark check below stays at the request order: it asks + * whether the request itself could succeed after reclaim. */ - if (did_some_progress && order <= PAGE_ALLOC_COSTLY_ORDER) + if (did_some_progress && order <= PAGE_ALLOC_COSTLY_ORDER && + nofrag_promote_order(order, alloc_flags, ac) == order) *no_progress_loops = 0; else (*no_progress_loops)++; @@ -5012,9 +5033,15 @@ __alloc_pages_slowpath(gfp_t gfp_mask, unsigned int order, &compaction_retries)) goto retry; - /* Reclaim/compaction failed to prevent the fallback */ + /* + * Reclaim/compaction failed to prevent the fallback. The retry + * budget was spent on making blocks, not on the request itself; + * give the fallback a fresh one before considering OOM. + */ if (defrag_mode && (alloc_flags & ALLOC_NOFRAGMENT)) { alloc_flags &= ~ALLOC_NOFRAGMENT; + no_progress_loops = 0; + compaction_retries = 0; goto retry; } From 179d2ae04df2329c3f8683baa73afcf8b7c0bfcd Mon Sep 17 00:00:00 2001 From: Daehyeon Ko <4ncienth@gmail.com> Date: Fri, 25 Sep 2026 17:05:48 +0900 Subject: [PATCH 0525/1352] assoc_array: discard shortcut when collapsing a leaf-only node assoc_array_delete() can collapse a subtree into a node that contains only leaves while retaining the shortcut that led to it. If that node later fills, all_leaves_cluster_together replaces it with another shortcut. The first shortcut then points directly to the second one. assoc_array_apply_edit() publishes this topology and propagates branch counts from the new child node. It skips the inner shortcut, encounters the outer shortcut where it requires a node and triggers the BUG_ON(). Linux v7.2 and v6.12.105 are affected. The same root remains at the base-commit below and in every supported stable branch checked down to 5.10. It requires CONFIG_KEYS, but no capability, user namespace or race. A UID/GID 1000 process produced: CONTROL_BEGIN mode=exact uid=1000 gid=1000 CONTROL_CapEff: 0000000000000000 kernel BUG at lib/assoc_array.c:1388! Oops: invalid opcode: 0000 [#1] SMP KASAN NOPTI CPU: 0 UID: 1000 PID: 154 Comm: exploit RIP: assoc_array_apply_edit+0x4aa/0x690 Call Trace: __key_link __key_instantiate_and_link __key_create_or_update __do_sys_add_key Kernel panic - not syncing: Fatal exception When deletion produces a leaf-only node, bypass its preceding shortcut as garbage collection already does. Retire the shortcut and old node together after an RCU grace period; reused leaves keep their references and the deleted leaf is still freed separately. The exact trigger reached the BUG in 3/3 unmodified v7.2 KASAN boots and completed cleanly in 3/3 fixed boots. Fixed v6.12.105 also passed 3/3. A source reproducer is available privately on request. Link: https://lore.kernel.org/20260925080548.2505640-1-4ncienth@gmail.com Fixes: 3cb989501c26 ("Add a generic associative array implementation.") Signed-off-by: Daehyeon Ko <4ncienth@gmail.com> Signed-off-by: Andrew Morton Reviewed-by: Jarkko Sakkinen Assisted-by: LLM Cc: David Howells Cc: --- lib/assoc_array.c | 36 ++++++++++++++++++++++-------------- 1 file changed, 22 insertions(+), 14 deletions(-) diff --git a/lib/assoc_array.c b/lib/assoc_array.c index b6c9723e12ced5..841dfe07dc962a 100644 --- a/lib/assoc_array.c +++ b/lib/assoc_array.c @@ -1210,8 +1210,22 @@ struct assoc_array_edit *assoc_array_delete(struct assoc_array *array, goto enomem; edit->new_meta[0] = assoc_array_node_to_ptr(new_n0); - new_n0->back_pointer = node->back_pointer; - new_n0->parent_slot = node->parent_slot; + /* A shortcut above a leaf-only node is redundant. Drop it as + * GC does so that a later split can't create two shortcuts in a row. + */ + ptr = node->back_pointer; + if (assoc_array_ptr_is_shortcut(ptr)) { + struct assoc_array_shortcut *s = + assoc_array_ptr_to_shortcut(ptr); + + new_n0->back_pointer = s->back_pointer; + new_n0->parent_slot = s->parent_slot; + edit->excised_subtree = ptr; + } else { + new_n0->back_pointer = ptr; + new_n0->parent_slot = node->parent_slot; + edit->excised_subtree = assoc_array_node_to_ptr(node); + } new_n0->nr_leaves_on_branch = node->nr_leaves_on_branch; edit->adjust_count_on = new_n0; @@ -1225,21 +1239,15 @@ struct assoc_array_edit *assoc_array_delete(struct assoc_array *array, pr_devel("collapsed %d,%lu\n", collapse.slot, new_n0->nr_leaves_on_branch); BUG_ON(collapse.slot != new_n0->nr_leaves_on_branch - 1); - if (!node->back_pointer) { + if (!new_n0->back_pointer) { edit->set[1].ptr = &array->root; - } else if (assoc_array_ptr_is_leaf(node->back_pointer)) { - BUG(); - } else if (assoc_array_ptr_is_node(node->back_pointer)) { - struct assoc_array_node *p = - assoc_array_ptr_to_node(node->back_pointer); - edit->set[1].ptr = &p->slots[node->parent_slot]; - } else if (assoc_array_ptr_is_shortcut(node->back_pointer)) { - struct assoc_array_shortcut *s = - assoc_array_ptr_to_shortcut(node->back_pointer); - edit->set[1].ptr = &s->next_node; + } else { + struct assoc_array_node *p; + + p = assoc_array_ptr_to_node(new_n0->back_pointer); + edit->set[1].ptr = &p->slots[new_n0->parent_slot]; } edit->set[1].to = assoc_array_node_to_ptr(new_n0); - edit->excised_subtree = assoc_array_node_to_ptr(node); } } From 127e8ac3d348bbd7d9caca71e241898ef083b1d7 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=EC=84=B1=EB=B3=91=EC=B0=AC?= Date: Fri, 2 Oct 2026 07:37:20 +0900 Subject: [PATCH 0526/1352] taskstats: restrict exit listener registration to init_net MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Taskstats exit listeners are stored globally per CPU as bare netlink port IDs, while exit records are always sent through init_net. Since commit edc73c7261ca ("kernel: make taskstats available from all net namespaces"), a task in the initial user and PID namespaces can register a listener from a child network namespace. An unprivileged init_net socket can bind the same namespace-local port ID and receive another UID task's complete exit record. Direct taskstats queries and listener registration by that receiver remain denied with EPERM. Reject CPU-mask listener registration and deregistration outside init_net. Keep PID and TGID queries available from non-initial network namespaces, preserving the getdelays use case added by the cited commit. AI-assisted source review helped identify this issue. The code path, security boundary, and three reproductions on both v7.2.8 and current mainline were manually verified in isolated QEMU guests. Link: https://lore.kernel.org/20261001223721.458667-2-tjdqudcks0424@naver.com Link: https://lore.kernel.org/all/20110630120831.GB7707@albatros/ Link: https://lore.kernel.org/all/87v8x678ph.fsf@email.froward.int.ebiederm.org/ Fixes: edc73c7261ca ("kernel: make taskstats available from all net namespaces") Signed-off-by: 성병찬 Signed-off-by: Andrew Morton Balbir Singh Cc: Cc: Cc: --- kernel/taskstats.c | 3 +++ 1 file changed, 3 insertions(+) diff --git a/kernel/taskstats.c b/kernel/taskstats.c index f31df72f0e9df1..44be8a0f46397b 100644 --- a/kernel/taskstats.c +++ b/kernel/taskstats.c @@ -454,6 +454,9 @@ static int cmd_attr_cpumask(struct genl_info *info, int attr, cpumask_var_t mask __free(free_cpumask_var) = CPUMASK_VAR_NULL; int rc; + if (!net_eq(genl_info_net(info), &init_net)) + return -EINVAL; + if (!alloc_cpumask_var(&mask, GFP_KERNEL)) return -ENOMEM; rc = parse(info->attrs[attr], mask); From 6965760ed9fe35dc90539bb241a043e0229d943d Mon Sep 17 00:00:00 2001 From: Salvatore Dipietro Date: Thu, 1 Oct 2026 08:21:51 +0000 Subject: [PATCH 0527/1352] mm/page_alloc: avoid direct reclaim and compaction for costly __GFP_NORETRY allocations Commit 5d8edfb900d5 ("iomap: Copy larger chunks from userspace") introduced high-order folio allocations in the iomap buffered write path. When memory is fragmented, each failed costly-order allocation enters __alloc_pages_slowpath() which runs direct reclaim, direct compaction and drain_all_pages(), causing a 0.38x throughput drop on PostgreSQL pgbench (simple-update) with 1024 clients on a 96-vCPU arm64 system. The root issue is that direct reclaim and direct compaction are too expensive for hot allocation paths that have fallbacks to smaller allocations. __filemap_get_folio_mpol() already marks higher-order allocations with __GFP_NORETRY | __GFP_NOWARN, signalling that the caller can handle failure. However, the page allocator still enters the full blocking slowpath for costly orders with __GFP_NORETRY, which is unnecessarily aggressive when the caller will simply retry at a lower order. For costly-order allocations with __GFP_NORETRY, clear __GFP_DIRECT_RECLAIM at the very start of the slowpath, before can_direct_reclaim, can_compact and the nofail checks are evaluated. This makes the entire slowpath treat the request as non-blocking: no direct reclaim, no direct compaction and no drain_all_pages() IPI across every CPU. Reclaim and compaction move to the background instead: kswapd is still woken for reclaim, and it wakes kcompactd for defragmentation. The system keeps its long-term health while the latency-critical direct allocation path no longer stalls. Allocations that also request __GFP_THISNODE are exempted. That flag pairing identifies the local-node-first THP attempt issued by alloc_pages_mpol() (mempolicy.c), which relies on direct compaction to form transparent huge pages. Clearing __GFP_DIRECT_RECLAIM makes these allocations reach the !can_direct_reclaim bailout, which since commit 4c0ed883e051 ("mm/page_alloc: fix defrag_mode for non-reclaimable allocations") drops ALLOC_NOFRAGMENT and retries instead of failing when defrag_mode is on. With vm.defrag_mode=1 a __GFP_NORETRY allocation could therefore succeed by fragmenting another migratetype's pageblock. That is the wrong trade for a caller that has already communicated it can handle failure and has a cheap lower-order fallback, so exclude __GFP_NORETRY from the relaxation. The exclusion is not restricted to costly orders: any __GFP_NORETRY caller has said it can cope with failure, so failing is preferable to fragmenting for all of them. Test environment: Hardware: AWS EC2 m8g.24xlarge (96 vCPU, arm64) 12x 1TB IO2 32000 IOPS RAID0 XFS OS: AL2023 Kernel: v7.3-rc1 Database: PostgreSQL 18.4 Workload: pgbench simple-update, 1024 clients, 96 threads, 1200s Results (average of 3 runs, TPS): Config Avg TPS % vs baseline v7.3-rc1 baseline (vm.defrag_mode=0) 58,602 - v7.3-rc1 + this patch (vm.defrag_mode=0) 157,246 +168.3% v7.3-rc1 baseline (vm.defrag_mode=1) 109,012 - v7.3-rc1 + this patch (vm.defrag_mode=1) 157,778 +44.7% Link: https://lore.kernel.org/20261001082152.2879289-1-dipiets@amazon.it Link: https://lore.kernel.org/all/20260403193535.9970-1-dipiets@amazon.it/T/#t [v1] Link: https://lore.kernel.org/linux-mm/20260420161404.642-1-dipiets@amazon.it/T/#u [v2] Link: https://lore.kernel.org/all/20260710143437.12379-1-dipiets@amazon.it/T/#u [v3] Link: https://lore.kernel.org/all/20260904115629.3993331-1-dipiets@amazon.it/T/#u [v4] Link: https://lore.kernel.org/all/20260911142102.2294202-1-dipiets@amazon.it/T/#u [v5] Fixes: 5d8edfb900d5 ("iomap: Copy larger chunks from userspace") Signed-off-by: Salvatore Dipietro Signed-off-by: Andrew Morton Acked-by: Vlastimil Babka (SUSE) Acked-by: Zi Yan Reviewed-by: Johannes Weiner Reviewed-by: Christoph Hellwig Cc: David Hildenbrand Cc: Michal Hocko Cc: Matthew Wilcox Cc: Dave Chinner Cc: Ritesh Harjani Cc: Dmitry Ilvokhin Cc: Hazem Mohamed Abuelfotoh Cc: Ali Saidi Cc: Geoff Blake Cc: Christian Brauner Cc: Cc: Darrick J. Wong Cc: Brendan Jackman Cc: Cc: Suren Baghdasaryan Cc: --- mm/page_alloc.c | 21 ++++++++++++++++++--- 1 file changed, 18 insertions(+), 3 deletions(-) diff --git a/mm/page_alloc.c b/mm/page_alloc.c index 608487672d933e..358fc0b6cdfe61 100644 --- a/mm/page_alloc.c +++ b/mm/page_alloc.c @@ -4805,10 +4805,10 @@ static inline struct page * __alloc_pages_slowpath(gfp_t gfp_mask, unsigned int order, struct alloc_context *ac) { - bool can_direct_reclaim = gfp_mask & __GFP_DIRECT_RECLAIM; - bool can_compact = can_direct_reclaim && gfp_compaction_allowed(gfp_mask); - bool nofail = gfp_mask & __GFP_NOFAIL; const bool costly_order = order > PAGE_ALLOC_COSTLY_ORDER; + bool can_direct_reclaim; + bool can_compact; + bool nofail; struct page *page = NULL; unsigned int alloc_flags; unsigned long did_some_progress; @@ -4823,6 +4823,18 @@ __alloc_pages_slowpath(gfp_t gfp_mask, unsigned int order, bool can_retry_reserves = true; unsigned long alloc_start_time = jiffies; + /* + * Costly __GFP_NORETRY callers have a cheap fallback, so don't stall + * them in reclaim or compaction. __GFP_THISNODE callers are exempt. + */ + if (costly_order && (gfp_mask & __GFP_NORETRY) && + !(gfp_mask & __GFP_THISNODE)) + gfp_mask &= ~__GFP_DIRECT_RECLAIM; + + can_direct_reclaim = gfp_mask & __GFP_DIRECT_RECLAIM; + can_compact = can_direct_reclaim && gfp_compaction_allowed(gfp_mask); + nofail = gfp_mask & __GFP_NOFAIL; + if (unlikely(nofail)) { /* * Also we don't support __GFP_NOFAIL without __GFP_DIRECT_RECLAIM, @@ -4939,8 +4951,11 @@ __alloc_pages_slowpath(gfp_t gfp_mask, unsigned int order, * Reclaim/compaction cannot run, so defrag_mode's strategy * of enforcing ALLOC_NOFRAGMENT cannot be fulfilled. Allow * fallbacks rather than failing the allocation outright. + * Not for __GFP_NORETRY: those have a cheap lower order + * fallback, so failing beats fragmenting. */ if (defrag_mode && (alloc_flags & ALLOC_NOFRAGMENT) && + !(gfp_mask & __GFP_NORETRY) && (gfp_mask & __GFP_KSWAPD_RECLAIM)) { alloc_flags &= ~ALLOC_NOFRAGMENT; goto retry; From 95a27e71ed9115a7315ef92fa077160007aa5b26 Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Wed, 19 Aug 2026 17:51:43 +0800 Subject: [PATCH 0528/1352] mm: use a folio in the softleaf_is_device_private path Use the folio APIs in the device_private migration path of do_swap_page(), replacing four calls to compound_head() with two page_folio() calls. The second one re-fetches the folio from vmf->page after migrate_to_ram(), which might have split the folio. Link: https://lore.kernel.org/20260819095144.45660-1-hongfu.li@linux.dev Signed-off-by: Hongfu Li Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand (Arm) Link: https://lore.kernel.org/all/e20678ed-3fa1-4677-a1d7-e2af481e8302@kernel.org/ Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Anshuman Khandual Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/memory.c | 15 +++++++++------ 1 file changed, 9 insertions(+), 6 deletions(-) diff --git a/mm/memory.c b/mm/memory.c index 8b0c2c735d3de7..9cbce5c90bffde 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -4926,18 +4926,21 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) goto unlock; /* - * Get a page reference while we know the page can't be - * freed. + * Get a folio reference while we know the folio can't + * be freed. */ - if (trylock_page(vmf->page)) { + folio = page_folio(vmf->page); + if (folio_trylock(folio)) { struct dev_pagemap *pgmap; - get_page(vmf->page); + folio_get(folio); pte_unmap_unlock(vmf->pte, vmf->ptl); pgmap = page_pgmap(vmf->page); ret = pgmap->ops->migrate_to_ram(vmf); - unlock_page(vmf->page); - put_page(vmf->page); + /* migrate_to_ram() might have split the folio. */ + folio = page_folio(vmf->page); + folio_unlock(folio); + folio_put(folio); } else { pte_unmap(vmf->pte); softleaf_entry_wait_on_locked(entry, vmf->ptl); From 3df2226724667b7889369e4cab9b6593094b3918 Mon Sep 17 00:00:00 2001 From: Qi Xi Date: Wed, 19 Aug 2026 16:20:52 +0800 Subject: [PATCH 0529/1352] mm: drop stale MAX_ORDER references The treewide rename in commit 5e0a760b4441 ("mm, treewide: rename MAX_ORDER to MAX_PAGE_ORDER") left a few spots still using the old name: - two comments in include/net/mana/mana.h and mm/page_alloc.c; - the gdb helper scripts/gdb/linux/mm.py, where self.MAX_ORDER is a local mirror of the kernel's MAX_ORDER define. Rename the leftover instances to MAX_PAGE_ORDER so the tree is consistent. No functional changes. Link: https://lore.kernel.org/20260819082052.3338603-1-xiqi2@huawei.com Signed-off-by: Qi Xi Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Cc: Jan Kiszka Cc: Johannes Weiner Cc: Kefeng Wang Cc: Kieran Bingham Cc: Konstantin Taranov Cc: Long Li Cc: Michal Hocko Cc: Nanyong Sun Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/net/mana/mana.h | 4 ++-- mm/page_alloc.c | 2 +- scripts/gdb/linux/mm.py | 8 ++++---- 3 files changed, 7 insertions(+), 7 deletions(-) diff --git a/include/net/mana/mana.h b/include/net/mana/mana.h index 83b7eff4646ead..e8fda092be37fb 100644 --- a/include/net/mana/mana.h +++ b/include/net/mana/mana.h @@ -47,8 +47,8 @@ enum mana_priv_flag_bits { #define COMP_ENTRY_SIZE 64 /* This Max value for RX buffers is derived from __alloc_page()'s max page - * allocation calculation. It allows maximum 2^(MAX_ORDER -1) pages. RX buffer - * size beyond this value gets rejected by __alloc_page() call. + * allocation calculation. It allows maximum 2^MAX_PAGE_ORDER pages. RX + * buffer size beyond this value gets rejected by __alloc_page() call. */ #define MAX_RX_BUFFERS_PER_QUEUE 8192 #define DEF_RX_BUFFERS_PER_QUEUE 1024 diff --git a/mm/page_alloc.c b/mm/page_alloc.c index 358fc0b6cdfe61..d622347c877bdf 100644 --- a/mm/page_alloc.c +++ b/mm/page_alloc.c @@ -8020,7 +8020,7 @@ static bool cond_accept_memory(struct zone *zone, unsigned int order, /* * Watermarks have not been initialized yet. * - * Accepting one MAX_ORDER page to ensure progress. + * Accepting one MAX_PAGE_ORDER page to ensure progress. */ if (!wmark) return try_to_accept_memory_one(zone); diff --git a/scripts/gdb/linux/mm.py b/scripts/gdb/linux/mm.py index dffadccbb01d24..28d33624c38bd6 100644 --- a/scripts/gdb/linux/mm.py +++ b/scripts/gdb/linux/mm.py @@ -56,7 +56,7 @@ def __init__(self): self.MAX_PHYSMEM_BITS = 46 self.SECTION_SIZE_BITS = 27 - self.MAX_ORDER = 10 + self.MAX_PAGE_ORDER = 10 self.SECTIONS_SHIFT = self.MAX_PHYSMEM_BITS - self.SECTION_SIZE_BITS self.NR_MEM_SECTIONS = 1 << self.SECTIONS_SHIFT @@ -233,11 +233,11 @@ def __init__(self): self.SECTIONS_SHIFT = self.MAX_PHYSMEM_BITS - self.SECTION_SIZE_BITS if str(constants.LX_CONFIG_ARCH_FORCE_MAX_ORDER).isdigit(): - self.MAX_ORDER = constants.LX_CONFIG_ARCH_FORCE_MAX_ORDER + self.MAX_PAGE_ORDER = constants.LX_CONFIG_ARCH_FORCE_MAX_ORDER else: - self.MAX_ORDER = 10 + self.MAX_PAGE_ORDER = 10 - self.MAX_ORDER_NR_PAGES = 1 << (self.MAX_ORDER) + self.MAX_ORDER_NR_PAGES = 1 << (self.MAX_PAGE_ORDER) self.PFN_SECTION_SHIFT = self.SECTION_SIZE_BITS - self.PAGE_SHIFT self.NR_MEM_SECTIONS = 1 << self.SECTIONS_SHIFT self.PAGES_PER_SECTION = 1 << self.PFN_SECTION_SHIFT From e7aefdf0d97488ddbe7c4959fa1f632f0f32dd72 Mon Sep 17 00:00:00 2001 From: Ridong Chen Date: Wed, 26 Aug 2026 20:44:09 +0800 Subject: [PATCH 0530/1352] mm/vmscan: drop the combined limit gate in __node_reclaim() __node_reclaim() is called from two paths: node_reclaim() and user_proactive_reclaim(). node_reclaim() already bails out early unless node_pagecache_reclaimable() is over pgdat->min_unmapped_pages or the reclaimable slab is over pgdat->min_slab_pages. The identical check inside __node_reclaim() that guards the shrink_node() loop is therefore redundant for this path. user_proactive_reclaim() is proactive reclaim driven by userspace and should not be gated by the per-node min_unmapped_pages / min_slab_pages limits at all [1]. With the gate in place, a proactive request is silently turned into a no-op whenever the node happens to sit below both thresholds. Drop the gate in __node_reclaim() and always run the shrink_node() loop. The node_reclaim() path is unchanged, since its caller has already applied the same test; the proactive path is no longer wrongly gated. Link: https://lore.kernel.org/20260826124409.35569-1-ridong.chen@linux.dev Link: https://sashiko.dev/#/patchset/20260723045718.2052070-1-ridong.chen@linux.dev [1] Fixes: b980077899ea ("mm: introduce per-node proactive reclaim interface") Acked-by: Johannes Weiner Signed-off-by: Ridong Chen Signed-off-by: Andrew Morton Acked-by: Michal Hocko Acked-by: Shakeel Butt Acked-by: Davidlohr Bueso Assisted-by: Claude:claude-opus-4-8 Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Roman Gushchin Cc: Wei Xu Cc: Yuanchu Xie --- mm/vmscan.c | 13 +++---------- 1 file changed, 3 insertions(+), 10 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index f11491ee9ed5c1..6dff207ad8c612 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -7851,16 +7851,9 @@ static unsigned long __node_reclaim(struct pglist_data *pgdat, noreclaim_flag = memalloc_noreclaim_save(); set_task_reclaim_state(p, &sc->reclaim_state); - if (node_pagecache_reclaimable(pgdat) > pgdat->min_unmapped_pages || - node_page_state_pages(pgdat, NR_SLAB_RECLAIMABLE_B) > pgdat->min_slab_pages) { - /* - * Free memory by calling shrink node with increasing - * priorities until we have enough memory freed. - */ - do { - shrink_node(pgdat, sc); - } while (sc->nr_reclaimed < nr_pages && --sc->priority >= 0); - } + do { + shrink_node(pgdat, sc); + } while (sc->nr_reclaimed < nr_pages && --sc->priority >= 0); set_task_reclaim_state(p, NULL); memalloc_noreclaim_restore(noreclaim_flag); From 7490f41d91d474be62d50a1194a40665ee9e14fa Mon Sep 17 00:00:00 2001 From: JonasZhou Date: Tue, 25 Aug 2026 18:46:59 +0800 Subject: [PATCH 0531/1352] mm/vmalloc: avoid false sharing with drain_vmap_work free_vmap_area_noflush() queues drain_vmap_work after the number of lazily freed pages exceeds lazy_max_pages(). Until the worker purges those pages, concurrent frees keep calling schedule_work(). Even if the work is already pending, queue_work_on() performs a locked test_and_set_bit() on the pending bit in the work item. On the tested x86-64 build, drain_vmap_work and vmap_nodes occupy the same 64-byte cache line. The work item starts at offset 0 and the vmap_nodes pointer at offset 32. The latter is read by vmap allocation and free paths, so updates to the work item invalidate a cache line read by all CPUs. Put drain_vmap_work in the cacheline-aligned data section. Tests were run on Linux 7.2. On a two-socket Intel Xeon Silver 4208 system using 16 workers, the runtimes of vmalloc.fix_align, vmalloc.fix_size, and vmalloc.no_block_alloc decreased by 12.61%, 5.78%, and 6.87%, respectively. HITM samples for the affected cache line and total HITM samples decreased by 96.55% and 13.36%, respectively. Link: https://lore.kernel.org/20260825104659.100134-1-jonaszhou-oc@zhaoxin.com Signed-off-by: JonasZhou Signed-off-by: Andrew Morton Reviewed-by: Uladzislau Rezki (Sony) Cc: --- mm/vmalloc.c | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index 89c327a6ce7d9f..b879260d31a577 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -1090,7 +1090,12 @@ RB_DECLARE_CALLBACKS_MAX(static, free_vmap_area_rb_augment_cb, static void reclaim_and_purge_vmap_areas(void); static BLOCKING_NOTIFIER_HEAD(vmap_notify_list); static void drain_vmap_area_work(struct work_struct *work); -static DECLARE_WORK(drain_vmap_work, drain_vmap_area_work); +/* + * Keep the work item, whose pending bit is updated by freeing CPUs, + * away from vmap metadata read by allocation and free paths. + */ +static __cacheline_aligned_in_smp +DECLARE_WORK(drain_vmap_work, drain_vmap_area_work); static struct workqueue_struct *drain_vmap_helpers_wq; static struct workqueue_struct *drain_vmap_wq; From c1b31095ff2f8460d69a5e50b0b90cdb63c07c69 Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Tue, 25 Aug 2026 10:10:13 +0800 Subject: [PATCH 0532/1352] mm/hugetlb: fix resv_huge_pages double decrement in memfd error path alloc_hugetlb_folio_reserve() decrements h->resv_huge_pages when dequeuing a folio, but unlike the use_global_reservation handling in hugetlb_alloc_folio(), it does not set HPageRestoreReserve on the folio. Its sole caller memfd_alloc_folio() pre-allocates a reservation via hugetlb_reserve_pages() before allocating. When hugetlb_add_to_page_cache() fails, folio_put() drops the folio without HPageRestoreReserve set, so free_huge_folio() does not restore the reservation. The subsequent hugetlb_unreserve_pages() on the err_unresv path decrements the counter a second time, leaving resv_huge_pages off by one for every failed allocation. Set HPageRestoreReserve when consuming the reservation in alloc_hugetlb_folio_reserve(). On the error path, free_huge_folio() then restores the reservation before hugetlb_unreserve_pages() releases it. The success path is unaffected, as hugetlb_add_to_page_cache() clears the flag once the folio is added to the page cache. Link: https://lore.kernel.org/20260825021013.25672-1-hongfu.li@linux.dev Fixes: 26a8ea80929c ("mm/hugetlb: fix memfd_pin_folios resv_huge_pages leak") Signed-off-by: Hongfu Li Signed-off-by: Andrew Morton Reviewed-by: Muchun Song Cc: David Hildenbrand Cc: Oscar Salvador Cc: Steven Sistare Cc: Vivek Kasireddy --- mm/hugetlb.c | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 49ffbcb54f8c08..3af46101f1bbce 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -2187,8 +2187,10 @@ struct folio *alloc_hugetlb_folio_reserve(struct hstate *h, int preferred_nid, folio = dequeue_hugetlb_folio_nodemask(h, gfp_mask, preferred_nid, nmask); - if (folio) + if (folio) { + folio_set_hugetlb_restore_reserve(folio); h->resv_huge_pages--; + } spin_unlock_irq(&hugetlb_lock); return folio; From be7573570f60b0d330cbf97d3511cf861b40ce6a Mon Sep 17 00:00:00 2001 From: Hemanth Selam Date: Tue, 25 Aug 2026 21:47:15 +0530 Subject: [PATCH 0533/1352] selftests/mm: remove the local PKEY_UNRESTRICTED fallback pkey-helpers.h defines PKEY_UNRESTRICTED itself when the macro is not already known, a stopgap from when the generic definition was still under review. It has been merged since, commit 6d61527d931b ("mm/pkey: Add PKEY_UNRESTRICTED macro"), so the guard is never taken and the FIXME can be honoured. The definition comes from tools/include/uapi/asm-generic/mman-common.h via TOOLS_INCLUDES, which commit e076eaca5906 ("selftests: break the dependency upon local header files") added so that the mm selftests build without "make headers". It is reached through the that the system includes. Building the pkey tests with KHDR_INCLUDES pointing at an empty directory confirms that; emptying TOOLS_INCLUDES as well is what makes the macro go missing. No functional change intended. Link: https://lore.kernel.org/20260825161715.2807297-1-hemanth.selam@gmail.com Signed-off-by: Hemanth Selam Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: SJ Park Assisted-by: Cursor:claude-opus-5 Cc: Kevin Brodsky Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Yury Khrustalev --- tools/testing/selftests/mm/pkey-helpers.h | 7 ------- 1 file changed, 7 deletions(-) diff --git a/tools/testing/selftests/mm/pkey-helpers.h b/tools/testing/selftests/mm/pkey-helpers.h index 46a8a1878dc1fd..9b949951743dd6 100644 --- a/tools/testing/selftests/mm/pkey-helpers.h +++ b/tools/testing/selftests/mm/pkey-helpers.h @@ -114,13 +114,6 @@ void record_pkey_malloc(void *ptr, long size, int prot); #define PKEY_MASK (PKEY_DISABLE_ACCESS | PKEY_DISABLE_WRITE) #endif -/* - * FIXME: Remove once the generic PKEY_UNRESTRICTED definition is merged. - */ -#ifndef PKEY_UNRESTRICTED -#define PKEY_UNRESTRICTED 0x0 -#endif - #ifndef set_pkey_bits static inline u64 set_pkey_bits(u64 reg, int pkey, u64 flags) { From 694de831d2ad4398d8e16c40b89fcbdc1dccf436 Mon Sep 17 00:00:00 2001 From: Ridong Chen Date: Fri, 21 Aug 2026 10:16:06 +0800 Subject: [PATCH 0534/1352] mm/mglru: preserve inactive placement when enabling MGLRU When the LRU is switched to MGLRU (echo y > /sys/kernel/mm/lru_gen/ enabled), fill_evictable() re-inserts every folio via lru_gen_add_folio(..., false). With reclaiming hardcoded to false, an inactive anonymous folio (no PG_active, not in the swapcache) takes the "gen = MIN_NR_GENS" branch in lru_gen_folio_seq() and is seeded at seq = max_seq - 1, which lru_gen_is_active() treats as active. Its inactive placement is lost and NR_INACTIVE_ANON is folded into NR_ACTIVE_ANON. Pass reclaiming=!active so a folio from an inactive list is seeded into an older generation. Folios from the active list carry PG_active and hit the first branch either way, so they are unchanged. Reclaiming also selects the insertion end in lru_gen_add_folio(): list_add_tail() for inactive folios, list_add() for active ones. Both the legacy LRU and a MGLRU generation keep the hottest folios at the head and the coldest at the tail, and reclaim takes from the tail. To preserve that order the folio must be taken from the end matching the insertion end, so take inactive folios from the head and active folios from the tail; otherwise hot/cold would be reversed within the generation. Tested on x86_64, next-20260812, 2G VM + 1G swap, ~1.5G anon pushed onto the inactive list before enabling MGLRU: Active(anon) Inactive(anon) before switch (legacy) 2952 1548792 kB after `echo y`, unpatched 1552052 0 kB after `echo y`, patched 15144 1536636 kB Inactive file folios stay inactive either way (NR_INACTIVE_FILE is preserved). Link: https://lore.kernel.org/20260821021606.877330-1-ridong.chen@linux.dev Fixes: 354ed5974429 ("mm: multi-gen LRU: kill switch") Signed-off-by: Ridong Chen Signed-off-by: Andrew Morton Suggested-by: Barry Song Acked-by: Barry Song Assisted-by: Claude:claude-opus-4-8 Cc: Axel Rasmussen Cc: David Hildenbrand Cc: Jan Alexander Steffens (heftig) Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oleksandr Natalenko Cc: Shakeel Butt Cc: Steven Barrett Cc: Suleiman Souhlal Cc: Wei Xu Cc: Yuanchu Xie Cc: Yu Zhao --- mm/vmscan.c | 18 ++++++++++++++++-- 1 file changed, 16 insertions(+), 2 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 6dff207ad8c612..fdd13299a04a93 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -5307,7 +5307,17 @@ static bool fill_evictable(struct lruvec *lruvec) while (!list_empty(head)) { bool success; - struct folio *folio = lru_to_folio(head); + struct folio *folio; + + /* + * lru_gen_add_folio() uses list_add_tail() rather + * than list_add() when reclaiming is true. Match + * its ordering to avoid cold/hot inversion. + */ + if (active) + folio = lru_to_folio(head); + else + folio = list_first_entry(head, struct folio, lru); VM_WARN_ON_ONCE_FOLIO(folio_test_unevictable(folio), folio); VM_WARN_ON_ONCE_FOLIO(folio_test_active(folio) != active, folio); @@ -5315,7 +5325,11 @@ static bool fill_evictable(struct lruvec *lruvec) VM_WARN_ON_ONCE_FOLIO(folio_lru_gen(folio) != -1, folio); lruvec_del_folio(lruvec, folio); - success = lru_gen_add_folio(lruvec, folio, false); + /* + * Borrow reclaiming=true to place inactive folios in + * the older gens. + */ + success = lru_gen_add_folio(lruvec, folio, !active); VM_WARN_ON_ONCE(!success); if (!--remaining) From 985e7e383a481db867cd1751f05d1c76a5839541 Mon Sep 17 00:00:00 2001 From: Chengfeng Ye Date: Mon, 24 Aug 2026 19:24:33 +0800 Subject: [PATCH 0535/1352] mm/ksm: mark migration stores with WRITE_ONCE() ksm_get_folio() deliberately samples stable_node->kpfn and folio->mapping without taking the folio lock because the KSM folio may be migrated concurrently. folio_migrate_ksm() updates the same state using plain assignments. The reader can load the old kpfn, then the migrator can store the new kpfn, execute smp_wmb(), and clear the old folio's mapping before the reader checks that mapping. Thus the initial kpfn load can overlap its update and the subsequent mapping load can overlap the clear, with no common lock. This leaves marked READ_ONCE() accesses racing with plain stores. The kernel reported: BUG: KCSAN: data-race in folio_migrate_ksm / ksm_get_folio read (marked) to 0xffff8ce401421330 of 8 bytes by task 48 on cpu 3: ksm_get_folio+0x7f/0x2a0 ksm_scan_thread+0x1635/0x3330 kthread+0x1af/0x1f0 write to 0xffff8ce401421330 of 8 bytes by task 102 on cpu 1: folio_migrate_ksm+0x6a/0xd0 folio_migrate_flags+0x193/0x420 __migrate_folio.isra.0+0x162/0x1a0 migrate_folio+0x4c/0x70 move_to_new_folio+0xd6/0x170 Use WRITE_ONCE() for both stores to pair them with the existing lockless reads. This preserves the existing smp_wmb()/smp_rmb() migration protocol and control flow while preventing compiler transformations of the shared accesses. Link: https://lore.kernel.org/20260824112433.191301-1-nicoyip.dev@gmail.com Fixes: c8d6553b9580 ("ksm: make KSM page migration possible") Signed-off-by: Chengfeng Ye Signed-off-by: Andrew Morton Acked-by: Xu Xin Acked-by: David Hildenbrand (Arm) Cc: Hugh Dickins Cc: --- mm/ksm.c | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/mm/ksm.c b/mm/ksm.c index 49d48d1e099801..fe42c17490b89c 100644 --- a/mm/ksm.c +++ b/mm/ksm.c @@ -1115,7 +1115,8 @@ static inline void folio_set_stable_node(struct folio *folio, struct ksm_stable_node *stable_node) { VM_WARN_ON_FOLIO(folio_test_anon(folio) && PageAnonExclusive(&folio->page), folio); - folio->mapping = (void *)((unsigned long)stable_node | FOLIO_MAPPING_KSM); + WRITE_ONCE(folio->mapping, + (void *)((unsigned long)stable_node | FOLIO_MAPPING_KSM)); } #ifdef CONFIG_SYSFS @@ -3316,7 +3317,7 @@ void folio_migrate_ksm(struct folio *newfolio, struct folio *folio) stable_node = folio_stable_node(folio); if (stable_node) { VM_BUG_ON_FOLIO(stable_node->kpfn != folio_pfn(folio), folio); - stable_node->kpfn = folio_pfn(newfolio); + WRITE_ONCE(stable_node->kpfn, folio_pfn(newfolio)); /* * newfolio->mapping was set in advance; now we need smp_wmb() * to make sure that the new stable_node->kpfn is visible From c2f340b1349a57c76646a24cc80abefdc59d6bc6 Mon Sep 17 00:00:00 2001 From: Kaitao Cheng Date: Mon, 24 Aug 2026 23:16:55 +0800 Subject: [PATCH 0536/1352] mm/hugetlb: use hugetlb_vmemmap_optimizable() in boolean contexts The two boot-time sites in hugetlb_hstate_alloc_pages_onenode() and hugetlb_pages_alloc_boot_node() only need to know whether HVO is applicable, not the exact optimizable size. Switch them from hugetlb_vmemmap_optimizable_size() to hugetlb_vmemmap_optimizable() to make intent explicit. No functional change intended. Link: https://lore.kernel.org/20260824151655.30840-1-kaitao.cheng@linux.dev Signed-off-by: Kaitao Cheng Signed-off-by: Andrew Morton Reviewed-by: Joshua Hahn Reviewed-by: Muchun Song Cc: David Hildenbrand Cc: Oscar Salvador --- mm/hugetlb.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 3af46101f1bbce..d73108393e4b76 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -3438,7 +3438,7 @@ static void __init hugetlb_hstate_alloc_pages_onenode(struct hstate *h, int nid) folio = only_alloc_fresh_hugetlb_folio(h, gfp_mask, nid, &node_states[N_MEMORY], NULL); if (!folio && !list_empty(&folio_list) && - hugetlb_vmemmap_optimizable_size(h)) { + hugetlb_vmemmap_optimizable(h)) { prep_and_add_allocated_folios(h, &folio_list); INIT_LIST_HEAD(&folio_list); folio = only_alloc_fresh_hugetlb_folio(h, gfp_mask, nid, @@ -3507,7 +3507,7 @@ static void __init hugetlb_pages_alloc_boot_node(unsigned long start, unsigned l for (i = 0; i < num; ++i) { struct folio *folio; - if (hugetlb_vmemmap_optimizable_size(h) && + if (hugetlb_vmemmap_optimizable(h) && (si_mem_available() == 0) && !list_empty(&folio_list)) { prep_and_add_allocated_folios(h, &folio_list); INIT_LIST_HEAD(&folio_list); From 24bb05a71509862953bc06dfaddd8efd77e51feb Mon Sep 17 00:00:00 2001 From: Jinjiang Tu Date: Mon, 24 Aug 2026 14:10:09 +0800 Subject: [PATCH 0537/1352] docs: ksm: fix typos in sysfs knob names Patch series "docs/ksm: fix advisor documentation and comment", v3. This series fixes two problems left in the KSM advisor documentation and code comment: - Patch 1 fixes two typos in sysfs knob names ("adivsor_max_cpu" and "adivsor_max_pages_to_scan") in ksm.rst that don't match the actual knob names. - Patch 2 fixes the description of advisor_min_pages_to_scan: it is described as the lower limit of pages_to_scan, but is actually only used as it's initial value, and the runtime pages_to_scan can drop below advisor_min_pages_to_scan. This patch (of 2): The sysfs knob names in mm/ksm.c are "advisor_max_cpu" and "advisor_max_pages_to_scan", but the ksm.rst documentation spelled both as "adivsor_*", fix the two typos. Link: https://lore.kernel.org/20260824061010.3343959-1-tujinjiang@huawei.com Link: https://lore.kernel.org/20260824061010.3343959-2-tujinjiang@huawei.com Signed-off-by: Jinjiang Tu Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: Randy Dunlap Cc: Chengming Zhou Cc: Jonathan Corbet Cc: Kefeng Wang Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Nanyong Sun Cc: Stefan Roesch Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: xu xin --- Documentation/admin-guide/mm/ksm.rst | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/Documentation/admin-guide/mm/ksm.rst b/Documentation/admin-guide/mm/ksm.rst index ad8e7a41f3b5d2..c9f533b10f6f9d 100644 --- a/Documentation/admin-guide/mm/ksm.rst +++ b/Documentation/admin-guide/mm/ksm.rst @@ -174,7 +174,7 @@ advisor_mode The section about ``advisor`` explains in detail how the scan time advisor works. -adivsor_max_cpu +advisor_max_cpu specifies the upper limit of the cpu percent usage of the ksmd background thread. The default is 70. @@ -186,7 +186,7 @@ advisor_min_pages_to_scan specifies the lower limit of the ``pages_to_scan`` parameter of the scan time advisor. The default is 500. -adivsor_max_pages_to_scan +advisor_max_pages_to_scan specifies the upper limit of the ``pages_to_scan`` parameter of the scan time advisor. The default is 30000. From a18bddcaef521595f1bc550a024c59fbf3146487 Mon Sep 17 00:00:00 2001 From: Jinjiang Tu Date: Mon, 24 Aug 2026 14:10:10 +0800 Subject: [PATCH 0538/1352] mm/ksm: fix advisor_min_pages_to_scan description Both Documentation/admin-guide/mm/ksm.rst and the comment next to the variable definition in mm/ksm.c describe advisor_min_pages_to_scan as a lower limit of the pages_to_scan parameter, but that is not how the scan-time advisor actually uses it. commit 4e5fa4f5eff6 ("mm/ksm: add ksm advisor") only uses it to initialize ksm_thread_pages_to_scan when the scan-time advisor is enabled. ksm_thread_pages_to_scan is adjusted by scan_time_advisor() after a full scan finishes. ksm_thread_pages_to_scan could be increased or decreased depend on the real scan time is longer or shorter than the target scan time. The min value of ksm_thread_pages_to_scan is only limited by KSM_ADVISOR_MIN_CPU, so ksm_thread_pages_to_scan could be smaller than ksm_advisor_min_pages_to_scan. The semantics of advisor_min_pages_to_scan was updated in the v2 patchset [1], but the documentation wasn't updated. Update the documentation and comment to match the semantics of advisor_min_pages_to_scan. Link: https://lore.kernel.org/linux-mm/20231028000945.2428830-2-shr@devkernel.io/ [1] Link: https://lore.kernel.org/20260824061010.3343959-3-tujinjiang@huawei.com Signed-off-by: Jinjiang Tu Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Cc: Chengming Zhou Cc: Jonathan Corbet Cc: Kefeng Wang Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Nanyong Sun Cc: Stefan Roesch Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: xu xin Cc: Randy Dunlap --- Documentation/admin-guide/mm/ksm.rst | 4 ++-- mm/ksm.c | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/Documentation/admin-guide/mm/ksm.rst b/Documentation/admin-guide/mm/ksm.rst index c9f533b10f6f9d..c329ca747b8c47 100644 --- a/Documentation/admin-guide/mm/ksm.rst +++ b/Documentation/admin-guide/mm/ksm.rst @@ -183,8 +183,8 @@ advisor_target_scan_time pages. The default value is 200 seconds. advisor_min_pages_to_scan - specifies the lower limit of the ``pages_to_scan`` parameter of the - scan time advisor. The default is 500. + specifies the initial value of the ``pages_to_scan`` parameter of + the scan time advisor. The default is 500. advisor_max_pages_to_scan specifies the upper limit of the ``pages_to_scan`` parameter of the diff --git a/mm/ksm.c b/mm/ksm.c index fe42c17490b89c..624f37975e1295 100644 --- a/mm/ksm.c +++ b/mm/ksm.c @@ -348,7 +348,7 @@ static enum ksm_advisor_type ksm_advisor; * Only called through the sysfs control interface: */ -/* At least scan this many pages per batch. */ +/* Initial number of pages to scan per batch. */ static unsigned long ksm_advisor_min_pages_to_scan = 500; static void set_advisor_defaults(void) From 939320955ae04d543f5749448e3f4bab6d793361 Mon Sep 17 00:00:00 2001 From: Anshuman Date: Wed, 26 Aug 2026 11:43:00 +0530 Subject: [PATCH 0539/1352] selftests/mm: fix line buffer leak in mremap_test is_range_mapped() is_range_mapped() uses getline() to read /proc/self/maps line by line, but never frees the buffer it allocates. Every exit path (parse failure, match found, or reaching EOF) returns without calling free(line), leaking the buffer on each call. The function is called multiple times in this test, so the leak accumulates across calls. Free line before returning. Link: https://lore.kernel.org/20260826061300.14038-1-anshumantewari123@gmail.com Signed-off-by: Anshuman Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: SJ Park Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/mm/mremap_test.c | 1 + 1 file changed, 1 insertion(+) diff --git a/tools/testing/selftests/mm/mremap_test.c b/tools/testing/selftests/mm/mremap_test.c index 131d9d6db86790..779ef2d5f1b9f7 100644 --- a/tools/testing/selftests/mm/mremap_test.c +++ b/tools/testing/selftests/mm/mremap_test.c @@ -156,6 +156,7 @@ static bool is_range_mapped(FILE *maps_fp, unsigned long start, } } + free(line); return success; } From ccb67a4ccba9d7c7cc967c4ea7976c25fe428fbc Mon Sep 17 00:00:00 2001 From: Jiayuan Chen Date: Thu, 27 Aug 2026 10:54:56 +0800 Subject: [PATCH 0540/1352] mm/memcontrol: fix data-race on reading jiffies_64 KCSAN reported a data-race between tick_do_update_jiffies64() updating jiffies_64 and mem_cgroup_flush_stats_ratelimited() reading it directly. Unlike jiffies, jiffies_64 is not volatile, so raw reads are plain accesses and can even be torn on 32-bit. Use get_jiffies_64() instead, and fix the same pattern in mem_cgroup_flush_foreign(). Link: https://lore.kernel.org/20260827025457.116191-1-jiayuan.chen@linux.dev Fixes: 508bed884767 ("mm: memcg: change flush_next_time to flush_last_time") Fixes: 97b27821b485 ("writeback, memcg: Implement foreign dirty flushing") Signed-off-by: Jiayuan Chen Signed-off-by: Andrew Morton Reported-by: syzbot+ced4d9a8cadb5ef3adae@syzkaller.appspotmail.com Acked-by: Muchun Song Acked-by: Johannes Weiner Acked-by: Michal Hocko Cc: Chris Li Cc: Jan Kara Cc: Jens Axboe Cc: Roman Gushchin Cc: Shakeel Butt Cc: Tejun Heo Cc: --- mm/memcontrol.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 856a7d07586ccc..d399710e9799ab 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -757,7 +757,7 @@ static void __mem_cgroup_flush_stats(struct mem_cgroup *memcg, bool force) return; if (mem_cgroup_is_root(memcg)) - WRITE_ONCE(flush_last_time, jiffies_64); + WRITE_ONCE(flush_last_time, get_jiffies_64()); css_rstat_flush(&memcg->css); } @@ -785,7 +785,7 @@ void mem_cgroup_flush_stats(struct mem_cgroup *memcg) void mem_cgroup_flush_stats_ratelimited(struct mem_cgroup *memcg) { /* Only flush if the periodic flusher is one full cycle late */ - if (time_after64(jiffies_64, READ_ONCE(flush_last_time) + 2*FLUSH_TIME)) + if (time_after64(get_jiffies_64(), READ_ONCE(flush_last_time) + 2 * FLUSH_TIME)) mem_cgroup_flush_stats(memcg); } @@ -3946,7 +3946,7 @@ void mem_cgroup_flush_foreign(struct bdi_writeback *wb) { struct mem_cgroup *memcg = mem_cgroup_from_css(wb->memcg_css); unsigned long intv = msecs_to_jiffies(dirty_expire_interval * 10); - u64 now = jiffies_64; + u64 now = get_jiffies_64(); int i; for (i = 0; i < MEMCG_CGWB_FRN_CNT; i++) { From 0d2c882beabe00b4fce3b20966e84fbe3c2746f1 Mon Sep 17 00:00:00 2001 From: Hui Zhu Date: Thu, 27 Aug 2026 15:05:46 +0800 Subject: [PATCH 0541/1352] mm/vmstat: annotate data race for per-cpu pageset fields zoneinfo_show_print() reads pcp->count, pcp->high, pcp->batch, pcp->high_min, pcp->high_max and the per-cpu stat_threshold while holding only zone->lock, which does not synchronize these fields. The writers are the page allocation and free fast paths under pcp->lock, decay_pcp_high() which updates pcp->high without any lock, pageset_update() which writes batch/high_min/high_max with WRITE_ONCE(), and refresh_zone_stat_thresholds() which writes stat_threshold locklessly. The race is benign: the values are only printed to /proc/zoneinfo, they are naturally aligned integers, and pageset_update() already documents that users of batch/high_min/high_max must cope with the fields changing asynchronously. Annotate the reads with data_race(), following commit af1c31acc853 ("mm/vmstat: annotate data race for zone->free_area[order].nr_free"). Found by KCSAN testing on an older kernel; the same race still exists on mainline. No functional change intended. BUG: KCSAN: data-race in zoneinfo_show_print+0x355/0x520 root/klinux/mm/vmstat.c:1774 race at unknown origin, with read to 0xffff8e1835410808 of 4 bytes by task 22653 on cpu 12: zoneinfo_show_print+0x355/0x520 root/klinux/mm/vmstat.c:1774 walk_zones_in_node root/klinux/mm/vmstat.c:1496 [inline] zoneinfo_show+0x41/0x70 root/klinux/mm/vmstat.c:1806 seq_read_iter+0x30c/0x970 root/klinux/fs/seq_file.c:230 proc_reg_read_iter+0x10c/0x170 root/klinux/fs/proc/inode.c:305 copy_splice_read+0x2a1/0x4e0 root/klinux/fs/splice.c:365 do_splice_read root/klinux/fs/splice.c:985 [inline] do_splice_read+0x139/0x1a0 root/klinux/fs/splice.c:959 splice_direct_to_actor+0x16b/0x540 root/klinux/fs/splice.c:1089 do_splice_direct_actor root/klinux/fs/splice.c:1207 [inline] do_splice_direct+0x10a/0x180 root/klinux/fs/splice.c:1233 do_sendfile+0x6ea/0x7e0 root/klinux/fs/read_write.c:1363 __do_sys_sendfile64 root/klinux/fs/read_write.c:1424 [inline] __se_sys_sendfile64 root/klinux/fs/read_write.c:1410 [inline] __x64_sys_sendfile64+0x117/0x130 root/klinux/fs/read_write.c:1410 x64_sys_call+0x1cc7/0x1ee0 root/klinux/./arch/x86/include/generated/asm/syscalls_64.h:41 do_syscall_x64 root/klinux/arch/x86/entry/common.c:46 [inline] do_syscall_64+0x75/0x2c0 root/klinux/arch/x86/entry/common.c:76 entry_SYSCALL_64_after_hwframe+0x76/0xe0 value changed: 0x000001b2 -> 0x000001b1 Reported by Kernel Concurrency Sanitizer on: CPU: 12 PID: 22653 Comm: syz-executor.12 Not tainted 6.6.140+ #672 Hardware name: QEMU Standard PC (i440FX + PIIX, 1996), BIOS rel-1.16.3-0-ga6ed6b701f0a-prebuilt.qemu.org 04/01/2014 Link: https://lore.kernel.org/20260827070546.1336383-1-hui.zhu@linux.dev Signed-off-by: Hui Zhu Signed-off-by: Andrew Morton Acked-by: Vlastimil Babka (SUSE) Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan --- mm/vmstat.c | 17 +++++++++++------ 1 file changed, 11 insertions(+), 6 deletions(-) diff --git a/mm/vmstat.c b/mm/vmstat.c index cb57714539fb5e..a3e809c57f295a 100644 --- a/mm/vmstat.c +++ b/mm/vmstat.c @@ -1837,6 +1837,11 @@ static void zoneinfo_show_print(struct seq_file *m, pg_data_t *pgdat, struct per_cpu_zonestat __maybe_unused *pzstats; pcp = per_cpu_ptr(zone->per_cpu_pageset, i); + /* + * Access to the per-cpu pageset fields is lockless as they + * are used only for printing purposes. Use data_race to + * avoid KCSAN warning. + */ seq_printf(m, "\n cpu: %i" "\n count: %i" @@ -1845,15 +1850,15 @@ static void zoneinfo_show_print(struct seq_file *m, pg_data_t *pgdat, "\n high_min: %i" "\n high_max: %i", i, - pcp->count, - pcp->high, - pcp->batch, - pcp->high_min, - pcp->high_max); + data_race(pcp->count), + data_race(pcp->high), + data_race(pcp->batch), + data_race(pcp->high_min), + data_race(pcp->high_max)); #ifdef CONFIG_SMP pzstats = per_cpu_ptr(zone->per_cpu_zonestats, i); seq_printf(m, "\n vm stats threshold: %d", - pzstats->stat_threshold); + data_race(pzstats->stat_threshold)); #endif } seq_printf(m, From 1b8bd7817356f02aeea0b3739e27ba67f272d4de Mon Sep 17 00:00:00 2001 From: Hao Li Date: Thu, 27 Aug 2026 15:18:06 +0800 Subject: [PATCH 0542/1352] mm: remove unused anon_vma_trylock_write() The last user of anon_vma_trylock_write() was removed by cc22b9978509, leaving the helper unused. Remove it. Link: https://lore.kernel.org/20260827071845.17636-1-hao.li@linux.dev Signed-off-by: Hao Li Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) Reviewed-by: SJ Park Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/internal.h | 5 ----- 1 file changed, 5 deletions(-) diff --git a/mm/internal.h b/mm/internal.h index 38b1165212c941..da833cafcd599d 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -295,11 +295,6 @@ static inline void anon_vma_lock_write(struct anon_vma *anon_vma) down_write(&anon_vma->root->rwsem); } -static inline int anon_vma_trylock_write(struct anon_vma *anon_vma) -{ - return down_write_trylock(&anon_vma->root->rwsem); -} - static inline void anon_vma_unlock_write(struct anon_vma *anon_vma) { up_write(&anon_vma->root->rwsem); From 8cff41cda6f86da9011d08f69ed23eb093b49905 Mon Sep 17 00:00:00 2001 From: Yue Haibing Date: Thu, 27 Aug 2026 16:27:22 +0800 Subject: [PATCH 0543/1352] mm/swap: remove unused declaration swapcache_clear() Commit c246d236b18b ("mm/shmem: never bypass the swap cache for SWP_SYNCHRONOUS_IO") removed the implementations but leave this. Link: https://lore.kernel.org/20260827082722.1809702-1-yuehaibing@huawei.com Signed-off-by: Yue Haibing Signed-off-by: Andrew Morton Reviewed-by: Baoquan He Acked-by: Kairui Song Acked-by: Nick Huang Reviewed-by: Barry Song Reviewed-by: SJ Park Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham --- mm/swap.h | 1 - 1 file changed, 1 deletion(-) diff --git a/mm/swap.h b/mm/swap.h index 90a551a88df63b..fddba7a87500a4 100644 --- a/mm/swap.h +++ b/mm/swap.h @@ -324,7 +324,6 @@ void __swap_cache_replace_folio(struct swap_cluster_info *ci, struct folio *old, struct folio *new); void show_swap_cache_info(void); -void swapcache_clear(struct swap_info_struct *si, swp_entry_t entry, int nr); struct folio *read_swap_cache_async(struct swap_io_ctx *ctx, swp_entry_t entry, gfp_t gfp_mask, struct vm_area_struct *vma, unsigned long addr); struct folio *swap_cluster_readahead(swp_entry_t entry, gfp_t flag, From 8f6c6ac922f23e59b421a30bcf6fed83276b2016 Mon Sep 17 00:00:00 2001 From: Tao Cui Date: Wed, 26 Aug 2026 10:17:53 +0800 Subject: [PATCH 0544/1352] docs: cgroup: document empty-write behavior of memory limit knobs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A maintenance script on a cluster wrote an unset variable into memory.max of a workload cgroup; the variable expanded to an empty string, the write succeeded, and the workload in the cgroup was OOM-killed. Nothing pointed back at the write, so it took quite some time to trace the OOM kills to that script. The memory controller documentation does not say what an empty write does; the cpuset controller documents its empty-value semantics. The actual behavior is that the empty string is accepted as 0. Reproduced on a k8s cluster (v1.29, cgroup v2, two-container pod, 384M limit): # LIMIT= # echo "$LIMIT" > $CG/memory.max # echo $? 0 m6demo 0/2 OOMKilled 0 Memory cgroup out of memory: Killed process 339529 (sleep) ... anon-rss:32kB State it where the interface files are introduced, alongside the existing notes on units and page rounding. Link: https://lore.kernel.org/all/aoVUlFdZYLFn_gvJ@tiehlicka/ Link: https://lore.kernel.org/20260826021753.197871-1-cui.tao@linux.dev Signed-off-by: Tao Cui Signed-off-by: Andrew Morton Acked-by: Michal Hocko Acked-by: Shakeel Butt Cc: Johannes Weiner Cc: Michal Koutný Cc: Muchun Song Cc: Roman Gushchin Cc: Tejun Heo --- Documentation/admin-guide/cgroup-v2.rst | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/Documentation/admin-guide/cgroup-v2.rst b/Documentation/admin-guide/cgroup-v2.rst index 86a2a0099178ea..8d2603751c51a1 100644 --- a/Documentation/admin-guide/cgroup-v2.rst +++ b/Documentation/admin-guide/cgroup-v2.rst @@ -1321,6 +1321,10 @@ All memory amounts are in bytes. If a value which is not aligned to PAGE_SIZE is written, the value may be rounded up to the closest PAGE_SIZE multiple when read back. +For the limit files described below, an empty or all-whitespace +write is accepted and sets the limit to 0. To disable a limit, +write "max"; to set it to zero explicitly, write "0". + memory.current A read-only single value file which exists on non-root cgroups. From aaeca9aae94cc0b15be1c3ac89b3eb3dfeb5fd12 Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Tue, 25 Aug 2026 13:20:59 +0200 Subject: [PATCH 0545/1352] selftests/mm: khugepaged: remove str_dup() usage We don't check str_dup() return value and never free it. While both things are irrelevant in practice, let's just clean it up by working on argv[0] directly and avoiding the str_dup(). Nobody after us needs these parts of the argv[0] string anyway. This patch is inspired by previous work from Anshuman Tewari [1]. Link: https://lore.kernel.org/r/20260821114416.12255-1-anshumantewari123@gmail.com [1] Link: https://lore.kernel.org/20260825-remove_str_dup-v1-1-0ba2121a820c@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Anshuman Tewari Reviewed-by: Zi Yan Reviewed-by: Lance Yang Reviewed-by: Dev Jain Acked-by: Usama Arif Reviewed-by: Baolin Wang Reviewed-by: Barry Song Reviewed-by: Andrew Morton --- tools/testing/selftests/mm/khugepaged.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 1d2d6bd72fd2af..83d27d069c4139 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -1227,7 +1227,7 @@ static void parse_test_type(int argc, char **argv) return; } - buf = strdup(argv[0]); + buf = argv[0]; token = strsep(&buf, ":"); if (!strcmp(token, "all")) { From 93470f40d0e91aeee56565b5cd223933eeeb2d37 Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Tue, 25 Aug 2026 20:01:53 +0800 Subject: [PATCH 0546/1352] mm/memcontrol: remove unused memcg parameter in calculate_high_delay() The memcg argument of calculate_high_delay() is never referenced in its function body. The delay calculation only depends on nr_pages and max_overage, and both callers have already obtained max_overage from the same memcg. Drop this unused parameter and update the two call sites inside __mem_cgroup_handle_over_high(). Link: https://lore.kernel.org/20260825120153.1405-1-hongfu.li@linux.dev Signed-off-by: Hongfu Li Signed-off-by: Andrew Morton Acked-by: Michal Hocko Reviewed-by: Gregory Price (Meta) Reviewed-by: SJ Park Acked-by: Shakeel Butt --- mm/memcontrol.c | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index d399710e9799ab..1709ac96bbdec5 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -2515,8 +2515,7 @@ static u64 swap_find_max_overage(struct mem_cgroup *memcg) * Get the number of jiffies that we should penalise a mischievous cgroup which * is exceeding its memory.high by checking both it and its ancestors. */ -static unsigned long calculate_high_delay(struct mem_cgroup *memcg, - unsigned int nr_pages, +static unsigned long calculate_high_delay(unsigned int nr_pages, u64 max_overage) { unsigned long penalty_jiffies; @@ -2594,10 +2593,10 @@ void __mem_cgroup_handle_over_high(gfp_t gfp_mask) * memory.high is breached and reclaim is unable to keep up. Throttle * allocators proactively to slow down excessive growth. */ - penalty_jiffies = calculate_high_delay(memcg, nr_pages, + penalty_jiffies = calculate_high_delay(nr_pages, mem_find_max_overage(memcg)); - penalty_jiffies += calculate_high_delay(memcg, nr_pages, + penalty_jiffies += calculate_high_delay(nr_pages, swap_find_max_overage(memcg)); /* From 86e8e2eac969cb6ca45719b84a86f1d74e0404ae Mon Sep 17 00:00:00 2001 From: Qi Xi Date: Tue, 25 Aug 2026 20:05:48 +0800 Subject: [PATCH 0547/1352] mm/page_isolation: fix UBSAN shift-out-of-bounds warning Patch series "mm/page_isolation: fix UBSAN shift-out-of-bounds in isolate_single_pageblock", v3. Patch 1 fixes a UBSAN shift-out-of-bounds warning in the PageBuddy branch of isolate_single_pageblock() triggered by concurrent buddy allocation. Patch 2 addresses the same class of issue in the PageCompound branch, where racy compound_order() and compound_head() reads could lead to out-of-range shifts or incorrect page skipping. This patch (of 2): A contig-range allocation racing with buddy allocation on the adjacent pageblock can trigger: UBSAN: shift-out-of-bounds in mm/page_isolation.c:393:15 shift exponent -749042176 is negative Call trace: isolate_single_pageblock start_isolate_page_range alloc_contig_frozen_range_noprof alloc_contig_range_noprof isolate_single_pageblock() first calls set_migratetype_isolate() with zone->lock held, which marks the pageblock MIGRATE_ISOLATE and moves any free page straddling the boundary out of the way. Once the lock is dropped, it scans the MAX_ORDER_NR_PAGES-aligned window [start_pfn, boundary_pfn) locklessly, only to skip the free pages already handled above and to detect in-use pages straddling the boundary. Since this scan only reads page state to decide how far to skip and returns -EBUSY on a straddling in-use page, it does not take the lock. The window also covers the adjacent pageblock, whose free pages stay on the normal movable/CMA freelist and can be allocated concurrently. So after the scan observes PageBuddy(page), another CPU can allocate the page, leaving a stale value in page->private that makes "1 << order" shift out of range. Use buddy_order_unsafe() with READ_ONCE to read the order, and validate it is within MAX_PAGE_ORDER before shifting to prevent UBSAN warnings. Since pageblock_isolate_and_move_free_pages() already handles free pages straddling boundary_pfn under zone->lock, bail out with -EBUSY instead of VM_WARN_ON_ONCE() when a PageBuddy page appears to cross the boundary during the lockless scan. Link: https://lore.kernel.org/20260825120549.966271-2-xiqi2@huawei.com Fixes: b2c9e2fbba32 ("mm: make alloc_contig_range work at pageblock granularity") Signed-off-by: Qi Xi Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Cc: Johannes Weiner Cc: Kefeng Wang Cc: Michal Hocko Cc: Nanyong Sun Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Brendan Jackman Cc: --- mm/page_isolation.c | 18 ++++++++++++------ 1 file changed, 12 insertions(+), 6 deletions(-) diff --git a/mm/page_isolation.c b/mm/page_isolation.c index e5dfc7bf494460..e4ee998c00f3bc 100644 --- a/mm/page_isolation.c +++ b/mm/page_isolation.c @@ -388,13 +388,19 @@ static int isolate_single_pageblock(unsigned long boundary_pfn, } if (PageBuddy(page)) { - int order = buddy_order(page); + unsigned int order = buddy_order_unsafe(page); - /* pageblock_isolate_and_move_free_pages() handled this */ - VM_WARN_ON_ONCE(pfn + (1 << order) > boundary_pfn); - - pfn += 1UL << order; - continue; + /* buddy_order_unsafe() is racy. Validate the order before shifting. */ + if (order <= MAX_PAGE_ORDER && + /* + * pageblock_isolate_and_move_free_pages() splits + * cross-boundary PageBuddy, verify it. + */ + pfn + (1UL << order) <= boundary_pfn) { + pfn += 1UL << order; + continue; + } + goto failed; } /* From e311dbbb63a742f28f708af82f5333d18b0b0fb5 Mon Sep 17 00:00:00 2001 From: Qi Xi Date: Tue, 25 Aug 2026 20:05:49 +0800 Subject: [PATCH 0548/1352] mm/page_isolation: guard compound_order() against racing The PageCompound branch reads compound_head() without holding a reference. A racing split or free can cause compound_head() to return a stale pointer, and compound_nr() reads the order from that stale head, leading to out-of-range shifts and making the skip distance meaningless. Read the order explicitly with compound_order() and validate it is within MAX_FOLIO_ORDER before shifting. Also verify the derived head_pfn against the legitimate pfn: the head must not be past pfn, must be aligned to nr_pages, and pfn must fall within the compound page. Bail out with -EBUSY if any check fails. Link: https://lore.kernel.org/20260825120549.966271-3-xiqi2@huawei.com Fixes: b2c9e2fbba32 ("mm: make alloc_contig_range work at pageblock granularity") Signed-off-by: Qi Xi Signed-off-by: Andrew Morton Suggested-by: Zi Yan Reviewed-by: Zi Yan Cc: Brendan Jackman Cc: Johannes Weiner Cc: Kefeng Wang Cc: Michal Hocko Cc: Nanyong Sun Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: --- mm/page_isolation.c | 22 ++++++++++++++++++++-- 1 file changed, 20 insertions(+), 2 deletions(-) diff --git a/mm/page_isolation.c b/mm/page_isolation.c index e4ee998c00f3bc..5aaad037384e44 100644 --- a/mm/page_isolation.c +++ b/mm/page_isolation.c @@ -419,10 +419,28 @@ static int isolate_single_pageblock(unsigned long boundary_pfn, if (PageCompound(page)) { struct page *head = compound_head(page); unsigned long head_pfn = page_to_pfn(head); - unsigned long nr_pages = compound_nr(head); + unsigned int order = compound_order(head); + unsigned long nr_pages; + + /* compound_order() is racy. Cap it at MAX_FOLIO_ORDER. */ + if (order > MAX_FOLIO_ORDER) + goto failed; + + nr_pages = 1UL << order; + + /* + * compound_head() is also racy, so the derived head_pfn + * needs additional checks to make sure it is valid. + * Otherwise, just fail the check. pfn comes from + * __first_valid_page() as a legitimate PFN, so use it to + * check head_pfn. + */ + if (head_pfn > pfn || !IS_ALIGNED(head_pfn, nr_pages) || + pfn - head_pfn >= nr_pages) + goto failed; if (head_pfn + nr_pages <= boundary_pfn || - PageHuge(page)) { + PageHuge(head)) { pfn = head_pfn + nr_pages; continue; } From 51d75e67082c4ddb3e405cd1ab7648d835ffd502 Mon Sep 17 00:00:00 2001 From: "Zenghui Yu (Huawei)" Date: Tue, 25 Aug 2026 20:30:23 +0800 Subject: [PATCH 0549/1352] selftests/mm: fix incorrect skip output in pkey_sighandler_tests When pkeys is not supported, ksft_exit_skip() runs with ksft_plan already set, which takes the "ok N # SKIP" branch intended for skipping a single test case. The result is a TAP plan of 5 but only one result line. $ ./pkey_sighandler_tests TAP version 13 1..5 ok 1 # SKIP pkeys not supported # 1 skipped test(s) detected. Consider enabling relevant config options to improve coverage. # Planned tests != run tests (5 != 1) # Totals: pass:0 fail:0 xfail:0 xpass:0 skip:1 error:0 Move ksft_set_plan() after the skip check so ksft_exit_skip() takes the "1..0 # SKIP" branch, the correct TAP representation for skipping an entire test file. $ ./pkey_sighandler_tests TAP version 13 1..0 # SKIP pkeys not supported Link: https://lore.kernel.org/20260825123023.64418-1-zenghui.yu@linux.dev Signed-off-by: Zenghui Yu (Huawei) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/mm/pkey_sighandler_tests.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/tools/testing/selftests/mm/pkey_sighandler_tests.c b/tools/testing/selftests/mm/pkey_sighandler_tests.c index c218d0510a2a42..74bf79a5399dac 100644 --- a/tools/testing/selftests/mm/pkey_sighandler_tests.c +++ b/tools/testing/selftests/mm/pkey_sighandler_tests.c @@ -543,11 +543,12 @@ static void (*pkey_tests[])(void) = { int main(int argc, char *argv[]) { ksft_print_header(); - ksft_set_plan(ARRAY_SIZE(pkey_tests)); if (!is_pkeys_supported()) ksft_exit_skip("pkeys not supported\n"); + ksft_set_plan(ARRAY_SIZE(pkey_tests)); + for (test_nr = 0; test_nr < ARRAY_SIZE(pkey_tests); test_nr++) { tracing_on(); (*pkey_tests[test_nr])(); From 45c3174579b3e01811fe5d944ee709b204c23910 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:40 +0800 Subject: [PATCH 0550/1352] mm/sparse: relax struct mem_section size constraints Patch series "mm: Introduce section-based vmemmap optimization for HugeTLB", v6. This series is split out from the earlier, larger series "mm: Generalize HVO for HugeTLB and device DAX" [1]. While the parent series generalizes vmemmap optimization across HugeTLB and device DAX, this subset addresses a single, self-contained step: making the generic sparse-vmemmap code section-based optimization aware and switching HugeTLB bootmem pages to this path. HugeTLB vmemmap optimization currently has its own early boot setup path. It pre-populates optimized vmemmap mappings before the normal sparse-vmemmap population code runs, and sparsemem carries SPARSEMEM_VMEMMAP_PREINIT only to support that special case. That makes the HugeTLB vmemmap optimization path harder to share with other users of sparse-vmemmap optimization and leaves a fair amount of HugeTLB-specific boot-time state in the generic memory initialization flow. This series introduces section-based vmemmap optimization support in the sparse-vmemmap code and switches HugeTLB bootmem pages over to it. Instead of having HugeTLB pre-populate optimized vmemmap mappings itself, HugeTLB now records the compound page order in the corresponding memory sections. The generic sparse-vmemmap population path can then allocate or reuse shared tail vmemmap pages based on section metadata. The patches are organized as follows: - patches 1-2 prepare sparsemem and vmemmap optimization metadata - patches 3-8 teach the common sparse-vmemmap paths to use that state - patches 9-10 switch HugeTLB bootmem optimization to the section-based path - patches 11-17 clean up sparsemem and HugeTLB bootmem code that is no longer needed after the conversion This is intended to be the second smaller step toward the broader HVO generalization. The device DAX conversion and the wider HVO consolidation are left for follow-up series. This patch (of 17): struct mem_section is currently forced to a power-of-2 size so the section-to-root lookup can use a mask instead of a modulo. That requirement makes future extensions harder than necessary: adding a small field can require configuration-dependent padding or layout checks just to preserve the lookup scheme. Keep the lookup correct for any struct mem_section size by using a plain modulo instead. Do not leave the layout entirely unconstrained, though. Keep struct mem_section double-word aligned so modest size changes, such as adding another word-sized field on 64-bit systems, still keep a compact and efficient layout. If future fields grow the structure beyond that sweet spot, the lookup remains correct; only the exact layout efficiency changes. Link: https://lore.kernel.org/20260910063256.64386-2-songmuchun@bytedance.com Link: https://lore.kernel.org/all/20260513130542.35604-1-songmuchun@bytedance.com/ [1] Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Qi Zheng --- include/linux/mmzone.h | 11 +++-------- mm/sparse.c | 2 -- scripts/gdb/linux/mm.py | 6 ++---- 3 files changed, 5 insertions(+), 14 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 94f9c3ff541604..0a2428714108c4 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -2027,13 +2027,9 @@ struct mem_section { * section. (see page_ext.h about this.) */ struct page_ext *page_ext; - unsigned long pad; #endif - /* - * WARNING: mem_section must be a power-of-2 in size for the - * calculation and use of SECTION_ROOT_MASK to make sense. - */ -}; +/* Sacrifice minor padding space for efficient lookup. */ +} __aligned(2 * sizeof(unsigned long)); #ifdef CONFIG_SPARSEMEM_EXTREME #define SECTIONS_PER_ROOT (PAGE_SIZE / sizeof (struct mem_section)) @@ -2043,7 +2039,6 @@ struct mem_section { #define SECTION_NR_TO_ROOT(sec) ((sec) / SECTIONS_PER_ROOT) #define NR_SECTION_ROOTS DIV_ROUND_UP(NR_MEM_SECTIONS, SECTIONS_PER_ROOT) -#define SECTION_ROOT_MASK (SECTIONS_PER_ROOT - 1) #ifdef CONFIG_SPARSEMEM_EXTREME extern struct mem_section **mem_section; @@ -2067,7 +2062,7 @@ static inline struct mem_section *__nr_to_section(unsigned long nr) if (!mem_section || !mem_section[root]) return NULL; #endif - return &mem_section[root][nr & SECTION_ROOT_MASK]; + return &mem_section[root][nr % SECTIONS_PER_ROOT]; } /* diff --git a/mm/sparse.c b/mm/sparse.c index 7c15406e77f5d2..c84b4c7b8c7069 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -322,8 +322,6 @@ void __init sparse_init(void) unsigned long pnum_end, pnum_begin, map_count = 1; int nid_begin; - /* see include/linux/mmzone.h 'struct mem_section' definition */ - BUILD_BUG_ON(!is_power_of_2(sizeof(struct mem_section))); memblocks_present(); if (compound_info_has_mask()) { diff --git a/scripts/gdb/linux/mm.py b/scripts/gdb/linux/mm.py index 28d33624c38bd6..193a88d763abf7 100644 --- a/scripts/gdb/linux/mm.py +++ b/scripts/gdb/linux/mm.py @@ -70,7 +70,6 @@ def __init__(self): self.SECTIONS_PER_ROOT = 1 self.NR_SECTION_ROOTS = DIV_ROUND_UP(self.NR_MEM_SECTIONS, self.SECTIONS_PER_ROOT) - self.SECTION_ROOT_MASK = self.SECTIONS_PER_ROOT - 1 try: self.SECTION_HAS_MEM_MAP = 1 << int(gdb.parse_and_eval('SECTION_HAS_MEM_MAP_BIT')) @@ -100,7 +99,7 @@ def SECTION_NR_TO_ROOT(self, sec): def __nr_to_section(self, nr): root = self.SECTION_NR_TO_ROOT(nr) mem_section = gdb.parse_and_eval("mem_section") - return mem_section[root][nr & self.SECTION_ROOT_MASK] + return mem_section[root][nr % self.SECTIONS_PER_ROOT] def pfn_to_section_nr(self, pfn): return pfn >> self.PFN_SECTION_SHIFT @@ -249,7 +248,6 @@ def __init__(self): self.SECTIONS_PER_ROOT = 1 self.NR_SECTION_ROOTS = DIV_ROUND_UP(self.NR_MEM_SECTIONS, self.SECTIONS_PER_ROOT) - self.SECTION_ROOT_MASK = self.SECTIONS_PER_ROOT - 1 self.SUBSECTION_SHIFT = 21 self.SEBSECTION_SIZE = 1 << self.SUBSECTION_SHIFT self.PFN_SUBSECTION_SHIFT = self.SUBSECTION_SHIFT - self.PAGE_SHIFT @@ -304,7 +302,7 @@ def SECTION_NR_TO_ROOT(self, sec): def __nr_to_section(self, nr): root = self.SECTION_NR_TO_ROOT(nr) mem_section = gdb.parse_and_eval("mem_section") - return mem_section[root][nr & self.SECTION_ROOT_MASK] + return mem_section[root][nr % self.SECTIONS_PER_ROOT] def pfn_to_section_nr(self, pfn): return pfn >> self.PFN_SECTION_SHIFT From 60aecbfa7e45460bfa84773373aac20aec6fddd6 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:41 +0800 Subject: [PATCH 0551/1352] mm/sparse-vmemmap: rename HVO order macros The macros VMEMMAP_TAIL_MIN_ORDER and NR_VMEMMAP_TAILS describe the order range where HVO can be applied, but their names tie that range to the tail-page cache implementation. Rename them with a VMEMMAP_OPTIMIZATION prefix and use the new names in the HVO paths. This makes the code describe the optimization requirements rather than the tail-page cache implementation detail. No functional change intended. Link: https://lore.kernel.org/20260910063256.64386-3-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/mmzone.h | 17 +++++++++-------- mm/hugetlb.c | 4 ++-- mm/hugetlb_vmemmap.c | 2 +- mm/sparse-vmemmap.c | 4 ++-- 4 files changed, 14 insertions(+), 13 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 0a2428714108c4..5fb9b37819d550 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -107,13 +107,14 @@ is_power_of_2(sizeof(struct page)) ? \ MAX_FOLIO_NR_PAGES * sizeof(struct page) : 0) -/* - * vmemmap optimization (like HVO) is only possible for page orders that fill - * two or more pages with struct pages. - */ -#define VMEMMAP_TAIL_MIN_ORDER (ilog2(2 * PAGE_SIZE / sizeof(struct page))) -#define __NR_VMEMMAP_TAILS (MAX_FOLIO_ORDER - VMEMMAP_TAIL_MIN_ORDER + 1) -#define NR_VMEMMAP_TAILS (__NR_VMEMMAP_TAILS > 0 ? __NR_VMEMMAP_TAILS : 0) +/* The number of struct pages covered by the retained vmemmap pages with HVO enabled. */ +#define VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES (PAGE_SIZE / sizeof(struct page)) +#define VMEMMAP_OPTIMIZATION_MIN_ORDER (ilog2(VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES) + 1) + +#define __VMEMMAP_OPTIMIZATION_NR_ORDERS \ + (MAX_FOLIO_ORDER - VMEMMAP_OPTIMIZATION_MIN_ORDER + 1) +#define VMEMMAP_OPTIMIZATION_NR_ORDERS \ + (__VMEMMAP_OPTIMIZATION_NR_ORDERS > 0 ? __VMEMMAP_OPTIMIZATION_NR_ORDERS : 0) enum migratetype { MIGRATE_UNMOVABLE, @@ -1158,7 +1159,7 @@ struct zone { atomic_long_t vm_stat[NR_VM_ZONE_STAT_ITEMS]; atomic_long_t vm_numa_event[NR_VM_NUMA_EVENT_ITEMS]; #ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP - struct page *vmemmap_tails[NR_VMEMMAP_TAILS]; + struct page *vmemmap_tails[VMEMMAP_OPTIMIZATION_NR_ORDERS]; #endif } ____cacheline_internodealigned_in_smp; diff --git a/mm/hugetlb.c b/mm/hugetlb.c index d73108393e4b76..f1c92e80c34672 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -3368,7 +3368,7 @@ void __init hugetlb_bootmem_struct_page_init(void) struct zone *zone; for_each_zone(zone) { - for (int i = 0; i < NR_VMEMMAP_TAILS; i++) { + for (int i = 0; i < VMEMMAP_OPTIMIZATION_NR_ORDERS; i++) { struct page *tail, *p; unsigned int order; @@ -3376,7 +3376,7 @@ void __init hugetlb_bootmem_struct_page_init(void) if (!tail) continue; - order = i + VMEMMAP_TAIL_MIN_ORDER; + order = i + VMEMMAP_OPTIMIZATION_MIN_ORDER; p = page_to_virt(tail); /* * prep_and_add_bootmem_folios() can access pageblock diff --git a/mm/hugetlb_vmemmap.c b/mm/hugetlb_vmemmap.c index 917db0984143c2..ae8fdaa4211891 100644 --- a/mm/hugetlb_vmemmap.c +++ b/mm/hugetlb_vmemmap.c @@ -494,7 +494,7 @@ static bool vmemmap_should_optimize_folio(const struct hstate *h, struct folio * static struct page *vmemmap_get_tail(unsigned int order, struct zone *zone) { - const unsigned int idx = order - VMEMMAP_TAIL_MIN_ORDER; + const unsigned int idx = order - VMEMMAP_OPTIMIZATION_MIN_ORDER; struct page *tail, *p; int node = zone_to_nid(zone); diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 5a2469fb1838c7..aa6a4a2fae9886 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -329,12 +329,12 @@ static __meminit struct page *vmemmap_get_tail(unsigned int order, struct zone * unsigned int idx; int node = zone_to_nid(zone); - if (WARN_ON_ONCE(order < VMEMMAP_TAIL_MIN_ORDER)) + if (WARN_ON_ONCE(order < VMEMMAP_OPTIMIZATION_MIN_ORDER)) return NULL; if (WARN_ON_ONCE(order > MAX_FOLIO_ORDER)) return NULL; - idx = order - VMEMMAP_TAIL_MIN_ORDER; + idx = order - VMEMMAP_OPTIMIZATION_MIN_ORDER; tail = zone->vmemmap_tails[idx]; if (tail) return tail; From 28f74e96bdf9a2845dc0228e46191364c7105a53 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:42 +0800 Subject: [PATCH 0552/1352] mm/mm_init: skip initializing shared vmemmap tail pages memmap_init_range() initializes every struct page in the target range. For compound pages with vmemmap optimization, the tail struct pages are backed by a shared vmemmap page. Initializing those tail struct pages would overwrite the shared vmemmap page contents, requiring users such as HugeTLB to restore the metadata afterwards. Track the compound page order for HVO-backed sections and use that metadata to detect struct pages that fall into the shared tail vmemmap range. Skip those shared tail pages in memmap_init_range(), then initialize pageblock migratetypes for the processed range with a helper after the per-page initialization loop. Keep direct mem_section access inside sparse helpers. Expose pfn_to_section_compound_order() for callers that only need the order associated with a PFN. This lets memmap_init_range() skip shared tail vmemmap pages without exposing __pfn_to_section() to !SPARSEMEM builds. This is a preparatory change for consolidating handling across users of vmemmap optimization, and it also avoids redundant initialization of shared tail vmemmap pages during early boot. That early-boot benefit appears only once HugeTLB is switched to this common handling, since HugeTLB is the early-boot user that creates those shared tail vmemmap pages. Link: https://lore.kernel.org/20260910063256.64386-4-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Reviewed-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/mmzone.h | 8 ++++++++ mm/mm_init.c | 40 +++++++++++++++++++++++----------------- mm/sparse.h | 33 +++++++++++++++++++++++++++++++++ 3 files changed, 64 insertions(+), 17 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 5fb9b37819d550..738bf3d272563b 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -2022,6 +2022,14 @@ struct mem_section { unsigned long section_mem_map; struct mem_section_usage *usage; +#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP + /* + * Normally, sections hold regular (order-0) pages. However, for + * sections with HVO enabled, this tracks the compound page order + * to enable deduplication of redundant vmemmap pages. + */ + unsigned int compound_page_order; +#endif #ifdef CONFIG_PAGE_EXTENSION /* * If SPARSEMEM, pgdat doesn't have page_ext pointer. We use diff --git a/mm/mm_init.c b/mm/mm_init.c index 1533aebafb6885..6655fe696e0f01 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -29,6 +29,7 @@ #include #include #include +#include #include #include #include @@ -677,21 +678,19 @@ static inline void fixup_hashdist(void) static inline void fixup_hashdist(void) {} #endif /* CONFIG_NUMA */ -#if defined(CONFIG_ZONE_DEVICE) || defined(CONFIG_DEFERRED_STRUCT_PAGE_INIT) static __meminit void pageblock_migratetype_init_range(unsigned long pfn, - unsigned long nr_pages, int migratetype, bool atomic) + unsigned long nr_pages, int migratetype, bool isolate, bool atomic) { const unsigned long end = pfn + nr_pages; for (pfn = pageblock_align(pfn); pfn < end; pfn += pageblock_nr_pages) { enum migratetype mt = kho_scratch_migratetype(pfn, migratetype); - init_pageblock_migratetype(pfn_to_page(pfn), mt, false); - if (!atomic && IS_ALIGNED(pfn, PAGES_PER_SECTION)) + init_pageblock_migratetype(pfn_to_page(pfn), mt, isolate); + if (!atomic && IS_ALIGNED(pfn, PFN_DOWN(SZ_1G))) cond_resched(); } } -#endif #ifdef CONFIG_DEFERRED_STRUCT_PAGE_INIT static inline void pgdat_set_deferred_range(pg_data_t *pgdat) @@ -886,6 +885,17 @@ void __meminit memmap_init_range(unsigned long size, int nid, unsigned long zone } } + /* + * Vmemmap-optimizable PFNs are backed by shared tail struct pages, + * which have already been initialized during vmemmap population. + */ + if (vmemmap_optimizable_pfn(pfn)) { + const unsigned int order = pfn_to_section_compound_order(pfn); + + pfn = min(ALIGN(pfn, 1UL << order), end_pfn); + continue; + } + page = pfn_to_page(pfn); __init_single_page(page, pfn, zone, nid); if (context == MEMINIT_HOTPLUG) { @@ -897,19 +907,13 @@ void __meminit memmap_init_range(unsigned long size, int nid, unsigned long zone __SetPageOffline(page); } - /* - * Usually, we want to mark the pageblock MIGRATE_MOVABLE, - * such that unmovable allocations won't be scattered all - * over the place during system boot. - */ - if (pageblock_aligned(pfn)) { - enum migratetype mt = kho_scratch_migratetype(pfn, migratetype); - - init_pageblock_migratetype(page, mt, isolate_pageblock); + if (pageblock_aligned(pfn)) cond_resched(); - } pfn++; } + + pageblock_migratetype_init_range(start_pfn, pfn - start_pfn, migratetype, + isolate_pageblock, /* atomic */ false); } static void __init memmap_init_zone_range(struct zone *zone, @@ -1112,7 +1116,8 @@ void __ref memmap_init_zone_device(struct zone *zone, compound_nr_pages(pfn, altmap, pgmap)); } - pageblock_migratetype_init_range(start_pfn, nr_pages, MIGRATE_MOVABLE, false); + pageblock_migratetype_init_range(start_pfn, nr_pages, MIGRATE_MOVABLE, + /* isolate */ false, /* atomic */ false); pr_debug("%s initialised %lu pages in %ums\n", __func__, nr_pages, jiffies_to_msecs(jiffies - start)); @@ -1921,7 +1926,8 @@ static void __init deferred_free_pages(unsigned long pfn, if (!nr_pages) return; - pageblock_migratetype_init_range(pfn, nr_pages, MIGRATE_MOVABLE, true); + pageblock_migratetype_init_range(pfn, nr_pages, MIGRATE_MOVABLE, + /* isolate */ false, /* atomic */ true); page = pfn_to_page(pfn); diff --git a/mm/sparse.h b/mm/sparse.h index 3b744667a7e624..da79c83adeae11 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -10,6 +10,39 @@ #include +#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP +static inline unsigned int section_compound_order(const struct mem_section *section) +{ + return section->compound_page_order; +} + +static inline unsigned int pfn_to_section_compound_order(unsigned long pfn) +{ + return section_compound_order(__pfn_to_section(pfn)); +} +#else +static inline unsigned int section_compound_order(const struct mem_section *section) +{ + return 0; +} + +static inline unsigned int pfn_to_section_compound_order(unsigned long pfn) +{ + return 0; +} +#endif + +static inline bool vmemmap_optimizable_pfn(unsigned long pfn) +{ + const unsigned int order = pfn_to_section_compound_order(pfn); + const unsigned long nr_pages = 1UL << order; + + if (!is_power_of_2(sizeof(struct page))) + return false; + + return (pfn & (nr_pages - 1)) >= VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES; +} + /* * mm/sparse.c */ From 98c6e837213512aa28c0bd8864f21bca4cc3202f Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:43 +0800 Subject: [PATCH 0553/1352] mm/sparse-vmemmap: initialize shared tail vmemmap pages on allocation The shared tail vmemmap page allocated in vmemmap_get_tail() used to be left uninitialized, because memmap_init_range() would later overwrite it. That forced users such as HugeTLB to defer the initialization to their own setup paths. Now that memmap_init_range() skips shared tail vmemmap pages, initialize them immediately in vmemmap_get_tail() with init_compound_tail() instead. This adds initialization at the point where the shared tail page is allocated. The existing HugeTLB initialization remains necessary until the compound page order is set and memmap_init_range() starts skipping shared tails. It is removed when HugeTLB switches to the section-based mechanism. Link: https://lore.kernel.org/20260910063256.64386-5-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/sparse-vmemmap.c | 12 ++---------- 1 file changed, 2 insertions(+), 10 deletions(-) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index aa6a4a2fae9886..107215cf8488ce 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -338,19 +338,11 @@ static __meminit struct page *vmemmap_get_tail(unsigned int order, struct zone * tail = zone->vmemmap_tails[idx]; if (tail) return tail; - - /* - * Only allocate the page, but do not initialize it. - * - * Any initialization done here will be overwritten by memmap_init(). - * - * hugetlb_bootmem_struct_page_init() will take care of initialization - * after memmap_init(). - */ - p = vmemmap_alloc_block_zero(PAGE_SIZE, node); if (!p) return NULL; + for (int i = 0; i < PAGE_SIZE / sizeof(struct page); i++) + init_compound_tail(p + i, NULL, order, zone); tail = virt_to_page(p); zone->vmemmap_tails[idx] = tail; From a9d7d1b55545b83146615a88d8cbb93d08ddc6fe Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:44 +0800 Subject: [PATCH 0554/1352] mm/sparse-vmemmap: support section-based vmemmap accounting section_nr_vmemmap_pages() can account ordinary sections and DAX sections, but section-based vmemmap optimization stores the compound page order in struct mem_section and retains a different number of vmemmap pages. Teach section_nr_vmemmap_pages() to recognize section-based optimized sections and calculate their vmemmap page count from the stored compound page order and the HVO retained page count. Link: https://lore.kernel.org/20260910063256.64386-6-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/mmzone.h | 6 ++++-- mm/sparse-vmemmap.c | 10 ++++++---- mm/sparse.h | 16 ++++++++++++++++ 3 files changed, 26 insertions(+), 6 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 738bf3d272563b..2c4f63e379da7f 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -107,8 +107,10 @@ is_power_of_2(sizeof(struct page)) ? \ MAX_FOLIO_NR_PAGES * sizeof(struct page) : 0) -/* The number of struct pages covered by the retained vmemmap pages with HVO enabled. */ -#define VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES (PAGE_SIZE / sizeof(struct page)) +/* The number of retained vmemmap pages with HVO enabled. */ +#define VMEMMAP_OPTIMIZATION_PAGES 1 +#define VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES \ + (VMEMMAP_OPTIMIZATION_PAGES * PAGE_SIZE / sizeof(struct page)) #define VMEMMAP_OPTIMIZATION_MIN_ORDER (ilog2(VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES) + 1) #define __VMEMMAP_OPTIMIZATION_NR_ORDERS \ diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 107215cf8488ce..dea7179fbc90b0 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -649,24 +649,26 @@ void offline_mem_sections(unsigned long start_pfn, unsigned long end_pfn) static int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, struct vmem_altmap *altmap, struct dev_pagemap *pgmap) { - const unsigned int order = pgmap ? pgmap->vmemmap_shift : 0; + const struct mem_section *ms = __pfn_to_section(pfn); + const int order = pgmap ? pgmap->vmemmap_shift : section_compound_order(ms); + const int vmemmap_pages = pgmap ? VMEMMAP_RESERVE_NR : VMEMMAP_OPTIMIZATION_PAGES; const unsigned long pages_per_compound = 1UL << order; VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SUBSECTION)); VM_WARN_ON_ONCE(nr_pages > PAGES_PER_SECTION); - if (!vmemmap_can_optimize(altmap, pgmap)) + if (!vmemmap_can_optimize(altmap, pgmap) && !section_vmemmap_optimizable(ms)) return DIV_ROUND_UP(nr_pages * sizeof(struct page), PAGE_SIZE); if (order < PFN_SECTION_SHIFT) { VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, pages_per_compound)); - return VMEMMAP_RESERVE_NR * nr_pages / pages_per_compound; + return vmemmap_pages * nr_pages / pages_per_compound; } VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SECTION)); if (IS_ALIGNED(pfn, pages_per_compound)) - return VMEMMAP_RESERVE_NR; + return vmemmap_pages; return 0; } diff --git a/mm/sparse.h b/mm/sparse.h index da79c83adeae11..563fc1f4d71778 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -43,6 +43,17 @@ static inline bool vmemmap_optimizable_pfn(unsigned long pfn) return (pfn & (nr_pages - 1)) >= VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES; } +static inline bool vmemmap_optimizable_order(unsigned int order) +{ + if (!IS_ENABLED(CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP)) + return false; + + if (!is_power_of_2(sizeof(struct page))) + return false; + + return order >= VMEMMAP_OPTIMIZATION_MIN_ORDER; +} + /* * mm/sparse.c */ @@ -86,6 +97,11 @@ static inline size_t mem_section_usage_size(void) return struct_size_t(struct mem_section_usage, pageblock_flags, BITS_TO_LONGS(SECTION_BLOCKFLAGS_BITS)); } + +static inline bool section_vmemmap_optimizable(const struct mem_section *ms) +{ + return vmemmap_optimizable_order(section_compound_order(ms)); +} #else static inline void sparse_init(void) {} #endif /* CONFIG_SPARSEMEM */ From d3659eb1489a6dfc5b71b95385e4e61d13151baf Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:45 +0800 Subject: [PATCH 0555/1352] mm/mm_init: factor out pfn_to_zone() pfn_to_zone() in hugetlb_vmemmap.c duplicates the zone lookup logic in __init_deferred_page(). Move it to mm_init.c, declare it in mm/mm_init.h, and reuse it from __init_deferred_page() and HugeTLB early vmemmap initialization instead of open-coding the zone walk there. Link: https://lore.kernel.org/20260910063256.64386-7-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/hugetlb_vmemmap.c | 17 ++--------------- mm/mm_init.c | 28 ++++++++++++++++++---------- mm/mm_init.h | 1 + 3 files changed, 21 insertions(+), 25 deletions(-) diff --git a/mm/hugetlb_vmemmap.c b/mm/hugetlb_vmemmap.c index ae8fdaa4211891..c48fcea076a522 100644 --- a/mm/hugetlb_vmemmap.c +++ b/mm/hugetlb_vmemmap.c @@ -19,6 +19,7 @@ #include #include "hugetlb_vmemmap.h" #include "internal.h" +#include "mm_init.h" /** * struct vmemmap_remap_walk - walk vmemmap page table @@ -744,20 +745,6 @@ static bool vmemmap_should_optimize_bootmem_page(struct huge_bootmem_page *m) return true; } -static struct zone *pfn_to_zone(unsigned nid, unsigned long pfn) -{ - struct zone *zone; - enum zone_type zone_type; - - for (zone_type = 0; zone_type < MAX_NR_ZONES; zone_type++) { - zone = &NODE_DATA(nid)->node_zones[zone_type]; - if (zone_spans_pfn(zone, pfn)) - return zone; - } - - return NULL; -} - /* * Initialize memmap section for a gigantic page, HVO-style. */ @@ -787,7 +774,7 @@ void __init hugetlb_vmemmap_init_early(int nid) map = pfn_to_page(pfn); start = (unsigned long)map; end = start + hugetlb_vmemmap_size(m->hstate); - zone = pfn_to_zone(nid, pfn); + zone = pfn_to_zone(pfn, nid); if (vmemmap_populate_hvo(start, end, huge_page_order(m->hstate), zone, HUGETLB_VMEMMAP_RESERVE_SIZE)) diff --git a/mm/mm_init.c b/mm/mm_init.c index 6655fe696e0f01..ee3bd792176890 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -692,6 +692,20 @@ static __meminit void pageblock_migratetype_init_range(unsigned long pfn, } } +struct zone __meminit *pfn_to_zone(unsigned long pfn, int nid) +{ + pg_data_t *pgdat = NODE_DATA(nid); + + for (enum zone_type zone_type = 0; zone_type < MAX_NR_ZONES; zone_type++) { + struct zone *zone = &pgdat->node_zones[zone_type]; + + if (zone_spans_pfn(zone, pfn)) + return zone; + } + + return NULL; +} + #ifdef CONFIG_DEFERRED_STRUCT_PAGE_INIT static inline void pgdat_set_deferred_range(pg_data_t *pgdat) { @@ -750,20 +764,14 @@ defer_init(int nid, unsigned long pfn, unsigned long end_pfn) static void __meminit __init_deferred_page(unsigned long pfn, int nid) { - pg_data_t *pgdat = NODE_DATA(nid); - int zid; + struct zone *zone; if (early_page_initialised(pfn, nid)) return; - for (zid = 0; zid < MAX_NR_ZONES; zid++) { - struct zone *zone = &pgdat->node_zones[zid]; - - if (zone_spans_pfn(zone, pfn)) - break; - } - __init_single_page(pfn_to_page(pfn), pfn, zid, nid); - + zone = pfn_to_zone(pfn, nid); + __init_single_page(pfn_to_page(pfn), pfn, + zone ? zone_idx(zone) : MAX_NR_ZONES, nid); if (pageblock_aligned(pfn)) { enum migratetype mt = kho_scratch_migratetype(pfn, MIGRATE_MOVABLE); diff --git a/mm/mm_init.h b/mm/mm_init.h index 39f75df9be1c26..c9fc35e7e9f1f3 100644 --- a/mm/mm_init.h +++ b/mm/mm_init.h @@ -39,6 +39,7 @@ void memmap_init_range(unsigned long size, int nid, unsigned long zone, enum meminit_context context, struct vmem_altmap *altmap, int migratetype, bool isolate_pageblock); +struct zone *pfn_to_zone(unsigned long pfn, int nid); #if defined CONFIG_COMPACTION || defined CONFIG_CMA /* Free whole pageblock and set its migration type to MIGRATE_CMA. */ From 63510c6bb3974cd970e9569a92e132ce95591092 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:46 +0800 Subject: [PATCH 0556/1352] mm/sparse-vmemmap: move helpers ahead of future callers Prepare for section-based vmemmap optimization by moving helpers that follow-up changes will use. vmemmap_get_tail() will be called from the PTE population path. section_nr_vmemmap_pages() will be made visible outside the memory hotplug code and called from sparse_init_nid(). Move vmemmap_alloc_block_zero() together with vmemmap_get_tail(), since the tail helper depends on it. Move section_nr_vmemmap_pages() earlier into its own CONFIG_MEMORY_HOTPLUG block. That lets the later patch change its visibility and callers without also moving the function body. No functional change is intended. Link: https://lore.kernel.org/20260910063256.64386-8-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/sparse-vmemmap.c | 134 +++++++++++++++++++++++--------------------- 1 file changed, 69 insertions(+), 65 deletions(-) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index dea7179fbc90b0..e2a2a5eab4dc43 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -148,6 +148,75 @@ void __meminit vmemmap_verify(pte_t *pte, int node, start, end - 1); } +#ifdef CONFIG_MEMORY_HOTPLUG +static int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, + struct vmem_altmap *altmap, struct dev_pagemap *pgmap) +{ + const struct mem_section *ms = __pfn_to_section(pfn); + const int order = pgmap ? pgmap->vmemmap_shift : section_compound_order(ms); + const int vmemmap_pages = pgmap ? VMEMMAP_RESERVE_NR : VMEMMAP_OPTIMIZATION_PAGES; + const unsigned long pages_per_compound = 1UL << order; + + VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SUBSECTION)); + VM_WARN_ON_ONCE(nr_pages > PAGES_PER_SECTION); + + if (!vmemmap_can_optimize(altmap, pgmap) && !section_vmemmap_optimizable(ms)) + return DIV_ROUND_UP(nr_pages * sizeof(struct page), PAGE_SIZE); + + if (order < PFN_SECTION_SHIFT) { + VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, pages_per_compound)); + return vmemmap_pages * nr_pages / pages_per_compound; + } + + VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SECTION)); + + if (IS_ALIGNED(pfn, pages_per_compound)) + return vmemmap_pages; + + return 0; +} +#endif + +static void * __meminit vmemmap_alloc_block_zero(unsigned long size, int node) +{ + void *p = vmemmap_alloc_block(size, node); + + if (!p) + return NULL; + memset(p, 0, size); + + return p; +} + +#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP +static __meminit struct page *vmemmap_get_tail(unsigned int order, struct zone *zone) +{ + struct page *p, *tail; + unsigned int idx; + int node = zone_to_nid(zone); + + if (WARN_ON_ONCE(order < VMEMMAP_OPTIMIZATION_MIN_ORDER)) + return NULL; + if (WARN_ON_ONCE(order > MAX_FOLIO_ORDER)) + return NULL; + + idx = order - VMEMMAP_OPTIMIZATION_MIN_ORDER; + tail = zone->vmemmap_tails[idx]; + if (tail) + return tail; + p = vmemmap_alloc_block_zero(PAGE_SIZE, node); + if (!p) + return NULL; + for (int i = 0; i < PAGE_SIZE / sizeof(struct page); i++) + init_compound_tail(p + i, NULL, order, zone); + + tail = virt_to_page(p); + zone->vmemmap_tails[idx] = tail; + + return tail; +} +#endif + static pte_t * __meminit vmemmap_pte_populate(pmd_t *pmd, unsigned long addr, int node, struct vmem_altmap *altmap, unsigned long ptpfn, unsigned long flags) @@ -181,17 +250,6 @@ static pte_t * __meminit vmemmap_pte_populate(pmd_t *pmd, unsigned long addr, in return pte; } -static void * __meminit vmemmap_alloc_block_zero(unsigned long size, int node) -{ - void *p = vmemmap_alloc_block(size, node); - - if (!p) - return NULL; - memset(p, 0, size); - - return p; -} - static pmd_t * __meminit vmemmap_pmd_populate(pud_t *pud, unsigned long addr, int node) { pmd_t *pmd = pmd_offset(pud, addr); @@ -323,33 +381,6 @@ void vmemmap_wrprotect_hvo(unsigned long addr, unsigned long end, } #ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP -static __meminit struct page *vmemmap_get_tail(unsigned int order, struct zone *zone) -{ - struct page *p, *tail; - unsigned int idx; - int node = zone_to_nid(zone); - - if (WARN_ON_ONCE(order < VMEMMAP_OPTIMIZATION_MIN_ORDER)) - return NULL; - if (WARN_ON_ONCE(order > MAX_FOLIO_ORDER)) - return NULL; - - idx = order - VMEMMAP_OPTIMIZATION_MIN_ORDER; - tail = zone->vmemmap_tails[idx]; - if (tail) - return tail; - p = vmemmap_alloc_block_zero(PAGE_SIZE, node); - if (!p) - return NULL; - for (int i = 0; i < PAGE_SIZE / sizeof(struct page); i++) - init_compound_tail(p + i, NULL, order, zone); - - tail = virt_to_page(p); - zone->vmemmap_tails[idx] = tail; - - return tail; -} - int __meminit vmemmap_populate_hvo(unsigned long addr, unsigned long end, unsigned int order, struct zone *zone, unsigned long headsize) @@ -646,33 +677,6 @@ void offline_mem_sections(unsigned long start_pfn, unsigned long end_pfn) } } -static int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, - struct vmem_altmap *altmap, struct dev_pagemap *pgmap) -{ - const struct mem_section *ms = __pfn_to_section(pfn); - const int order = pgmap ? pgmap->vmemmap_shift : section_compound_order(ms); - const int vmemmap_pages = pgmap ? VMEMMAP_RESERVE_NR : VMEMMAP_OPTIMIZATION_PAGES; - const unsigned long pages_per_compound = 1UL << order; - - VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SUBSECTION)); - VM_WARN_ON_ONCE(nr_pages > PAGES_PER_SECTION); - - if (!vmemmap_can_optimize(altmap, pgmap) && !section_vmemmap_optimizable(ms)) - return DIV_ROUND_UP(nr_pages * sizeof(struct page), PAGE_SIZE); - - if (order < PFN_SECTION_SHIFT) { - VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, pages_per_compound)); - return vmemmap_pages * nr_pages / pages_per_compound; - } - - VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SECTION)); - - if (IS_ALIGNED(pfn, pages_per_compound)) - return vmemmap_pages; - - return 0; -} - static struct page * __meminit populate_section_memmap(unsigned long pfn, unsigned long nr_pages, int nid, struct vmem_altmap *altmap, struct dev_pagemap *pgmap) From 08e0dd3cf718b797a6d5a943207c1289ad7020df Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:47 +0800 Subject: [PATCH 0557/1352] mm/sparse-vmemmap: support section-based vmemmap optimization Teach sparse-vmemmap population code to use the compound page order when deciding whether a vmemmap page can be optimized. With this information, the common sparse-vmemmap population path can allocate or reuse shared tail vmemmap pages directly instead of relying on HugeTLB-specific handling. This centralizes vmemmap optimization logic in the sparse-vmemmap code, based on section metadata, and prepares for sharing the same mechanism across different users of vmemmap optimization, including HugeTLB and DAX. Link: https://lore.kernel.org/20260910063256.64386-9-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/sparse-vmemmap.c | 46 +++++++++++++++++++++++++++++++++++++-------- mm/sparse.c | 4 ++-- mm/sparse.h | 7 +++++++ 3 files changed, 47 insertions(+), 10 deletions(-) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index e2a2a5eab4dc43..faebf344cdac9b 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -148,8 +148,7 @@ void __meminit vmemmap_verify(pte_t *pte, int node, start, end - 1); } -#ifdef CONFIG_MEMORY_HOTPLUG -static int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, +int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, struct vmem_altmap *altmap, struct dev_pagemap *pgmap) { const struct mem_section *ms = __pfn_to_section(pfn); @@ -175,7 +174,6 @@ static int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long n return 0; } -#endif static void * __meminit vmemmap_alloc_block_zero(unsigned long size, int node) { @@ -215,19 +213,44 @@ static __meminit struct page *vmemmap_get_tail(unsigned int order, struct zone * return tail; } +#else +static inline struct page *vmemmap_get_tail(unsigned int order, struct zone *zone) +{ + return NULL; +} #endif +static __meminit void *vmemmap_alloc_pte(unsigned long pfn, int node, + struct vmem_altmap *altmap) +{ + struct zone *zone; + struct page *page; + const unsigned int order = pfn_to_section_compound_order(pfn); + + if (!vmemmap_optimizable_pfn(pfn)) + return vmemmap_alloc_block_buf(PAGE_SIZE, node, altmap); + + zone = pfn_to_zone(pfn, node); + page = vmemmap_get_tail(order, zone); + if (!page) + return NULL; + + return page_address(page); +} + static pte_t * __meminit vmemmap_pte_populate(pmd_t *pmd, unsigned long addr, int node, struct vmem_altmap *altmap, unsigned long ptpfn, unsigned long flags) { pte_t *pte = pte_offset_kernel(pmd, addr); + unsigned long pfn = page_to_pfn((struct page *)addr); + if (pte_none(ptep_get(pte))) { pte_t entry; - void *p; if (ptpfn == (unsigned long)-1) { - p = vmemmap_alloc_block_buf(PAGE_SIZE, node, altmap); + void *p = vmemmap_alloc_pte(pfn, node, altmap); + if (!p) return NULL; ptpfn = PHYS_PFN(__pa(p)); @@ -246,7 +269,8 @@ static pte_t * __meminit vmemmap_pte_populate(pmd_t *pmd, unsigned long addr, in } entry = pfn_pte(ptpfn, PAGE_KERNEL); set_pte_at(&init_mm, addr, pte, entry); - } + } else if (WARN_ON_ONCE(vmemmap_optimizable_pfn(pfn))) + return NULL; return pte; } @@ -435,6 +459,9 @@ int __meminit vmemmap_populate_hugepages(unsigned long start, unsigned long end, pmd_t *pmd; for (addr = start; addr < end; addr = next) { + unsigned long pfn = page_to_pfn((struct page *)addr); + const struct mem_section *ms = __pfn_to_section(pfn); + next = pmd_addr_end(addr, end); pgd = vmemmap_pgd_populate(addr, node); @@ -450,7 +477,7 @@ int __meminit vmemmap_populate_hugepages(unsigned long start, unsigned long end, return -ENOMEM; pmd = pmd_offset(pud, addr); - if (pmd_none(pmdp_get(pmd))) { + if (pmd_none(pmdp_get(pmd)) && !section_vmemmap_optimizable(ms)) { void *p; p = vmemmap_alloc_block_buf(PMD_SIZE, node, altmap); @@ -468,8 +495,11 @@ int __meminit vmemmap_populate_hugepages(unsigned long start, unsigned long end, */ return -ENOMEM; } - } else if (vmemmap_check_pmd(pmd, node, addr, next)) + } else if (vmemmap_check_pmd(pmd, node, addr, next)) { + if (WARN_ON_ONCE(section_vmemmap_optimizable(ms))) + return -EOPNOTSUPP; continue; + } if (vmemmap_populate_basepages(addr, next, node, altmap)) return -ENOMEM; } diff --git a/mm/sparse.c b/mm/sparse.c index c84b4c7b8c7069..e6cb67ca9c8d10 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -305,8 +305,8 @@ static void __init sparse_init_nid(int nid, unsigned long pnum_begin, nid, NULL, NULL); if (!map) panic("Failed to allocate memmap for section %lu\n", pnum); - memmap_boot_pages_add(DIV_ROUND_UP(PAGES_PER_SECTION * sizeof(struct page), - PAGE_SIZE)); + memmap_boot_pages_add(section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION, + NULL, NULL)); sparse_init_early_section(nid, map, pnum, 0); } } diff --git a/mm/sparse.h b/mm/sparse.h index 563fc1f4d71778..74dd79c51e7598 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -111,8 +111,15 @@ static inline void sparse_init(void) {} */ #ifdef CONFIG_SPARSEMEM_VMEMMAP void sparse_init_subsection_map(void); +int section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, + struct vmem_altmap *altmap, struct dev_pagemap *pgmap); #else static inline void sparse_init_subsection_map(void) {} +static inline int section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, + struct vmem_altmap *altmap, struct dev_pagemap *pgmap) +{ + return DIV_ROUND_UP(nr_pages * sizeof(struct page), PAGE_SIZE); +} #endif /* CONFIG_SPARSEMEM_VMEMMAP */ #endif /* __MM_SPARSE_H */ From 5770a92b4965f853c1af886f751f43b0d69b5cba Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:48 +0800 Subject: [PATCH 0558/1352] mm/sparse: initialize memory sections earlier Upcoming HugeTLB bootmem changes need sparsemem section metadata before the HugeTLB bootmem allocation path runs. The memory sections are initialized from sparse_init(), which is called too late for that setup. Move the code that initializes sparsemem section metadata for memblock ranges into mm_core_init_early(), before free_area_init() and the HugeTLB bootmem setup. Rename the helper to sparse_sections_init() so the new caller describes the sparsemem-specific initialization step. This is a preparatory change. Link: https://lore.kernel.org/20260910063256.64386-10-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Reviewed-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/mm_init.c | 1 + mm/sparse.c | 9 +-------- mm/sparse.h | 2 ++ 3 files changed, 4 insertions(+), 8 deletions(-) diff --git a/mm/mm_init.c b/mm/mm_init.c index ee3bd792176890..304c88da5cce21 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -2642,6 +2642,7 @@ void __init mm_core_init_early(void) { kho_memory_init_early(); + sparse_sections_init(); free_area_init(); hugetlb_cma_reserve(); diff --git a/mm/sparse.c b/mm/sparse.c index e6cb67ca9c8d10..428d81838ece4a 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -191,12 +191,7 @@ static void __init memory_present(int nid, unsigned long start, unsigned long en } } -/* - * Mark all memblocks as present using memory_present(). - * This is a convenience function that is useful to mark all of the systems - * memory as present during initialization. - */ -static void __init memblocks_present(void) +void __init sparse_sections_init(void) { unsigned long start, end; int i, nid; @@ -322,8 +317,6 @@ void __init sparse_init(void) unsigned long pnum_end, pnum_begin, map_count = 1; int nid_begin; - memblocks_present(); - if (compound_info_has_mask()) { VM_WARN_ON_ONCE(!IS_ALIGNED((unsigned long) pfn_to_page(0), MAX_FOLIO_VMEMMAP_ALIGN)); diff --git a/mm/sparse.h b/mm/sparse.h index 74dd79c51e7598..e998347867f4cf 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -59,6 +59,7 @@ static inline bool vmemmap_optimizable_order(unsigned int order) */ #ifdef CONFIG_SPARSEMEM void sparse_init(void); +void sparse_sections_init(void); int sparse_index_init(unsigned long section_nr, int nid); static inline void sparse_init_one_section(struct mem_section *ms, @@ -104,6 +105,7 @@ static inline bool section_vmemmap_optimizable(const struct mem_section *ms) } #else static inline void sparse_init(void) {} +static inline void sparse_sections_init(void) {} #endif /* CONFIG_SPARSEMEM */ /* From c49153a480e0f2d49fceb8e181cbc0acc1d63024 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:49 +0800 Subject: [PATCH 0559/1352] mm/hugetlb: switch HugeTLB to section-based vmemmap optimization HugeTLB bootmem vmemmap optimization still carries its own early setup path, including pre-populating optimized mappings before the generic sparse-vmemmap code runs. Now that the section-based vmemmap optimization can derive HugeTLB vmemmap deduplication from section metadata, HugeTLB only needs to mark the bootmem huge page range with the appropriate order. The generic sparse-vmemmap population path can then allocate and map the shared tail vmemmap pages without any HugeTLB-specific early population code. Do that by recording the compound page order when a bootmem huge page is allocated and dropping the dedicated pre-HVO helpers and related special-casing. This removes duplicate early setup logic and switches HugeTLB to the section-based vmemmap optimization path. Link: https://lore.kernel.org/20260910063256.64386-11-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/hugetlb.h | 1 - include/linux/mm.h | 3 -- mm/hugetlb.c | 31 +++----------- mm/hugetlb_vmemmap.c | 91 ++++------------------------------------- mm/hugetlb_vmemmap.h | 14 +++---- mm/sparse-vmemmap.c | 31 -------------- mm/sparse.h | 30 ++++++++++++++ 7 files changed, 47 insertions(+), 154 deletions(-) diff --git a/include/linux/hugetlb.h b/include/linux/hugetlb.h index 16c4c4caa126ce..fe28f98e1b220e 100644 --- a/include/linux/hugetlb.h +++ b/include/linux/hugetlb.h @@ -171,7 +171,6 @@ struct address_space *hugetlb_folio_mapping_lock_write(struct folio *folio); extern int movable_gigantic_pages __read_mostly; extern int sysctl_hugetlb_shm_group __read_mostly; -extern struct list_head huge_boot_pages[MAX_NUMNODES]; void hugetlb_bootmem_struct_page_init(void); void hugetlb_bootmem_alloc(void); diff --git a/include/linux/mm.h b/include/linux/mm.h index dd09c438fa23ec..441bd39eab7343 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -5159,9 +5159,6 @@ int vmemmap_populate_hugepages(unsigned long start, unsigned long end, int node, struct vmem_altmap *altmap); int vmemmap_populate(unsigned long start, unsigned long end, int node, struct vmem_altmap *altmap); -int vmemmap_populate_hvo(unsigned long start, unsigned long end, - unsigned int order, struct zone *zone, - unsigned long headsize); void vmemmap_wrprotect_hvo(unsigned long start, unsigned long end, int node, unsigned long headsize); void vmemmap_populate_print_last(void); diff --git a/mm/hugetlb.c b/mm/hugetlb.c index f1c92e80c34672..89d7d684b8d968 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -52,6 +52,7 @@ #include "hugetlb_cma.h" #include "hugetlb_internal.h" #include "mm_init.h" +#include "sparse.h" #include int hugetlb_max_hstate __read_mostly; @@ -59,7 +60,7 @@ unsigned int default_hstate_idx; struct hstate hstates[HUGE_MAX_HSTATE]; __initdata nodemask_t hugetlb_bootmem_nodes; -__initdata struct list_head huge_boot_pages[MAX_NUMNODES]; +static struct list_head huge_boot_pages[MAX_NUMNODES] __initdata; /* * Due to ordering constraints across the init code for various @@ -3161,6 +3162,7 @@ static bool __init alloc_bootmem_huge_page(struct hstate *h, int nid) } else { list_add_tail(&m->list, &huge_boot_pages[nid]); m->flags |= HUGE_BOOTMEM_ZONES_VALID; + hugetlb_vmemmap_optimize_bootmem_page(m); /* * Only initialize the head struct page in memmap_init_reserved_pages, * rest of the struct pages will be initialized by the HugeTLB @@ -3321,6 +3323,8 @@ static void __init gather_bootmem_prealloc_node(unsigned long nid) * this folio. */ folio_set_hugetlb_vmemmap_optimized(folio); + section_set_compound_order_range(folio_pfn(folio), + folio_nr_pages(folio), 0); if (hugetlb_bootmem_page_earlycma(m)) folio_set_hugetlb_cma(folio); @@ -3364,31 +3368,6 @@ void __init hugetlb_bootmem_struct_page_init(void) .max_threads = num_node_state(N_MEMORY), .numa_aware = true, }; -#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP - struct zone *zone; - - for_each_zone(zone) { - for (int i = 0; i < VMEMMAP_OPTIMIZATION_NR_ORDERS; i++) { - struct page *tail, *p; - unsigned int order; - - tail = zone->vmemmap_tails[i]; - if (!tail) - continue; - - order = i + VMEMMAP_OPTIMIZATION_MIN_ORDER; - p = page_to_virt(tail); - /* - * prep_and_add_bootmem_folios() can access pageblock - * flags on bootmem HugeTLB pages, so initialize the - * shared tail struct pages here before bootmem folios - * start using them. - */ - for (int j = 0; j < PAGE_SIZE / sizeof(struct page); j++) - init_compound_tail(p + j, NULL, order, zone); - } - } -#endif padata_do_multithreaded(&job); } diff --git a/mm/hugetlb_vmemmap.c b/mm/hugetlb_vmemmap.c index c48fcea076a522..d1b031dcd17740 100644 --- a/mm/hugetlb_vmemmap.c +++ b/mm/hugetlb_vmemmap.c @@ -18,8 +18,7 @@ #include #include "hugetlb_vmemmap.h" -#include "internal.h" -#include "mm_init.h" +#include "sparse.h" /** * struct vmemmap_remap_walk - walk vmemmap page table @@ -706,95 +705,19 @@ void hugetlb_vmemmap_optimize_bootmem_folios(struct hstate *h, struct list_head __hugetlb_vmemmap_optimize_folios(h, folio_list, true); } -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT - -/* Return true of a bootmem allocated HugeTLB page should be pre-HVO-ed */ -static bool vmemmap_should_optimize_bootmem_page(struct huge_bootmem_page *m) +void __init hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m) { - unsigned long section_size, psize, pmd_vmemmap_size; - phys_addr_t paddr; - - if (!READ_ONCE(vmemmap_optimize_enabled)) - return false; - - if (!hugetlb_vmemmap_optimizable(m->hstate)) - return false; - - psize = huge_page_size(m->hstate); - paddr = virt_to_phys(m); - - /* - * Pre-HVO only works if the bootmem huge page - * is aligned to the section size. - */ - section_size = (1UL << PA_SECTION_SHIFT); - if (!IS_ALIGNED(paddr, section_size) || - !IS_ALIGNED(psize, section_size)) - return false; - - /* - * The pre-HVO code does not deal with splitting PMDS, - * so the bootmem page must be aligned to the number - * of base pages that can be mapped with one vmemmap PMD. - */ - pmd_vmemmap_size = (PMD_SIZE / (sizeof(struct page))) << PAGE_SHIFT; - if (!IS_ALIGNED(paddr, pmd_vmemmap_size) || - !IS_ALIGNED(psize, pmd_vmemmap_size)) - return false; - - return true; -} - -/* - * Initialize memmap section for a gigantic page, HVO-style. - */ -void __init hugetlb_vmemmap_init_early(int nid) -{ - unsigned long psize, paddr, section_size; - unsigned long ns, i, pnum, pfn, nr_pages; - unsigned long start, end; - struct huge_bootmem_page *m = NULL; - void *map; + struct hstate *h = m->hstate; + unsigned long pfn = PHYS_PFN(__pa(m)); if (!READ_ONCE(vmemmap_optimize_enabled)) return; - section_size = (1UL << PA_SECTION_SHIFT); - - list_for_each_entry(m, &huge_boot_pages[nid], list) { - struct zone *zone; - - if (!vmemmap_should_optimize_bootmem_page(m)) - continue; - - nr_pages = pages_per_huge_page(m->hstate); - psize = nr_pages << PAGE_SHIFT; - paddr = virt_to_phys(m); - pfn = PHYS_PFN(paddr); - map = pfn_to_page(pfn); - start = (unsigned long)map; - end = start + hugetlb_vmemmap_size(m->hstate); - zone = pfn_to_zone(pfn, nid); - - if (vmemmap_populate_hvo(start, end, huge_page_order(m->hstate), - zone, HUGETLB_VMEMMAP_RESERVE_SIZE)) - panic("Failed to allocate memmap for HugeTLB page\n"); - memmap_boot_pages_add(DIV_ROUND_UP(HUGETLB_VMEMMAP_RESERVE_SIZE, PAGE_SIZE)); - - pnum = pfn_to_section_nr(pfn); - ns = psize / section_size; - - for (i = 0; i < ns; i++) { - sparse_init_early_section(nid, map, pnum, - SECTION_IS_VMEMMAP_PREINIT); - map += section_map_size(); - pnum++; - } - + section_set_compound_order_range(pfn, pages_per_huge_page(h), + huge_page_order(h)); + if (vmemmap_optimizable_order(pfn_to_section_compound_order(pfn))) m->flags |= HUGE_BOOTMEM_HVO; - } } -#endif static const struct ctl_table hugetlb_vmemmap_sysctls[] = { { diff --git a/mm/hugetlb_vmemmap.h b/mm/hugetlb_vmemmap.h index 7ac49c52457ddf..20eb03df542a43 100644 --- a/mm/hugetlb_vmemmap.h +++ b/mm/hugetlb_vmemmap.h @@ -9,8 +9,7 @@ #ifndef _LINUX_HUGETLB_VMEMMAP_H #define _LINUX_HUGETLB_VMEMMAP_H #include -#include -#include +#include "internal.h" /* * Reserve one vmemmap page, all vmemmap addresses are mapped to it. See @@ -27,10 +26,7 @@ long hugetlb_vmemmap_restore_folios(const struct hstate *h, void hugetlb_vmemmap_optimize_folio(const struct hstate *h, struct folio *folio); void hugetlb_vmemmap_optimize_folios(struct hstate *h, struct list_head *folio_list); void hugetlb_vmemmap_optimize_bootmem_folios(struct hstate *h, struct list_head *folio_list); -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT -void hugetlb_vmemmap_init_early(int nid); -#endif - +void hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m); static inline unsigned int hugetlb_vmemmap_size(const struct hstate *h) { @@ -76,13 +72,13 @@ static inline void hugetlb_vmemmap_optimize_bootmem_folios(struct hstate *h, { } -static inline void hugetlb_vmemmap_init_early(int nid) +static inline unsigned int hugetlb_vmemmap_optimizable_size(const struct hstate *h) { + return 0; } -static inline unsigned int hugetlb_vmemmap_optimizable_size(const struct hstate *h) +static inline void hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m) { - return 0; } #endif /* CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP */ diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index faebf344cdac9b..8c2b9bed236d15 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -32,8 +32,6 @@ #include #include -#include "hugetlb_vmemmap.h" - /* * Flags for vmemmap_populate_range and friends. */ @@ -404,34 +402,6 @@ void vmemmap_wrprotect_hvo(unsigned long addr, unsigned long end, } } -#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP -int __meminit vmemmap_populate_hvo(unsigned long addr, unsigned long end, - unsigned int order, struct zone *zone, - unsigned long headsize) -{ - unsigned long maddr; - struct page *tail; - pte_t *pte; - int node = zone_to_nid(zone); - - tail = vmemmap_get_tail(order, zone); - if (!tail) - return -ENOMEM; - - for (maddr = addr; maddr < addr + headsize; maddr += PAGE_SIZE) { - pte = vmemmap_populate_address(maddr, node, NULL, -1, 0); - if (!pte) - return -ENOMEM; - } - - /* - * Reuse the last page struct page mapped above for the rest. - */ - return vmemmap_populate_range(maddr, end, node, NULL, - page_to_pfn(tail), 0); -} -#endif - void __weak __meminit vmemmap_set_pmd(pmd_t *pmd, void *p, int node, unsigned long addr, unsigned long next) { @@ -634,7 +604,6 @@ struct page * __meminit __populate_section_memmap(unsigned long pfn, */ void __init sparse_vmemmap_init_nid_early(int nid) { - hugetlb_vmemmap_init_early(nid); } #endif diff --git a/mm/sparse.h b/mm/sparse.h index e998347867f4cf..d3a71ef4fad0fe 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -16,6 +16,26 @@ static inline unsigned int section_compound_order(const struct mem_section *sect return section->compound_page_order; } +static inline void section_set_compound_order(struct mem_section *section, + unsigned int order) +{ + VM_WARN_ON(section_compound_order(section) && order && + section_compound_order(section) != order); + section->compound_page_order = order; +} + +static inline void section_set_compound_order_range(unsigned long pfn, + unsigned long nr_pages, unsigned int order) +{ + unsigned long section_nr = pfn_to_section_nr(pfn); + + if (!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SECTION)) + return; + + for (unsigned long i = 0; i < nr_pages / PAGES_PER_SECTION; i++) + section_set_compound_order(__nr_to_section(section_nr + i), order); +} + static inline unsigned int pfn_to_section_compound_order(unsigned long pfn) { return section_compound_order(__pfn_to_section(pfn)); @@ -26,6 +46,16 @@ static inline unsigned int section_compound_order(const struct mem_section *sect return 0; } +static inline void section_set_compound_order(struct mem_section *section, + unsigned int order) +{ +} + +static inline void section_set_compound_order_range(unsigned long pfn, + unsigned long nr_pages, unsigned int order) +{ +} + static inline unsigned int pfn_to_section_compound_order(unsigned long pfn) { return 0; From 8a47591db9cbd6d15fbb04cb82909e0d24fa7980 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:50 +0800 Subject: [PATCH 0560/1352] mm/sparse-vmemmap: remove SPARSEMEM_VMEMMAP_PREINIT support SPARSEMEM_VMEMMAP_PREINIT existed only to support HugeTLB's early vmemmap optimization setup. Now that HugeTLB bootmem vmemmap optimization uses the common section-based sparse-vmemmap path, the sparse initialization code no longer needs a separate pre-initialization mechanism for vmemmap population. Remove the related Kconfig symbols, section flag, and empty early hook, so present sections always go through the normal sparse setup path. Link: https://lore.kernel.org/20260910063256.64386-12-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Acked-by: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- arch/x86/Kconfig | 1 - fs/Kconfig | 1 - include/linux/mmzone.h | 25 ------------------------- mm/Kconfig | 5 ----- mm/sparse-vmemmap.c | 13 ------------- mm/sparse.c | 23 ++++++++--------------- 6 files changed, 8 insertions(+), 60 deletions(-) diff --git a/arch/x86/Kconfig b/arch/x86/Kconfig index 15fd9ec5ecacb7..7aa74bcc72f9db 100644 --- a/arch/x86/Kconfig +++ b/arch/x86/Kconfig @@ -150,7 +150,6 @@ config X86 select ARCH_WANT_LD_ORPHAN_WARN select ARCH_WANT_OPTIMIZE_DAX_VMEMMAP if X86_64 select ARCH_WANT_OPTIMIZE_HUGETLB_VMEMMAP if X86_64 - select ARCH_WANT_HUGETLB_VMEMMAP_PREINIT if X86_64 select ARCH_WANTS_THP_SWAP if X86_64 select ARCH_HAS_PARANOID_L1D_FLUSH select ARCH_WANT_IRQS_OFF_ACTIVATE_MM diff --git a/fs/Kconfig b/fs/Kconfig index e05917adcd608e..d1c210c6508f0a 100644 --- a/fs/Kconfig +++ b/fs/Kconfig @@ -278,7 +278,6 @@ config HUGETLB_PAGE_OPTIMIZE_VMEMMAP def_bool HUGETLB_PAGE depends on ARCH_WANT_OPTIMIZE_HUGETLB_VMEMMAP depends on SPARSEMEM_VMEMMAP - select SPARSEMEM_VMEMMAP_PREINIT if ARCH_WANT_HUGETLB_VMEMMAP_PREINIT config HUGETLB_PMD_PAGE_TABLE_SHARING def_bool HUGETLB_PAGE diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 2c4f63e379da7f..f2d39a888eac51 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -2095,9 +2095,6 @@ enum { SECTION_IS_EARLY_BIT, #ifdef CONFIG_ZONE_DEVICE SECTION_TAINT_ZONE_DEVICE_BIT, -#endif -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT - SECTION_IS_VMEMMAP_PREINIT_BIT, #endif SECTION_MAP_LAST_BIT, }; @@ -2109,9 +2106,6 @@ enum { #ifdef CONFIG_ZONE_DEVICE #define SECTION_TAINT_ZONE_DEVICE BIT(SECTION_TAINT_ZONE_DEVICE_BIT) #endif -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT -#define SECTION_IS_VMEMMAP_PREINIT BIT(SECTION_IS_VMEMMAP_PREINIT_BIT) -#endif #define SECTION_MAP_MASK (~(BIT(SECTION_MAP_LAST_BIT) - 1)) #define SECTION_NID_SHIFT SECTION_MAP_LAST_BIT @@ -2166,24 +2160,6 @@ static inline int online_device_section(const struct mem_section *section) } #endif -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT -static inline int preinited_vmemmap_section(const struct mem_section *section) -{ - return (section && - (section->section_mem_map & SECTION_IS_VMEMMAP_PREINIT)); -} - -void sparse_vmemmap_init_nid_early(int nid); -#else -static inline int preinited_vmemmap_section(const struct mem_section *section) -{ - return 0; -} -static inline void sparse_vmemmap_init_nid_early(int nid) -{ -} -#endif - static inline int online_section_nr(unsigned long nr) { return online_section(__nr_to_section(nr)); @@ -2385,7 +2361,6 @@ static inline unsigned long next_present_section_nr(unsigned long section_nr) #endif #else -#define sparse_vmemmap_init_nid_early(_nid) do {} while (0) #define pfn_in_present_section pfn_valid #endif /* CONFIG_SPARSEMEM */ diff --git a/mm/Kconfig b/mm/Kconfig index 604c58199acbf8..2c385f8b29445e 100644 --- a/mm/Kconfig +++ b/mm/Kconfig @@ -461,8 +461,6 @@ config SPARSEMEM_VMEMMAP pfn_to_page and page_to_pfn operations. This is the most efficient option when sufficient kernel resources are available. -config SPARSEMEM_VMEMMAP_PREINIT - bool # # Select this config option from the architecture Kconfig, if it is preferred # to enable the feature of HugeTLB/dev_dax vmemmap optimization. @@ -473,9 +471,6 @@ config ARCH_WANT_OPTIMIZE_DAX_VMEMMAP config ARCH_WANT_OPTIMIZE_HUGETLB_VMEMMAP bool -config ARCH_WANT_HUGETLB_VMEMMAP_PREINIT - bool - config HAVE_MEMBLOCK_PHYS_MAP bool diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 8c2b9bed236d15..f22d815d7af04e 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -594,19 +594,6 @@ struct page * __meminit __populate_section_memmap(unsigned long pfn, return pfn_to_page(pfn); } -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT -/* - * This is called just before initializing sections for a NUMA node. - * Any special initialization that needs to be done before the - * generic initialization can be done from here. Sections that - * are initialized in hooks called from here will be skipped by - * the generic initialization. - */ -void __init sparse_vmemmap_init_nid_early(int nid) -{ -} -#endif - static void subsection_mask_set(unsigned long *map, unsigned long pfn, unsigned long nr_pages) { diff --git a/mm/sparse.c b/mm/sparse.c index 428d81838ece4a..f4f393a033a424 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -283,27 +283,20 @@ static void __init sparse_init_nid(int nid, unsigned long pnum_begin, if (sparse_usage_init(nid, map_count)) panic("Failed to allocate usemap for node %d\n", nid); - sparse_vmemmap_init_nid_early(nid); - for_each_present_section_nr(pnum_begin, pnum) { - struct mem_section *ms; unsigned long pfn = section_nr_to_pfn(pnum); + struct page *map; if (pnum >= pnum_end) break; - ms = __nr_to_section(pnum); - if (!preinited_vmemmap_section(ms)) { - struct page *map; - - map = __populate_section_memmap(pfn, PAGES_PER_SECTION, - nid, NULL, NULL); - if (!map) - panic("Failed to allocate memmap for section %lu\n", pnum); - memmap_boot_pages_add(section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION, - NULL, NULL)); - sparse_init_early_section(nid, map, pnum, 0); - } + map = __populate_section_memmap(pfn, PAGES_PER_SECTION, + nid, NULL, NULL); + if (!map) + panic("Failed to allocate memmap for section %lu\n", pnum); + memmap_boot_pages_add(section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION, + NULL, NULL)); + sparse_init_early_section(nid, map, pnum, 0); } sparse_usage_fini(); } From c6186c1ef18cae7c3d64b442c8fd4fa62edd4755 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:51 +0800 Subject: [PATCH 0561/1352] mm/sparse: inline usemap allocation into sparse_init_nid() After removing SPARSEMEM_VMEMMAP_PREINIT, sparse_init_nid() no longer needs the transient sparse_usagebuf state and its helper wrappers. Allocate the usemap buffer directly in sparse_init_nid(), pass it to sparse_init_one_section(), and drop sparse_usage_init(), sparse_usage_fini(), and sparse_init_early_section(). Link: https://lore.kernel.org/20260910063256.64386-13-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Qi Zheng Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/mmzone.h | 3 --- mm/sparse.c | 46 +++++++----------------------------------- 2 files changed, 7 insertions(+), 42 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index f2d39a888eac51..6acc14b169bbe6 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -2223,9 +2223,6 @@ static inline bool pfn_section_first_valid(struct mem_section *ms, unsigned long } #endif -void sparse_init_early_section(int nid, struct page *map, unsigned long pnum, - unsigned long flags); - #ifndef CONFIG_HAVE_ARCH_PFN_VALID /** * pfn_valid - check if there is a valid memory map entry for a PFN diff --git a/mm/sparse.c b/mm/sparse.c index f4f393a033a424..d19c173b40feb8 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -234,42 +234,6 @@ void __weak __meminit vmemmap_populate_print_last(void) { } -static void *sparse_usagebuf __initdata; -static void *sparse_usagebuf_end __initdata; - -/* - * Helper function that is used for generic section initialization, and - * can also be used by any hooks added above. - */ -void __init sparse_init_early_section(int nid, struct page *map, - unsigned long pnum, unsigned long flags) -{ - BUG_ON(!sparse_usagebuf || sparse_usagebuf >= sparse_usagebuf_end); - sparse_init_one_section(__nr_to_section(pnum), pnum, map, - sparse_usagebuf, SECTION_IS_EARLY | flags); - sparse_usagebuf = (void *)sparse_usagebuf + mem_section_usage_size(); -} - -static int __init sparse_usage_init(int nid, unsigned long map_count) -{ - unsigned long size; - - size = mem_section_usage_size() * map_count; - sparse_usagebuf = memblock_alloc_node(size, SMP_CACHE_BYTES, nid); - if (!sparse_usagebuf) { - sparse_usagebuf_end = NULL; - return -ENOMEM; - } - - sparse_usagebuf_end = sparse_usagebuf + size; - return 0; -} - -static void __init sparse_usage_fini(void) -{ - sparse_usagebuf = sparse_usagebuf_end = NULL; -} - /* * Initialize sparse on a specific node. The node spans [pnum_begin, pnum_end) * And number of present sections in this node is map_count. @@ -279,8 +243,11 @@ static void __init sparse_init_nid(int nid, unsigned long pnum_begin, unsigned long map_count) { unsigned long pnum; + struct mem_section_usage *usage; - if (sparse_usage_init(nid, map_count)) + usage = memblock_alloc_node(map_count * mem_section_usage_size(), + SMP_CACHE_BYTES, nid); + if (!usage) panic("Failed to allocate usemap for node %d\n", nid); for_each_present_section_nr(pnum_begin, pnum) { @@ -296,9 +263,10 @@ static void __init sparse_init_nid(int nid, unsigned long pnum_begin, panic("Failed to allocate memmap for section %lu\n", pnum); memmap_boot_pages_add(section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION, NULL, NULL)); - sparse_init_early_section(nid, map, pnum, 0); + sparse_init_one_section(__nr_to_section(pnum), pnum, map, usage, + SECTION_IS_EARLY); + usage = (void *)usage + mem_section_usage_size(); } - sparse_usage_fini(); } /* From 2b476f8c5e9107296f9f74a7da093d135b26052f Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:52 +0800 Subject: [PATCH 0562/1352] mm/sparse: remove section_map_size() section_map_size() no longer provides any shared logic. After the sparse-vmemmap changes, its only remaining user is the !CONFIG_SPARSEMEM_VMEMMAP path in __populate_section_memmap(), which can compute the size inline with PAGE_ALIGN(sizeof(struct page) * PAGES_PER_SECTION). Remove section_map_size() and inline the remaining calculation. Link: https://lore.kernel.org/20260910063256.64386-14-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Qi Zheng Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/mm.h | 1 - mm/sparse.c | 15 ++------------- 2 files changed, 2 insertions(+), 14 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index 441bd39eab7343..b19711b6dbc69a 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -5140,7 +5140,6 @@ static inline void print_vma_addr(char *prefix, unsigned long rip) } #endif -unsigned long section_map_size(void); struct page * __populate_section_memmap(unsigned long pfn, unsigned long nr_pages, int nid, struct vmem_altmap *altmap, struct dev_pagemap *pgmap); diff --git a/mm/sparse.c b/mm/sparse.c index d19c173b40feb8..cc28bb41fdb1f1 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -208,23 +208,12 @@ void __init sparse_sections_init(void) memory_present(nid, start, end); } -#ifdef CONFIG_SPARSEMEM_VMEMMAP -unsigned long __init section_map_size(void) -{ - return ALIGN(sizeof(struct page) * PAGES_PER_SECTION, PMD_SIZE); -} - -#else -unsigned long __init section_map_size(void) -{ - return PAGE_ALIGN(sizeof(struct page) * PAGES_PER_SECTION); -} - +#ifndef CONFIG_SPARSEMEM_VMEMMAP struct page __init *__populate_section_memmap(unsigned long pfn, unsigned long nr_pages, int nid, struct vmem_altmap *altmap, struct dev_pagemap *pgmap) { - unsigned long size = section_map_size(); + const unsigned long size = PAGE_ALIGN(sizeof(struct page) * PAGES_PER_SECTION); return memmap_alloc(size, size, __pa(MAX_DMA_ADDRESS), nid, false); } From d4dcfa30ea28e4a7b9fa466a0eec61bbe5dfce42 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:53 +0800 Subject: [PATCH 0563/1352] mm/hugetlb: remove HUGE_BOOTMEM_HVO The HUGE_BOOTMEM_HVO flag tracked whether a bootmem huge page had already gone through the old early vmemmap optimization path. Now that HugeTLB uses section-based vmemmap optimization, that state is already reflected in the compound page order stored in section metadata. Remove HUGE_BOOTMEM_HVO and its helper, and use the section state directly when deciding whether to mark a folio as vmemmap-optimized. Link: https://lore.kernel.org/20260910063256.64386-15-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/hugetlb.h | 5 ++--- mm/hugetlb.c | 16 +++------------- mm/hugetlb_vmemmap.c | 2 -- 3 files changed, 5 insertions(+), 18 deletions(-) diff --git a/include/linux/hugetlb.h b/include/linux/hugetlb.h index fe28f98e1b220e..3559041a5a5780 100644 --- a/include/linux/hugetlb.h +++ b/include/linux/hugetlb.h @@ -675,9 +675,8 @@ struct hstate { char name[HSTATE_NAME_LEN]; }; -#define HUGE_BOOTMEM_HVO 0x0001 -#define HUGE_BOOTMEM_ZONES_VALID 0x0002 -#define HUGE_BOOTMEM_CMA 0x0004 +#define HUGE_BOOTMEM_ZONES_VALID BIT(0) +#define HUGE_BOOTMEM_CMA BIT(1) int isolate_or_dissolve_huge_folio(struct folio *folio, struct list_head *list); int replace_free_hugepage_folios(unsigned long start_pfn, unsigned long end_pfn); diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 89d7d684b8d968..73999023cfdd39 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -3220,11 +3220,6 @@ static void __init hugetlb_folio_init_vmemmap(struct folio *folio, prep_compound_head(&folio->page, huge_page_order(h)); } -static bool __init hugetlb_bootmem_page_prehvo(struct huge_bootmem_page *m) -{ - return m->flags & HUGE_BOOTMEM_HVO; -} - static bool __init hugetlb_bootmem_page_earlycma(struct huge_bootmem_page *m) { return m->flags & HUGE_BOOTMEM_CMA; @@ -3299,6 +3294,7 @@ static void __init gather_bootmem_prealloc_node(unsigned long nid) list_for_each_entry_safe(m, tm, &huge_boot_pages[nid], list) { struct page *page = virt_to_page(m); struct folio *folio = (void *)page; + const unsigned long pfn = folio_pfn(folio); h = m->hstate; /* @@ -3316,15 +3312,9 @@ static void __init gather_bootmem_prealloc_node(unsigned long nid) HUGETLB_VMEMMAP_RESERVE_PAGES); init_new_hugetlb_folio(folio); - if (hugetlb_bootmem_page_prehvo(m)) - /* - * If pre-HVO was done, just set the - * flag, the HVO code will then skip - * this folio. - */ + if (vmemmap_optimizable_order(pfn_to_section_compound_order(pfn))) folio_set_hugetlb_vmemmap_optimized(folio); - section_set_compound_order_range(folio_pfn(folio), - folio_nr_pages(folio), 0); + section_set_compound_order_range(pfn, folio_nr_pages(folio), 0); if (hugetlb_bootmem_page_earlycma(m)) folio_set_hugetlb_cma(folio); diff --git a/mm/hugetlb_vmemmap.c b/mm/hugetlb_vmemmap.c index d1b031dcd17740..fba0c5a7d46fc4 100644 --- a/mm/hugetlb_vmemmap.c +++ b/mm/hugetlb_vmemmap.c @@ -715,8 +715,6 @@ void __init hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m) section_set_compound_order_range(pfn, pages_per_huge_page(h), huge_page_order(h)); - if (vmemmap_optimizable_order(pfn_to_section_compound_order(pfn))) - m->flags |= HUGE_BOOTMEM_HVO; } static const struct ctl_table hugetlb_vmemmap_sysctls[] = { From 8264b54431afb5429f5b115058dcc804c4a799c2 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:54 +0800 Subject: [PATCH 0564/1352] mm/hugetlb: remove HUGE_BOOTMEM_CMA Track early CMA hugetlb pages from the hstate instead of storing a redundant bootmem flag. This removes the unused helper and keeps the bootmem metadata minimal. Link: https://lore.kernel.org/20260910063256.64386-16-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/hugetlb.h | 1 - mm/hugetlb.c | 14 ++++---------- 2 files changed, 4 insertions(+), 11 deletions(-) diff --git a/include/linux/hugetlb.h b/include/linux/hugetlb.h index 3559041a5a5780..255a258f11d13c 100644 --- a/include/linux/hugetlb.h +++ b/include/linux/hugetlb.h @@ -676,7 +676,6 @@ struct hstate { }; #define HUGE_BOOTMEM_ZONES_VALID BIT(0) -#define HUGE_BOOTMEM_CMA BIT(1) int isolate_or_dissolve_huge_folio(struct folio *folio, struct list_head *list); int replace_free_hugepage_folios(unsigned long start_pfn, unsigned long end_pfn); diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 73999023cfdd39..4ecf5db73de451 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -3144,7 +3144,7 @@ static bool __init alloc_bootmem_huge_page(struct hstate *h, int nid) */ INIT_LIST_HEAD(&m->list); m->hstate = h; - m->flags = hugetlb_early_cma(h) ? HUGE_BOOTMEM_CMA : 0; + m->flags = 0; /* CMA pages: zone-crossing is validated in hugetlb_cma_reserve(). */ if (!hugetlb_early_cma(h) && @@ -3220,11 +3220,6 @@ static void __init hugetlb_folio_init_vmemmap(struct folio *folio, prep_compound_head(&folio->page, huge_page_order(h)); } -static bool __init hugetlb_bootmem_page_earlycma(struct huge_bootmem_page *m) -{ - return m->flags & HUGE_BOOTMEM_CMA; -} - /* * memblock-allocated pageblocks might not have the migrate type set * if marked with the 'noinit' flag. Set it to the default (MIGRATE_MOVABLE) @@ -3316,9 +3311,6 @@ static void __init gather_bootmem_prealloc_node(unsigned long nid) folio_set_hugetlb_vmemmap_optimized(folio); section_set_compound_order_range(pfn, folio_nr_pages(folio), 0); - if (hugetlb_bootmem_page_earlycma(m)) - folio_set_hugetlb_cma(folio); - list_add(&folio->lru, &folio_list); /* @@ -3329,7 +3321,9 @@ static void __init gather_bootmem_prealloc_node(unsigned long nid) * For CMA pages, this is done in init_cma_pageblock * (via hugetlb_bootmem_init_migratetype), so skip it here. */ - if (!folio_test_hugetlb_cma(folio)) + if (hugetlb_early_cma(h)) + folio_set_hugetlb_cma(folio); + else adjust_managed_page_count(page, pages_per_huge_page(h)); cond_resched(); } From 1c6cf76a6301a0d50f649bb37c28c2987f93995c Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:55 +0800 Subject: [PATCH 0565/1352] mm/hugetlb: localize struct huge_bootmem_page struct huge_bootmem_page is only used by hugetlb boot-time allocation code, but its definition currently lives in mm/internal.h because hugetlb_vmemmap_optimize_bootmem_page() takes it as an argument. This exposes a hugetlb-specific internal type more broadly than needed. Change hugetlb_vmemmap_optimize_bootmem_page() to take the information it actually needs. With that interface, mm/hugetlb_vmemmap.h no longer needs to include mm/internal.h, and struct huge_bootmem_page can move into mm/hugetlb.c. Link: https://lore.kernel.org/20260910063256.64386-17-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/hugetlb.c | 8 +++++++- mm/hugetlb_vmemmap.c | 9 +++------ mm/hugetlb_vmemmap.h | 5 ++--- mm/internal.h | 7 ------- 4 files changed, 12 insertions(+), 17 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 4ecf5db73de451..cdd061953ab130 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -55,6 +55,12 @@ #include "sparse.h" #include +struct huge_bootmem_page { + struct list_head list; + struct hstate *hstate; + unsigned long flags; +}; + int hugetlb_max_hstate __read_mostly; unsigned int default_hstate_idx; struct hstate hstates[HUGE_MAX_HSTATE]; @@ -3162,7 +3168,7 @@ static bool __init alloc_bootmem_huge_page(struct hstate *h, int nid) } else { list_add_tail(&m->list, &huge_boot_pages[nid]); m->flags |= HUGE_BOOTMEM_ZONES_VALID; - hugetlb_vmemmap_optimize_bootmem_page(m); + hugetlb_vmemmap_optimize_bootmem_page(pfn, huge_page_order(h)); /* * Only initialize the head struct page in memmap_init_reserved_pages, * rest of the struct pages will be initialized by the HugeTLB diff --git a/mm/hugetlb_vmemmap.c b/mm/hugetlb_vmemmap.c index fba0c5a7d46fc4..f977d0a7e00274 100644 --- a/mm/hugetlb_vmemmap.c +++ b/mm/hugetlb_vmemmap.c @@ -19,6 +19,7 @@ #include #include "hugetlb_vmemmap.h" #include "sparse.h" +#include "internal.h" /** * struct vmemmap_remap_walk - walk vmemmap page table @@ -705,16 +706,12 @@ void hugetlb_vmemmap_optimize_bootmem_folios(struct hstate *h, struct list_head __hugetlb_vmemmap_optimize_folios(h, folio_list, true); } -void __init hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m) +void __init hugetlb_vmemmap_optimize_bootmem_page(unsigned long pfn, unsigned int order) { - struct hstate *h = m->hstate; - unsigned long pfn = PHYS_PFN(__pa(m)); - if (!READ_ONCE(vmemmap_optimize_enabled)) return; - section_set_compound_order_range(pfn, pages_per_huge_page(h), - huge_page_order(h)); + section_set_compound_order_range(pfn, 1UL << order, order); } static const struct ctl_table hugetlb_vmemmap_sysctls[] = { diff --git a/mm/hugetlb_vmemmap.h b/mm/hugetlb_vmemmap.h index 20eb03df542a43..464192e32decf4 100644 --- a/mm/hugetlb_vmemmap.h +++ b/mm/hugetlb_vmemmap.h @@ -9,7 +9,6 @@ #ifndef _LINUX_HUGETLB_VMEMMAP_H #define _LINUX_HUGETLB_VMEMMAP_H #include -#include "internal.h" /* * Reserve one vmemmap page, all vmemmap addresses are mapped to it. See @@ -26,7 +25,7 @@ long hugetlb_vmemmap_restore_folios(const struct hstate *h, void hugetlb_vmemmap_optimize_folio(const struct hstate *h, struct folio *folio); void hugetlb_vmemmap_optimize_folios(struct hstate *h, struct list_head *folio_list); void hugetlb_vmemmap_optimize_bootmem_folios(struct hstate *h, struct list_head *folio_list); -void hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m); +void hugetlb_vmemmap_optimize_bootmem_page(unsigned long pfn, unsigned int order); static inline unsigned int hugetlb_vmemmap_size(const struct hstate *h) { @@ -77,7 +76,7 @@ static inline unsigned int hugetlb_vmemmap_optimizable_size(const struct hstate return 0; } -static inline void hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m) +static inline void hugetlb_vmemmap_optimize_bootmem_page(unsigned long pfn, unsigned int order) { } #endif /* CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP */ diff --git a/mm/internal.h b/mm/internal.h index da833cafcd599d..5cc220db907668 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -23,13 +23,6 @@ #include "vma.h" struct folio_batch; -struct hstate; - -struct huge_bootmem_page { - struct list_head list; - struct hstate *hstate; - unsigned long flags; -}; /* mm/workingset.c */ bool workingset_test_recent(void *shadow, bool file, bool *workingset, From cb0cdf4f3d55f4af8c3e30d8cb94825b25e75828 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:56 +0800 Subject: [PATCH 0566/1352] mm/hugetlb: localize HUGE_BOOTMEM_ZONES_VALID HUGE_BOOTMEM_ZONES_VALID is only used by the huge_bootmem_page flag handling in mm/hugetlb.c. Keep the definition next to that private data structure instead of exposing it through the public hugetlb header. No functional change is intended. Link: https://lore.kernel.org/20260910063256.64386-18-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/hugetlb.h | 2 -- mm/hugetlb.c | 2 ++ 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/include/linux/hugetlb.h b/include/linux/hugetlb.h index 255a258f11d13c..900c95e346b2e3 100644 --- a/include/linux/hugetlb.h +++ b/include/linux/hugetlb.h @@ -675,8 +675,6 @@ struct hstate { char name[HSTATE_NAME_LEN]; }; -#define HUGE_BOOTMEM_ZONES_VALID BIT(0) - int isolate_or_dissolve_huge_folio(struct folio *folio, struct list_head *list); int replace_free_hugepage_folios(unsigned long start_pfn, unsigned long end_pfn); void wait_for_freed_hugetlb_folios(void); diff --git a/mm/hugetlb.c b/mm/hugetlb.c index cdd061953ab130..74de391c39730d 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -55,6 +55,8 @@ #include "sparse.h" #include +#define HUGE_BOOTMEM_ZONES_VALID BIT(0) + struct huge_bootmem_page { struct list_head list; struct hstate *hstate; From 8ed8078ec001436f5e7cf8c83d6f493ec2bc47ec Mon Sep 17 00:00:00 2001 From: Song Hu Date: Tue, 25 Aug 2026 16:57:54 +0800 Subject: [PATCH 0567/1352] selftests/mm: emit TAP header in uffd-wp-mremap Patch series "selftests/mm: TAP output and global-state fixes", v4. uffd-wp-mremap and mremap_test never print the TAP header (and mremap_test skips with a bare exit(KSFT_SKIP) rather than a KTAP skip), so their output is not valid KTAP; hugetlb-soft-offline toggles enable_soft_offline during the run and leaves it disabled afterwards. This patch (of 3): uffd-wp-mremap calls ksft_set_plan() without ksft_print_header(), so its output is not valid KTAP. Add the header, like the sibling uffd tests (uffd-stress, uffd-unit-tests). Link: https://lore.kernel.org/20260825085756.63030-1-husong@kylinos.cn Link: https://lore.kernel.org/20260825085756.63030-2-husong@kylinos.cn Signed-off-by: Song Hu Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Reviewed-by: Sarthak Sharma Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: Muhammad Usama Anjum Tested-by: Muhammad Usama Anjum Acked-by: David Hildenbrand (Arm) Cc: Liam R. Howlett Cc: Michal Hocko Cc: Peter Xu Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/mm/uffd-wp-mremap.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/tools/testing/selftests/mm/uffd-wp-mremap.c b/tools/testing/selftests/mm/uffd-wp-mremap.c index c973d6722720c3..572c2516e874d7 100644 --- a/tools/testing/selftests/mm/uffd-wp-mremap.c +++ b/tools/testing/selftests/mm/uffd-wp-mremap.c @@ -347,6 +347,8 @@ int main(int argc, char **argv) struct thp_settings settings; int i, j, plan = 0; + ksft_print_header(); + hugepage_save_settings(true, true); check_uffd_wp_feature_supported(); From d3622a401c635a667f6435eaa4a720b32606bedd Mon Sep 17 00:00:00 2001 From: Song Hu Date: Tue, 25 Aug 2026 16:57:55 +0800 Subject: [PATCH 0568/1352] selftests/mm: emit TAP header and use TAP skip in mremap_test mremap_test calls ksft_set_plan() without ksft_print_header(), and its get_mmap_min_addr() skip path uses a bare exit(KSFT_SKIP) that prints no TAP line, so its output is not valid KTAP. Add the header and switch the skip to ksft_exit_skip(). Also fix two more KTAP compliance issues spotted in review: - get_mmap_min_addr() calls strerror(errno) after fclose(), which may clobber errno; save errno before fclose() instead. - Some ksft_*() messages embed "\n\t", so the text after each embedded newline is printed without the "# " prefix. Split those into separate messages. And cache mmap_min_addr in main() before ksft_set_plan(), so that the skip paths in get_mmap_min_addr() are taken before the plan is set; a skip after the plan leaves the run with fewer tests than planned. Link: https://lore.kernel.org/20260825085756.63030-3-husong@kylinos.cn Signed-off-by: Song Hu Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Reviewed-by: Sarthak Sharma Reviewed-by: Muhammad Usama Anjum Tested-by: Muhammad Usama Anjum Acked-by: Lorenzo Stoakes (ARM) Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Michal Hocko Cc: Peter Xu Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/mm/mremap_test.c | 43 ++++++++++++++---------- 1 file changed, 25 insertions(+), 18 deletions(-) diff --git a/tools/testing/selftests/mm/mremap_test.c b/tools/testing/selftests/mm/mremap_test.c index 779ef2d5f1b9f7..97abf4713cc595 100644 --- a/tools/testing/selftests/mm/mremap_test.c +++ b/tools/testing/selftests/mm/mremap_test.c @@ -111,18 +111,17 @@ static unsigned long long get_mmap_min_addr(void) return addr; fp = fopen("/proc/sys/vm/mmap_min_addr", "r"); - if (fp == NULL) { - ksft_print_msg("Failed to open /proc/sys/vm/mmap_min_addr: %s\n", - strerror(errno)); - exit(KSFT_SKIP); - } + if (!fp) + ksft_exit_skip("Failed to open /proc/sys/vm/mmap_min_addr: %s\n", + strerror(errno)); n_matched = fscanf(fp, "%llu", &addr); if (n_matched != 1) { - ksft_print_msg("Failed to read /proc/sys/vm/mmap_min_addr: %s\n", - strerror(errno)); + int err = errno; + fclose(fp); - exit(KSFT_SKIP); + ksft_exit_skip("Failed to read /proc/sys/vm/mmap_min_addr: %s\n", + strerror(err)); } fclose(fp); @@ -1165,10 +1164,11 @@ static void run_mremap_test_case(struct test test_case, int *failures, rand_addr); if (remap_time < 0) { - if (test_case.expect_failure) - ksft_test_result_xfail("%s\n\tExpected mremap failure\n", - test_case.name); - else { + if (test_case.expect_failure) { + ksft_print_msg("%s: expected mremap failure\n", + test_case.name); + ksft_test_result_xfail("%s\n", test_case.name); + } else { ksft_test_result_fail("%s\n", test_case.name); *failures += 1; } @@ -1178,11 +1178,13 @@ static void run_mremap_test_case(struct test test_case, int *failures, * was faulted in. */ if (threshold_mb == VALIDATION_NO_THRESHOLD || - test_case.config.region_size <= threshold_mb * _1MB) - ksft_test_result_pass("%s\n\tmremap time: %12lldns\n", - test_case.name, remap_time); - else + test_case.config.region_size <= threshold_mb * _1MB) { + ksft_print_msg("%s: mremap time: %12lldns\n", + test_case.name, remap_time); ksft_test_result_pass("%s\n", test_case.name); + } else { + ksft_test_result_pass("%s\n", test_case.name); + } } } @@ -1251,13 +1253,18 @@ int main(int argc, char **argv) time_t t; FILE *maps_fp; + ksft_print_header(); + + get_mmap_min_addr(); + pattern_seed = (unsigned int) time(&t); if (parse_args(argc, argv, &threshold_mb, &pattern_seed) < 0) exit(EXIT_FAILURE); - ksft_print_msg("Test configs:\n\tthreshold_mb=%u\n\tpattern_seed=%u\n\n", - threshold_mb, pattern_seed); + ksft_print_msg("Test configs:\n"); + ksft_print_msg("threshold_mb=%u\n", threshold_mb); + ksft_print_msg("pattern_seed=%u\n", pattern_seed); /* * set preallocated random array according to test configs; see the From 3ecf2c2f0a609fb135987f56110de5229690c348 Mon Sep 17 00:00:00 2001 From: Song Hu Date: Tue, 25 Aug 2026 16:57:56 +0800 Subject: [PATCH 0569/1352] selftests/mm: restore enable_soft_offline in hugetlb-soft-offline hugetlb-soft-offline toggles /proc/sys/vm/enable_soft_offline between 1 and 0 (test_soft_offline_common(1) then (0)) and leaves it at 0 when it finishes, silently disabling soft offlining for the whole system after the run. Save the original value before the test and restore it from an atexit() handler, as hugepage_restore_settings_atexit() in hugepage_settings.c already does. Use read_num()/write_num() from vm_util instead of hand-rolled popen()/fopen() helpers. The restore handler must not call write_num(): on failure it re-enters exit() through ksft_exit_fail_msg(), which is undefined behavior from inside an atexit handler. A non-root run hits it directly - the restore write fails the same way the write that triggered the exit did. Restore with plain open()/write(), best effort. Link: https://lore.kernel.org/20260825085756.63030-4-husong@kylinos.cn Signed-off-by: Song Hu Signed-off-by: Andrew Morton Reviewed-by: Muhammad Usama Anjum Acked-by: David Hildenbrand (Arm) Cc: Liam R. Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Peter Xu Cc: Sarthak Sharma Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- .../selftests/mm/hugetlb-soft-offline.c | 49 +++++++++++-------- 1 file changed, 28 insertions(+), 21 deletions(-) diff --git a/tools/testing/selftests/mm/hugetlb-soft-offline.c b/tools/testing/selftests/mm/hugetlb-soft-offline.c index bc202e4ed2bda7..4af9d3db7b5b6f 100644 --- a/tools/testing/selftests/mm/hugetlb-soft-offline.c +++ b/tools/testing/selftests/mm/hugetlb-soft-offline.c @@ -11,6 +11,7 @@ #define _GNU_SOURCE #include +#include #include #include #include @@ -23,6 +24,7 @@ #include #include "kselftest.h" +#include "vm_util.h" #include "hugepage_settings.h" #ifndef MADV_SOFT_OFFLINE @@ -31,6 +33,8 @@ #define EPREFIX " !!! " +#define ENABLE_SOFT_OFFLINE_PATH "/proc/sys/vm/enable_soft_offline" + static int do_soft_offline(int fd, size_t len, int expect_errno) { char *filemap = NULL; @@ -77,26 +81,29 @@ static int do_soft_offline(int fd, size_t len, int expect_errno) return ret; } -static int set_enable_soft_offline(int value) -{ - char cmd[256] = {0}; - FILE *cmdfile = NULL; - - if (value != 0 && value != 1) - return -EINVAL; +static unsigned long orig_enable_soft_offline = -1UL; - sprintf(cmd, "echo %d > /proc/sys/vm/enable_soft_offline", value); - cmdfile = popen(cmd, "r"); +/* + * Runs from an atexit handler, so it must not call anything that + * exits on failure: write_num() would re-enter exit() through + * ksft_exit_fail_msg(). + */ +static void restore_enable_soft_offline(void) +{ + char buf[24]; + int fd, len; - if (cmdfile) - ksft_print_msg("enable_soft_offline => %d\n", value); - else { - ksft_perror(EPREFIX "failed to set enable_soft_offline"); - return errno; - } + if (orig_enable_soft_offline == -1UL) + return; - pclose(cmdfile); - return 0; + len = snprintf(buf, sizeof(buf), "%lu", orig_enable_soft_offline); + fd = open(ENABLE_SOFT_OFFLINE_PATH, O_WRONLY); + if (fd < 0) + return; + if (write(fd, buf, len) != len) + ksft_print_msg("failed to restore enable_soft_offline: %s\n", + strerror(errno)); + close(fd); } static int create_hugetlbfs_file(struct statfs *file_stat) @@ -145,10 +152,7 @@ static void test_soft_offline_common(int enable_soft_offline) hugepagesize_kb = file_stat.f_bsize / 1024; ksft_print_msg("Hugepagesize is %ldkB\n", hugepagesize_kb); - if (set_enable_soft_offline(enable_soft_offline) != 0) { - close(fd); - ksft_exit_fail_msg("Failed to set enable_soft_offline\n"); - } + write_num(ENABLE_SOFT_OFFLINE_PATH, enable_soft_offline); nr_hugepages_before = hugetlb_nr_default_pages(); @@ -192,6 +196,9 @@ int main(int argc, char **argv) ksft_set_plan(2); + orig_enable_soft_offline = read_num(ENABLE_SOFT_OFFLINE_PATH); + atexit(restore_enable_soft_offline); + test_soft_offline_common(1); test_soft_offline_common(0); From 1b456e88f76b44448b4a02e3dad7c3c04c67748e Mon Sep 17 00:00:00 2001 From: Longlong Xia Date: Sun, 23 Aug 2026 12:40:52 +0800 Subject: [PATCH 0570/1352] mm/hugetlb: warn instead of silently bailing gigantic pages without runtime support remove_hugetlb_folio() and __update_and_free_hugetlb_folio() silently return for gigantic hstates that lack runtime freeing support. All callers should already filter such hstates upstream, so turn the silent bail into a VM_WARN_ON_ONCE to catch caller regressions instead of masking them. Link: https://lore.kernel.org/20260823044118.1097121-3-xialonglong2025@163.com Signed-off-by: Longlong Xia Signed-off-by: Andrew Morton Acked-by: Muchun Song Assisted-by: Codex:gpt-5.6-sol Cc: David Hildenbrand Cc: Miaohe Lin Cc: Michal Hocko Cc: Oscar Salvador --- mm/hugetlb.c | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 74de391c39730d..03182cc28a7dbc 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -1396,8 +1396,11 @@ void remove_hugetlb_folio(struct hstate *h, struct folio *folio, VM_BUG_ON_FOLIO(hugetlb_cgroup_from_folio_rsvd(folio), folio); lockdep_assert_held(&hugetlb_lock); - if (hstate_is_gigantic_no_runtime(h)) + if (hstate_is_gigantic_no_runtime(h)) { + /* Callers must filter gigantic_no_runtime upstream. */ + VM_WARN_ON_ONCE(1); return; + } list_del(&folio->lru); @@ -1458,8 +1461,11 @@ static void __update_and_free_hugetlb_folio(struct hstate *h, { bool clear_flag = folio_test_hugetlb_vmemmap_optimized(folio); - if (hstate_is_gigantic_no_runtime(h)) + if (hstate_is_gigantic_no_runtime(h)) { + /* Callers must filter gigantic_no_runtime upstream. */ + VM_WARN_ON_ONCE(1); return; + } /* * If we don't know which subpages are hwpoisoned, we can't free From 2929b3289f4a6bea374b541ce706dbf011ace4c9 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Thu, 3 Sep 2026 12:28:27 +0300 Subject: [PATCH 0571/1352] set_memory: add number of pages parameter to set_direct_map APIs Patch series "arch, mm/execmem: resolve confusion about set_direct_map_valid_noflush()", v3. Recent discussion about implementation of execmem's ROX caches on arm64 revealed a confusion about how set_direct_map_valid_noflush() implemented on different architectures. On arm64 it sets or clears the PTE_VALID bit marking a PTE as present or not present. On other architectures it's a range version of set_direct_map_invalid_noflush() and set_direct_map_default_noflush() Unlike arm64::set_direct_map_valid_noflush(), set_direct_map_default_noflush() not only marks PTE as present, but also sets its default protection mode. Other than that, initial design of execmem ROX caches didn't rely on restoration of large mappings that's now available on x86, but completely removed the memory allocated for the ROX cache from the direct map to ensure that large mappings are not split. This precluded usage of VM_FLUSH_RESET_PERMS for the ROX cache allocations and required execmem to implement manipulation of the direct map alias. Current implementation of ROX caches does not remove the direct map alias but simply calls set_memory_rox() that updates the permissions in both vmalloc address space and the direct map and relies on collapse_large_pages() in x86 CPA to keep large mappings. This allow using VM_FLUSH_RESET_PERMS for execmem ROX cache allocations with small adjustments to set_direct_map APIs and vmalloc::reset_perms() behaviour: adding number of pages parameter to set_direct_map APIs and making resetting of the direct map permissions in vmalloc VMAP_HUGE friendly. Implement these adjustments, make execmem always use VM_FLUSH_RESET_PERMS and revert set_direct_map_valid_noflush() changes. This patch (of 6): When set_direct_map APIs were introduced by the commit d253ca0c3865 ("x86/mm/cpa: Add set_direct_map_*() functions") the single page parameter made sense because the initial callers (vmalloc and hibernation) had sets of unsorted struct pages that required changes of their mappings in the direct map. Since there is an increasing demand for direct map manipulation and it is also desirable to be able to update larger physically contiguous mappings, for example an entire large folio, extend set_direct_map APIs to receive number of pages parameter. As there is still only a handful of callers, change the existing functions directly and update all the call sites rather than adding wrappers for single page case. Link: https://lore.kernel.org/20260903-execmem-set-vm-perms-v0-2-v3-0-949b64a9f755@kernel.org Link: https://lore.kernel.org/20260903-execmem-set-vm-perms-v0-2-v3-1-949b64a9f755@kernel.org Link: https://lore.kernel.org/all/20260611130144.1385343-4-abarnas@google.com [1] Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Kevin Brodsky Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Andy Lutomirski Cc: "Borislav Petkov (AMD)" Cc: Brendan Jackman Cc: Catalin Marinas Cc: Christian Borntraeger Cc: Dave Hansen Cc: Gerald Schaefer Cc: Heiko Carstens Cc: "H. Peter Anvin" Cc: Huacai Chen Cc: Ingo Molnar Cc: Len Brown Cc: Palmer Dabbelt Cc: Peter Zijlstra Cc: "Rafael J. Wysocki" Cc: Ryan Roberts Cc: Sven Schnelle Cc: "Uladzislau Rezki (Sony)" Cc: Vasily Gorbik Cc: WANG Xuerui Cc: Will Deacon Cc: Dev Jain --- arch/arm64/include/asm/set_memory.h | 4 ++-- arch/arm64/mm/pageattr.c | 8 ++++---- arch/loongarch/include/asm/set_memory.h | 4 ++-- arch/loongarch/mm/pageattr.c | 8 ++++---- arch/riscv/include/asm/set_memory.h | 4 ++-- arch/riscv/mm/pageattr.c | 8 ++++---- arch/s390/include/asm/set_memory.h | 4 ++-- arch/s390/mm/pageattr.c | 8 ++++---- arch/x86/include/asm/set_memory.h | 4 ++-- arch/x86/mm/pat/set_memory.c | 8 ++++---- include/linux/set_memory.h | 6 ++++-- kernel/power/snapshot.c | 4 ++-- mm/secretmem.c | 6 +++--- mm/vmalloc.c | 5 +++-- 14 files changed, 42 insertions(+), 39 deletions(-) diff --git a/arch/arm64/include/asm/set_memory.h b/arch/arm64/include/asm/set_memory.h index 90f61b17275e1b..b07fd4e026eac0 100644 --- a/arch/arm64/include/asm/set_memory.h +++ b/arch/arm64/include/asm/set_memory.h @@ -11,8 +11,8 @@ bool can_set_direct_map(void); int set_memory_valid(unsigned long addr, int numpages, int enable); -int set_direct_map_invalid_noflush(struct page *page); -int set_direct_map_default_noflush(struct page *page); +int set_direct_map_invalid_noflush(struct page *page, unsigned int numpages); +int set_direct_map_default_noflush(struct page *page, unsigned int numpages); int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); bool kernel_page_present(struct page *page); diff --git a/arch/arm64/mm/pageattr.c b/arch/arm64/mm/pageattr.c index bbe98ac9ad8c67..db8d60a84d1441 100644 --- a/arch/arm64/mm/pageattr.c +++ b/arch/arm64/mm/pageattr.c @@ -251,7 +251,7 @@ int set_memory_valid(unsigned long addr, int numpages, int enable) __pgprot(PTE_PRESENT_VALID_KERNEL)); } -int set_direct_map_invalid_noflush(struct page *page) +int set_direct_map_invalid_noflush(struct page *page, unsigned int numpages) { pgprot_t clear_mask = __pgprot(PTE_PRESENT_VALID_KERNEL); pgprot_t set_mask = __pgprot(PTE_PRESENT_INVALID); @@ -260,10 +260,10 @@ int set_direct_map_invalid_noflush(struct page *page) return 0; return update_range_prot((unsigned long)page_address(page), - PAGE_SIZE, set_mask, clear_mask); + PAGE_SIZE * numpages, set_mask, clear_mask); } -int set_direct_map_default_noflush(struct page *page) +int set_direct_map_default_noflush(struct page *page, unsigned int numpages) { pgprot_t set_mask = __pgprot(PTE_PRESENT_VALID_KERNEL | PTE_WRITE); pgprot_t clear_mask = __pgprot(PTE_PRESENT_INVALID | PTE_RDONLY); @@ -272,7 +272,7 @@ int set_direct_map_default_noflush(struct page *page) return 0; return update_range_prot((unsigned long)page_address(page), - PAGE_SIZE, set_mask, clear_mask); + PAGE_SIZE * numpages, set_mask, clear_mask); } static int __set_memory_enc_dec(unsigned long addr, diff --git a/arch/loongarch/include/asm/set_memory.h b/arch/loongarch/include/asm/set_memory.h index 55dfaefd02c8a6..563aab92896e9b 100644 --- a/arch/loongarch/include/asm/set_memory.h +++ b/arch/loongarch/include/asm/set_memory.h @@ -15,8 +15,8 @@ int set_memory_ro(unsigned long addr, int numpages); int set_memory_rw(unsigned long addr, int numpages); bool kernel_page_present(struct page *page); -int set_direct_map_default_noflush(struct page *page); -int set_direct_map_invalid_noflush(struct page *page); +int set_direct_map_default_noflush(struct page *page, unsigned int nr); +int set_direct_map_invalid_noflush(struct page *page, unsigned int nr); int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); #endif /* _ASM_LOONGARCH_SET_MEMORY_H */ diff --git a/arch/loongarch/mm/pageattr.c b/arch/loongarch/mm/pageattr.c index 614ccc7afccbea..43ad2a104f19df 100644 --- a/arch/loongarch/mm/pageattr.c +++ b/arch/loongarch/mm/pageattr.c @@ -198,24 +198,24 @@ bool kernel_page_present(struct page *page) return pte_present(ptep_get(pte)); } -int set_direct_map_default_noflush(struct page *page) +int set_direct_map_default_noflush(struct page *page, unsigned int nr) { unsigned long addr = (unsigned long)page_address(page); if (addr < vm_map_base) return 0; - return __set_memory(addr, 1, PAGE_KERNEL, __pgprot(0)); + return __set_memory(addr, nr, PAGE_KERNEL, __pgprot(0)); } -int set_direct_map_invalid_noflush(struct page *page) +int set_direct_map_invalid_noflush(struct page *page, unsigned int nr) { unsigned long addr = (unsigned long)page_address(page); if (addr < vm_map_base) return 0; - return __set_memory(addr, 1, __pgprot(0), __pgprot(_PAGE_PRESENT | _PAGE_VALID)); + return __set_memory(addr, nr, __pgprot(0), __pgprot(_PAGE_PRESENT | _PAGE_VALID)); } int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid) diff --git a/arch/riscv/include/asm/set_memory.h b/arch/riscv/include/asm/set_memory.h index ef59e1716a2cfd..db1d0ed82b6962 100644 --- a/arch/riscv/include/asm/set_memory.h +++ b/arch/riscv/include/asm/set_memory.h @@ -40,8 +40,8 @@ static inline int set_kernel_memory(char *startp, char *endp, } #endif -int set_direct_map_invalid_noflush(struct page *page); -int set_direct_map_default_noflush(struct page *page); +int set_direct_map_invalid_noflush(struct page *page, unsigned int nr); +int set_direct_map_default_noflush(struct page *page, unsigned int nr); int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); bool kernel_page_present(struct page *page); diff --git a/arch/riscv/mm/pageattr.c b/arch/riscv/mm/pageattr.c index 3f76db3d276992..20ef95b1d0c36e 100644 --- a/arch/riscv/mm/pageattr.c +++ b/arch/riscv/mm/pageattr.c @@ -374,15 +374,15 @@ int set_memory_nx(unsigned long addr, int numpages) return __set_memory(addr, numpages, __pgprot(0), __pgprot(_PAGE_EXEC)); } -int set_direct_map_invalid_noflush(struct page *page) +int set_direct_map_invalid_noflush(struct page *page, unsigned int nr) { - return __set_memory((unsigned long)page_address(page), 1, + return __set_memory((unsigned long)page_address(page), nr, __pgprot(0), __pgprot(_PAGE_PRESENT)); } -int set_direct_map_default_noflush(struct page *page) +int set_direct_map_default_noflush(struct page *page, unsigned int nr) { - return __set_memory((unsigned long)page_address(page), 1, + return __set_memory((unsigned long)page_address(page), nr, PAGE_KERNEL, __pgprot(_PAGE_EXEC)); } diff --git a/arch/s390/include/asm/set_memory.h b/arch/s390/include/asm/set_memory.h index 94092f4ae76499..6b0aa9147ed8e6 100644 --- a/arch/s390/include/asm/set_memory.h +++ b/arch/s390/include/asm/set_memory.h @@ -60,8 +60,8 @@ __SET_MEMORY_FUNC(set_memory_rox, SET_MEMORY_RO | SET_MEMORY_X) __SET_MEMORY_FUNC(set_memory_rwnx, SET_MEMORY_RW | SET_MEMORY_NX) __SET_MEMORY_FUNC(set_memory_4k, SET_MEMORY_4K) -int set_direct_map_invalid_noflush(struct page *page); -int set_direct_map_default_noflush(struct page *page); +int set_direct_map_invalid_noflush(struct page *page, unsigned int nr); +int set_direct_map_default_noflush(struct page *page, unsigned int nr); int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); bool kernel_page_present(struct page *page); diff --git a/arch/s390/mm/pageattr.c b/arch/s390/mm/pageattr.c index 1e202e3d08e75f..7549543d624125 100644 --- a/arch/s390/mm/pageattr.c +++ b/arch/s390/mm/pageattr.c @@ -382,14 +382,14 @@ int __set_memory(unsigned long addr, unsigned long numpages, unsigned long flags return rc; } -int set_direct_map_invalid_noflush(struct page *page) +int set_direct_map_invalid_noflush(struct page *page, unsigned int nr) { - return __set_memory((unsigned long)page_to_virt(page), 1, SET_MEMORY_INV); + return __set_memory((unsigned long)page_to_virt(page), nr, SET_MEMORY_INV); } -int set_direct_map_default_noflush(struct page *page) +int set_direct_map_default_noflush(struct page *page, unsigned int nr) { - return __set_memory((unsigned long)page_to_virt(page), 1, SET_MEMORY_DEF); + return __set_memory((unsigned long)page_to_virt(page), nr, SET_MEMORY_DEF); } int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid) diff --git a/arch/x86/include/asm/set_memory.h b/arch/x86/include/asm/set_memory.h index 4362c26aa992db..0c4235d159f483 100644 --- a/arch/x86/include/asm/set_memory.h +++ b/arch/x86/include/asm/set_memory.h @@ -86,8 +86,8 @@ int set_pages_wb(struct page *page, int numpages); int set_pages_ro(struct page *page, int numpages); int set_pages_rw(struct page *page, int numpages); -int set_direct_map_invalid_noflush(struct page *page); -int set_direct_map_default_noflush(struct page *page); +int set_direct_map_invalid_noflush(struct page *page, unsigned int nr); +int set_direct_map_default_noflush(struct page *page, unsigned int nr); int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); bool kernel_page_present(struct page *page); diff --git a/arch/x86/mm/pat/set_memory.c b/arch/x86/mm/pat/set_memory.c index 4652487b5572b0..d36a58ae94db64 100644 --- a/arch/x86/mm/pat/set_memory.c +++ b/arch/x86/mm/pat/set_memory.c @@ -2685,14 +2685,14 @@ static int __set_pages_np(struct page *page, int numpages, unsigned int cpa_flag return __change_page_attr_set_clr(&cpa, 1); } -int set_direct_map_invalid_noflush(struct page *page) +int set_direct_map_invalid_noflush(struct page *page, unsigned int nr) { - return __set_pages_np(page, 1, 0); + return __set_pages_np(page, nr, 0); } -int set_direct_map_default_noflush(struct page *page) +int set_direct_map_default_noflush(struct page *page, unsigned int nr) { - return __set_pages_p(page, 1, 0); + return __set_pages_p(page, nr, 0); } int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid) diff --git a/include/linux/set_memory.h b/include/linux/set_memory.h index 3030d9245f5ac8..0b77f1d7d8b9ce 100644 --- a/include/linux/set_memory.h +++ b/include/linux/set_memory.h @@ -25,11 +25,13 @@ static inline int set_memory_rox(unsigned long addr, int numpages) #endif #ifndef CONFIG_ARCH_HAS_SET_DIRECT_MAP -static inline int set_direct_map_invalid_noflush(struct page *page) +static inline int set_direct_map_invalid_noflush(struct page *page, + unsigned int nr) { return 0; } -static inline int set_direct_map_default_noflush(struct page *page) +static inline int set_direct_map_default_noflush(struct page *page, + unsigned int nr) { return 0; } diff --git a/kernel/power/snapshot.c b/kernel/power/snapshot.c index b209712cb2c3ac..d5dba0e50b2eb6 100644 --- a/kernel/power/snapshot.c +++ b/kernel/power/snapshot.c @@ -88,7 +88,7 @@ static inline int hibernate_restore_unprotect_page(void *page_address) {return 0 static inline void hibernate_map_page(struct page *page) { if (IS_ENABLED(CONFIG_ARCH_HAS_SET_DIRECT_MAP)) { - int ret = set_direct_map_default_noflush(page); + int ret = set_direct_map_default_noflush(page, 1); if (ret) pr_warn_once("Failed to remap page\n"); @@ -101,7 +101,7 @@ static inline void hibernate_unmap_page(struct page *page) { if (IS_ENABLED(CONFIG_ARCH_HAS_SET_DIRECT_MAP)) { unsigned long addr = (unsigned long)page_address(page); - int ret = set_direct_map_invalid_noflush(page); + int ret = set_direct_map_invalid_noflush(page, 1); if (ret) pr_warn_once("Failed to remap page\n"); diff --git a/mm/secretmem.c b/mm/secretmem.c index 384f5cfc457f9e..6cbb8efc994a4d 100644 --- a/mm/secretmem.c +++ b/mm/secretmem.c @@ -139,7 +139,7 @@ static vm_fault_t secretmem_fault(struct vm_fault *vmf) goto out; } - err = set_direct_map_invalid_noflush(folio_page(folio, 0)); + err = set_direct_map_invalid_noflush(folio_page(folio, 0), 1); if (err) { secretmem_unaccount_folio(state, folio); folio_put(folio); @@ -156,7 +156,7 @@ static vm_fault_t secretmem_fault(struct vm_fault *vmf) * already happened when we marked the page invalid * which guarantees that this call won't fail */ - set_direct_map_default_noflush(folio_page(folio, 0)); + set_direct_map_default_noflush(folio_page(folio, 0), 1); folio_put(folio); if (err == -EEXIST) goto retry; @@ -228,7 +228,7 @@ static int secretmem_migrate_folio(struct address_space *mapping, static void secretmem_free_folio(struct folio *folio) { - set_direct_map_default_noflush(folio_page(folio, 0)); + set_direct_map_default_noflush(folio_page(folio, 0), 1); folio_zero_segment(folio, 0, folio_size(folio)); } diff --git a/mm/vmalloc.c b/mm/vmalloc.c index b879260d31a577..33ffc978ccbe8a 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -3371,14 +3371,15 @@ struct vm_struct *remove_vm_area(const void *addr) } static inline void set_area_direct_map(const struct vm_struct *area, - int (*set_direct_map)(struct page *page)) + int (*set_direct_map)(struct page *page, + unsigned int nr)) { unsigned long i; /* HUGE_VMALLOC passes small pages to set_direct_map */ for (i = 0; i < area->nr_pages; i++) if (page_address(area->pages[i])) - set_direct_map(area->pages[i]); + set_direct_map(area->pages[i], 1); } /* From dece4da4d8f4d666daf49faada002740407f89e6 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Thu, 3 Sep 2026 12:28:28 +0300 Subject: [PATCH 0572/1352] mm/vmalloc: set area's page_order after allocation succeeds __vmalloc_area_node() calls set_vm_area_page_order() to set area's page_order before actually allocating pages to populate the area. If allocation of large pages in HUGE_VMAP case fails midway, this leaves the area with elevated page_order throughout the cleanup path. There is no actual issue with this because the only place that currently relies on area->page_order on the cleanup path is the loop calculating the direct map alias range in vm_reset_perms() and it anyway skips unpopulated pages. But having set_vm_area_page_order() in the middle of __vmalloc_area_node() makes things very obscure, hard to reason about and error prone against future changes of the cleanup path. Move the call to set_vm_area_page_order() just before the successful return from __vmalloc_area_node() where page order is guaranteed. While on it, initialize local page_order variable with its declaration. Link: https://lore.kernel.org/20260903-execmem-set-vm-perms-v0-2-v3-2-949b64a9f755@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Reviewed-by: Uladzislau Rezki (Sony) Reviewed-by: Dev Jain Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Andy Lutomirski Cc: "Borislav Petkov (AMD)" Cc: Brendan Jackman Cc: Catalin Marinas Cc: Christian Borntraeger Cc: Dave Hansen Cc: David Hildenbrand Cc: Gerald Schaefer Cc: Heiko Carstens Cc: "H. Peter Anvin" Cc: Huacai Chen Cc: Ingo Molnar Cc: Len Brown Cc: Palmer Dabbelt Cc: Peter Zijlstra Cc: "Rafael J. Wysocki" Cc: Ryan Roberts Cc: Sven Schnelle Cc: Vasily Gorbik Cc: WANG Xuerui Cc: Will Deacon Cc: Kevin Brodsky --- mm/vmalloc.c | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index 33ffc978ccbe8a..67c0fba2470f1b 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -3886,7 +3886,7 @@ static void *__vmalloc_area_node(struct vm_struct *area, gfp_t gfp_mask, unsigned long size = get_vm_area_size(area); unsigned long array_size; unsigned long nr_small_pages = size >> PAGE_SHIFT; - unsigned int page_order; + unsigned int page_order = page_shift - PAGE_SHIFT; unsigned int flags; int ret; @@ -3914,9 +3914,6 @@ static void *__vmalloc_area_node(struct vm_struct *area, gfp_t gfp_mask, goto fail; } - set_vm_area_page_order(area, page_shift - PAGE_SHIFT); - page_order = vm_area_page_order(area); - /* * High-order nofail allocations are really expensive and * potentially dangerous (pre-mature OOM, disruptive reclaim @@ -3971,6 +3968,7 @@ static void *__vmalloc_area_node(struct vm_struct *area, gfp_t gfp_mask, goto fail; } + set_vm_area_page_order(area, page_order); return area->addr; fail: From 1e4523da596553e110dc7dc026d1d3ea4af4c556 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Thu, 3 Sep 2026 12:28:29 +0300 Subject: [PATCH 0573/1352] mm/vmalloc: constify vm parameter of get_vm_area_page_order() get_vm_area_page_order() and vm_area_page_order() do not need to modify struct vm_struct passed to them. Constify the parameter. Link: https://lore.kernel.org/20260903-execmem-set-vm-perms-v0-2-v3-3-949b64a9f755@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Reviewed-by: Uladzislau Rezki (Sony) Reviewed-by: Dev Jain Reviewed-by: David Hildenbrand (Arm) Reviewed-by: Kevin Brodsky Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Andy Lutomirski Cc: "Borislav Petkov (AMD)" Cc: Brendan Jackman Cc: Catalin Marinas Cc: Christian Borntraeger Cc: Dave Hansen Cc: Gerald Schaefer Cc: Heiko Carstens Cc: "H. Peter Anvin" Cc: Huacai Chen Cc: Ingo Molnar Cc: Len Brown Cc: Palmer Dabbelt Cc: Peter Zijlstra Cc: "Rafael J. Wysocki" Cc: Ryan Roberts Cc: Sven Schnelle Cc: Vasily Gorbik Cc: WANG Xuerui Cc: Will Deacon --- mm/vmalloc.c | 4 ++-- mm/vmalloc.h | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index 67c0fba2470f1b..b6dd287a0be33b 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -3137,7 +3137,7 @@ EXPORT_SYMBOL(vm_map_ram); static struct vm_struct *vmlist __initdata; -static inline unsigned int vm_area_page_order(struct vm_struct *vm) +static inline unsigned int vm_area_page_order(const struct vm_struct *vm) { #ifdef CONFIG_HAVE_ARCH_HUGE_VMALLOC return vm->page_order; @@ -3146,7 +3146,7 @@ static inline unsigned int vm_area_page_order(struct vm_struct *vm) #endif } -unsigned int get_vm_area_page_order(struct vm_struct *vm) +unsigned int get_vm_area_page_order(const struct vm_struct *vm) { return vm_area_page_order(vm); } diff --git a/mm/vmalloc.h b/mm/vmalloc.h index 8866ddcff6681b..211869f365095f 100644 --- a/mm/vmalloc.h +++ b/mm/vmalloc.h @@ -12,7 +12,7 @@ void __init vmalloc_init(void); int __must_check vmap_pages_range_noflush(unsigned long addr, unsigned long end, pgprot_t prot, struct page **pages, unsigned int page_shift, gfp_t gfp_mask); -unsigned int get_vm_area_page_order(struct vm_struct *vm); +unsigned int get_vm_area_page_order(const struct vm_struct *vm); #else static inline void vmalloc_init(void) {} From 3b875d93e7d4fa8d869f9f1e4656a865e28a0b0e Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Thu, 3 Sep 2026 12:28:30 +0300 Subject: [PATCH 0574/1352] mm/vmalloc: make set_area_direct_map HUGE_VMAP friendly set_area_direct_map() always updates direct map alias permissions in single page increments. For HUGE_VMAP areas it's suboptimal. Not only the loop in set_area_direct_map() needlessly has more iterations (e.g times 512 on x86), but it also causes fragmentation of the direct map that could be avoided for the HUGE_VMAP areas populated with large pages. All pages in an area are always of the same order: either same-order large pages when VM_ALLOW_HUGE_VMAP is set and all huge pages were successfully allocated, or order-0 page when VM_ALLOW_HUGE_VMAP is cleared or when huge pages allocation fails and fallback path is taken. Instead of updating the direct map permissions for every order-0 page in an area, use the area's page_order as the loop increment and update the large pages in one call to set_direct_map_{invalid,default}_noflush(). Link: https://lore.kernel.org/20260903-execmem-set-vm-perms-v0-2-v3-4-949b64a9f755@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Reviewed-by: Dev Jain Reviewed-by: Kevin Brodsky Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Andy Lutomirski Cc: "Borislav Petkov (AMD)" Cc: Brendan Jackman Cc: Catalin Marinas Cc: Christian Borntraeger Cc: Dave Hansen Cc: David Hildenbrand Cc: Gerald Schaefer Cc: Heiko Carstens Cc: "H. Peter Anvin" Cc: Huacai Chen Cc: Ingo Molnar Cc: Len Brown Cc: Palmer Dabbelt Cc: Peter Zijlstra Cc: "Rafael J. Wysocki" Cc: Ryan Roberts Cc: Sven Schnelle Cc: "Uladzislau Rezki (Sony)" Cc: Vasily Gorbik Cc: WANG Xuerui Cc: Will Deacon --- mm/vmalloc.c | 13 ++++++++----- 1 file changed, 8 insertions(+), 5 deletions(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index b6dd287a0be33b..db357a9bdd1250 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -3374,12 +3374,15 @@ static inline void set_area_direct_map(const struct vm_struct *area, int (*set_direct_map)(struct page *page, unsigned int nr)) { - unsigned long i; + unsigned int nr = (1U << vm_area_page_order(area)); + + for (unsigned long i = 0; i < area->nr_pages; i += nr) { + if (page_address(area->pages[i])) { + int err = set_direct_map(area->pages[i], nr); - /* HUGE_VMALLOC passes small pages to set_direct_map */ - for (i = 0; i < area->nr_pages; i++) - if (page_address(area->pages[i])) - set_direct_map(area->pages[i], 1); + WARN_ON_ONCE(err); + } + } } /* From f6281c2d8b0ba96ee6db1134c142d7ab489cca2b Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Thu, 3 Sep 2026 12:28:31 +0300 Subject: [PATCH 0575/1352] mm/execmem: use VM_FLUSH_RESET_PERMS for ROX cache allocations Initially execmem completely removed direct map alias for the memory allocated for the ROX cache in PMD_SIZE chunks. When that memory was freed, its direct map was restored also in PMD_SIZE chunks to avoid fragmentation of the direct map caused by vmalloc::vm_reset_perms(). This required execmem to implement the wrappers for set_direct_map APIs for proper sequencing of removal and restoration of the direct map aliases. Since then x86's CPA gained support for collapsing the direct map page tables for ROX pages and execmem switched from removing ROX caches from the direct map to making them ROX there, so execmem only needs to update direct map alias permissions when freeing the ROX cache memory. vmalloc already handles those updates for areas with VM_FLUSH_RESET_PERMS set and vmalloc::vm_reset_perms() does not force split of the direct map for PMD_SIZE chunks. Set the area permissions with set_vm_flush_reset_perms() when populating the execmem cache just before flipping the area to ROX. This way freeing an allocated area on an error path won't incur two updates of the direct map alias of that area and TLB flushing in vm_reset_perms(). Link: https://lore.kernel.org/20260903-execmem-set-vm-perms-v0-2-v3-5-949b64a9f755@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Reviewed-by: Kevin Brodsky Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Andy Lutomirski Cc: "Borislav Petkov (AMD)" Cc: Brendan Jackman Cc: Catalin Marinas Cc: Christian Borntraeger Cc: Dave Hansen Cc: David Hildenbrand Cc: Gerald Schaefer Cc: Heiko Carstens Cc: "H. Peter Anvin" Cc: Huacai Chen Cc: Ingo Molnar Cc: Len Brown Cc: Palmer Dabbelt Cc: Peter Zijlstra Cc: "Rafael J. Wysocki" Cc: Ryan Roberts Cc: Sven Schnelle Cc: "Uladzislau Rezki (Sony)" Cc: Vasily Gorbik Cc: WANG Xuerui Cc: Will Deacon Cc: Dev Jain --- mm/execmem.c | 40 +++++++--------------------------------- 1 file changed, 7 insertions(+), 33 deletions(-) diff --git a/mm/execmem.c b/mm/execmem.c index 74a178a87e7581..ad07cae9ed5854 100644 --- a/mm/execmem.c +++ b/mm/execmem.c @@ -113,28 +113,6 @@ static inline unsigned long mas_range_len(struct ma_state *mas) return mas->last - mas->index + 1; } -static int execmem_set_direct_map_valid(struct vm_struct *vm, bool valid) -{ - unsigned int nr = (1 << get_vm_area_page_order(vm)); - unsigned int updated = 0; - int err = 0; - - for (int i = 0; i < vm->nr_pages; i += nr) { - err = set_direct_map_valid_noflush(vm->pages[i], nr, valid); - if (err) - goto err_restore; - updated += nr; - } - - return 0; - -err_restore: - for (int i = 0; i < updated; i += nr) - set_direct_map_valid_noflush(vm->pages[i], nr, !valid); - - return err; -} - static int execmem_force_rw(void *ptr, size_t size) { unsigned int nr = PAGE_ALIGN(size) >> PAGE_SHIFT; @@ -169,9 +147,6 @@ static void execmem_cache_clean(struct work_struct *work) if (IS_ALIGNED(size, PMD_SIZE) && IS_ALIGNED(mas.index, PMD_SIZE)) { - struct vm_struct *vm = find_vm_area(area); - - execmem_set_direct_map_valid(vm, true); mas_store_gfp(&mas, NULL, GFP_KERNEL); vfree(area); } @@ -301,6 +276,8 @@ static void *execmem_cache_populate_alloc(struct execmem_range *range, size_t si /* fill memory with instructions that will trap */ execmem_fill_trapping_insns(p, alloc_size); + set_vm_flush_reset_perms(p); + err = set_memory_rox((unsigned long)p, vm->nr_pages); if (err) goto err_free_mem; @@ -312,18 +289,15 @@ static void *execmem_cache_populate_alloc(struct execmem_range *range, size_t si */ mutex_lock(mutex); err = execmem_cache_add_locked(p, alloc_size, GFP_KERNEL); - if (err) - goto err_reset_direct_map; - - p = execmem_cache_alloc_locked(range, size); - + if (!err) + p = execmem_cache_alloc_locked(range, size); mutex_unlock(mutex); + if (err) + goto err_free_mem; + return p; -err_reset_direct_map: - mutex_unlock(mutex); - execmem_set_direct_map_valid(vm, true); err_free_mem: vfree(p); return NULL; From 01addb14b9f9e02c7edd79b4763a3596f7ed6135 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Thu, 3 Sep 2026 12:28:32 +0300 Subject: [PATCH 0576/1352] Revert "arch: introduce set_direct_map_valid_noflush()" Commit 0c6378a71574 ("arch: introduce set_direct_map_valid_noflush()") added set_direct_map_valid_noflush() to allow updating the direct map for a physically contiguous range in execmem. As Brendan recently pointed out [1], this API is confusing because on arm64 it means that is sets VALID bit in ptes, while on other architectures it is an analog of set_direct_map_default_noflush(). The only user of set_direct_map_valid_noflush() was execmem's ROX cache freeing path and it was switched to utilize VM_FLUSH_RESET_PERMS for resetting permissions of the direct map alias. With the last user gone and with set_direct_map_{invalid,default}_noflush() accepting number of pages as a parameter, set_direct_map_valid_noflush() become a copy of set_memory_valid() on arm64 and a duplicate of set_direct_map_{invalid,default}_noflush() on other architecture, it is safe to remove set_direct_map_valid_noflush(). Also drop a stale comment in arm64::__kernel_map_pages() that Linus bothered to add when merging changes containing set_direct_map_valid_noflush() to his tree. This reverts commit 0c6378a71574daa6cd1534ad42a956e3262756c7. Link: https://lore.kernel.org/20260903-execmem-set-vm-perms-v0-2-v3-6-949b64a9f755@kernel.org Link: https://lore.kernel.org/all/DJ69RCVRBO0Y.3JCYSW50IC4RC@linux.dev [1] Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Reviewed-by: Brendan Jackman Acked-by: David Hildenbrand (Arm) Reviewed-by: Kevin Brodsky Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Andy Lutomirski Cc: "Borislav Petkov (AMD)" Cc: Catalin Marinas Cc: Christian Borntraeger Cc: Dave Hansen Cc: Gerald Schaefer Cc: Heiko Carstens Cc: "H. Peter Anvin" Cc: Huacai Chen Cc: Ingo Molnar Cc: Len Brown Cc: Palmer Dabbelt Cc: Peter Zijlstra Cc: "Rafael J. Wysocki" Cc: Ryan Roberts Cc: Sven Schnelle Cc: "Uladzislau Rezki (Sony)" Cc: Vasily Gorbik Cc: WANG Xuerui Cc: Will Deacon Cc: Dev Jain --- arch/arm64/include/asm/set_memory.h | 1 - arch/arm64/mm/pageattr.c | 16 ---------------- arch/loongarch/include/asm/set_memory.h | 1 - arch/loongarch/mm/pageattr.c | 19 ------------------- arch/riscv/include/asm/set_memory.h | 1 - arch/riscv/mm/pageattr.c | 15 --------------- arch/s390/include/asm/set_memory.h | 1 - arch/s390/mm/pageattr.c | 12 ------------ arch/x86/include/asm/set_memory.h | 1 - arch/x86/mm/pat/set_memory.c | 8 -------- include/linux/set_memory.h | 6 ------ 11 files changed, 81 deletions(-) diff --git a/arch/arm64/include/asm/set_memory.h b/arch/arm64/include/asm/set_memory.h index b07fd4e026eac0..0091ba12200e68 100644 --- a/arch/arm64/include/asm/set_memory.h +++ b/arch/arm64/include/asm/set_memory.h @@ -13,7 +13,6 @@ int set_memory_valid(unsigned long addr, int numpages, int enable); int set_direct_map_invalid_noflush(struct page *page, unsigned int numpages); int set_direct_map_default_noflush(struct page *page, unsigned int numpages); -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); bool kernel_page_present(struct page *page); int set_memory_encrypted(unsigned long addr, int numpages); diff --git a/arch/arm64/mm/pageattr.c b/arch/arm64/mm/pageattr.c index db8d60a84d1441..132938b32eb16f 100644 --- a/arch/arm64/mm/pageattr.c +++ b/arch/arm64/mm/pageattr.c @@ -355,23 +355,7 @@ int realm_register_memory_enc_ops(void) return arm64_mem_crypt_ops_register(&realm_crypt_ops); } -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid) -{ - unsigned long addr = (unsigned long)page_address(page); - - if (!can_set_direct_map()) - return 0; - - return set_memory_valid(addr, nr, valid); -} - #ifdef CONFIG_DEBUG_PAGEALLOC -/* - * This is - apart from the return value - doing the same - * thing as the new set_direct_map_valid_noflush() function. - * - * Unify? Explain the conceptual differences? - */ void __kernel_map_pages(struct page *page, int numpages, int enable) { if (!can_set_direct_map()) diff --git a/arch/loongarch/include/asm/set_memory.h b/arch/loongarch/include/asm/set_memory.h index 563aab92896e9b..4bb01172fbc245 100644 --- a/arch/loongarch/include/asm/set_memory.h +++ b/arch/loongarch/include/asm/set_memory.h @@ -17,6 +17,5 @@ int set_memory_rw(unsigned long addr, int numpages); bool kernel_page_present(struct page *page); int set_direct_map_default_noflush(struct page *page, unsigned int nr); int set_direct_map_invalid_noflush(struct page *page, unsigned int nr); -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); #endif /* _ASM_LOONGARCH_SET_MEMORY_H */ diff --git a/arch/loongarch/mm/pageattr.c b/arch/loongarch/mm/pageattr.c index 43ad2a104f19df..a7dcff40f75982 100644 --- a/arch/loongarch/mm/pageattr.c +++ b/arch/loongarch/mm/pageattr.c @@ -217,22 +217,3 @@ int set_direct_map_invalid_noflush(struct page *page, unsigned int nr) return __set_memory(addr, nr, __pgprot(0), __pgprot(_PAGE_PRESENT | _PAGE_VALID)); } - -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid) -{ - unsigned long addr = (unsigned long)page_address(page); - pgprot_t set, clear; - - if (addr < vm_map_base) - return 0; - - if (valid) { - set = PAGE_KERNEL; - clear = __pgprot(0); - } else { - set = __pgprot(0); - clear = __pgprot(_PAGE_PRESENT | _PAGE_VALID); - } - - return __set_memory(addr, nr, set, clear); -} diff --git a/arch/riscv/include/asm/set_memory.h b/arch/riscv/include/asm/set_memory.h index db1d0ed82b6962..e9f9960c194772 100644 --- a/arch/riscv/include/asm/set_memory.h +++ b/arch/riscv/include/asm/set_memory.h @@ -42,7 +42,6 @@ static inline int set_kernel_memory(char *startp, char *endp, int set_direct_map_invalid_noflush(struct page *page, unsigned int nr); int set_direct_map_default_noflush(struct page *page, unsigned int nr); -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); bool kernel_page_present(struct page *page); #endif /* __ASSEMBLER__ */ diff --git a/arch/riscv/mm/pageattr.c b/arch/riscv/mm/pageattr.c index 20ef95b1d0c36e..5b3cf326455db1 100644 --- a/arch/riscv/mm/pageattr.c +++ b/arch/riscv/mm/pageattr.c @@ -386,21 +386,6 @@ int set_direct_map_default_noflush(struct page *page, unsigned int nr) PAGE_KERNEL, __pgprot(_PAGE_EXEC)); } -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid) -{ - pgprot_t set, clear; - - if (valid) { - set = PAGE_KERNEL; - clear = __pgprot(_PAGE_EXEC); - } else { - set = __pgprot(0); - clear = __pgprot(_PAGE_PRESENT); - } - - return __set_memory((unsigned long)page_address(page), nr, set, clear); -} - #ifdef CONFIG_DEBUG_PAGEALLOC static int debug_pagealloc_set_page(pte_t *pte, unsigned long addr, void *data) { diff --git a/arch/s390/include/asm/set_memory.h b/arch/s390/include/asm/set_memory.h index 6b0aa9147ed8e6..e3562bf0c1aa5e 100644 --- a/arch/s390/include/asm/set_memory.h +++ b/arch/s390/include/asm/set_memory.h @@ -62,7 +62,6 @@ __SET_MEMORY_FUNC(set_memory_4k, SET_MEMORY_4K) int set_direct_map_invalid_noflush(struct page *page, unsigned int nr); int set_direct_map_default_noflush(struct page *page, unsigned int nr); -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); bool kernel_page_present(struct page *page); #endif diff --git a/arch/s390/mm/pageattr.c b/arch/s390/mm/pageattr.c index 7549543d624125..80e834e8b8e1b5 100644 --- a/arch/s390/mm/pageattr.c +++ b/arch/s390/mm/pageattr.c @@ -392,18 +392,6 @@ int set_direct_map_default_noflush(struct page *page, unsigned int nr) return __set_memory((unsigned long)page_to_virt(page), nr, SET_MEMORY_DEF); } -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid) -{ - unsigned long flags; - - if (valid) - flags = SET_MEMORY_DEF; - else - flags = SET_MEMORY_INV; - - return __set_memory((unsigned long)page_to_virt(page), nr, flags); -} - bool kernel_page_present(struct page *page) { unsigned long addr; diff --git a/arch/x86/include/asm/set_memory.h b/arch/x86/include/asm/set_memory.h index 0c4235d159f483..39271a5ea92527 100644 --- a/arch/x86/include/asm/set_memory.h +++ b/arch/x86/include/asm/set_memory.h @@ -88,7 +88,6 @@ int set_pages_rw(struct page *page, int numpages); int set_direct_map_invalid_noflush(struct page *page, unsigned int nr); int set_direct_map_default_noflush(struct page *page, unsigned int nr); -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); bool kernel_page_present(struct page *page); extern int kernel_set_to_readonly; diff --git a/arch/x86/mm/pat/set_memory.c b/arch/x86/mm/pat/set_memory.c index d36a58ae94db64..7b6d983088ea9a 100644 --- a/arch/x86/mm/pat/set_memory.c +++ b/arch/x86/mm/pat/set_memory.c @@ -2695,14 +2695,6 @@ int set_direct_map_default_noflush(struct page *page, unsigned int nr) return __set_pages_p(page, nr, 0); } -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid) -{ - if (valid) - return __set_pages_p(page, nr, 0); - - return __set_pages_np(page, nr, 0); -} - #ifdef CONFIG_DEBUG_PAGEALLOC void __kernel_map_pages(struct page *page, int numpages, int enable) { diff --git a/include/linux/set_memory.h b/include/linux/set_memory.h index 0b77f1d7d8b9ce..3fe293cfed8cc8 100644 --- a/include/linux/set_memory.h +++ b/include/linux/set_memory.h @@ -36,12 +36,6 @@ static inline int set_direct_map_default_noflush(struct page *page, return 0; } -static inline int set_direct_map_valid_noflush(struct page *page, - unsigned nr, bool valid) -{ - return 0; -} - static inline bool kernel_page_present(struct page *page) { return true; From a8eac8feb502c97351530fee0652913fc234bd34 Mon Sep 17 00:00:00 2001 From: Hao Jia Date: Fri, 28 Aug 2026 16:31:49 +0800 Subject: [PATCH 0577/1352] zram: fix idle age_sec underflow in idle_store() After commit 2e8ff2f51dde ("zram: use u32 for entry ac_time tracking"), idle_store() computes the idle cutoff as: cutoff = ktime_sub((u32)ktime_get_boottime_seconds(), age_sec); Because the left operand is cast to u32, when age_sec exceeds the current uptime the subtraction wraps modulo 2^32 and the huge result is zero-extended into the s64 cutoff. mark_idle() then marks every entry as idle instead of matching nothing. For instance, running echo 86400 > /sys/block/zramX/idle on a machine up for only two minutes marks all newly written pages idle and hands them to idle writeback and recompression. No slot can have been accessed before the system booted, so an age_sec that reaches back past uptime cannot match any slot. Return early in that case, without walking the table or taking any slot locks. Track the cutoff as time64_t rather than ktime_t. Both cutoff and ac_time are boot-time values in seconds, so a plain arithmetic comparison against ac_time in mark_idle() is correct and no ktime helpers are needed. Link: https://lore.kernel.org/20260828083149.45760-1-jiahao.kernel@gmail.com Fixes: 2e8ff2f51dde ("zram: use u32 for entry ac_time tracking") Signed-off-by: Hao Jia Signed-off-by: Andrew Morton Suggested-by: Sergey Senozhatsky Cc: Brian Geffon Cc: Jens Axboe Cc: Minchan Kim Cc: --- drivers/block/zram/zram_drv.c | 21 +++++++++++++-------- 1 file changed, 13 insertions(+), 8 deletions(-) diff --git a/drivers/block/zram/zram_drv.c b/drivers/block/zram/zram_drv.c index a9b3bb1d3bef35..4ba0f77b2abd80 100644 --- a/drivers/block/zram/zram_drv.c +++ b/drivers/block/zram/zram_drv.c @@ -415,7 +415,7 @@ static ssize_t mem_used_max_store(struct device *dev, * Mark all pages which are older than or equal to cutoff as IDLE. * Callers should hold the zram init lock in read mode */ -static void mark_idle(struct zram *zram, ktime_t cutoff) +static void mark_idle(struct zram *zram, time64_t cutoff) { int is_idle = 1; unsigned long nr_pages = zram->disksize >> PAGE_SHIFT; @@ -439,7 +439,7 @@ static void mark_idle(struct zram *zram, ktime_t cutoff) #ifdef CONFIG_ZRAM_TRACK_ENTRY_ACTIME is_idle = !cutoff || - ktime_after(cutoff, zram->table[index].attr.ac_time); + cutoff > zram->table[index].attr.ac_time; #endif if (is_idle) set_slot_flag(zram, index, ZRAM_IDLE); @@ -453,21 +453,26 @@ static ssize_t idle_store(struct device *dev, struct device_attribute *attr, const char *buf, size_t len) { struct zram *zram = dev_to_zram(dev); - ktime_t cutoff = 0; + time64_t cutoff = 0; if (!sysfs_streq(buf, "all")) { /* * If it did not parse as 'all' try to treat it as an integer * when we have memory tracking enabled. */ + time64_t uptime; u32 age_sec; - if (IS_ENABLED(CONFIG_ZRAM_TRACK_ENTRY_ACTIME) && - !kstrtouint(buf, 0, &age_sec)) - cutoff = ktime_sub((u32)ktime_get_boottime_seconds(), - age_sec); - else + if (!IS_ENABLED(CONFIG_ZRAM_TRACK_ENTRY_ACTIME) || + kstrtouint(buf, 0, &age_sec)) return -EINVAL; + + /* No slot can be older than the system uptime */ + uptime = ktime_get_boottime_seconds(); + if (age_sec >= uptime) + return len; + + cutoff = uptime - age_sec; } guard(rwsem_read)(&zram->dev_lock); From a6ed4552e77a3dfaea5cc210e06c9a20d1c0797d Mon Sep 17 00:00:00 2001 From: Vernon Yang Date: Wed, 9 Sep 2026 10:58:01 +0800 Subject: [PATCH 0578/1352] mm: khugepaged: fix swap entry value to folio_pfn() Patch series "mm: khugepaged: fix tracepoint UAF", v5. The khugepaged tracepoints take a folio pointer and call folio_pfn(), but by then the folio may no longer be valid: freed after folio_put(), folio_unlock() or pte_unmap_unlock(), or not a folio at all but an xarray-encoded swap entry. On classic SPARSEMEM, dereferencing it oopses khugepaged as soon as the trace event is enabled; on other memory models it merely prints a bogus pfn. Pass the pfn to the tracepoints directly, captured while the folio is still pinned, closing the use-after-free windows in mm_khugepaged_scan_file(), mm_khugepaged_scan_pmd() and mm_khugepaged_collapse_file(). This patch (of 3): When the swap entries found exceed max_ptes_swap, the loop is left via break with folio still holding the xarray value that encodes the swap entry, not valid folio pointer. That value is passed to trace_mm_khugepaged_scan_file(), which feeds it to folio_pfn(). On FLATMEM and SPARSEMEM_VMEMMAP, the page_to_pfn() is plain pointer arithmetic, so the trace event merely prints bogus scan_pfn. On classic SPARSEMEM, the page_to_pfn() reads page->flags, dereferencing the tiny encoded integer and oopsing khugepaged whenever the trace event is enabled. So when folio is the swap entry value, simply set pfn to -1, just like exhausted scan naturally. And the folio_put() has maybe dropped the last reference of folio. The trace_mm_khugepaged_scan_file() is left with a dangling folio pointer. so using the folio_pfn() before dropping the reference, closing use-after-free window. Link: https://lore.kernel.org/20260909025804.3233645-1-vernon2gm@gmail.com Link: https://lore.kernel.org/20260909025804.3233645-2-vernon2gm@gmail.com Fixes: d41fd2016ed0 ("mm/khugepaged: add tracepoint to hpage_collapse_scan_file()") Signed-off-by: Vernon Yang Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Barry Song Cc: Dev Jain Cc: Lance Yang Cc: Lorenzo Stoakes Cc: Ryan Roberts Cc: Zach O'Keefe Cc: --- include/trace/events/huge_memory.h | 6 +++--- mm/khugepaged.c | 7 ++++++- 2 files changed, 9 insertions(+), 4 deletions(-) diff --git a/include/trace/events/huge_memory.h b/include/trace/events/huge_memory.h index 5a48c5406cce49..7b526528f85b80 100644 --- a/include/trace/events/huge_memory.h +++ b/include/trace/events/huge_memory.h @@ -178,10 +178,10 @@ TRACE_EVENT(mm_collapse_huge_page_swapin, TRACE_EVENT(mm_khugepaged_scan_file, - TP_PROTO(struct mm_struct *mm, struct folio *folio, struct file *file, + TP_PROTO(struct mm_struct *mm, unsigned long pfn, struct file *file, int present, int swap, int result), - TP_ARGS(mm, folio, file, present, swap, result), + TP_ARGS(mm, pfn, file, present, swap, result), TP_STRUCT__entry( __field(struct mm_struct *, mm) @@ -194,7 +194,7 @@ TRACE_EVENT(mm_khugepaged_scan_file, TP_fast_assign( __entry->mm = mm; - __entry->pfn = folio ? folio_pfn(folio) : -1; + __entry->pfn = pfn; __assign_str(filename); __entry->present = present; __entry->swap = swap; diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 75639298efc271..6380a12b8eee05 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -2683,6 +2683,7 @@ static enum scan_result collapse_scan_file(struct mm_struct *mm, int present, swap; int node = NUMA_NO_NODE; enum scan_result result = SCAN_SUCCEED; + unsigned long failed_pfn = -1; present = 0; swap = 0; @@ -2715,6 +2716,7 @@ static enum scan_result collapse_scan_file(struct mm_struct *mm, if (is_pmd_order(folio_order(folio))) { result = SCAN_PTE_MAPPED_HUGEPAGE; + failed_pfn = folio_pfn(folio); /* * PMD-sized THP implies that we can only try * retracting the PTE table. @@ -2726,6 +2728,7 @@ static enum scan_result collapse_scan_file(struct mm_struct *mm, node = folio_nid(folio); if (collapse_scan_abort(node, cc)) { result = SCAN_SCAN_ABORT; + failed_pfn = folio_pfn(folio); folio_put(folio); break; } @@ -2733,12 +2736,14 @@ static enum scan_result collapse_scan_file(struct mm_struct *mm, if (!folio_test_lru(folio)) { result = SCAN_PAGE_LRU; + failed_pfn = folio_pfn(folio); folio_put(folio); break; } if (folio_expected_ref_count(folio) + 1 != folio_ref_count(folio)) { result = SCAN_PAGE_COUNT; + failed_pfn = folio_pfn(folio); folio_put(folio); break; } @@ -2773,7 +2778,7 @@ static enum scan_result collapse_scan_file(struct mm_struct *mm, } } - trace_mm_khugepaged_scan_file(mm, folio, file, present, swap, result); + trace_mm_khugepaged_scan_file(mm, failed_pfn, file, present, swap, result); return result; } From c5f29ecac03c1e3e743f233b6e36e2d59e2d1fb3 Mon Sep 17 00:00:00 2001 From: Vernon Yang Date: Wed, 9 Sep 2026 10:58:02 +0800 Subject: [PATCH 0579/1352] mm: khugepaged: fix folio is used after pte_unmap_unlock() After the page table lock has dropped, the folio can be freed concurrently. The trace_mm_khugepaged_scan_pmd() is left with a dangling folio pointer. So using the folio_pfn() before dropping the page table lock, closing use-after-free window. And other pre-existing bug, When the `for (i = 0; i < HPAGE_PMD_NR; i++)` iteration to terminate and the folio operation preceding is normal, but pfn will be incorrect. so we really only trace the PFN if it really was problematic. Link: https://lore.kernel.org/20260909025804.3233645-3-vernon2gm@gmail.com Fixes: 7d2eba0557c1 ("mm: add tracepoint for scanning pages") Signed-off-by: Vernon Yang Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Lorenzo Stoakes (ARM) Cc: Barry Song Cc: Dev Jain Cc: Lance Yang Cc: Ryan Roberts Cc: Zach O'Keefe Cc: --- include/trace/events/huge_memory.h | 6 +++--- mm/khugepaged.c | 10 +++++++++- 2 files changed, 12 insertions(+), 4 deletions(-) diff --git a/include/trace/events/huge_memory.h b/include/trace/events/huge_memory.h index 7b526528f85b80..fa828967e1fb0e 100644 --- a/include/trace/events/huge_memory.h +++ b/include/trace/events/huge_memory.h @@ -55,10 +55,10 @@ SCAN_STATUS TRACE_EVENT(mm_khugepaged_scan_pmd, - TP_PROTO(struct mm_struct *mm, struct folio *folio, + TP_PROTO(struct mm_struct *mm, unsigned long pfn, int referenced, int none_or_zero, int status, int unmapped), - TP_ARGS(mm, folio, referenced, none_or_zero, status, unmapped), + TP_ARGS(mm, pfn, referenced, none_or_zero, status, unmapped), TP_STRUCT__entry( __field(struct mm_struct *, mm) @@ -71,7 +71,7 @@ TRACE_EVENT(mm_khugepaged_scan_pmd, TP_fast_assign( __entry->mm = mm; - __entry->pfn = folio ? folio_pfn(folio) : -1; + __entry->pfn = pfn; __entry->referenced = referenced; __entry->none_or_zero = none_or_zero; __entry->status = status; diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 6380a12b8eee05..730829947b4ce2 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -1612,6 +1612,7 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, enum scan_result result = SCAN_FAIL; struct page *page = NULL; struct folio *folio = NULL; + unsigned long failed_pfn = -1; unsigned long addr; unsigned long enabled_orders; spinlock_t *ptl; @@ -1706,11 +1707,13 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, if (cc->is_khugepaged && !(vma->vm_flags & VM_DROPPABLE) && folio_test_lazyfree(folio) && !pte_dirty(pteval)) { result = SCAN_PAGE_LAZYFREE; + failed_pfn = folio_pfn(folio); goto out_unmap; } if (!folio_test_anon(folio)) { result = SCAN_PAGE_ANON; + failed_pfn = folio_pfn(folio); goto out_unmap; } @@ -1721,6 +1724,7 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, if (folio_maybe_mapped_shared(folio)) { if (++shared > max_ptes_shared) { result = SCAN_EXCEED_SHARED_PTE; + failed_pfn = folio_pfn(folio); count_collapse_event(HPAGE_PMD_ORDER, THP_SCAN_EXCEED_SHARED_PTE, MTHP_STAT_COLLAPSE_EXCEED_SHARED); goto out_unmap; @@ -1738,15 +1742,18 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, node = folio_nid(folio); if (collapse_scan_abort(node, cc)) { result = SCAN_SCAN_ABORT; + failed_pfn = folio_pfn(folio); goto out_unmap; } cc->node_load[node]++; if (!folio_test_lru(folio)) { result = SCAN_PAGE_LRU; + failed_pfn = folio_pfn(folio); goto out_unmap; } if (folio_test_locked(folio)) { result = SCAN_PAGE_LOCK; + failed_pfn = folio_pfn(folio); goto out_unmap; } @@ -1759,6 +1766,7 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, */ if (folio_expected_ref_count(folio) != folio_ref_count(folio)) { result = SCAN_PAGE_COUNT; + failed_pfn = folio_pfn(folio); goto out_unmap; } @@ -1784,7 +1792,7 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, *lock_dropped = true; } out: - trace_mm_khugepaged_scan_pmd(mm, folio, referenced, + trace_mm_khugepaged_scan_pmd(mm, failed_pfn, referenced, none_or_zero, result, unmapped); return result; } From 1e3d476b89a0b76c27be9bc5de2a1db3a88c92fa Mon Sep 17 00:00:00 2001 From: Vernon Yang Date: Wed, 9 Sep 2026 10:58:03 +0800 Subject: [PATCH 0580/1352] mm: khugepaged: fix folio is used after folio_put/unlock() On the rollback path, folio_put() has already dropped the last reference of new_folio. On the success path, new_folio is already unlocked and can be freed concurrently. The trace_mm_khugepaged_collapse_file() is left with a dangling folio pointer. So using the folio_pfn() before dropping the reference, closing use-after-free window. Link: https://lore.kernel.org/20260909025804.3233645-4-vernon2gm@gmail.com Fixes: 4c9473e87e75 ("mm/khugepaged: add tracepoint to collapse_file()") Signed-off-by: Vernon Yang Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Lorenzo Stoakes (ARM) Cc: Barry Song Cc: Dev Jain Cc: Lance Yang Cc: Ryan Roberts Cc: Zach O'Keefe Cc: --- include/trace/events/huge_memory.h | 6 +++--- mm/khugepaged.c | 4 +++- 2 files changed, 6 insertions(+), 4 deletions(-) diff --git a/include/trace/events/huge_memory.h b/include/trace/events/huge_memory.h index fa828967e1fb0e..5fb4d92cfd8408 100644 --- a/include/trace/events/huge_memory.h +++ b/include/trace/events/huge_memory.h @@ -211,10 +211,10 @@ TRACE_EVENT(mm_khugepaged_scan_file, ); TRACE_EVENT(mm_khugepaged_collapse_file, - TP_PROTO(struct mm_struct *mm, struct folio *new_folio, pgoff_t index, + TP_PROTO(struct mm_struct *mm, unsigned long new_pfn, pgoff_t index, unsigned long addr, bool is_shmem, struct file *file, int nr, int result), - TP_ARGS(mm, new_folio, index, addr, is_shmem, file, nr, result), + TP_ARGS(mm, new_pfn, index, addr, is_shmem, file, nr, result), TP_STRUCT__entry( __field(struct mm_struct *, mm) __field(unsigned long, hpfn) @@ -228,7 +228,7 @@ TRACE_EVENT(mm_khugepaged_collapse_file, TP_fast_assign( __entry->mm = mm; - __entry->hpfn = new_folio ? folio_pfn(new_folio) : -1; + __entry->hpfn = new_pfn; __entry->index = index; __entry->addr = addr; __entry->is_shmem = is_shmem; diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 730829947b4ce2..792166950bba81 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -2253,6 +2253,7 @@ static enum scan_result collapse_file(struct mm_struct *mm, unsigned long addr, struct address_space *mapping = file->f_mapping; struct page *dst; struct folio *folio, *tmp, *new_folio; + unsigned long new_pfn = -1; pgoff_t index = 0, end = start + HPAGE_PMD_NR; LIST_HEAD(pagelist); XA_STATE_ORDER(xas, &mapping->i_pages, start, HPAGE_PMD_ORDER); @@ -2272,6 +2273,7 @@ static enum scan_result collapse_file(struct mm_struct *mm, unsigned long addr, result = alloc_charge_folio(&new_folio, mm, cc, HPAGE_PMD_ORDER); if (result != SCAN_SUCCEED) goto out; + new_pfn = folio_pfn(new_folio); mapping_set_update(&xas, mapping); @@ -2675,7 +2677,7 @@ static enum scan_result collapse_file(struct mm_struct *mm, unsigned long addr, folio_put(new_folio); out: VM_BUG_ON(!list_empty(&pagelist)); - trace_mm_khugepaged_collapse_file(mm, new_folio, index, addr, is_shmem, file, HPAGE_PMD_NR, result); + trace_mm_khugepaged_collapse_file(mm, new_pfn, index, addr, is_shmem, file, HPAGE_PMD_NR, result); return result; } From d5fc5bcbbb5afb45dc9d086aba8b6be60f80b37c Mon Sep 17 00:00:00 2001 From: Qi Zheng Date: Mon, 17 Aug 2026 17:03:26 +0800 Subject: [PATCH 0581/1352] mm: memcontrol: make obj_cgroup_memcg() handle NULL objcg Patch series "make unused huge shrinker memcg aware", v4. The shmem unused huge shrinker maintains a per-superblock list of inodes whose tail huge folio extends beyond i_size. Because this list is not memcg aware, reclaim triggered by memcg A can scan inodes across the entire superblock and split huge folios charged to unrelated memcg B, causing unexpected impact on it. In the worst case, memcg A has no reclaimable shmem at all, making the reclaim entirely useless and incurring unnecessary latency. We observed this in production, where page lock contention during split caused multi-hundred-millisecond stalls: tid 11340 comm scanner locked a page for 182264 us! kstack: unlock_page+1 split_huge_page_to_list+3135 shmem_unused_huge_shrink+767 super_cache_scan+329 do_shrink_slab+291 shrink_slab+533 shrink_node+400 do_try_to_free_pages+206 try_to_free_mem_cgroup_pages+262 try_charge_memcg+591 mem_cgroup_charge+136 __handle_mm_fault+2431 handle_mm_fault+194 do_user_addr_fault+462 __do_page_fault+176 do_page_fault+48 page_fault+62 Usama's recent patch [1] prevents the shmem unused shrinker from being invoked during memcg-level reclaim altogether, but this is overly conservative: we can do better by reclaiming only the shmem charged to the reclaiming memcg. This series converts the shrinker list to a memcg-aware list_lru, so that non-root memcg reclaim walks only candidates charged to the reclaiming memcg. Global reclaim, root memcg reclaim and shmem quota reclaim retain their existing global semantics. To avoid pinning a dying memcg through a long-lived CSS reference, each inode stores an obj_cgroup reference instead of a mem_cgroup reference. The list_lru add/delete paths resolve the current memcg from the objcg under RCU, staying consistent with list_lru's own memcg migration on offline. This patch (of 3): obj_cgroup_memcg() currently requires a non-NULL objcg, so callers that may hold a NULL objcg must guard the call with an explicit NULL check. This pattern is duplicated in folio_memcg(), folio_memcg_check(), mm/page_owner.c, and mm/zswap.c. Teach obj_cgroup_memcg() to accept NULL and return NULL in that case, then remove the redundant NULL checks at the call sites. Also remove the mem_cgroup_from_entry() wrapper in zswap, which existed solely to provide this NULL-safe behaviour, and replace its two callers with direct obj_cgroup_memcg() calls. No functional change intended. Link: https://lore.kernel.org/cover.1786955972.git.zhengqi.arch@bytedance.com Link: https://lore.kernel.org/09bcf74312246a6e4146be8a0cb9787f8beddb28.1786955972.git.zhengqi.arch@bytedance.com Signed-off-by: Qi Zheng Signed-off-by: Andrew Morton Acked-by: Shakeel Butt Cc: Baolin Wang Cc: Christian Brauner Cc: David Hildenbrand Cc: Hugh Dickins Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin --- include/linux/memcontrol.h | 11 ++++++++--- mm/page_owner.c | 2 +- mm/zswap.c | 17 ++--------------- 3 files changed, 11 insertions(+), 19 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 7d1c0ce189a887..da625d2edb3bab 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -380,7 +380,7 @@ enum objext_flags { static inline struct mem_cgroup *obj_cgroup_memcg(struct obj_cgroup *objcg) { lockdep_assert_once(rcu_read_lock_held() || lockdep_is_held(&cgroup_mutex)); - return READ_ONCE(objcg->memcg); + return objcg ? READ_ONCE(objcg->memcg) : NULL; } /* @@ -433,7 +433,7 @@ static inline struct mem_cgroup *folio_memcg(struct folio *folio) { struct obj_cgroup *objcg = folio_objcg(folio); - return objcg ? obj_cgroup_memcg(objcg) : NULL; + return obj_cgroup_memcg(objcg); } /* @@ -476,7 +476,7 @@ static inline struct mem_cgroup *folio_memcg_check(struct folio *folio) objcg = (void *)(memcg_data & ~OBJEXTS_FLAGS_MASK); - return objcg ? obj_cgroup_memcg(objcg) : NULL; + return obj_cgroup_memcg(objcg); } static inline struct mem_cgroup *page_memcg_check(struct page *page) @@ -1050,6 +1050,11 @@ void mem_cgroup_flush_workqueue(void); extern int mem_cgroup_init(void); #else /* CONFIG_MEMCG */ +static inline struct mem_cgroup *obj_cgroup_memcg(struct obj_cgroup *objcg) +{ + return NULL; +} + #define MEM_CGROUP_ID_SHIFT 0 #define root_mem_cgroup (NULL) diff --git a/mm/page_owner.c b/mm/page_owner.c index fbbda7ba914ba5..3fc37d9b908ef0 100644 --- a/mm/page_owner.c +++ b/mm/page_owner.c @@ -575,7 +575,7 @@ static inline int print_page_owner_memcg(char *kbuf, size_t count, int ret, } objcg = (void *)(memcg_data & ~OBJEXTS_FLAGS_MASK); - memcg = objcg ? obj_cgroup_memcg(objcg) : NULL; + memcg = obj_cgroup_memcg(objcg); if (!memcg) goto out_unlock; diff --git a/mm/zswap.c b/mm/zswap.c index 37f34e406c8e3b..c1dc60926bad99 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -647,19 +647,6 @@ static int zswap_enabled_param_set(const char *val, * lru functions **********************************/ -/* should be called under RCU */ -#ifdef CONFIG_MEMCG -static inline struct mem_cgroup *mem_cgroup_from_entry(struct zswap_entry *entry) -{ - return entry->objcg ? obj_cgroup_memcg(entry->objcg) : NULL; -} -#else -static inline struct mem_cgroup *mem_cgroup_from_entry(struct zswap_entry *entry) -{ - return NULL; -} -#endif - static inline int entry_to_nid(struct zswap_entry *entry) { return page_to_nid(virt_to_page(entry)); @@ -682,7 +669,7 @@ static void zswap_lru_add(struct zswap_entry *entry) * Similar reasoning holds for list_lru_del(). */ rcu_read_lock(); - memcg = mem_cgroup_from_entry(entry); + memcg = obj_cgroup_memcg(entry->objcg); /* will always succeed */ list_lru_add(&zswap_list_lru, &entry->lru, nid, memcg); rcu_read_unlock(); @@ -694,7 +681,7 @@ static void zswap_lru_del(struct zswap_entry *entry) struct mem_cgroup *memcg; rcu_read_lock(); - memcg = mem_cgroup_from_entry(entry); + memcg = obj_cgroup_memcg(entry->objcg); /* will always succeed */ list_lru_del(&zswap_list_lru, &entry->lru, nid, memcg); rcu_read_unlock(); From 61b6dc191391c57fbb823b31075e5bdb4e012eb1 Mon Sep 17 00:00:00 2001 From: Qi Zheng Date: Mon, 17 Aug 2026 17:03:27 +0800 Subject: [PATCH 0582/1352] mm: shmem: move unused huge shrinklist queuing past the truncation check The shmem_get_folio_gfp() adds the inode to the unused huge shrinker list at the alloced label, but a subsequent truncation check may still fail and remove the folio, leaving the inode on the list with a stale folio. The original code works because the shrinker re-looks-up the folio and drops stale entries, but it is cleaner to queue the inode only after all checks that might remove the folio have passed. So just make the pure structural move with no functional change, and it serves as preparation for the memcg-aware shrinker conversion. Link: https://lore.kernel.org/17fcf64faec0dfbbe6cf8a97924e3f39cd3c51a4.1786955972.git.zhengqi.arch@bytedance.com Signed-off-by: Qi Zheng Signed-off-by: Andrew Morton Reviewed-by: Baolin Wang Cc: Christian Brauner Cc: David Hildenbrand Cc: Hugh Dickins Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Cc: Shakeel Butt --- mm/shmem.c | 48 +++++++++++++++++++++++++++--------------------- 1 file changed, 27 insertions(+), 21 deletions(-) diff --git a/mm/shmem.c b/mm/shmem.c index 848316eaa7f4fb..9f2585d6dbfb06 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -2539,27 +2539,6 @@ static int shmem_get_folio_gfp(struct inode *inode, pgoff_t index, alloced: alloced = true; - if (folio_test_large(folio) && - DIV_ROUND_UP(i_size_read(inode), PAGE_SIZE) < - folio_next_index(folio)) { - struct shmem_sb_info *sbinfo = SHMEM_SB(inode->i_sb); - struct shmem_inode_info *info = SHMEM_I(inode); - /* - * Part of the large folio is beyond i_size: subject - * to shrink under memory pressure. - */ - spin_lock(&sbinfo->shrinklist_lock); - /* - * _careful to defend against unlocked access to - * ->shrink_list in shmem_unused_huge_shrink() - */ - if (list_empty_careful(&info->shrinklist)) { - list_add_tail(&info->shrinklist, - &sbinfo->shrinklist); - sbinfo->shrinklist_len++; - } - spin_unlock(&sbinfo->shrinklist_lock); - } if (sgp == SGP_WRITE) folio_set_referenced(folio); @@ -2589,6 +2568,33 @@ static int shmem_get_folio_gfp(struct inode *inode, pgoff_t index, error = -EINVAL; goto unlock; } + + /* + * Queue the inode on the shrink list only after all checks that might + * remove the folio have passed. Otherwise the inode could be left on + * the shrinker list with a stale folio. + */ + if (alloced && folio_test_large(folio) && + DIV_ROUND_UP(i_size_read(inode), PAGE_SIZE) < folio_next_index(folio)) { + struct shmem_sb_info *sbinfo = SHMEM_SB(inode->i_sb); + struct shmem_inode_info *info = SHMEM_I(inode); + /* + * Part of the large folio is beyond i_size: subject + * to shrink under memory pressure. + */ + spin_lock(&sbinfo->shrinklist_lock); + /* + * _careful to defend against unlocked access to + * ->shrink_list in shmem_unused_huge_shrink() + */ + if (list_empty_careful(&info->shrinklist)) { + list_add_tail(&info->shrinklist, + &sbinfo->shrinklist); + sbinfo->shrinklist_len++; + } + spin_unlock(&sbinfo->shrinklist_lock); + } + out: *foliop = folio; return 0; From 036c2588f74666a7cc013c7fba8f5cfa687d6e54 Mon Sep 17 00:00:00 2001 From: Qi Zheng Date: Mon, 17 Aug 2026 17:03:28 +0800 Subject: [PATCH 0583/1352] mm: shmem: make unused huge shrinker memcg aware The shmem unused huge shrinker keeps a per-superblock list of inodes whose tail huge folio extends beyond i_size. Since that list is not memcg aware, reclaim triggered by one memcg can scan inodes from the whole superblock and split shmem huge folios charged to unrelated memcgs. Convert the shrink list to a memcg-aware list_lru. Queue each inode on the list_lru sublist matching the memcg and node of the current tail huge folio, so non-root memcg reclaim only walks candidates charged to the reclaiming memcg. Global reclaim, root memcg reclaim and shmem quota reclaim keep global semantics. Rather than pinning a struct mem_cgroup reference in shmem_inode_info, store a struct obj_cgroup reference instead. The list_lru add and delete paths resolve the current memcg from the objcg under RCU, so that memcg offline and list_lru entry migration remain consistent: list_lru migrates entries to the parent memcg sublist on offline, and obj_cgroup_memcg() follows the same reparenting, ensuring the correct sublist is always found at delete time. This avoids pinning a dying memcg through a long-lived CSS reference. The list_lru still tracks inodes while the actual split target is the current tail huge folio, so validate the folio memcg/node during scan. If the folio no longer matches the reclaim context or splitting cannot proceed, requeue the inode according to the current tail folio; if the inode is no longer shrinkable, drop the scan entry. This can be tested with the shrinker debugfs interface by allocating 32 tmpfs tail THPs in each of two memcgs, then scanning the sb-tmpfs shrinker with memcg A's cgroup id: before A scan after A scan base A=64M, B=64M A=64M, B=64M (per-memcg count is skipped) patched A=64M, B=64M A=0, B=64M Link: https://lore.kernel.org/94cc7fe1fd645254ae90effc1c0687678438d940.1786955972.git.zhengqi.arch@bytedance.com Signed-off-by: Qi Zheng Signed-off-by: Andrew Morton Reviewed-by: Baolin Wang Cc: Christian Brauner Cc: David Hildenbrand Cc: Hugh Dickins Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Cc: Shakeel Butt Cc: Qinyun Tan --- include/linux/shmem_fs.h | 12 +- mm/shmem.c | 362 ++++++++++++++++++++++++++++++--------- 2 files changed, 289 insertions(+), 85 deletions(-) diff --git a/include/linux/shmem_fs.h b/include/linux/shmem_fs.h index 5663dff53186e2..a7c7a96a7cbf90 100644 --- a/include/linux/shmem_fs.h +++ b/include/linux/shmem_fs.h @@ -11,6 +11,7 @@ #include #include #include +#include /* inode in-kernel data */ @@ -54,6 +55,11 @@ struct shmem_inode_info { struct dquot __rcu *i_dquot[MAXQUOTAS]; #endif struct inode vfs_inode; + +#ifdef CONFIG_TRANSPARENT_HUGEPAGE + struct obj_cgroup *shrinklist_objcg; + int shrinklist_nid; +#endif }; #define SHMEM_FL_USER_VISIBLE (FS_FL_USER_VISIBLE | FS_CASEFOLD_FL) @@ -83,9 +89,9 @@ struct shmem_sb_info { ino_t next_ino; /* The next per-sb inode number to use */ ino_t __percpu *ino_batch; /* The next per-cpu inode number to use */ struct mempolicy *mpol; /* default memory policy for mappings */ - spinlock_t shrinklist_lock; /* Protects shrinklist */ - struct list_head shrinklist; /* List of shinkable inodes */ - unsigned long shrinklist_len; /* Length of shrinklist */ +#ifdef CONFIG_TRANSPARENT_HUGEPAGE + struct list_lru shrinklist; /* List of shrinkable inodes */ +#endif struct shmem_quota_limits qlimits; /* Default quota limits */ struct simple_xattr_cache xa_cache; }; diff --git a/mm/shmem.c b/mm/shmem.c index 9f2585d6dbfb06..ae39966fc7304e 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -725,51 +725,258 @@ static const char *shmem_format_huge(int huge) } #endif -static unsigned long shmem_unused_huge_shrink(struct shmem_sb_info *sbinfo, - struct shrink_control *sc, unsigned long nr_to_free) +static bool is_shmem_unused_huge_isolated(struct shmem_inode_info *info) { - LIST_HEAD(list), *pos, *next; - struct inode *inode; + + return info->shrinklist_nid == -1; +} + +static void set_shmem_unused_huge_isolated(struct shmem_inode_info *info) +{ + info->shrinklist_nid = -1; +} + +static struct obj_cgroup *shmem_get_and_clear_objcg(struct shmem_inode_info *info) +{ + struct obj_cgroup *objcg = info->shrinklist_objcg; + + info->shrinklist_objcg = NULL; + + return objcg; +} + +#ifdef CONFIG_MEMCG +static struct obj_cgroup * +shmem_unused_huge_alloc_lru(struct shmem_sb_info *sbinfo, struct folio *folio, + gfp_t gfp) +{ + int ret; + + ret = folio_memcg_list_lru_alloc(folio, &sbinfo->shrinklist, gfp); + if (ret) + return ERR_PTR(ret); + + return get_obj_cgroup_from_folio(folio); +} +#else +static struct obj_cgroup * +shmem_unused_huge_alloc_lru(struct shmem_sb_info *sbinfo, struct folio *folio, + gfp_t gfp) +{ + return NULL; +} +#endif + +static void shmem_unused_huge_lru_add(struct shmem_sb_info *sbinfo, + struct list_head *item, int nid, + struct obj_cgroup *objcg) +{ + struct mem_cgroup *memcg; + + rcu_read_lock(); + memcg = obj_cgroup_memcg(objcg); + list_lru_add(&sbinfo->shrinklist, item, nid, memcg); + rcu_read_unlock(); +} + +static void shmem_unused_huge_lru_del(struct shmem_sb_info *sbinfo, + struct list_head *item, int nid, + struct obj_cgroup *objcg) +{ + struct mem_cgroup *memcg; + + rcu_read_lock(); + memcg = obj_cgroup_memcg(objcg); + list_lru_del(&sbinfo->shrinklist, item, nid, memcg); + rcu_read_unlock(); +} + +static void shmem_unused_huge_add(struct inode *inode, struct folio *folio, + gfp_t gfp) +{ + struct shmem_inode_info *info = SHMEM_I(inode); + struct shmem_sb_info *sbinfo = SHMEM_SB(inode->i_sb); + int nid = folio_nid(folio); + struct obj_cgroup *objcg = NULL, *old_objcg = NULL; + + objcg = shmem_unused_huge_alloc_lru(sbinfo, folio, gfp); + if (IS_ERR(objcg)) + return; + + spin_lock(&info->lock); + if (!list_empty(&info->shrinklist)) { + /* isolated on scan list, let shrink handle it */ + if (is_shmem_unused_huge_isolated(info)) + goto unlock; + + if (info->shrinklist_nid == nid && + info->shrinklist_objcg == objcg) + goto unlock; + + shmem_unused_huge_lru_del(sbinfo, &info->shrinklist, + info->shrinklist_nid, + info->shrinklist_objcg); + old_objcg = shmem_get_and_clear_objcg(info); + } + + info->shrinklist_objcg = objcg; + info->shrinklist_nid = nid; + shmem_unused_huge_lru_add(sbinfo, &info->shrinklist, nid, objcg); + objcg = NULL; +unlock: + spin_unlock(&info->lock); + obj_cgroup_put(old_objcg); + obj_cgroup_put(objcg); +} + +static void shmem_unused_huge_del(struct inode *inode) +{ + struct shmem_inode_info *info = SHMEM_I(inode); + struct shmem_sb_info *sbinfo = SHMEM_SB(inode->i_sb); + struct obj_cgroup *objcg = NULL; + + spin_lock(&info->lock); + if (!list_empty(&info->shrinklist)) { + shmem_unused_huge_lru_del(sbinfo, &info->shrinklist, + info->shrinklist_nid, + info->shrinklist_objcg); + objcg = shmem_get_and_clear_objcg(info); + } + spin_unlock(&info->lock); + + obj_cgroup_put(objcg); +} + +struct shmem_unused_huge_scan { + struct list_head list; + struct shrink_control *sc; +}; + +static enum lru_status shmem_unused_huge_isolate(struct list_head *item, + struct list_lru_one *lru, + void *arg) +{ + struct shmem_unused_huge_scan *scan = arg; struct shmem_inode_info *info; - struct folio *folio; - unsigned long batch = sc ? sc->nr_to_scan : 128; - unsigned long split = 0, freed = 0; + struct inode *inode; + struct obj_cgroup *objcg = NULL; - if (list_empty(&sbinfo->shrinklist)) - return SHRINK_STOP; + info = list_entry(item, struct shmem_inode_info, shrinklist); - spin_lock(&sbinfo->shrinklist_lock); - list_for_each_safe(pos, next, &sbinfo->shrinklist) { - info = list_entry(pos, struct shmem_inode_info, shrinklist); + /* + * Use trylock to avoid ABBA deadlock: add/del path takes info->lock + * before the list_lru bucket lock, while here the order is reversed. + */ + if (!spin_trylock(&info->lock)) + return LRU_SKIP; - /* pin the inode */ - inode = igrab(&info->vfs_inode); + /* pin the inode */ + inode = igrab(&info->vfs_inode); + /* inode is about to be evicted */ + if (!inode) { + list_lru_isolate(lru, item); + objcg = shmem_get_and_clear_objcg(info); + spin_unlock(&info->lock); + obj_cgroup_put(objcg); + return LRU_REMOVED; + } - /* inode is about to be evicted */ - if (!inode) { - list_del_init(&info->shrinklist); - goto next; - } + list_lru_isolate(lru, item); + objcg = shmem_get_and_clear_objcg(info); + set_shmem_unused_huge_isolated(info); + list_add_tail(&info->shrinklist, &scan->list); + spin_unlock(&info->lock); + obj_cgroup_put(objcg); - list_move(&info->shrinklist, &list); -next: - sbinfo->shrinklist_len--; - if (!--batch) - break; + return LRU_REMOVED; +} + +static bool is_shmem_unused_huge_match(struct folio *folio, + struct shrink_control *sc) +{ + struct mem_cgroup *memcg = NULL; + bool match; + + /* shmem quota reclaim has no NUMA node or memcg restriction */ + if (!sc) + return true; + + if (folio_nid(folio) != sc->nid) + return false; + + /* + * Only non-root memcg reclaim needs to match the folio charge against + * sc->memcg. Skip the folio memcg check for global shrinker reclaim and + * root memcg reclaim. + */ + if (!sc->memcg || mem_cgroup_is_root(sc->memcg)) + return true; + + memcg = get_mem_cgroup_from_folio(folio); + match = memcg == sc->memcg; + mem_cgroup_put(memcg); + + return match; +} + +static void shmem_unused_huge_drop(struct inode *inode) +{ + struct shmem_inode_info *info = SHMEM_I(inode); + + spin_lock(&info->lock); + list_del_init(&info->shrinklist); + spin_unlock(&info->lock); +} + +static void shmem_unused_huge_requeue(struct inode *inode, struct folio *folio) +{ + struct shmem_inode_info *info = SHMEM_I(inode); + struct shmem_sb_info *sbinfo = SHMEM_SB(inode->i_sb); + struct obj_cgroup *objcg; + int nid = folio_nid(folio); + + objcg = shmem_unused_huge_alloc_lru(sbinfo, folio, GFP_NOWAIT); + if (IS_ERR(objcg)) { + shmem_unused_huge_drop(inode); + return; } - spin_unlock(&sbinfo->shrinklist_lock); - list_for_each_safe(pos, next, &list) { - pgoff_t next, end; + spin_lock(&info->lock); + /* Requeue the inode to shrinklist */ + list_del_init(&info->shrinklist); + shmem_unused_huge_lru_add(sbinfo, &info->shrinklist, nid, objcg); + info->shrinklist_objcg = objcg; + info->shrinklist_nid = nid; + spin_unlock(&info->lock); +} + +static unsigned long shmem_unused_huge_shrink(struct shmem_sb_info *sbinfo, + struct shrink_control *sc, unsigned long nr_to_free) +{ + struct shmem_unused_huge_scan scan; + struct inode *inode; + struct shmem_inode_info *info; + struct folio *folio; + struct list_head *pos, *next; + unsigned long split = 0, freed = 0; + + INIT_LIST_HEAD(&scan.list); + scan.sc = sc; + if (sc) + list_lru_shrink_walk(&sbinfo->shrinklist, sc, + shmem_unused_huge_isolate, &scan); + else + list_lru_walk(&sbinfo->shrinklist, shmem_unused_huge_isolate, + &scan, 128); + + list_for_each_safe(pos, next, &scan.list) { + pgoff_t folio_end, end; loff_t i_size; int ret; info = list_entry(pos, struct shmem_inode_info, shrinklist); inode = &info->vfs_inode; - if (nr_to_free && freed >= nr_to_free) - goto move_back; - i_size = i_size_read(inode); folio = filemap_get_entry(inode->i_mapping, i_size / PAGE_SIZE); if (!folio || xa_is_value(folio)) @@ -782,13 +989,19 @@ static unsigned long shmem_unused_huge_shrink(struct shmem_sb_info *sbinfo, } /* Check if there is anything to gain from splitting */ - next = folio_next_index(folio); + folio_end = folio_next_index(folio); end = shmem_fallocend(inode, DIV_ROUND_UP(i_size, PAGE_SIZE)); - if (end <= folio->index || end >= next) { + if (end <= folio->index || end >= folio_end) { folio_put(folio); goto drop; } + if (!is_shmem_unused_huge_match(folio, scan.sc)) + goto move_back; + + if (nr_to_free && freed >= nr_to_free) + goto move_back; + /* * Move the inode on the list back to shrinklist if we failed * to lock the page at this time. @@ -796,35 +1009,30 @@ static unsigned long shmem_unused_huge_shrink(struct shmem_sb_info *sbinfo, * Waiting for the lock may lead to deadlock in the * reclaim path. */ - if (!folio_trylock(folio)) { - folio_put(folio); + if (!folio_trylock(folio)) + goto move_back; + + if (!is_shmem_unused_huge_match(folio, scan.sc)) { + folio_unlock(folio); goto move_back; } ret = split_folio(folio); folio_unlock(folio); - folio_put(folio); /* If split failed move the inode on the list back to shrinklist */ if (ret) goto move_back; - freed += next - end; + freed += folio_end - end; split++; + folio_put(folio); drop: - list_del_init(&info->shrinklist); + shmem_unused_huge_drop(inode); goto put; move_back: - /* - * Make sure the inode is either on the global list or deleted - * from any local list before iput() since it could be deleted - * in another thread once we put the inode (then the local list - * is corrupted). - */ - spin_lock(&sbinfo->shrinklist_lock); - list_move(&info->shrinklist, &sbinfo->shrinklist); - sbinfo->shrinklist_len++; - spin_unlock(&sbinfo->shrinklist_lock); + shmem_unused_huge_requeue(inode, folio); + folio_put(folio); put: iput(inode); } @@ -837,7 +1045,7 @@ static long shmem_unused_huge_scan(struct super_block *sb, { struct shmem_sb_info *sbinfo = SHMEM_SB(sb); - if (!READ_ONCE(sbinfo->shrinklist_len)) + if (!list_lru_shrink_count(&sbinfo->shrinklist, sc)) return SHRINK_STOP; return shmem_unused_huge_shrink(sbinfo, sc, 0); @@ -848,21 +1056,21 @@ static long shmem_unused_huge_count(struct super_block *sb, { struct shmem_sb_info *sbinfo = SHMEM_SB(sb); - /* - * The per-superblock shrinklist is filesystem-global and does not - * honour sc->memcg, so it is only meaningful on the global (kswapd or - * root direct reclaim) shrink path. Skip the per-memcg iterations of - * shrink_slab_memcg() to avoid queueing duplicate global work. - */ - if (!mem_cgroup_shrink_is_root(sc)) - return 0; - - return READ_ONCE(sbinfo->shrinklist_len); + return list_lru_shrink_count(&sbinfo->shrinklist, sc); } #else /* !CONFIG_TRANSPARENT_HUGEPAGE */ #define shmem_huge SHMEM_HUGE_DENY +static void shmem_unused_huge_add(struct inode *inode, struct folio *folio, + gfp_t gfp) +{ +} + +static void shmem_unused_huge_del(struct inode *inode) +{ +} + static unsigned long shmem_unused_huge_shrink(struct shmem_sb_info *sbinfo, struct shrink_control *sc, unsigned long nr_to_free) { @@ -1418,14 +1626,7 @@ static void shmem_evict_inode(struct inode *inode) inode->i_size = 0; mapping_set_exiting(inode->i_mapping); shmem_truncate_range(inode, 0, (loff_t)-1); - if (!list_empty(&info->shrinklist)) { - spin_lock(&sbinfo->shrinklist_lock); - if (!list_empty(&info->shrinklist)) { - list_del_init(&info->shrinklist); - sbinfo->shrinklist_len--; - } - spin_unlock(&sbinfo->shrinklist_lock); - } + shmem_unused_huge_del(inode); while (!list_empty(&info->swaplist)) { /* Wait while shmem_unuse() is scanning this inode... */ wait_var_event(&info->stop_eviction, @@ -2576,25 +2777,12 @@ static int shmem_get_folio_gfp(struct inode *inode, pgoff_t index, */ if (alloced && folio_test_large(folio) && DIV_ROUND_UP(i_size_read(inode), PAGE_SIZE) < folio_next_index(folio)) { - struct shmem_sb_info *sbinfo = SHMEM_SB(inode->i_sb); - struct shmem_inode_info *info = SHMEM_I(inode); /* * Part of the large folio is beyond i_size: subject * to shrink under memory pressure. */ - spin_lock(&sbinfo->shrinklist_lock); - /* - * _careful to defend against unlocked access to - * ->shrink_list in shmem_unused_huge_shrink() - */ - if (list_empty_careful(&info->shrinklist)) { - list_add_tail(&info->shrinklist, - &sbinfo->shrinklist); - sbinfo->shrinklist_len++; - } - spin_unlock(&sbinfo->shrinklist_lock); + shmem_unused_huge_add(inode, folio, gfp); } - out: *foliop = folio; return 0; @@ -3071,6 +3259,10 @@ static struct inode *__shmem_get_inode(struct mnt_idmap *idmap, if (info->fsflags) shmem_set_inode_flags(inode, info->fsflags, NULL); INIT_LIST_HEAD(&info->shrinklist); +#ifdef CONFIG_TRANSPARENT_HUGEPAGE + info->shrinklist_objcg = NULL; + info->shrinklist_nid = -1; +#endif INIT_LIST_HEAD(&info->swaplist); cache_no_acl(inode); if (sbinfo->noswap) @@ -4945,6 +5137,9 @@ static void shmem_put_super(struct super_block *sb) #endif free_percpu(sbinfo->ino_batch); percpu_counter_destroy(&sbinfo->used_blocks); +#ifdef CONFIG_TRANSPARENT_HUGEPAGE + list_lru_destroy(&sbinfo->shrinklist); +#endif mpol_put(sbinfo->mpol); #ifdef CONFIG_TMPFS_XATTR simple_xattr_cache_cleanup(&sbinfo->xa_cache); @@ -5038,8 +5233,11 @@ static int shmem_fill_super(struct super_block *sb, struct fs_context *fc) raw_spin_lock_init(&sbinfo->stat_lock); if (percpu_counter_init(&sbinfo->used_blocks, 0, GFP_KERNEL)) goto failed; - spin_lock_init(&sbinfo->shrinklist_lock); - INIT_LIST_HEAD(&sbinfo->shrinklist); + +#ifdef CONFIG_TRANSPARENT_HUGEPAGE + if (list_lru_init_memcg(&sbinfo->shrinklist, sb->s_shrink)) + goto failed; +#endif sb->s_maxbytes = MAX_LFS_FILESIZE; sb->s_blocksize = PAGE_SIZE; From 4749daa6b81bcfb031806219b23f7d21640ac13f Mon Sep 17 00:00:00 2001 From: Qinyun Tan Date: Wed, 2 Sep 2026 17:32:02 +0800 Subject: [PATCH 0584/1352] mm/list_lru: disable memcg awareness under cgroup_disable=memory __list_lru_init() only collapses a memcg-aware list_lru into plain per-node lists when kmem accounting is disabled (cgroup.memory=nokmem). When the memory controller is disabled entirely (cgroup_disable=memory), mem_cgroup_kmem_disabled() is false, so the lru stays memcg aware even though no object will ever be charged to a memcg. This is more than a semantic inconsistency. folio_memcg_list_lru_alloc() trusts list_lru_memcg_aware() and dereferences the folio's memcg, which is always NULL with the controller disabled. The only mainline caller, folio_memcg_alloc_deferred(), papers over this with an explicit mem_cgroup_disabled() check. The shmem unused-huge shrinker conversion ("mm: shmem: make unused huge shrinker memcg aware") adds a second caller without such a guard, so booting with cgroup_disable=memory and writing to a huge=always tmpfs oopses: BUG: unable to handle page fault for address: 0000000000000488 RIP: 0010:folio_memcg_list_lru_alloc+0x41/0xf0 Call Trace: shmem_get_folio_gfp+0x1cd/0x7c0 shmem_write_begin+0x5d/0x100 generic_perform_write+0x89/0x2a0 shmem_file_write_iter+0x82/0x90 vfs_write+0x256/0x410 ksys_write+0x61/0xe0 do_syscall_64+0x8d/0x460 entry_SYSCALL_64_after_hwframe+0x76/0x7e The faulting address is the offset of mem_cgroup->kmemcg_id, dereferenced on a NULL memcg in memcg_list_lru_allocated(): folio_memcg_list_lru_alloc() list_lru_memcg_aware() <- true, only nokmem checked memcg = folio_memcg(folio) <- NULL memcg_list_lru_allocated(memcg, lru) memcg->kmemcg_id <- NULL pointer dereference Check mem_cgroup_disabled() in __list_lru_init() so that all list_lrus fall back to plain per-node lists when the controller is disabled, matching what the shrinker side already does (shrinker_memcg_alloc() bails out on mem_cgroup_disabled()). This makes the mem_cgroup_disabled() check in callers unnecessary rather than mandatory. Link: https://lore.kernel.org/20260902093202.609559-1-qinyuntan@linux.alibaba.com Signed-off-by: Qinyun Tan Signed-off-by: Andrew Morton Reviewed-by: Baolin Wang Cc: Christian Brauner Cc: David Hildenbrand Cc: Hugh Dickins Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Qi Zheng Cc: Roman Gushchin Cc: Shakeel Butt --- mm/list_lru.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/list_lru.c b/mm/list_lru.c index 36662d02ff9631..a4522ca93ebcb9 100644 --- a/mm/list_lru.c +++ b/mm/list_lru.c @@ -671,7 +671,7 @@ int __list_lru_init(struct list_lru *lru, bool memcg_aware, struct shrinker *shr else lru->shrinker_id = -1; - if (mem_cgroup_kmem_disabled()) + if (mem_cgroup_disabled() || mem_cgroup_kmem_disabled()) memcg_aware = false; #endif From 0097e63d89f3616d9c3611ae1a20912c7097c749 Mon Sep 17 00:00:00 2001 From: Wilson Felipe Pereira Date: Fri, 28 Aug 2026 03:37:31 +0000 Subject: [PATCH 0585/1352] selftests/cgroup: test_zswap: wait for cgroup to unpopulate in test_zswap_writeback MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Patch series "selftests/cgroup: fixes for test_zswap on single core VM", v4. This series fixes two test failures in test_zswap observed when running on a single-core VM (-smp 1) with 4GB of RAM. Patch 1 addresses a race condition in test_zswap_writeback() where waitpid() returns before the exiting child process is switched away by the kernel, causing an immediate write of "+memory" to cgroup.subtree_control to fail with -EBUSY. We fix this by waiting for cgroup.events to report "populated 0". Patch 2 fixes an implicit unsigned conversion bug in test_no_kmem_bypass() where small negative timing differences between debugfs stored_pages and cgroup zswapped bytes caused the comparison to falsely fail due to unsigned promotion. This patch (of 2): When running test_zswap on a single-core VM (-smp 1) with 4GB of RAM, test_zswap_writeback intermittently fails on the initial run after boot. In test_zswap_writeback(), after waitpid() reaps the child process created by test_zswap_writeback_one(), writing "+memory" to cgroup.subtree_control can fail with -EBUSY. Under cgroup v2, enabling domain subtree controllers is forbidden while any tasks remain in cgroup.procs. When a child process exits, exit_notify() wakes the parent process, allowing waitpid() to return immediately. However, the cgroup populated task count (nr_populated_csets) is only decremented when the exiting task is switched away via finish_task_switch() -> cgroup_task_dead(). On single-core systems, the parent runs before the dead child has been switched out, causing "+memory" to fail with -EBUSY if written immediately after waitpid() returns. Fix this by waiting for cgroup.events to report "populated 0\n" via cg_read_strcmp_wait() before enabling subtree control. Link: https://lore.kernel.org/20260828033741.2184560-1-wfelipe@google.com Link: https://lore.kernel.org/20260828033741.2184560-2-wfelipe@google.com Signed-off-by: Wilson Felipe Pereira Signed-off-by: Andrew Morton Acked-by: Michal Koutný Cc: Chengming Zhou Cc: Johannes Weiner Cc: Nhat Pham Cc: Shuah Khan Cc: Tejun Heo --- tools/testing/selftests/cgroup/test_zswap.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/tools/testing/selftests/cgroup/test_zswap.c b/tools/testing/selftests/cgroup/test_zswap.c index 8df54b59513a8c..b1abc94317e4f7 100644 --- a/tools/testing/selftests/cgroup/test_zswap.c +++ b/tools/testing/selftests/cgroup/test_zswap.c @@ -409,6 +409,8 @@ static int test_zswap_writeback(const char *root, bool wb) * Thus, the parent's setting shall be what's in effect. */ if (cg_write(test_group, "memory.zswap.max", "max")) goto out; + if (cg_read_strcmp_wait(test_group, "cgroup.events", "populated 0\n")) + goto out; if (cg_write(test_group, "cgroup.subtree_control", "+memory")) goto out; From f1be3641f4dc7bdc947218136e8e9476131993b2 Mon Sep 17 00:00:00 2001 From: Wilson Felipe Pereira Date: Fri, 28 Aug 2026 03:37:32 +0000 Subject: [PATCH 0586/1352] selftests/cgroup: test_zswap: fix implicit unsigned promotion bug in test_no_kmem_bypass MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit In test_no_kmem_bypass(), delta (stored_pages * page_size - zswapped) is checked against stored_pages * page_size / 4 to verify that the pages pushed to zswap belong to the test memory cgroup. Due to slight stat update timing differences, delta can evaluate to a small negative number (e.g. -5MB out of 1GB). Because delta is declared as a signed int and stored_pages is an unsigned size_t, C's usual arithmetic conversions implicitly promote a negative delta to a large unsigned 64-bit integer, causing `delta < stored_pages * page_size / 4` to falsely evaluate to 0 and fail the test. Fix this by declaring zswapped and delta as signed long long and comparing against a signed threshold, ensuring negative deltas correctly evaluate to true. Link: https://lore.kernel.org/20260828033741.2184560-3-wfelipe@google.com Fixes: a549f9f31561a ("selftests: cgroup: add test_zswap with no kmem bypass test") Signed-off-by: Wilson Felipe Pereira Signed-off-by: Andrew Morton Acked-by: Michal Koutný Cc: Chengming Zhou Cc: Johannes Weiner Cc: Nhat Pham Cc: Shuah Khan Cc: Tejun Heo --- tools/testing/selftests/cgroup/test_zswap.c | 13 ++++++++----- 1 file changed, 8 insertions(+), 5 deletions(-) diff --git a/tools/testing/selftests/cgroup/test_zswap.c b/tools/testing/selftests/cgroup/test_zswap.c index b1abc94317e4f7..1ac77907277570 100644 --- a/tools/testing/selftests/cgroup/test_zswap.c +++ b/tools/testing/selftests/cgroup/test_zswap.c @@ -654,11 +654,14 @@ static int test_no_kmem_bypass(const char *root) break; /* If memory was pushed to zswap, verify it belongs to memcg */ if (stored_pages > stored_pages_threshold) { - int zswapped = cg_read_key_long(test_group, "memory.stat", "zswapped "); - int delta = stored_pages * page_size - zswapped; - int result_ok = delta < stored_pages * page_size / 4; - - ret = result_ok ? KSFT_PASS : KSFT_FAIL; + long zswapped = cg_read_key_long( + test_group, "memory.stat", "zswapped "); + long long delta = + (long long)stored_pages * page_size - zswapped; + long long max_delta = + (long long)stored_pages * page_size / 4; + + ret = (delta < max_delta) ? KSFT_PASS : KSFT_FAIL; break; } } From c04bec1e39b620cf140f580b6c569a073595f166 Mon Sep 17 00:00:00 2001 From: Rik van Riel Date: Fri, 28 Aug 2026 13:50:36 -0400 Subject: [PATCH 0587/1352] mm/memcontrol: fix stuck FLUSHING_CACHED_CHARGE bit on isolated cpus When drain_all_stock() sets FLUSHING_CACHED_CHARGE before checking isolation, schedule_drain_work() can drop the work in a separate RCU critical section, and housekeeping_update()'s synchronize_rcu() can race that second check, leaving the flag set. drain_local_stock() only clears the bit for work that ran, so the flag remains set and the stock is never drained again. Have schedule_drain_work() return whether the work was queued, and clear FLUSHING_CACHED_CHARGE in drain_all_stock() when the remote CPU is isolated, so future drains can retry. Link: https://lore.kernel.org/20260828135036.7d44361f@fangorn Fixes: 6a792697a53a ("memcg: do not drain charge pcp caches on remote isolated cpus") Suggested-by: Michal Hocko Suggested-by: Shakeel Butt Signed-off-by: Rik van Riel Signed-off-by: Andrew Morton Acked-by: Shakeel Butt Acked-by: Michal Hocko Cc: Johannes Weiner Cc: Muchun Song Cc: Roman Gushchin Cc: --- mm/memcontrol.c | 19 ++++++++++++------- 1 file changed, 12 insertions(+), 7 deletions(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 1709ac96bbdec5..c6b85e3a5a0b3e 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -2306,7 +2306,7 @@ static bool is_memcg_drain_needed(struct memcg_stock_pcp *stock, return flush; } -static void schedule_drain_work(int cpu, struct work_struct *work) +static bool schedule_drain_work(int cpu, struct work_struct *work) { /* * Protect housekeeping cpumask read and work enqueue together @@ -2315,8 +2315,11 @@ static void schedule_drain_work(int cpu, struct work_struct *work) * pending work on newly isolated CPUs. */ guard(rcu)(); - if (!cpu_is_isolated(cpu)) - queue_work_on(cpu, memcg_wq, work); + if (cpu_is_isolated(cpu)) + return false; + + queue_work_on(cpu, memcg_wq, work); + return true; } /* @@ -2348,8 +2351,9 @@ void drain_all_stock(struct mem_cgroup *root_memcg) &memcg_st->flags)) { if (cpu == curcpu) drain_local_memcg_stock(&memcg_st->work); - else - schedule_drain_work(cpu, &memcg_st->work); + else if (!schedule_drain_work(cpu, &memcg_st->work)) + clear_bit(FLUSHING_CACHED_CHARGE, + &memcg_st->flags); } if (!test_bit(FLUSHING_CACHED_CHARGE, &obj_st->flags) && @@ -2358,8 +2362,9 @@ void drain_all_stock(struct mem_cgroup *root_memcg) &obj_st->flags)) { if (cpu == curcpu) drain_local_obj_stock(&obj_st->work); - else - schedule_drain_work(cpu, &obj_st->work); + else if (!schedule_drain_work(cpu, &obj_st->work)) + clear_bit(FLUSHING_CACHED_CHARGE, + &obj_st->flags); } } migrate_enable(); From f4d29bb6a6beeb55017344be68b03de56014c2bd Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Fri, 28 Aug 2026 12:24:19 -0700 Subject: [PATCH 0588/1352] memcg: clear FLUSHING_CACHED_CHARGE on cpu offline Sashiko [1] reported that memcg_hotplug_cpu_dead() drains the stocks of the CPU which went away but leaves FLUSHING_CACHED_CHARGE alone. The flag can be set at that point: drain_all_stock() may have claimed the stock and queued the drain work shortly before the CPU went down. workqueue_offline_cpu() unbinds the per-cpu workers, so such a pending work item is executed by an unbound worker on some other CPU, where drain_local_memcg_stock() operates on this_cpu_ptr() and thus drains and clears the flag of that other CPU instead. Nothing clears the flag of the dead CPU, so drain_all_stock() would skip its stock forever once the CPU comes back online. Clear the flag of both stocks after draining them. Link: https://lore.kernel.org/20260828192419.3057939-1-shakeel.butt@linux.dev Link: https://sashiko.dev/#/patchset/20260828135036.7d44361f%40fangorn [1] Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Reviewed-by: Rik van Riel Reported-by: Sashiko Acked-by: Michal Hocko Cc: Johannes Weiner Cc: Muchun Song Cc: Roman Gushchin --- mm/memcontrol.c | 16 ++++++++++++++-- 1 file changed, 14 insertions(+), 2 deletions(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index c6b85e3a5a0b3e..05f5338765a995 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -2373,9 +2373,21 @@ void drain_all_stock(struct mem_cgroup *root_memcg) static int memcg_hotplug_cpu_dead(unsigned int cpu) { + struct memcg_stock_pcp *memcg_st = &per_cpu(memcg_stock, cpu); + struct obj_stock_pcp *obj_st = &per_cpu(obj_stock, cpu); + /* no need for the local lock */ - drain_obj_stock(&per_cpu(obj_stock, cpu)); - drain_stock_fully(&per_cpu(memcg_stock, cpu)); + drain_obj_stock(obj_st); + drain_stock_fully(memcg_st); + + /* + * A drain work queued before the CPU went away is executed by an + * unbound worker on some other CPU and clears that CPU's flag, so + * clear the flags here to make these stocks drainable again once + * the CPU comes back online. + */ + clear_bit(FLUSHING_CACHED_CHARGE, &memcg_st->flags); + clear_bit(FLUSHING_CACHED_CHARGE, &obj_st->flags); return 0; } From 235c1f066aa401a51d6157eb0376e26141f3a800 Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Fri, 28 Aug 2026 15:31:11 -0400 Subject: [PATCH 0589/1352] mm/mempolicy: take a cpuset cookie for the interleave node count alloc_pages_bulk_interleave() counts pol->nodes without a cpuset cookie: nodes = nodes_weight(pol->nodes); nr_pages_per_node = nr_pages / nodes; nodemask_t spans several words once MAX_NUMNODES exceeds BITS_PER_LONG, so a concurrent cpuset rebind can tear that read and yield an empty mask even though neither version of it was empty. The call then allocates nothing and returns 0. Some compilers will hoist the loop entry test above the division, because nr_pages_per_node is dead when the loop does not run. 682e: call ... <- nodes_weight() 6838: test %eax,%eax 683a: jle 692d <- nodes <= 0 skips the loop 684a: div %rcx So in most deployments, this div/0 is unreachable - but nothing in the source guarantees that, it's just not easily exercised. Take the cookie around the count and bail if the mask really is empty. Only the count needs it, interleave_nodes() takes the cookie itself so so a torn read there is already retried. A rebind landing mid-loop can still leave the count disagreeing with the mask, so the loop may revisit a node or skip one - but a rebind where nodes change causes migration, so a handful of misplaced pages isn't catastrophic in any sense. Measured on a 72 node VM (NODES_SHIFT=10) with a cgroup v2 cpuset flipping cpuset.mems between a word 0 and a word 1 node set, and the two word read artificially widened: 330 zero counts in 130414 calls without the cookie, and 401 retries with it. Link: https://lore.kernel.org/20260828193111.1023497-1-gourry@gourry.net Fixes: c00b6b961099 ("mm/vmalloc: introduce alloc_pages_bulk_array_mempolicy to accelerate memory allocation") Signed-off-by: Gregory Price (Meta) Signed-off-by: Andrew Morton Reported-by: Chelsy Ratnawat Closes: https://lore.kernel.org/all/20250907160829.91628-1-chelsyratnawat2001@gmail.com/ Reviewed-by: Huang Ying Assisted-by: Claude:claude-opus-5 Cc: Alistair Popple Cc: Byungchul Park Cc: Chenwandun Cc: David Hildenbrand Cc: Joshua Hahn Cc: Matthew Brost Cc: Rakie Kim Cc: "Uladzislau Rezki (Sony)" Cc: Zi Yan Cc: --- mm/mempolicy.c | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/mm/mempolicy.c b/mm/mempolicy.c index 79053ece02cd48..060a0eb2691709 100644 --- a/mm/mempolicy.c +++ b/mm/mempolicy.c @@ -2592,6 +2592,7 @@ static unsigned long alloc_pages_bulk_interleave(gfp_t gfp, struct mempolicy *pol, unsigned long nr_pages, struct page **page_array) { + unsigned int cpuset_mems_cookie; int nodes; unsigned long nr_pages_per_node; int delta; @@ -2599,7 +2600,16 @@ static unsigned long alloc_pages_bulk_interleave(gfp_t gfp, unsigned long nr_allocated; unsigned long total_allocated = 0; - nodes = nodes_weight(pol->nodes); + /* count the nodes, retry if a rebind happened during the read */ + do { + cpuset_mems_cookie = read_mems_allowed_begin(); + nodes = nodes_weight(pol->nodes); + } while (read_mems_allowed_retry(cpuset_mems_cookie)); + + /* if the nodemask has become invalid, we cannot do anything */ + if (!nodes) + return 0; + nr_pages_per_node = nr_pages / nodes; delta = nr_pages - nodes * nr_pages_per_node; From 268cc80cb4af7882650a45826d8eccc7d5e3c3b5 Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Wed, 19 Aug 2026 10:16:09 +0800 Subject: [PATCH 0590/1352] tools/mm/page_owner_sort: fix --sort option being silently ignored Patch series "tools/mm/page_owner_sort: fix --sort, add module filter, improve usage", v3. This series improves the page_owner_sort tool with a bug fix, a new module-name feature, and better usage text. Patch 1 fixes a long-standing bug where --sort was silently ignored when used without a short option (-a, -m, -p, etc.). The COMP_NO_FLAG case fell through to COMP_NUM and overwrote the sort conditions configured by parse_sort_args(). Patch 2 adds kernel module name support for sort, cull, and filter operations. Page owner stack traces already contain module names in the "function+0xNN/0xNN [module]" format produced by %pS, but page_owner_sort had no way to use them. Records without module frames are assigned "vmlinux". # Aggregate page usage per module ./page_owner_sort input.txt output.txt --cull=mod # Filter to records from xfs module only ./page_owner_sort input.txt output.txt --module xfs # Sort by module name, then by pid descending ./page_owner_sort input.txt output.txt --sort=mod,-pid Patch 3 lists all available sort keys with abbreviations and examples directly in the --sort help section so users no longer need to read the source to discover valid keys. This patch (of 3): When --sort is used without any short option (-a, -m, -p, etc.), compare_flag remains COMP_NO_FLAG. The switch (compare_flag) then falls through to the COMP_NUM case and calls set_single_cmp(), which unconditionally overwrites the sort conditions that parse_sort_args() already configured. This makes --sort silently ineffective unless a short option is also supplied. Split COMP_NO_FLAG out of the COMP_NUM fallthrough so that --sort is respected when no short option is present. Reproduction: # Before fix: ascending order (ignored --sort=-pid) ./page_owner_sort --sort=-pid input.txt output.txt # After fix: descending order as expected Link: https://lore.kernel.org/20260819021611.2910835-1-ye.liu@linux.dev Link: https://lore.kernel.org/20260819021611.2910835-2-ye.liu@linux.dev Signed-off-by: Ye Liu Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Liu Jing Cc: Yichong Chen --- tools/mm/page_owner_sort.c | 3 +++ 1 file changed, 3 insertions(+) diff --git a/tools/mm/page_owner_sort.c b/tools/mm/page_owner_sort.c index 22b3b500d33a47..f1dc0763c25cef 100644 --- a/tools/mm/page_owner_sort.c +++ b/tools/mm/page_owner_sort.c @@ -821,6 +821,9 @@ int main(int argc, char **argv) set_single_cmp(compare_stacktrace, SORT_ASC); break; case COMP_NO_FLAG: + if (sc.size > 0) + break; + /* fallthrough */ case COMP_NUM: set_single_cmp(compare_num, SORT_DESC); break; From af2d9a4faa34f2a0fa460c84a911b5914521839e Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Wed, 19 Aug 2026 10:16:10 +0800 Subject: [PATCH 0591/1352] tools/mm/page_owner_sort: add module name sort/cull/filter support Page owner stack traces already contain kernel module names in the "[module]" format produced by %pS, but page_owner_sort has no way to sort, cull, or filter by module. Extract the first module name from each record's stack trace using the regex \[([a-zA-Z0-9_]+)\]. Records whose stack traces contain no module frames are assigned "vmlinux". New options: -M Sort by module name --sort=mod Sort by module name (supports +/- prefix) --cull=mod Cull (aggregate) by module name --module Filter to records matching the given module(s) The module field is also printed in cull output when relevant. Link: https://lore.kernel.org/20260819021611.2910835-3-ye.liu@linux.dev Signed-off-by: Ye Liu Signed-off-by: Andrew Morton Cc: David Hildenbrand (Arm) Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Liu Jing Cc: Yichong Chen --- Documentation/mm/page_owner.rst | 8 ++- tools/mm/page_owner_sort.c | 122 +++++++++++++++++++++++++++----- 2 files changed, 110 insertions(+), 20 deletions(-) diff --git a/Documentation/mm/page_owner.rst b/Documentation/mm/page_owner.rst index a6bd3fe6423ad6..bd027377dff6ee 100644 --- a/Documentation/mm/page_owner.rst +++ b/Documentation/mm/page_owner.rst @@ -199,6 +199,7 @@ Usage -p Sort by pid. -P Sort by tgid. -n Sort by task command name. + -M Sort by module name. -r Sort by memory release time. -s Sort by stack trace. -t Sort by times (default). @@ -240,8 +241,10 @@ Usage group ID numbers appear in . --name Select by task command name. This selects the blocks whose task command name appear in . + --module Select by module name. This selects the blocks whose + module name appear in . - , , are single arguments in the form of a comma-separated list, + , , , are single arguments in the form of a comma-separated list, which offers a way to specify individual selecting rules. @@ -249,6 +252,7 @@ Usage ./page_owner_sort --pid=1 ./page_owner_sort --tgid=1,2,3 ./page_owner_sort --name name1,name2 + ./page_owner_sort --module xfs,ext4 STANDARD FORMAT SPECIFIERS ========================== @@ -265,6 +269,7 @@ STANDARD FORMAT SPECIFIERS ft free_ts timestamp of the page when it was released at alloc_ts timestamp of the page when it was allocated ator allocator memory allocator for pages + mod module kernel module name For --cull option: @@ -275,6 +280,7 @@ STANDARD FORMAT SPECIFIERS f free whether the page has been released or not st stacktrace stack trace of the page allocation ator allocator memory allocator for pages + mod module kernel module name Filtering page_owner output ============================ diff --git a/tools/mm/page_owner_sort.c b/tools/mm/page_owner_sort.c index f1dc0763c25cef..a5b61b4abc2f08 100644 --- a/tools/mm/page_owner_sort.c +++ b/tools/mm/page_owner_sort.c @@ -25,11 +25,13 @@ #include #define TASK_COMM_LEN 16 +#define MODULE_NAME_LEN 64 struct block_list { char *txt; char *comm; // task command name char *stacktrace; + char *module; // kernel module name __u64 ts_nsec; int len; int num; @@ -41,7 +43,8 @@ struct block_list { enum FILTER_BIT { FILTER_PID = 1<<1, FILTER_TGID = 1<<2, - FILTER_COMM = 1<<3 + FILTER_COMM = 1<<3, + FILTER_MODULE = 1<<4 }; enum FILTER_RESULT { @@ -55,7 +58,8 @@ enum CULL_BIT { CULL_TGID = 1<<2, CULL_COMM = 1<<3, CULL_STACKTRACE = 1<<4, - CULL_ALLOCATOR = 1<<5 + CULL_ALLOCATOR = 1<<5, + CULL_MODULE = 1<<6 }; enum ALLOCATOR_BIT { ALLOCATOR_CMA = 1<<1, @@ -65,7 +69,8 @@ enum ALLOCATOR_BIT { }; enum ARG_TYPE { ARG_TXT, ARG_COMM, ARG_STACKTRACE, ARG_ALLOC_TS, ARG_CULL_TIME, - ARG_PAGE_NUM, ARG_PID, ARG_TGID, ARG_UNKNOWN, ARG_ALLOCATOR + ARG_PAGE_NUM, ARG_PID, ARG_TGID, ARG_UNKNOWN, ARG_ALLOCATOR, + ARG_MODULE }; enum SORT_ORDER { SORT_ASC = 1, @@ -79,15 +84,18 @@ enum COMP_FLAG { COMP_STACK = 1<<3, COMP_NUM = 1<<4, COMP_TGID = 1<<5, - COMP_COMM = 1<<6 + COMP_COMM = 1<<6, + COMP_MODULE = 1<<7 }; struct filter_condition { pid_t *pids; pid_t *tgids; char **comms; + char **modules; int pids_size; int tgids_size; int comms_size; + int modules_size; }; struct sort_condition { int (**cmps)(const void *, const void *); @@ -101,6 +109,7 @@ static regex_t pid_pattern; static regex_t tgid_pattern; static regex_t comm_pattern; static regex_t ts_nsec_pattern; +static regex_t module_pattern; static struct block_list *list; static int list_size; static int max_size; @@ -184,6 +193,13 @@ static int compare_comm(const void *p1, const void *p2) return strcmp(l1->comm, l2->comm); } +static int compare_module(const void *p1, const void *p2) +{ + const struct block_list *l1 = p1, *l2 = p2; + + return strcmp(l1->module, l2->module); +} + static int compare_ts(const void *p1, const void *p2) { const struct block_list *l1 = p1, *l2 = p2; @@ -207,6 +223,8 @@ static int compare_cull_condition(const void *p1, const void *p2) return compare_tgid(p1, p2); if ((cull & CULL_COMM) && compare_comm(p1, p2)) return compare_comm(p1, p2); + if ((cull & CULL_MODULE) && compare_module(p1, p2)) + return compare_module(p1, p2); if ((cull & CULL_ALLOCATOR) && compare_allocator(p1, p2)) return compare_allocator(p1, p2); return 0; @@ -411,9 +429,33 @@ static char *get_comm(char *buf) return comm_str; } +static char *get_module(char *buf) +{ + char *module_str = malloc(MODULE_NAME_LEN); + regmatch_t pmatch[2]; + int val_len; + + if (!module_str) + return NULL; + memset(module_str, 0, MODULE_NAME_LEN); + if (regexec(&module_pattern, buf, 2, pmatch, REG_NOTBOL) != 0 || pmatch[1].rm_so == -1) { + strcpy(module_str, "vmlinux"); + return module_str; + } + + val_len = pmatch[1].rm_eo - pmatch[1].rm_so; + if ((size_t)val_len >= MODULE_NAME_LEN) + val_len = MODULE_NAME_LEN - 1; + memcpy(module_str, buf + pmatch[1].rm_so, val_len); + module_str[val_len] = '\0'; + + return module_str; +} + static void free_block_list(struct block_list *block) { free(block->comm); + free(block->module); free(block->txt); } @@ -433,6 +475,8 @@ static int get_arg_type(const char *arg) return ARG_ALLOC_TS; else if (!strcmp(arg, "allocator") || !strcmp(arg, "ator")) return ARG_ALLOCATOR; + else if (!strcmp(arg, "module") || !strcmp(arg, "mod")) + return ARG_MODULE; else { return ARG_UNKNOWN; } @@ -483,25 +527,36 @@ static bool match_str_list(const char *str, char **list, int list_size) static enum FILTER_RESULT filter_record(char *buf) { - char *comm; + char *comm, *module; if ((filter & FILTER_PID) && !match_num_list(get_pid(buf), fc.pids, fc.pids_size)) return FILTER_SKIP; if ((filter & FILTER_TGID) && !match_num_list(get_tgid(buf), fc.tgids, fc.tgids_size)) return FILTER_SKIP; - if (!(filter & FILTER_COMM)) + if (!(filter & (FILTER_COMM | FILTER_MODULE))) return FILTER_MATCH; - comm = get_comm(buf); - if (!comm) - return FILTER_ERROR; - - if (!match_str_list(comm, fc.comms, fc.comms_size)) { + if (filter & FILTER_COMM) { + comm = get_comm(buf); + if (!comm) + return FILTER_ERROR; + if (!match_str_list(comm, fc.comms, fc.comms_size)) { + free(comm); + return FILTER_SKIP; + } free(comm); - return FILTER_SKIP; } - free(comm); + if (filter & FILTER_MODULE) { + module = get_module(buf); + if (!module) + return FILTER_ERROR; + if (!match_str_list(module, fc.modules, fc.modules_size)) { + free(module); + return FILTER_SKIP; + } + free(module); + } return FILTER_MATCH; } @@ -547,6 +602,12 @@ static bool add_list(char *buf, int len, char *ext_buf) list[list_size].stacktrace++; list[list_size].ts_nsec = get_ts_nsec(buf); list[list_size].allocator = get_allocator(buf, ext_buf); + list[list_size].module = get_module(buf); + if (!list[list_size].module) { + fprintf(stderr, "Out of memory\n"); + free_block_list(&list[list_size]); + return false; + } list_size++; if (list_size % 1000 == 0) { printf("loaded %d\r", list_size); @@ -573,6 +634,8 @@ static bool parse_cull_args(const char *arg_str) cull |= CULL_STACKTRACE; else if (arg_type == ARG_ALLOCATOR) cull |= CULL_ALLOCATOR; + else if (arg_type == ARG_MODULE) + cull |= CULL_MODULE; else { free_explode(args, size); return false; @@ -635,6 +698,8 @@ static bool parse_sort_args(const char *arg_str) sc.cmps[i] = compare_txt; else if (arg_type == ARG_ALLOCATOR) sc.cmps[i] = compare_allocator; + else if (arg_type == ARG_MODULE) + sc.cmps[i] = compare_module; else { free_explode(args, size); sc.size = 0; @@ -691,7 +756,8 @@ static void usage(void) "-p\t\t\tSort by pid.\n" "-P\t\t\tSort by tgid.\n" "-s\t\t\tSort by the stacktrace.\n" - "-t\t\t\tSort by number of times record is seen (default).\n\n" + "-t\t\t\tSort by number of times record is seen (default).\n" + "-M\t\t\tSort by module name.\n\n" "--pid \t\tSelect by pid. This selects the information" " of\n\t\t\tblocks whose process ID numbers appear in .\n" "--tgid \tSelect by tgid. This selects the information" @@ -700,10 +766,11 @@ static void usage(void) "--name \tSelect by command name. This selects the" " information\n\t\t\tof blocks whose command name appears in" " .\n" - "--cull \t\tCull by user-defined rules. is a " - "single\n\t\t\targument in the form of a comma-separated list " - "with some\n\t\t\tcommon fields predefined (pid, tgid, comm, " - "stacktrace, allocator)\n" + "--module \tSelect by module name. This selects the information\n" + "\t\t\tof blocks whose module name appears in .\n" + "--cull \t\tCull by user-defined rules. is a single\n" + "\t\t\targument in the form of a comma-separated list with some\n" + "\t\t\tcommon fields predefined (pid, tgid, comm, stacktrace, allocator, module)\n" "--sort \t\tSpecify sort order as: [+|-]key[,[+|-]key[,...]]\n" ); } @@ -721,13 +788,14 @@ int main(int argc, char **argv) { "name", required_argument, NULL, 3 }, { "cull", required_argument, NULL, 4 }, { "sort", required_argument, NULL, 5 }, + { "module", required_argument, NULL, 6 }, { "help", no_argument, NULL, 'h' }, { 0, 0, 0, 0}, }; compare_flag = COMP_NO_FLAG; - while ((opt = getopt_long(argc, argv, "admnpstPh", longopts, NULL)) != -1) + while ((opt = getopt_long(argc, argv, "admnpstPMh", longopts, NULL)) != -1) switch (opt) { case 'a': compare_flag |= COMP_ALLOC; @@ -753,6 +821,9 @@ int main(int argc, char **argv) case 'n': compare_flag |= COMP_COMM; break; + case 'M': + compare_flag |= COMP_MODULE; + break; case 'h': usage(); exit(0); @@ -792,6 +863,10 @@ int main(int argc, char **argv) exit(1); } break; + case 6: + filter = filter | FILTER_MODULE; + fc.modules = explode(',', optarg, &fc.modules_size); + break; default: usage(); exit(1); @@ -833,6 +908,9 @@ int main(int argc, char **argv) case COMP_COMM: set_single_cmp(compare_comm, SORT_ASC); break; + case COMP_MODULE: + set_single_cmp(compare_module, SORT_ASC); + break; default: usage(); exit(1); @@ -855,6 +933,8 @@ int main(int argc, char **argv) goto out_comm; if (!check_regcomp(&ts_nsec_pattern, "ts\\s*([0-9]*)\\s*ns")) goto out_ts; + if (!check_regcomp(&module_pattern, "\\+0x[0-9a-f]+/0x[0-9a-f]+\\s*\\[([a-zA-Z0-9_-]+)\\]")) + goto out_module; fstat(fileno(fin), &st); max_size = st.st_size / 100; /* hack ... */ @@ -920,6 +1000,8 @@ int main(int argc, char **argv) fprintf(fout, ", TGID %d", list[i].tgid); if (cull & CULL_COMM || filter & FILTER_COMM) fprintf(fout, ", task_comm_name: %s", list[i].comm); + if (cull & CULL_MODULE || filter & FILTER_MODULE) + fprintf(fout, ", module: %s", list[i].module); if (cull & CULL_ALLOCATOR) { fprintf(fout, ", "); print_allocator(fout, list[i].allocator); @@ -940,6 +1022,8 @@ int main(int argc, char **argv) free_block_list(&list[i]); free(list); } +out_module: + regfree(&module_pattern); out_ts: regfree(&ts_nsec_pattern); out_comm: From d2dc3ca78685a2e45828ea62ee9736674e892490 Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Wed, 19 Aug 2026 10:16:11 +0800 Subject: [PATCH 0592/1352] tools/mm/page_owner_sort: show available sort keys in usage text The --sort option accepts abbreviated or complete key names, but the usage text never listed them. Users had to read the source or the documentation to discover valid keys. List all available keys (full form and abbreviation) with a brief description and examples directly in the --sort help section. Link: https://lore.kernel.org/20260819021611.2910835-4-ye.liu@linux.dev Signed-off-by: Ye Liu Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Yichong Chen Cc: Liu Jing --- tools/mm/page_owner_sort.c | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/tools/mm/page_owner_sort.c b/tools/mm/page_owner_sort.c index a5b61b4abc2f08..6c2ac5d9dc4b6e 100644 --- a/tools/mm/page_owner_sort.c +++ b/tools/mm/page_owner_sort.c @@ -772,6 +772,15 @@ static void usage(void) "\t\t\targument in the form of a comma-separated list with some\n" "\t\t\tcommon fields predefined (pid, tgid, comm, stacktrace, allocator, module)\n" "--sort \t\tSpecify sort order as: [+|-]key[,[+|-]key[,...]]\n" + "\t\t\tAvailable keys:\n" + "\t\t\t pid(p), tgid(tg), name(n), stacktrace(st),\n" + "\t\t\t txt(T), alloc_ts(at), allocator(ator), module(mod)\n" + "\t\t\tThe \"+\" is optional since default direction is\n" + "\t\t\tincreasing numerical or lexicographic order.\n" + "\t\t\tMixed use of abbreviated and complete-form is allowed.\n" + "\t\t\tExamples:\n" + "\t\t\t --sort=n,+pid,-tgid\n" + "\t\t\t --sort=mod,at\n" ); } From 568899efec8bb4050624fd1ae3792060409df972 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 19 Aug 2026 18:20:10 -0700 Subject: [PATCH 0593/1352] memcg: trim the per-cpu charge stock instead of draining it Joy reported that an application generating a request/response traffic pattern spends 44.6% to 57.0% of CPU in the memcg charge/uncharge path for a range of message sizes, against 0.27% to 0.71% outside that range. Running from the root memcg, where socket memory accounting is skipped, recovers the performance. Tracing the charge path showed that the application generates a pattern where the write syscall charges one page and the read syscall uncharges two pages on the same CPU. This hits a corner case in the memcg percpu stock code that thrashes the stock continuously. In the memcg percpu stock code, MEMCG_CHARGE_BATCH (64) is both the high watermark and the emptying target, i.e. on a request to charge one page the kernel charges MEMCG_CHARGE_BATCH pages and caches (MEMCG_CHARGE_BATCH - 1) of them in the percpu stock. The following uncharge of 2 pages takes the cached count to (MEMCG_CHARGE_BATCH + 1), and refill_stock() then empties the cache completely. With such a pattern the percpu stock becomes completely ineffective. Instead of a single boundary point for charges, use the technique the page allocator uses for its own percpu caches, which keeps the watermark and the emptying target apart: nr_pcp_free() frees between batch and high - batch pages, leaving at least pcp->batch on the list. Add a high watermark MEMCG_STOCK_HIGH and, once the cached count goes over it, return only the pages above MEMCG_STOCK_LOW. The watermarks are MEMCG_CHARGE_BATCH apart, so a page_counter update still covers a full batch. For now, keep MEMCG_STOCK_HIGH same as MEMCG_CHARGE_BATCH and in future we will reevaluate if it makes sense to increase it. Link: https://lore.kernel.org/20260820012010.2016086-1-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Reported-by: Joy Chaoyue Xiong Acked-by: Michal Hocko Cc: Jakub Kacinski Cc: Johannes Weiner Cc: Joshua Hahn Cc: Muchun Song Cc: Roman Gushchin --- mm/memcontrol.c | 25 +++++++++++++++++++------ 1 file changed, 19 insertions(+), 6 deletions(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 05f5338765a995..bfd0a74fac9239 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -2048,6 +2048,15 @@ void mem_cgroup_print_oom_group(struct mem_cgroup *memcg) * nr_pages in a single cacheline. This may change in future. */ #define NR_MEMCG_STOCK 7 + +/* + * Watermarks for a charge stock slot, in the spirit of pcp->high and + * pcp->batch: MEMCG_STOCK_HIGH is the high watermark at which a slot is + * trimmed, and it is trimmed down to MEMCG_STOCK_LOW rather than emptied. + */ +#define MEMCG_STOCK_LOW (MEMCG_CHARGE_BATCH / 2) +#define MEMCG_STOCK_HIGH (MEMCG_CHARGE_BATCH) + #define FLUSHING_CACHED_CHARGE 0 struct memcg_stock_pcp { local_trylock_t lock; @@ -2228,17 +2237,18 @@ static void refill_stock(struct mem_cgroup *memcg, unsigned int nr_pages) { struct memcg_stock_pcp *stock; struct mem_cgroup *cached; - uint8_t stock_pages; + unsigned int stock_pages; bool success = false; int empty_slot = -1; int i; /* - * For now limit MEMCG_CHARGE_BATCH to 127 and less. In future if we - * decide to increase it more than 127 then we will need more careful - * handling of nr_pages[] in struct memcg_stock_pcp. + * nr_pages[] is a uint8_t and a slot's count is capped at + * MEMCG_STOCK_HIGH. Raising MEMCG_CHARGE_BATCH beyond 127 would need + * more careful handling of nr_pages[] in struct memcg_stock_pcp. */ BUILD_BUG_ON(MEMCG_CHARGE_BATCH > S8_MAX); + BUILD_BUG_ON(MEMCG_STOCK_HIGH > U8_MAX); VM_WARN_ON_ONCE(mem_cgroup_is_root(memcg)); @@ -2259,9 +2269,12 @@ static void refill_stock(struct mem_cgroup *memcg, unsigned int nr_pages) empty_slot = i; if (memcg == READ_ONCE(stock->cached[i])) { stock_pages = READ_ONCE(stock->nr_pages[i]) + nr_pages; + if (stock_pages > MEMCG_STOCK_HIGH) { + memcg_uncharge(memcg, + stock_pages - MEMCG_STOCK_LOW); + stock_pages = MEMCG_STOCK_LOW; + } WRITE_ONCE(stock->nr_pages[i], stock_pages); - if (stock_pages > MEMCG_CHARGE_BATCH) - drain_stock(stock, i); success = true; break; } From 0d744684705ef9b4746c530aa781a530c8641773 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 2 Sep 2026 10:43:04 -0700 Subject: [PATCH 0594/1352] memcg: remove v1 soft limit reclaim Patch series "memcg: remove the v1 soft limit", v2. The v1 soft limit was deprecated in v6.12 by commit 569c4f62d84a ("memcg: initiate deprecation of v1 soft limit") and nobody has reported depending on it in the ~21 months since. memory.low and memory.min in v2 have covered the same ground for far longer. The knob has since been made inert by "memcg: make the v1 soft limit knob inert", already queued in mm-hotfixes as a backportable fix for a syzbot report [1]. Nothing can enter the soft limit rbtree anymore, so this series just deletes the machinery that is now dead: the reclaim pass in kswapd and direct reclaim, mem_cgroup_shrink_node() and its tracepoints, the per-node rbtree, lru_gen_soft_reclaim() and the MEMCG_LRU_HEAD op, the per-node tree fields, mem_cgroup->soft_limit, and finally the v1 event ratelimiting which is now down to a single target. memory.soft_limit_in_bytes itself is untouched: writes stay ignored and reads keep returning the maximum value. This patch (of 8): Nothing can put a cgroup on the soft limit rbtree anymore, so the tree is always empty and both callers of memcg1_soft_limit_reclaim() are guaranteed no-ops. Remove the reclaim pass from direct reclaim and from kswapd, along with its implementation. In shrink_zones() this leaves the global reclaim branch with a last_pgdat check that is now redundant with the identical check right below it, so drop it and move the explaining comment down to the check that remains. That check could only ever fire once last_pgdat was set, which implies first_pgdat had already been assigned, so skipping it does not change which node consider_reclaim_throttle() gets. Link: https://lore.kernel.org/20260902174311.1772372-1-shakeel.butt@linux.dev Link: https://lore.kernel.org/20260902174311.1772372-2-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Acked-by: Michal Hocko Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Muchun Song Cc: Roman Gushchin Cc: T.J. Mercier --- include/linux/memcontrol.h | 12 --- mm/memcontrol-v1.c | 175 ------------------------------------- mm/vmscan.c | 39 ++------- 3 files changed, 6 insertions(+), 220 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index da625d2edb3bab..11c1fa88d6fd0c 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -1924,10 +1924,6 @@ static inline bool mem_cgroup_zswap_writeback_enabled(struct mem_cgroup *memcg) /* Cgroup v1-related declarations */ #ifdef CONFIG_MEMCG_V1 -unsigned long memcg1_soft_limit_reclaim(pg_data_t *pgdat, int order, - gfp_t gfp_mask, - unsigned long *total_scanned); - bool mem_cgroup_oom_synchronize(bool wait); static inline bool task_in_memcg_oom(struct task_struct *p) @@ -1948,14 +1944,6 @@ static inline void mem_cgroup_exit_user_fault(void) } #else /* CONFIG_MEMCG_V1 */ -static inline -unsigned long memcg1_soft_limit_reclaim(pg_data_t *pgdat, int order, - gfp_t gfp_mask, - unsigned long *total_scanned) -{ - return 0; -} - static inline bool task_in_memcg_oom(struct task_struct *p) { return false; diff --git a/mm/memcontrol-v1.c b/mm/memcontrol-v1.c index 05ef55cae4dc61..b38b8d0f7f51c5 100644 --- a/mm/memcontrol-v1.c +++ b/mm/memcontrol-v1.c @@ -34,13 +34,6 @@ struct mem_cgroup_tree { static struct mem_cgroup_tree soft_limit_tree __read_mostly; -/* - * Maximum loops in mem_cgroup_soft_reclaim(), used for soft - * limit reclaim to prevent infinite loops, if they ever occur. - */ -#define MEM_CGROUP_MAX_RECLAIM_LOOPS 100 -#define MEM_CGROUP_MAX_SOFT_LIMIT_RECLAIM_LOOPS 2 - /* for OOM */ struct mem_cgroup_eventfd_list { struct list_head list; @@ -233,174 +226,6 @@ void memcg1_remove_from_trees(struct mem_cgroup *memcg) } } -static struct mem_cgroup_per_node * -__mem_cgroup_largest_soft_limit_node(struct mem_cgroup_tree_per_node *mctz) -{ - struct mem_cgroup_per_node *mz; - -retry: - mz = NULL; - if (!mctz->rb_rightmost) - goto done; /* Nothing to reclaim from */ - - mz = rb_entry(mctz->rb_rightmost, - struct mem_cgroup_per_node, tree_node); - /* - * Remove the node now but someone else can add it back, - * we will to add it back at the end of reclaim to its correct - * position in the tree. - */ - __mem_cgroup_remove_exceeded(mz, mctz); - if (!soft_limit_excess(mz->memcg) || - !css_tryget(&mz->memcg->css)) - goto retry; -done: - return mz; -} - -static struct mem_cgroup_per_node * -mem_cgroup_largest_soft_limit_node(struct mem_cgroup_tree_per_node *mctz) -{ - struct mem_cgroup_per_node *mz; - - spin_lock_irq(&mctz->lock); - mz = __mem_cgroup_largest_soft_limit_node(mctz); - spin_unlock_irq(&mctz->lock); - return mz; -} - -static int mem_cgroup_soft_reclaim(struct mem_cgroup *root_memcg, - pg_data_t *pgdat, - gfp_t gfp_mask, - unsigned long *total_scanned) -{ - struct mem_cgroup *victim = NULL; - int total = 0; - int loop = 0; - unsigned long excess; - unsigned long nr_scanned; - struct mem_cgroup_reclaim_cookie reclaim = { - .pgdat = pgdat, - }; - - excess = soft_limit_excess(root_memcg); - - while (1) { - victim = mem_cgroup_iter(root_memcg, victim, &reclaim); - if (!victim) { - loop++; - if (loop >= 2) { - /* - * If we have not been able to reclaim - * anything, it might because there are - * no reclaimable pages under this hierarchy - */ - if (!total) - break; - /* - * We want to do more targeted reclaim. - * excess >> 2 is not to excessive so as to - * reclaim too much, nor too less that we keep - * coming back to reclaim from this cgroup - */ - if (total >= (excess >> 2) || - (loop > MEM_CGROUP_MAX_RECLAIM_LOOPS)) - break; - } - continue; - } - total += mem_cgroup_shrink_node(victim, gfp_mask, false, - pgdat, &nr_scanned); - *total_scanned += nr_scanned; - if (!soft_limit_excess(root_memcg)) - break; - } - mem_cgroup_iter_break(root_memcg, victim); - return total; -} - -unsigned long memcg1_soft_limit_reclaim(pg_data_t *pgdat, int order, - gfp_t gfp_mask, - unsigned long *total_scanned) -{ - unsigned long nr_reclaimed = 0; - struct mem_cgroup_per_node *mz, *next_mz = NULL; - unsigned long reclaimed; - int loop = 0; - struct mem_cgroup_tree_per_node *mctz; - unsigned long excess; - - if (lru_gen_enabled()) - return 0; - - if (order > 0) - return 0; - - mctz = soft_limit_tree.rb_tree_per_node[pgdat->node_id]; - - /* - * Do not even bother to check the largest node if the root - * is empty. Do it lockless to prevent lock bouncing. Races - * are acceptable as soft limit is best effort anyway. - */ - if (!mctz || RB_EMPTY_ROOT(&mctz->rb_root)) - return 0; - - /* - * This loop can run a while, specially if mem_cgroup's continuously - * keep exceeding their soft limit and putting the system under - * pressure - */ - do { - if (next_mz) - mz = next_mz; - else - mz = mem_cgroup_largest_soft_limit_node(mctz); - if (!mz) - break; - - reclaimed = mem_cgroup_soft_reclaim(mz->memcg, pgdat, - gfp_mask, total_scanned); - nr_reclaimed += reclaimed; - spin_lock_irq(&mctz->lock); - - /* - * If we failed to reclaim anything from this memory cgroup - * it is time to move on to the next cgroup - */ - next_mz = NULL; - if (!reclaimed) - next_mz = __mem_cgroup_largest_soft_limit_node(mctz); - - excess = soft_limit_excess(mz->memcg); - /* - * One school of thought says that we should not add - * back the node to the tree if reclaim returns 0. - * But our reclaim could return 0, simply because due - * to priority we are exposing a smaller subset of - * memory to reclaim from. Consider this as a longer - * term TODO. - */ - /* If excess == 0, no tree ops */ - __mem_cgroup_insert_exceeded(mz, mctz, excess); - spin_unlock_irq(&mctz->lock); - css_put(&mz->memcg->css); - loop++; - /* - * Could not reclaim anything and there are no more - * mem cgroups to try or we seem to be looping without - * reclaiming anything. - */ - if (!nr_reclaimed && - (next_mz == NULL || - loop > MEM_CGROUP_MAX_SOFT_LIMIT_RECLAIM_LOOPS)) - break; - } while (!nr_reclaimed); - if (next_mz) - css_put(&next_mz->memcg->css); - return nr_reclaimed; -} - static u64 mem_cgroup_move_charge_read(struct cgroup_subsys_state *css, struct cftype *cft) { diff --git a/mm/vmscan.c b/mm/vmscan.c index fdd13299a04a93..0e04eaf64af3ae 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -6439,8 +6439,6 @@ static void shrink_zones(struct zonelist *zonelist, struct scan_control *sc) { struct zoneref *z; struct zone *zone; - unsigned long nr_soft_reclaimed; - unsigned long nr_soft_scanned; gfp_t orig_mask; pg_data_t *last_pgdat = NULL; pg_data_t *first_pgdat = NULL; @@ -6482,35 +6480,17 @@ static void shrink_zones(struct zonelist *zonelist, struct scan_control *sc) sc->compaction_ready = true; continue; } - - /* - * Shrink each node in the zonelist once. If the - * zonelist is ordered by zone (not the default) then a - * node may be shrunk multiple times but in that case - * the user prefers lower zones being preserved. - */ - if (zone->zone_pgdat == last_pgdat) - continue; - - /* - * This steals pages from memory cgroups over softlimit - * and returns the number of reclaimed pages and - * scanned pages. This works for global memory pressure - * and balancing, not for a memcg's limit. - */ - nr_soft_scanned = 0; - nr_soft_reclaimed = memcg1_soft_limit_reclaim(zone->zone_pgdat, - sc->order, sc->gfp_mask, - &nr_soft_scanned); - sc->nr_reclaimed += nr_soft_reclaimed; - sc->nr_scanned += nr_soft_scanned; - /* need some check for avoid more shrink_zone() */ } if (!first_pgdat) first_pgdat = zone->zone_pgdat; - /* See comment about same check for global reclaim above */ + /* + * Shrink each node in the zonelist once. If the zonelist is + * ordered by zone (not the default) then a node may be shrunk + * multiple times but in that case the user prefers lower zones + * being preserved. + */ if (zone->zone_pgdat == last_pgdat) continue; last_pgdat = zone->zone_pgdat; @@ -7171,8 +7151,6 @@ clear_reclaim_active(pg_data_t *pgdat, int highest_zoneidx) static int balance_pgdat(pg_data_t *pgdat, int order, int highest_zoneidx) { int i; - unsigned long nr_soft_reclaimed; - unsigned long nr_soft_scanned; unsigned long pflags; unsigned long nr_boost_reclaim; unsigned long zone_boosts[MAX_NR_ZONES] = { 0, }; @@ -7278,12 +7256,7 @@ static int balance_pgdat(pg_data_t *pgdat, int order, int highest_zoneidx) */ kswapd_age_node(pgdat, &sc); - /* Call soft limit reclaim before calling shrink_node. */ sc.nr_scanned = 0; - nr_soft_scanned = 0; - nr_soft_reclaimed = memcg1_soft_limit_reclaim(pgdat, sc.order, - sc.gfp_mask, &nr_soft_scanned); - sc.nr_reclaimed += nr_soft_reclaimed; /* * There should be no need to raise the scanning priority if From 5503a2e74bf310cc5913bd3ce934e2e66aeb34c0 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 2 Sep 2026 10:43:05 -0700 Subject: [PATCH 0595/1352] memcg: remove mem_cgroup_shrink_node() Its only caller was soft limit reclaim, which is gone. Link: https://lore.kernel.org/20260902174311.1772372-3-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Acked-by: Michal Hocko Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Muchun Song Cc: Roman Gushchin Cc: T.J. Mercier --- mm/internal.h | 4 ---- mm/vmscan.c | 41 ----------------------------------------- 2 files changed, 45 deletions(-) diff --git a/mm/internal.h b/mm/internal.h index 5cc220db907668..e16f1250b25c80 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -78,10 +78,6 @@ unsigned long try_to_free_mem_cgroup_pages(struct mem_cgroup *memcg, gfp_t gfp_mask, unsigned int reclaim_options, int *swappiness); -unsigned long mem_cgroup_shrink_node(struct mem_cgroup *memcg, - gfp_t gfp_mask, bool noswap, - pg_data_t *pgdat, - unsigned long *nr_scanned); #ifdef CONFIG_NUMA extern int sysctl_min_unmapped_ratio; diff --git a/mm/vmscan.c b/mm/vmscan.c index 0e04eaf64af3ae..d66b5cd167d6f2 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -6805,47 +6805,6 @@ unsigned long try_to_free_pages(struct zonelist *zonelist, int order, #ifdef CONFIG_MEMCG -/* Only used by soft limit reclaim. Do not reuse for anything else. */ -unsigned long mem_cgroup_shrink_node(struct mem_cgroup *memcg, - gfp_t gfp_mask, bool noswap, - pg_data_t *pgdat, - unsigned long *nr_scanned) -{ - struct lruvec *lruvec = mem_cgroup_lruvec(memcg, pgdat); - struct scan_control sc = { - .nr_to_reclaim = SWAP_CLUSTER_MAX, - .target_mem_cgroup = memcg, - .may_writepage = 1, - .may_unmap = 1, - .reclaim_idx = MAX_NR_ZONES - 1, - .may_swap = !noswap, - }; - - WARN_ON_ONCE(!current->reclaim_state); - - sc.gfp_mask = (gfp_mask & GFP_RECLAIM_MASK) | - (GFP_HIGHUSER_MOVABLE & ~GFP_RECLAIM_MASK); - - trace_mm_vmscan_memcg_softlimit_reclaim_begin(sc.gfp_mask, - sc.order, - memcg); - - /* - * NOTE: Although we can get the priority field, using it - * here is not a good idea, since it limits the pages we can scan. - * if we don't reclaim here, the shrink_node from balance_pgdat - * will pick up pages from other mem cgroup's as well. We hack - * the priority and make it zero. - */ - shrink_lruvec(lruvec, &sc); - - trace_mm_vmscan_memcg_softlimit_reclaim_end(sc.nr_reclaimed, memcg); - - *nr_scanned = sc.nr_scanned; - - return sc.nr_reclaimed; -} - unsigned long try_to_free_mem_cgroup_pages(struct mem_cgroup *memcg, unsigned long nr_pages, gfp_t gfp_mask, From e90747f77d6263b0a0009c6afa769437c9abd106 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 2 Sep 2026 10:43:06 -0700 Subject: [PATCH 0596/1352] memcg: remove the soft limit reclaim tracepoints mm_vmscan_memcg_softlimit_reclaim_begin and mm_vmscan_memcg_softlimit_reclaim_end were only emitted by mem_cgroup_shrink_node(), which is gone, so they can never fire again. Link: https://lore.kernel.org/20260902174311.1772372-4-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Acked-by: Michal Hocko Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Muchun Song Cc: Roman Gushchin Cc: T.J. Mercier --- include/trace/events/vmscan.h | 14 -------------- 1 file changed, 14 deletions(-) diff --git a/include/trace/events/vmscan.h b/include/trace/events/vmscan.h index b4bf7b8def1f5f..8a872990b4bee6 100644 --- a/include/trace/events/vmscan.h +++ b/include/trace/events/vmscan.h @@ -214,13 +214,6 @@ DEFINE_EVENT(mm_vmscan_direct_reclaim_begin_template, mm_vmscan_memcg_reclaim_be TP_ARGS(gfp_flags, order, memcg) ); - -DEFINE_EVENT(mm_vmscan_direct_reclaim_begin_template, mm_vmscan_memcg_softlimit_reclaim_begin, - - TP_PROTO(gfp_t gfp_flags, int order, struct mem_cgroup *memcg), - - TP_ARGS(gfp_flags, order, memcg) -); #endif /* CONFIG_MEMCG */ DECLARE_EVENT_CLASS(mm_vmscan_direct_reclaim_end_template, @@ -260,13 +253,6 @@ DEFINE_EVENT(mm_vmscan_direct_reclaim_end_template, mm_vmscan_memcg_reclaim_end, TP_ARGS(nr_reclaimed, memcg) ); - -DEFINE_EVENT(mm_vmscan_direct_reclaim_end_template, mm_vmscan_memcg_softlimit_reclaim_end, - - TP_PROTO(unsigned long nr_reclaimed, struct mem_cgroup *memcg), - - TP_ARGS(nr_reclaimed, memcg) -); #endif /* CONFIG_MEMCG */ TRACE_EVENT(mm_shrink_slab_start, From 132f2c422700863c72f6552205a17293e1378c5a Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 2 Sep 2026 10:43:07 -0700 Subject: [PATCH 0597/1352] memcg: remove the soft limit rbtree With soft limit reclaim gone, the per-node rbtree of cgroups in excess has no readers left. Remove the tree, the helpers maintaining it, and the subsys_initcall that existed only to allocate it. memcg1_check_events() no longer needs to feed it, which also drops the last caller of lru_gen_soft_reclaim(). Link: https://lore.kernel.org/20260902174311.1772372-5-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Acked-by: Michal Hocko Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Muchun Song Cc: Roman Gushchin Cc: T.J. Mercier --- mm/memcontrol-v1.c | 176 +-------------------------------------------- mm/memcontrol-v1.h | 2 - mm/memcontrol.c | 1 - 3 files changed, 2 insertions(+), 177 deletions(-) diff --git a/mm/memcontrol-v1.c b/mm/memcontrol-v1.c index b38b8d0f7f51c5..475f998b764318 100644 --- a/mm/memcontrol-v1.c +++ b/mm/memcontrol-v1.c @@ -17,23 +17,6 @@ #include "swap_table.h" #include "memcontrol-v1.h" -/* - * Cgroups above their limits are maintained in a RB-Tree, independent of - * their hierarchy representation - */ - -struct mem_cgroup_tree_per_node { - struct rb_root rb_root; - struct rb_node *rb_rightmost; - spinlock_t lock; -}; - -struct mem_cgroup_tree { - struct mem_cgroup_tree_per_node *rb_tree_per_node[MAX_NUMNODES]; -}; - -static struct mem_cgroup_tree soft_limit_tree __read_mostly; - /* for OOM */ struct mem_cgroup_eventfd_list { struct list_head list; @@ -99,133 +82,6 @@ static struct lockdep_map memcg_oom_lock_dep_map = { DEFINE_SPINLOCK(memcg_oom_lock); -static void __mem_cgroup_insert_exceeded(struct mem_cgroup_per_node *mz, - struct mem_cgroup_tree_per_node *mctz, - unsigned long new_usage_in_excess) -{ - struct rb_node **p = &mctz->rb_root.rb_node; - struct rb_node *parent = NULL; - struct mem_cgroup_per_node *mz_node; - bool rightmost = true; - - if (mz->on_tree) - return; - - mz->usage_in_excess = new_usage_in_excess; - if (!mz->usage_in_excess) - return; - while (*p) { - parent = *p; - mz_node = rb_entry(parent, struct mem_cgroup_per_node, - tree_node); - if (mz->usage_in_excess < mz_node->usage_in_excess) { - p = &(*p)->rb_left; - rightmost = false; - } else { - p = &(*p)->rb_right; - } - } - - if (rightmost) - mctz->rb_rightmost = &mz->tree_node; - - rb_link_node(&mz->tree_node, parent, p); - rb_insert_color(&mz->tree_node, &mctz->rb_root); - mz->on_tree = true; -} - -static void __mem_cgroup_remove_exceeded(struct mem_cgroup_per_node *mz, - struct mem_cgroup_tree_per_node *mctz) -{ - if (!mz->on_tree) - return; - - if (&mz->tree_node == mctz->rb_rightmost) - mctz->rb_rightmost = rb_prev(&mz->tree_node); - - rb_erase(&mz->tree_node, &mctz->rb_root); - mz->on_tree = false; -} - -static void mem_cgroup_remove_exceeded(struct mem_cgroup_per_node *mz, - struct mem_cgroup_tree_per_node *mctz) -{ - unsigned long flags; - - spin_lock_irqsave(&mctz->lock, flags); - __mem_cgroup_remove_exceeded(mz, mctz); - spin_unlock_irqrestore(&mctz->lock, flags); -} - -static unsigned long soft_limit_excess(struct mem_cgroup *memcg) -{ - unsigned long nr_pages = page_counter_read(&memcg->memory); - unsigned long soft_limit = READ_ONCE(memcg->soft_limit); - unsigned long excess = 0; - - if (nr_pages > soft_limit) - excess = nr_pages - soft_limit; - - return excess; -} - -static void memcg1_update_tree(struct mem_cgroup *memcg, int nid) -{ - unsigned long excess; - struct mem_cgroup_per_node *mz; - struct mem_cgroup_tree_per_node *mctz; - - if (lru_gen_enabled()) { - if (soft_limit_excess(memcg)) - lru_gen_soft_reclaim(memcg, nid); - return; - } - - mctz = soft_limit_tree.rb_tree_per_node[nid]; - if (!mctz) - return; - /* - * Necessary to update all ancestors when hierarchy is used. - * because their event counter is not touched. - */ - for (; memcg; memcg = parent_mem_cgroup(memcg)) { - mz = memcg->nodeinfo[nid]; - excess = soft_limit_excess(memcg); - /* - * We have to update the tree if mz is on RB-tree or - * mem is over its softlimit. - */ - if (excess || mz->on_tree) { - unsigned long flags; - - spin_lock_irqsave(&mctz->lock, flags); - /* if on-tree, remove it */ - if (mz->on_tree) - __mem_cgroup_remove_exceeded(mz, mctz); - /* - * Insert again. mz->usage_in_excess will be updated. - * If excess is 0, no tree ops. - */ - __mem_cgroup_insert_exceeded(mz, mctz, excess); - spin_unlock_irqrestore(&mctz->lock, flags); - } - } -} - -void memcg1_remove_from_trees(struct mem_cgroup *memcg) -{ - struct mem_cgroup_tree_per_node *mctz; - struct mem_cgroup_per_node *mz; - int nid; - - for_each_node(nid) { - mz = memcg->nodeinfo[nid]; - mctz = soft_limit_tree.rb_tree_per_node[nid]; - if (mctz) - mem_cgroup_remove_exceeded(mz, mctz); - } -} - static u64 mem_cgroup_move_charge_read(struct cgroup_subsys_state *css, struct cftype *cft) { @@ -336,7 +192,7 @@ static void mem_cgroup_threshold(struct mem_cgroup *memcg) } } -/* Cgroup1: threshold notifications & softlimit tree updates */ +/* Cgroup1: threshold notifications */ /* * Per memcg event counter is incremented at every pagein/pageout. With THP, @@ -405,17 +261,8 @@ static void memcg1_check_events(struct mem_cgroup *memcg, int nid) if (IS_ENABLED(CONFIG_PREEMPT_RT)) return; - /* threshold event is triggered in finer grain than soft limit */ - if (unlikely(memcg1_event_ratelimit(memcg, - MEM_CGROUP_TARGET_THRESH))) { - bool do_softlimit; - - do_softlimit = memcg1_event_ratelimit(memcg, - MEM_CGROUP_TARGET_SOFTLIMIT); + if (unlikely(memcg1_event_ratelimit(memcg, MEM_CGROUP_TARGET_THRESH))) mem_cgroup_threshold(memcg); - if (unlikely(do_softlimit)) - memcg1_update_tree(memcg, nid); - } } void memcg1_commit_charge(struct folio *folio, struct mem_cgroup *memcg) @@ -2391,22 +2238,3 @@ void memcg1_free_events(struct mem_cgroup *memcg) { free_percpu(memcg->events_percpu); } - -static int __init memcg1_init(void) -{ - int node; - - for_each_node(node) { - struct mem_cgroup_tree_per_node *rtpn; - - rtpn = kzalloc_node(sizeof(*rtpn), GFP_KERNEL, node); - - rtpn->rb_root = RB_ROOT; - rtpn->rb_rightmost = NULL; - spin_lock_init(&rtpn->lock); - soft_limit_tree.rb_tree_per_node[node] = rtpn; - } - - return 0; -} -subsys_initcall(memcg1_init); diff --git a/mm/memcontrol-v1.h b/mm/memcontrol-v1.h index 1e394269c613d8..fd611e66859a32 100644 --- a/mm/memcontrol-v1.h +++ b/mm/memcontrol-v1.h @@ -41,7 +41,6 @@ bool memcg1_alloc_events(struct mem_cgroup *memcg); void memcg1_free_events(struct mem_cgroup *memcg); void memcg1_memcg_init(struct mem_cgroup *memcg); -void memcg1_remove_from_trees(struct mem_cgroup *memcg); static inline void memcg1_soft_limit_reset(struct mem_cgroup *memcg) { @@ -98,7 +97,6 @@ static inline bool memcg1_alloc_events(struct mem_cgroup *memcg) { return true; static inline void memcg1_free_events(struct mem_cgroup *memcg) {} static inline void memcg1_memcg_init(struct mem_cgroup *memcg) {} -static inline void memcg1_remove_from_trees(struct mem_cgroup *memcg) {} static inline void memcg1_soft_limit_reset(struct mem_cgroup *memcg) {} static inline void memcg1_css_offline(struct mem_cgroup *memcg) {} diff --git a/mm/memcontrol.c b/mm/memcontrol.c index bfd0a74fac9239..29def037681943 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -4429,7 +4429,6 @@ static void mem_cgroup_css_free(struct cgroup_subsys_state *css) vmpressure_cleanup(&memcg->vmpressure); cancel_work_sync(&memcg->high_work); - memcg1_remove_from_trees(memcg); free_shrinker_info(memcg); mem_cgroup_free(memcg); } From e6a4798cf2968ceabb8eab182868f817815d049f Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 2 Sep 2026 10:43:08 -0700 Subject: [PATCH 0598/1352] memcg: remove lru_gen_soft_reclaim() The soft limit rbtree was the only caller. Dropping it leaves MEMCG_LRU_HEAD unreachable, since nothing else ever rotates a memcg with that op, so remove the op too and update the memcg LRU comment. Link: https://lore.kernel.org/20260902174311.1772372-6-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Reviewed-by: T.J. Mercier Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin --- include/linux/mmzone.h | 30 +++++++++++------------------- mm/vmscan.c | 16 ++-------------- 2 files changed, 13 insertions(+), 33 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 6acc14b169bbe6..c070b867e2f349 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -638,35 +638,32 @@ struct lru_gen_mm_walk { * For each node, memcgs are divided into two generations: the old and the * young. For each generation, memcgs are randomly sharded into multiple bins * to improve scalability. For each bin, the hlist_nulls is virtually divided - * into three segments: the head, the tail and the default. + * into two segments: the tail and the default. * * An onlining memcg is added to the tail of a random bin in the old generation. * The eviction starts at the head of a random bin in the old generation. The * per-node memcg generation counter, whose reminder (mod MEMCG_NR_GENS) indexes * the old generation, is incremented when all its bins become empty. * - * There are four operations: - * 1. MEMCG_LRU_HEAD, which moves a memcg to the head of a random bin in its - * current generation (old or young) and updates its "seg" to "head"; - * 2. MEMCG_LRU_TAIL, which moves a memcg to the tail of a random bin in its + * There are three operations: + * 1. MEMCG_LRU_TAIL, which moves a memcg to the tail of a random bin in its * current generation (old or young) and updates its "seg" to "tail"; - * 3. MEMCG_LRU_OLD, which moves a memcg to the head of a random bin in the old + * 2. MEMCG_LRU_OLD, which moves a memcg to the head of a random bin in the old * generation, updates its "gen" to "old" and resets its "seg" to "default"; - * 4. MEMCG_LRU_YOUNG, which moves a memcg to the tail of a random bin in the + * 3. MEMCG_LRU_YOUNG, which moves a memcg to the tail of a random bin in the * young generation, updates its "gen" to "young" and resets its "seg" to * "default". * * The events that trigger the above operations are: - * 1. Exceeding the soft limit, which triggers MEMCG_LRU_HEAD; - * 2. The first attempt to reclaim a memcg below low, which triggers + * 1. The first attempt to reclaim a memcg below low, which triggers * MEMCG_LRU_TAIL; - * 3. The first attempt to reclaim a memcg offlined or below reclaimable size + * 2. The first attempt to reclaim a memcg offlined or below reclaimable size * threshold, which triggers MEMCG_LRU_TAIL; - * 4. The second attempt to reclaim a memcg offlined or below reclaimable size + * 3. The second attempt to reclaim a memcg offlined or below reclaimable size * threshold, which triggers MEMCG_LRU_YOUNG; - * 5. Attempting to reclaim a memcg below min, which triggers MEMCG_LRU_YOUNG; - * 6. Finishing the aging on the eviction path, which triggers MEMCG_LRU_YOUNG; - * 7. Offlining a memcg, which triggers MEMCG_LRU_OLD. + * 4. Attempting to reclaim a memcg below min, which triggers MEMCG_LRU_YOUNG; + * 5. Finishing the aging on the eviction path, which triggers MEMCG_LRU_YOUNG; + * 6. Offlining a memcg, which triggers MEMCG_LRU_OLD. * * Notes: * 1. Memcg LRU only applies to global reclaim, and the round-robin incrementing @@ -699,7 +696,6 @@ void lru_gen_exit_memcg(struct mem_cgroup *memcg); void lru_gen_online_memcg(struct mem_cgroup *memcg); void lru_gen_offline_memcg(struct mem_cgroup *memcg); void lru_gen_release_memcg(struct mem_cgroup *memcg); -void lru_gen_soft_reclaim(struct mem_cgroup *memcg, int nid); void max_lru_gen_memcg(struct mem_cgroup *memcg, int nid); bool recheck_lru_gen_max_memcg(struct mem_cgroup *memcg, int nid); void lru_gen_reparent_memcg(struct mem_cgroup *memcg, struct mem_cgroup *parent, int nid); @@ -740,10 +736,6 @@ static inline void lru_gen_release_memcg(struct mem_cgroup *memcg) { } -static inline void lru_gen_soft_reclaim(struct mem_cgroup *memcg, int nid) -{ -} - static inline void max_lru_gen_memcg(struct mem_cgroup *memcg, int nid) { } diff --git a/mm/vmscan.c b/mm/vmscan.c index d66b5cd167d6f2..deb087c57007db 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -4373,7 +4373,6 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr) /* see the comment on MEMCG_NR_GENS */ enum { MEMCG_LRU_NOP, - MEMCG_LRU_HEAD, MEMCG_LRU_TAIL, MEMCG_LRU_OLD, MEMCG_LRU_YOUNG, @@ -4395,9 +4394,7 @@ static void lru_gen_rotate_memcg(struct lruvec *lruvec, int op) new = old = lruvec->lrugen.gen; /* see the comment on MEMCG_NR_GENS */ - if (op == MEMCG_LRU_HEAD) - seg = MEMCG_LRU_HEAD; - else if (op == MEMCG_LRU_TAIL) + if (op == MEMCG_LRU_TAIL) seg = MEMCG_LRU_TAIL; else if (op == MEMCG_LRU_OLD) new = get_memcg_gen(pgdat->memcg_lru.seq); @@ -4411,7 +4408,7 @@ static void lru_gen_rotate_memcg(struct lruvec *lruvec, int op) hlist_nulls_del_rcu(&lruvec->lrugen.list); - if (op == MEMCG_LRU_HEAD || op == MEMCG_LRU_OLD) + if (op == MEMCG_LRU_OLD) hlist_nulls_add_head_rcu(&lruvec->lrugen.list, &pgdat->memcg_lru.fifo[new][bin]); else hlist_nulls_add_tail_rcu(&lruvec->lrugen.list, &pgdat->memcg_lru.fifo[new][bin]); @@ -4489,15 +4486,6 @@ void lru_gen_release_memcg(struct mem_cgroup *memcg) } } -void lru_gen_soft_reclaim(struct mem_cgroup *memcg, int nid) -{ - struct lruvec *lruvec = get_lruvec(memcg, nid); - - /* see the comment on MEMCG_NR_GENS */ - if (READ_ONCE(lruvec->lrugen.seg) != MEMCG_LRU_HEAD) - lru_gen_rotate_memcg(lruvec, MEMCG_LRU_HEAD); -} - bool recheck_lru_gen_max_memcg(struct mem_cgroup *memcg, int nid) { struct lruvec *lruvec = get_lruvec(memcg, nid); From c36e3ed6cd20a8255da922d9908fd1cdb5a5dca0 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 2 Sep 2026 10:43:09 -0700 Subject: [PATCH 0599/1352] memcg: remove the per-node soft limit tree fields tree_node, usage_in_excess and on_tree only existed for the soft limit rbtree. They also doubled as the buffer between the read-mostly head of struct mem_cgroup_per_node and its update-often tail, so replace them with the explicit padding that CONFIG_MEMCG_V1=n already used. Link: https://lore.kernel.org/20260902174311.1772372-7-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Acked-by: Michal Hocko Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Muchun Song Cc: Roman Gushchin Cc: T.J. Mercier --- include/linux/memcontrol.h | 13 ------------- 1 file changed, 13 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 11c1fa88d6fd0c..1eababed16f532 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -95,20 +95,7 @@ struct mem_cgroup_per_node { struct lruvec_stats *lruvec_stats; struct shrinker_info __rcu *shrinker_info; -#ifdef CONFIG_MEMCG_V1 - /* - * Memcg-v1 only stuff in middle as buffer between read mostly fields - * and update often fields to avoid false sharing. If v1 stuff is - * not present, an explicit padding is needed. - */ - - struct rb_node tree_node; /* RB tree node */ - unsigned long usage_in_excess;/* Set to the value by which */ - /* the soft limit is exceeded*/ - bool on_tree; -#else CACHELINE_PADDING(_pad1_); -#endif /* Fields which get updated often at the end. */ struct lruvec lruvec; From 3ad4018e4884f4488b0f02f7b84b9017ff879bc1 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 2 Sep 2026 10:43:10 -0700 Subject: [PATCH 0600/1352] memcg: remove mem_cgroup->soft_limit Nothing reads it anymore, so the field and the helper that reset it on css alloc and css reset can go. Link: https://lore.kernel.org/20260902174311.1772372-8-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Acked-by: Michal Hocko Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Muchun Song Cc: Roman Gushchin Cc: T.J. Mercier --- include/linux/memcontrol.h | 2 -- mm/memcontrol-v1.h | 6 ------ mm/memcontrol.c | 2 -- 3 files changed, 10 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 1eababed16f532..c799926435560f 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -280,8 +280,6 @@ struct mem_cgroup { struct memcg1_events_percpu __percpu *events_percpu; - unsigned long soft_limit; - /* protected by memcg_oom_lock */ bool oom_lock; int under_oom; diff --git a/mm/memcontrol-v1.h b/mm/memcontrol-v1.h index fd611e66859a32..f48d0e22e615b6 100644 --- a/mm/memcontrol-v1.h +++ b/mm/memcontrol-v1.h @@ -42,11 +42,6 @@ void memcg1_free_events(struct mem_cgroup *memcg); void memcg1_memcg_init(struct mem_cgroup *memcg); -static inline void memcg1_soft_limit_reset(struct mem_cgroup *memcg) -{ - WRITE_ONCE(memcg->soft_limit, PAGE_COUNTER_MAX); -} - struct cgroup_taskset; void memcg1_css_offline(struct mem_cgroup *memcg); @@ -97,7 +92,6 @@ static inline bool memcg1_alloc_events(struct mem_cgroup *memcg) { return true; static inline void memcg1_free_events(struct mem_cgroup *memcg) {} static inline void memcg1_memcg_init(struct mem_cgroup *memcg) {} -static inline void memcg1_soft_limit_reset(struct mem_cgroup *memcg) {} static inline void memcg1_css_offline(struct mem_cgroup *memcg) {} static inline bool memcg1_oom_prepare(struct mem_cgroup *memcg, bool *locked) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 29def037681943..bce3962dba5752 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -4257,7 +4257,6 @@ mem_cgroup_css_alloc(struct cgroup_subsys_state *parent_css) return ERR_CAST(memcg); page_counter_set_high(&memcg->memory, PAGE_COUNTER_MAX); - memcg1_soft_limit_reset(memcg); #ifdef CONFIG_ZSWAP memcg->zswap_max = PAGE_COUNTER_MAX; WRITE_ONCE(memcg->zswap_writeback, true); @@ -4464,7 +4463,6 @@ static void mem_cgroup_css_reset(struct cgroup_subsys_state *css) page_counter_set_min(&memcg->memory, 0); page_counter_set_low(&memcg->memory, 0); page_counter_set_high(&memcg->memory, PAGE_COUNTER_MAX); - memcg1_soft_limit_reset(memcg); page_counter_set_high(&memcg->swap, PAGE_COUNTER_MAX); memcg_wb_domain_size_changed(memcg); } From 15a7ee98e0d1d4e6e76e9875be088791c5d13be6 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 2 Sep 2026 10:43:11 -0700 Subject: [PATCH 0601/1352] memcg: simplify v1 event ratelimiting Thresholds are the only periodic v1 event left, so the target enum, the per-cpu target array and the switch in memcg1_event_ratelimit() all collapse to a single counter. memcg1_check_events() no longer needs a node id either, which lets memcg1_uncharge_batch() drop its nid argument and struct uncharge_gather drop the field feeding it. Link: https://lore.kernel.org/20260902174311.1772372-9-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Acked-by: Michal Hocko Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Muchun Song Cc: Roman Gushchin Cc: T.J. Mercier --- mm/memcontrol-v1.c | 43 +++++++++++-------------------------------- mm/memcontrol-v1.h | 4 ++-- mm/memcontrol.c | 4 +--- 3 files changed, 14 insertions(+), 37 deletions(-) diff --git a/mm/memcontrol-v1.c b/mm/memcontrol-v1.c index 475f998b764318..bf2c7d53b01b1c 100644 --- a/mm/memcontrol-v1.c +++ b/mm/memcontrol-v1.c @@ -200,15 +200,9 @@ static void mem_cgroup_threshold(struct mem_cgroup *memcg) * to trigger some periodic events. This is straightforward and better * than using jiffies etc. to handle periodic memcg event. */ -enum mem_cgroup_events_target { - MEM_CGROUP_TARGET_THRESH, - MEM_CGROUP_TARGET_SOFTLIMIT, - MEM_CGROUP_NTARGETS, -}; - struct memcg1_events_percpu { unsigned long nr_page_events; - unsigned long targets[MEM_CGROUP_NTARGETS]; + unsigned long threshold_target; }; static void memcg1_charge_statistics(struct mem_cgroup *memcg, int nr_pages) @@ -225,43 +219,28 @@ static void memcg1_charge_statistics(struct mem_cgroup *memcg, int nr_pages) } #define THRESHOLDS_EVENTS_TARGET 128 -#define SOFTLIMIT_EVENTS_TARGET 1024 -static bool memcg1_event_ratelimit(struct mem_cgroup *memcg, - enum mem_cgroup_events_target target) +static bool memcg1_event_ratelimit(struct mem_cgroup *memcg) { unsigned long val, next; val = __this_cpu_read(memcg->events_percpu->nr_page_events); - next = __this_cpu_read(memcg->events_percpu->targets[target]); + next = __this_cpu_read(memcg->events_percpu->threshold_target); /* from time_after() in jiffies.h */ if ((long)(next - val) < 0) { - switch (target) { - case MEM_CGROUP_TARGET_THRESH: - next = val + THRESHOLDS_EVENTS_TARGET; - break; - case MEM_CGROUP_TARGET_SOFTLIMIT: - next = val + SOFTLIMIT_EVENTS_TARGET; - break; - default: - break; - } - __this_cpu_write(memcg->events_percpu->targets[target], next); + __this_cpu_write(memcg->events_percpu->threshold_target, + val + THRESHOLDS_EVENTS_TARGET); return true; } return false; } -/* - * Check events in order. - * - */ -static void memcg1_check_events(struct mem_cgroup *memcg, int nid) +static void memcg1_check_events(struct mem_cgroup *memcg) { if (IS_ENABLED(CONFIG_PREEMPT_RT)) return; - if (unlikely(memcg1_event_ratelimit(memcg, MEM_CGROUP_TARGET_THRESH))) + if (unlikely(memcg1_event_ratelimit(memcg))) mem_cgroup_threshold(memcg); } @@ -271,7 +250,7 @@ void memcg1_commit_charge(struct folio *folio, struct mem_cgroup *memcg) local_irq_save(flags); memcg1_charge_statistics(memcg, folio_nr_pages(folio)); - memcg1_check_events(memcg, folio_nid(folio)); + memcg1_check_events(memcg); local_irq_restore(flags); } @@ -344,7 +323,7 @@ void __memcg1_swapout(struct folio *folio, struct swap_cluster_info *ci) VM_WARN_ON_IRQS_ENABLED(); memcg1_charge_statistics(memcg, -folio_nr_pages(folio)); preempt_enable_nested(); - memcg1_check_events(memcg, folio_nid(folio)); + memcg1_check_events(memcg); rcu_read_unlock(); obj_cgroup_put(objcg); @@ -398,14 +377,14 @@ void memcg1_swapin(struct folio *folio) #endif void memcg1_uncharge_batch(struct mem_cgroup *memcg, unsigned long pgpgout, - unsigned long nr_memory, int nid) + unsigned long nr_memory) { unsigned long flags; local_irq_save(flags); count_memcg_events(memcg, PGPGOUT, pgpgout); __this_cpu_add(memcg->events_percpu->nr_page_events, nr_memory); - memcg1_check_events(memcg, nid); + memcg1_check_events(memcg); local_irq_restore(flags); } diff --git a/mm/memcontrol-v1.h b/mm/memcontrol-v1.h index f48d0e22e615b6..b9a21f0fd2c3ac 100644 --- a/mm/memcontrol-v1.h +++ b/mm/memcontrol-v1.h @@ -59,7 +59,7 @@ void memcg1_oom_recover(struct mem_cgroup *memcg); void memcg1_commit_charge(struct folio *folio, struct mem_cgroup *memcg); void memcg1_uncharge_batch(struct mem_cgroup *memcg, unsigned long pgpgout, - unsigned long nr_memory, int nid); + unsigned long nr_memory); void memcg1_stat_format(struct mem_cgroup *memcg, struct seq_buf *s); void reparent_memcg1_state_local(struct mem_cgroup *memcg, struct mem_cgroup *parent); @@ -107,7 +107,7 @@ static inline void memcg1_commit_charge(struct folio *folio, static inline void memcg1_uncharge_batch(struct mem_cgroup *memcg, unsigned long pgpgout, - unsigned long nr_memory, int nid) {} + unsigned long nr_memory) {} static inline void memcg1_stat_format(struct mem_cgroup *memcg, struct seq_buf *s) {} diff --git a/mm/memcontrol.c b/mm/memcontrol.c index bce3962dba5752..30636b9d96739e 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -5328,7 +5328,6 @@ struct uncharge_gather { unsigned long nr_memory; unsigned long pgpgout; unsigned long nr_kmem; - int nid; }; static inline void uncharge_gather_clear(struct uncharge_gather *ug) @@ -5351,7 +5350,7 @@ static void uncharge_batch(const struct uncharge_gather *ug) memcg1_oom_recover(memcg); } - memcg1_uncharge_batch(memcg, ug->pgpgout, ug->nr_memory, ug->nid); + memcg1_uncharge_batch(memcg, ug->pgpgout, ug->nr_memory); rcu_read_unlock(); /* drop reference from uncharge_folio */ @@ -5380,7 +5379,6 @@ static void uncharge_folio(struct folio *folio, struct uncharge_gather *ug) uncharge_gather_clear(ug); } ug->objcg = objcg; - ug->nid = folio_nid(folio); /* pairs with obj_cgroup_put in uncharge_batch */ obj_cgroup_get(objcg); From def18718e59450ca76e126063aaacd0ed69de61d Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Sat, 29 Aug 2026 15:09:00 -0400 Subject: [PATCH 0602/1352] mm/page_io: convert write completion handlers to folios Patch series "mm/page_io: folio conversion cleanups", v2. Convert the remaining struct page usage in mm/page_io.c to folios. This removes one of the last callers of end_page_writeback(), along with the last caller of ClearPageReclaim(). This allows us to remove the PG_reclaim page accessors entirely. Clean up a few other stale references to pages throughout while at it. The rename of mm/page_io.c to mm/swap_io.c is deferred to a separate series. This patch (of 6): Convert swap_write_end() and swap_fs_write_complete() to operate on folios directly instead of going through the folio-compat page APIs. This removes calls to end_page_writeback() and set_page_dirty(), and the last caller of ClearPageReclaim(), saving two calls to compound_head() per folio on the write error path. Link: https://lore.kernel.org/20260829-b4-page_io-folios-v2-0-649728091117@columbia.edu Link: https://lore.kernel.org/20260829-b4-page_io-folios-v2-1-649728091117@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: Johannes Weiner Reviewed-by: Matthew Wilcox (Oracle) Reviewed-by: Lorenzo Stoakes (ARM) Cc: Baoquan He Cc: Barry Song Cc: Chengming Zhou Cc: Chris Li Cc: Christoph Hellwig Cc: David Hildenbrand Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/page_io.c | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/mm/page_io.c b/mm/page_io.c index 88962571cb931e..fbcf58ff292d8b 100644 --- a/mm/page_io.c +++ b/mm/page_io.c @@ -497,13 +497,13 @@ static void swap_write_end(struct swap_iocb *sio, bool failed) int p; for (p = 0; p < sio->nr_bvecs; p++) { - struct page *page = sio->bvecs[p].bv_page; + struct folio *folio = bvec_folio(&sio->bvecs[p]); if (failed) { - set_page_dirty(page); - ClearPageReclaim(page); + folio_mark_dirty(folio); + folio_clear_reclaim(folio); } - end_page_writeback(page); + folio_end_writeback(folio); } mempool_free(sio, sio_pool); } @@ -514,16 +514,16 @@ static void swap_fs_write_complete(struct kiocb *iocb, long ret) bool failed = ret != sio->len; if (failed) { - struct page *page = sio->bvecs[0].bv_page; + struct folio *folio = bvec_folio(&sio->bvecs[0]); /* * In the case of swap-over-nfs, this can be a temporary failure * if the system has limited memory for allocating transmit - * buffers. Mark the page dirty and avoid + * buffers. Mark the folio dirty and avoid * folio_rotate_reclaimable but rate-limit the messages. */ pr_err_ratelimited("Write error %ld on dio swapfile (%llu)\n", - ret, swap_dev_pos(page_swap_entry(page))); + ret, swap_dev_pos(folio->swap)); } swap_write_end(sio, failed); From 0fec8efadd399da933a2a054706d79c59af5d190 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Sat, 29 Aug 2026 15:09:01 -0400 Subject: [PATCH 0603/1352] mm: remove PageReclaim This flag is now only used on folios, so we can remove all the page accessors. folio_test_clear_reclaim() is not used, so don't add FOLIO_TEST_CLEAR_FLAG() for it. Link: https://lore.kernel.org/20260829-b4-page_io-folios-v2-2-649728091117@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: Johannes Weiner Reviewed-by: Matthew Wilcox (Oracle) Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) --- include/linux/page-flags.h | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/include/linux/page-flags.h b/include/linux/page-flags.h index 7a863572adce79..ae2ebaed6d4d96 100644 --- a/include/linux/page-flags.h +++ b/include/linux/page-flags.h @@ -593,8 +593,7 @@ TESTPAGEFLAG(Writeback, writeback, PF_NO_TAIL) FOLIO_FLAG(mappedtodisk, FOLIO_HEAD_PAGE) /* PG_readahead is only used for reads; PG_reclaim is only for writes */ -PAGEFLAG(Reclaim, reclaim, PF_NO_TAIL) - TESTCLEARFLAG(Reclaim, reclaim, PF_NO_TAIL) +FOLIO_FLAG(reclaim, FOLIO_HEAD_PAGE) FOLIO_FLAG(readahead, FOLIO_HEAD_PAGE) FOLIO_TEST_CLEAR_FLAG(readahead, FOLIO_HEAD_PAGE) From 91750337acd2a32f852e3b63c0810e36d752716e Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Sat, 29 Aug 2026 15:09:02 -0400 Subject: [PATCH 0604/1352] mm/page_io: use swap entries directly in zeromap helpers Increment swp_entry_t::val directly instead of recomputing each entry with page_swap_entry(). This removes the last struct page usage in page_io.c and saves one call to compound_head() per page. Link: https://lore.kernel.org/20260829-b4-page_io-folios-v2-3-649728091117@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: Johannes Weiner Reviewed-by: Matthew Wilcox (Oracle) Reviewed-by: Lorenzo Stoakes (ARM) --- mm/page_io.c | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/mm/page_io.c b/mm/page_io.c index fbcf58ff292d8b..dd95cdb0cd7a18 100644 --- a/mm/page_io.c +++ b/mm/page_io.c @@ -160,7 +160,7 @@ static void swap_zeromap_folio_set(struct folio *folio) struct obj_cgroup *objcg = get_obj_cgroup_from_folio(folio); int nr_pages = folio_nr_pages(folio); struct swap_cluster_info *ci; - swp_entry_t entry; + swp_entry_t entry = folio->swap; unsigned int i; VM_WARN_ON_ONCE_FOLIO(!folio_test_swapcache(folio), folio); @@ -168,8 +168,8 @@ static void swap_zeromap_folio_set(struct folio *folio) ci = swap_cluster_get_and_lock(folio); for (i = 0; i < folio_nr_pages(folio); i++) { - entry = page_swap_entry(folio_page(folio, i)); __swap_table_set_zero(ci, swp_cluster_offset(entry)); + entry.val++; } swap_cluster_unlock(ci); @@ -183,7 +183,7 @@ static void swap_zeromap_folio_set(struct folio *folio) static void swap_zeromap_folio_clear(struct folio *folio) { struct swap_cluster_info *ci; - swp_entry_t entry; + swp_entry_t entry = folio->swap; unsigned int i; VM_WARN_ON_ONCE_FOLIO(!folio_test_swapcache(folio), folio); @@ -191,8 +191,8 @@ static void swap_zeromap_folio_clear(struct folio *folio) ci = swap_cluster_get_and_lock(folio); for (i = 0; i < folio_nr_pages(folio); i++) { - entry = page_swap_entry(folio_page(folio, i)); __swap_table_clear_zero(ci, swp_cluster_offset(entry)); + entry.val++; } swap_cluster_unlock(ci); } From 9058d8c4d2b8d2a6879976f88e425482c65a82b6 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Sat, 29 Aug 2026 15:09:03 -0400 Subject: [PATCH 0605/1352] mm/page_io: rename bio_associate_blkg_from_page() This function takes a folio. Rename it to bio_associate_blkg_from_folio() accordingly. While at it, convert the macro in the !CONFIG_MEMCG || !CONFIG_BLK_CGROUP case to a function. Link: https://lore.kernel.org/20260829-b4-page_io-folios-v2-4-649728091117@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: Johannes Weiner Reviewed-by: Matthew Wilcox (Oracle) Reviewed-by: Lorenzo Stoakes (ARM) --- mm/page_io.c | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/mm/page_io.c b/mm/page_io.c index dd95cdb0cd7a18..295cc6ac6244af 100644 --- a/mm/page_io.c +++ b/mm/page_io.c @@ -277,7 +277,7 @@ static bool folio_blkg_can_merge(struct folio *folio, struct folio *prev_folio) return can_merge; } -static void bio_associate_blkg_from_page(struct bio *bio, struct folio *folio) +static void bio_associate_blkg_from_folio(struct bio *bio, struct folio *folio) { struct cgroup_subsys_state *css; @@ -298,7 +298,9 @@ static bool folio_blkg_can_merge(struct folio *folio, struct folio *prev_folio) { return true; } -#define bio_associate_blkg_from_page(bio, folio) do { } while (0) +static void bio_associate_blkg_from_folio(struct bio *bio, struct folio *folio) +{ +} #endif /* CONFIG_MEMCG && CONFIG_BLK_CGROUP */ static mempool_t *sio_pool; @@ -596,7 +598,7 @@ static void swap_bdev_submit_write(struct swap_io_ctx *ctx) REQ_OP_WRITE | REQ_SWAP); bio->bi_iter.bi_size = sio->len; bio->bi_iter.bi_sector = swap_folio_sector(bio_first_folio_all(bio)); - bio_associate_blkg_from_page(bio, bio_first_folio_all(bio)); + bio_associate_blkg_from_folio(bio, bio_first_folio_all(bio)); if (ctx->sis->flags & SWP_SYNCHRONOUS_IO) { submit_bio_wait(bio); From 20b44a589986cb2c90bc27dfd91b0f4549963b11 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Sat, 29 Aug 2026 15:09:04 -0400 Subject: [PATCH 0606/1352] mm/page_io: refer to folios in swap_writeout() comments swap_writeout() operates on folios, not pages. Update its comments accordingly. Link: https://lore.kernel.org/20260829-b4-page_io-folios-v2-5-649728091117@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: Johannes Weiner Reviewed-by: Matthew Wilcox (Oracle) Reviewed-by: Lorenzo Stoakes (ARM) --- mm/page_io.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/mm/page_io.c b/mm/page_io.c index 295cc6ac6244af..36466159cf191e 100644 --- a/mm/page_io.c +++ b/mm/page_io.c @@ -209,7 +209,7 @@ int swap_writeout(struct swap_io_ctx *ctx, struct folio *folio) goto out_unlock; /* - * Arch code may have to preserve more data than just the page + * Arch code may have to preserve more data than just the folio * contents, e.g. memory tags. */ ret = arch_prepare_to_swap(folio); @@ -220,7 +220,7 @@ int swap_writeout(struct swap_io_ctx *ctx, struct folio *folio) /* * Use the swap table zero mark to avoid doing IO for zero-filled - * pages. The zero mark is protected by the cluster lock, which is + * folios. The zero mark is protected by the cluster lock, which is * acquired internally by swap_zeromap_folio_set/clear. */ if (is_folio_zero_filled(folio)) { From 903aaf1fcf43eff8e075d9fdc08f2af0b2a7a1ca Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Sat, 29 Aug 2026 15:09:05 -0400 Subject: [PATCH 0607/1352] mm/swap: rename __swap_writepage() to __swap_writeout() Commit 84798514db50 ("mm: Remove swap_writepage() and shmem_writepage()") renamed swap_writepage() to swap_writeout(). Rename __swap_writepage(), which operates on a folio, to match its caller. Update a stale reference to swap_writepage() in swapfile.c as well. Link: https://lore.kernel.org/20260829-b4-page_io-folios-v2-6-649728091117@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Reviewed-by: Matthew Wilcox (Oracle) Reviewed-by: Lorenzo Stoakes (ARM) --- mm/page_io.c | 4 ++-- mm/swap.h | 2 +- mm/swapfile.c | 2 +- mm/zswap.c | 2 +- 4 files changed, 5 insertions(+), 5 deletions(-) diff --git a/mm/page_io.c b/mm/page_io.c index 36466159cf191e..1da4ff484f0971 100644 --- a/mm/page_io.c +++ b/mm/page_io.c @@ -248,7 +248,7 @@ int swap_writeout(struct swap_io_ctx *ctx, struct folio *folio) } rcu_read_unlock(); - __swap_writepage(ctx, folio); + __swap_writeout(ctx, folio); return 0; out_unlock: folio_unlock(folio); @@ -369,7 +369,7 @@ static void swap_add_folio(struct swap_io_ctx *ctx, struct folio *folio, int rw) } } -void __swap_writepage(struct swap_io_ctx *ctx, struct folio *folio) +void __swap_writeout(struct swap_io_ctx *ctx, struct folio *folio) { VM_BUG_ON_FOLIO(!folio_test_swapcache(folio), folio); diff --git a/mm/swap.h b/mm/swap.h index fddba7a87500a4..0b5d507739bcb6 100644 --- a/mm/swap.h +++ b/mm/swap.h @@ -258,7 +258,7 @@ void swap_read_folio(struct swap_io_ctx *ctx, struct folio *folio); void swap_read_submit(struct swap_io_ctx *ctx); void swap_write_submit(struct swap_io_ctx *ctx); int swap_writeout(struct swap_io_ctx *ctx, struct folio *folio); -void __swap_writepage(struct swap_io_ctx *ctx, struct folio *folio); +void __swap_writeout(struct swap_io_ctx *ctx, struct folio *folio); /* linux/mm/swap_state.c */ extern struct address_space swap_space __read_mostly; diff --git a/mm/swapfile.c b/mm/swapfile.c index 601979b97f95b2..3b2279a16d1cf8 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -2928,7 +2928,7 @@ EXPORT_SYMBOL_GPL(add_swap_extent); /* * A `swap extent' is a simple thing which maps a contiguous range of pages * onto a contiguous range of disk blocks. A rbtree of swap extents is - * built at swapon time and is then used at swap_writepage/swap_read_folio + * built at swapon time and is then used at swap_writeout/swap_read_folio * time for locating where on disk a page belongs. * * If the swapfile is an S_ISBLK block device, a single extent is installed. diff --git a/mm/zswap.c b/mm/zswap.c index c1dc60926bad99..fc869d60ef1b48 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -1037,7 +1037,7 @@ static int zswap_writeback_entry(struct zswap_entry *entry, folio_set_reclaim(folio); /* start writeback */ - __swap_writepage(&ctx, folio); + __swap_writeout(&ctx, folio); swap_write_submit(&ctx); out: From 84bf81ee2c22be985ff82edef80cb86c05db7dc2 Mon Sep 17 00:00:00 2001 From: Ridong Chen Date: Sat, 29 Aug 2026 15:42:03 +0800 Subject: [PATCH 0608/1352] mm/mglru: make type fallback logic explicit in isolate_folios() Patch series "mm/mglru: clean up isolate_folios for readability and clarity", v2. Right now, isolate_folios() is quite difficult to follow: 1. It uses for_each_evictable_type(i, swappiness) to iterate over the types, but 'i' is not actually used as the type within the loop body. 2. It retries the same type when folios were scanned but none could be isolated, but the retry is implemented in a rather subtle way that is difficult to understand. This patchset makes both behaviors explicit and much easier to follow. There are no functional changes for swappiness values from 1 to 200. There is a slight functional change for 0 and 201: with the existing code, there is no chance to retry for these values because for_each_evictable_type() only iterates once. After this patch, 0 and 201 have behavior that is more consistent with the 1-200 range. This patch (of 2): The for_each_evictable_type() loop in isolate_folios() is misleading: it does not actually iterate over each evictable type. Instead, get_type_to_scan() selects the type to scan, while the iterator `i` merely bounds the number of attempts. Make the fallback behavior explicit in the code and remove the opaque for_each_evictable_type(i, swappiness). Link: https://lore.kernel.org/20260829074204.45304-1-baohua@kernel.org Link: https://lore.kernel.org/20260829074204.45304-2-baohua@kernel.org Signed-off-by: Ridong Chen Co-developed-by: Barry Song (Xiaomi) Signed-off-by: Barry Song (Xiaomi) Signed-off-by: Andrew Morton Reviewed-by: Lian Wang Reviewed-by: Baoquan He Cc: Axel Rasmussen Cc: Baolin Wang Cc: David Hildenbrand Cc: David Stevens Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie --- mm/vmscan.c | 46 ++++++++++++++++++++++++++-------------------- 1 file changed, 26 insertions(+), 20 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index deb087c57007db..1b40c63706f2ed 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -4826,35 +4826,41 @@ static int get_type_to_scan(struct lruvec *lruvec, int swappiness) return positive_ctrl_err(&sp, &pv); } +static inline bool is_single_type_reclaim(int swappiness) +{ + return swappiness == MIN_SWAPPINESS || + swappiness == SWAPPINESS_ANON_ONLY; +} + static int isolate_folios(unsigned long nr_to_scan, struct lruvec *lruvec, struct scan_control *sc, int swappiness, struct list_head *list, int *isolated, int *isolate_type, int *isolate_scanned) { - int i; - int total_scanned = 0; + bool type_fallback_allowed = !is_single_type_reclaim(swappiness); int type = get_type_to_scan(lruvec, swappiness); + int total_scanned = 0, scanned, tier; - for_each_evictable_type(i, swappiness) { - int scanned; - int tier = get_tier_idx(lruvec, type); +retry: + tier = get_tier_idx(lruvec, type); + scanned = scan_folios(nr_to_scan, lruvec, sc, + type, tier, list, isolated); - scanned = scan_folios(nr_to_scan, lruvec, sc, - type, tier, list, isolated); + total_scanned += scanned; + if (*isolated) { + *isolate_type = type; + *isolate_scanned = scanned; + return total_scanned; + } - total_scanned += scanned; - if (*isolated) { - *isolate_type = type; - *isolate_scanned = scanned; - break; - } - /* - * If scanned > 0 and isolated == 0, avoid falling back to the - * other type, as this type remains sufficient. Falling back - * too readily can disrupt the positive_ctrl_err() bias. - */ - if (!scanned) - type = !type; + /* + * We are running out of the current reclaim type. Fall back to + * the other type if allowed. + */ + if (!scanned && type_fallback_allowed) { + type = !type; + type_fallback_allowed = false; + goto retry; } return total_scanned; From 1978f3f192ebb600be9a7ba86d79f74910170cfa Mon Sep 17 00:00:00 2001 From: "Barry Song (Xiaomi)" Date: Sat, 29 Aug 2026 15:42:04 +0800 Subject: [PATCH 0609/1352] mm/mglru: make retry logic explicit in isolate_folios() The existing mainline code retries the same type once in a rather subtle way. `for_each_evictable_type()` may provide one more iteration, allowing the same type to be retried if we scanned some folios but failed to isolate any due to protections, promotions, or races. This patch makes the retry behavior explicit. Link: https://lore.kernel.org/20260829074204.45304-3-baohua@kernel.org Signed-off-by: Barry Song (Xiaomi) Signed-off-by: Andrew Morton Reviewed-by: Baolin Wang Cc: Axel Rasmussen Cc: Baoquan He Cc: David Hildenbrand Cc: David Stevens Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Ridong Chen Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie --- mm/vmscan.c | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/mm/vmscan.c b/mm/vmscan.c index 1b40c63706f2ed..413efe44d1f695 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -4840,6 +4840,7 @@ static int isolate_folios(unsigned long nr_to_scan, struct lruvec *lruvec, bool type_fallback_allowed = !is_single_type_reclaim(swappiness); int type = get_type_to_scan(lruvec, swappiness); int total_scanned = 0, scanned, tier; + bool tried = false; retry: tier = get_tier_idx(lruvec, type); @@ -4859,9 +4860,18 @@ static int isolate_folios(unsigned long nr_to_scan, struct lruvec *lruvec, */ if (!scanned && type_fallback_allowed) { type = !type; + tried = true; type_fallback_allowed = false; goto retry; } + /* + * We scanned some folios but failed to isolate any due to promotions, + * protections, or races. Retry once to avoid a larger loop. + */ + if (scanned && !tried) { + tried = true; + goto retry; + } return total_scanned; } From 5326bd11ddb7ca032365a706781dd28f0e09b9a4 Mon Sep 17 00:00:00 2001 From: "Barry Song (Xiaomi)" Date: Thu, 3 Sep 2026 14:48:52 +0800 Subject: [PATCH 0610/1352] mm: revert slight behavior change for swappiness 1-200 Baoquan's review found that we unexpectedly introduced a slight behavior change for swappiness 1-200. We could now have a case like: 1. First scan -> `scanned != 0` 2. Second scan -> `scanned = 0` 3. Type fallback Step 3 was impossible before. Let's remove this possibility. Link: https://lore.kernel.org/20260903070500.76379-1-baohua@kernel.org Signed-off-by: Barry Song (Xiaomi) Signed-off-by: Andrew Morton Reported-by: Baoquan He Closes: https://lore.kernel.org/linux-mm/apfZQE1X6zGAsBb_@fedora/ Cc: Axel Rasmussen Cc: Baolin Wang Cc: David Hildenbrand Cc: David Stevens Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Ridong Chen Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie --- mm/vmscan.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 413efe44d1f695..42fcdcd3d2e49d 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -4858,7 +4858,7 @@ static int isolate_folios(unsigned long nr_to_scan, struct lruvec *lruvec, * We are running out of the current reclaim type. Fall back to * the other type if allowed. */ - if (!scanned && type_fallback_allowed) { + if (!scanned && !tried && type_fallback_allowed) { type = !type; tried = true; type_fallback_allowed = false; From 9a10c333be139426abafd80ba4b4ecc358cd6c25 Mon Sep 17 00:00:00 2001 From: Avi Weiss Date: Sat, 29 Aug 2026 20:11:12 +0300 Subject: [PATCH 0611/1352] mm/memory: simplify error handling in insert_pages() Patch series "mm/memory: improve insert_pages() error handling", v3. Improve insert_pages() error handling. The first patch simplifies error handling by initializing the error status to zero and assigning error codes at their respective failure sites. The second patch returns -ENOMEM when walk_to_pmd() fails. A NULL return from walk_to_pmd() indicates failure to allocate an upper page-table level, so -ENOMEM is more appropriate than -EFAULT and is consistent with the subsequent pte_alloc() failure. This patch (of 2): Initialize error return status to zero and then set it as needed at each point of failure. Assign -ENOMEM explicitly when pte_alloc() fails as the pte_alloc() macro returns a boolean. Link: https://lore.kernel.org/cover.1788022178.git.thnkslprpt@gmail.com Link: https://lore.kernel.org/dd3a672c858b38c7525541b19a919e120c4e5a0e.1788022178.git.thnkslprpt@gmail.com Signed-off-by: Avi Weiss Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/memory.c | 22 +++++++++++----------- 1 file changed, 11 insertions(+), 11 deletions(-) diff --git a/mm/memory.c b/mm/memory.c index 9cbce5c90bffde..2561dc6bdde635 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -2565,20 +2565,22 @@ static int insert_pages(struct vm_area_struct *vma, unsigned long addr, unsigned long curr_page_idx = 0; unsigned long remaining_pages_total = *num; unsigned long pages_to_write_in_pmd; - int ret; + int err = 0; more: - ret = -EFAULT; pmd = walk_to_pmd(mm, addr); - if (!pmd) + if (!pmd) { + err = -EFAULT; goto out; + } pages_to_write_in_pmd = min_t(unsigned long, remaining_pages_total, PTRS_PER_PTE - pte_index(addr)); /* Allocate the PTE if necessary; takes PMD lock once only. */ - ret = -ENOMEM; - if (pte_alloc(mm, pmd)) + if (pte_alloc(mm, pmd)) { + err = -ENOMEM; goto out; + } while (pages_to_write_in_pmd) { int pte_idx = 0; @@ -2586,15 +2588,14 @@ static int insert_pages(struct vm_area_struct *vma, unsigned long addr, start_pte = pte_offset_map_lock(mm, pmd, addr, &pte_lock); if (!start_pte) { - ret = -EFAULT; + err = -EFAULT; goto out; } for (pte = start_pte; pte_idx < batch_size; ++pte, ++pte_idx) { - int err = insert_page_in_batch_locked(vma, pte, - addr, pages[curr_page_idx], prot); + err = insert_page_in_batch_locked(vma, pte, addr, + pages[curr_page_idx], prot); if (unlikely(err)) { pte_unmap_unlock(start_pte, pte_lock); - ret = err; remaining_pages_total -= pte_idx; goto out; } @@ -2607,10 +2608,9 @@ static int insert_pages(struct vm_area_struct *vma, unsigned long addr, } if (remaining_pages_total) goto more; - ret = 0; out: *num = remaining_pages_total; - return ret; + return err; } /** From ca5a78fbb07ef849608c35f606258093881eefd6 Mon Sep 17 00:00:00 2001 From: Avi Weiss Date: Sat, 29 Aug 2026 20:11:13 +0300 Subject: [PATCH 0612/1352] mm/memory: return -ENOMEM for page-table allocation failure in insert_pages() walk_to_pmd() returns NULL only when p4d_alloc(), pud_alloc(), or pmd_alloc() fails. These are page-table allocation failures, but insert_pages() currently reports them as -EFAULT. Return -ENOMEM instead, consistent with the subsequent pte_alloc() failure and with the single-page insert_page() path, which reports failure of the same page-table allocation chain as -ENOMEM. Address and range validation failures in vm_insert_pages() continue to return -EFAULT. Keep the later -EFAULT return for pte_offset_map_lock(), which is not an allocation failure. Link: https://lore.kernel.org/9d990c3ed43608e674d4b12a8c221a09fd200f49.1788022178.git.thnkslprpt@gmail.com Signed-off-by: Avi Weiss Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/memory.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/memory.c b/mm/memory.c index 2561dc6bdde635..27ffe1a99a08b0 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -2569,7 +2569,7 @@ static int insert_pages(struct vm_area_struct *vma, unsigned long addr, more: pmd = walk_to_pmd(mm, addr); if (!pmd) { - err = -EFAULT; + err = -ENOMEM; goto out; } From bbe6aa1b89a7786140d0443476454e5df58a1c90 Mon Sep 17 00:00:00 2001 From: Wei Yang Date: Sat, 29 Aug 2026 02:58:47 +0000 Subject: [PATCH 0613/1352] mm: adjust out-dated document of __GFP_NOFAIL Commit ee040cbd6e48 ("mm/page_alloc: don't warn about large allocations with __GFP_NOFAIL") remove a warning on allocating large folio with __GFP_NOFAIL, which was placed there by commit 903edea6c53f ("mm: warn about illegal __GFP_NOFAIL usage in a more appropriate location and manner"). While in that commit, it also documented this behavior which is out-dated now. Adjust the document to align to current code, and adjust the comment while at it. Link: https://lore.kernel.org/20260829025847.26779-1-richard.weiyang@gmail.com Signed-off-by: Wei Yang Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Zi Yan Reviewed-by: Vlastimil Babka (SUSE) Cc: Brendan Jackman Cc: David Hildenbrand Cc: Johannes Weiner Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan --- include/linux/gfp_types.h | 2 +- mm/page_alloc.c | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/include/linux/gfp_types.h b/include/linux/gfp_types.h index 190191411009f2..bfd4c43ed77794 100644 --- a/include/linux/gfp_types.h +++ b/include/linux/gfp_types.h @@ -244,7 +244,7 @@ enum { * definitely preferable to use the flag rather than opencode endless * loop around allocator. * Allocating pages from the buddy with __GFP_NOFAIL and order > 1 is - * not supported. Please consider using kvmalloc() instead. + * discouraged. Please consider using kvmalloc() instead if possible. */ #define __GFP_IO ((__force gfp_t)___GFP_IO) #define __GFP_FS ((__force gfp_t)___GFP_FS) diff --git a/mm/page_alloc.c b/mm/page_alloc.c index d622347c877bdf..242edcaa915b33 100644 --- a/mm/page_alloc.c +++ b/mm/page_alloc.c @@ -4837,7 +4837,7 @@ __alloc_pages_slowpath(gfp_t gfp_mask, unsigned int order, if (unlikely(nofail)) { /* - * Also we don't support __GFP_NOFAIL without __GFP_DIRECT_RECLAIM, + * We don't support __GFP_NOFAIL without __GFP_DIRECT_RECLAIM, * otherwise, we may result in lockup. */ WARN_ON_ONCE(!can_direct_reclaim); From 2a7864e1d090d05691bf1b257e9ce0b9f5d2b332 Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Thu, 17 Sep 2026 20:12:02 -0400 Subject: [PATCH 0614/1352] mm/mempolicy: use SRCU for the weighted interleave state Patch series "mm/mempolicy: stop copying state in the interleave paths", v2. The interleave node selectors and bulk allocators take copies of nodemasks and node weights (for weighted interleave) in the fault path. Both of these copies can be entirely eliminated. For node weights, use SRCU to pin the weights in place. This eliminates a copy and a kmalloc from the bulk allocator path. For nodemasks, we can operate directly on pol->nodes as long as we bounds check the walk. A concurrent rebind can shrink the mask, or tear the read of it so the mask appears empty. - The interleave node selectors fall back to numa_node_id() when that happens, which is what they already did when a copy came back empty. - The bulk allocator simply returns what it managed to allocate. The node count and weight totals are read separately from the nodemask walk that consumes them - creating a time-of-check / time-of-use race. Just clamp the walk to a single pass (number of nodes), and clamp each bulk allocation chunk to the space left in the request. The cost is distribution accuracy during a rebind. The copies never corrected for that either - they only kept the code from dividing by zero and overrunning the allocation request. This patch (of 2): alloc_pages_bulk_weighted_interleave() copies iw_table into a scratch array on every call so it can walk the weights outside of RCU. The copy exists only because the loop may sleep in the page allocator and so cannot hold rcu_read_lock(). Use SRCU to pin the global iw_table object and use it in-place instead. Retire through both flavors - call_srcu() for the sleeping readers, then kfree_rcu() for the reference-less ones - so writers no longer block on synchronize_rcu() either. Tested in a VM with KASAN, PROVE_LOCKING and DEBUG_OBJECTS_RCU_HEAD, with a udelay() injected into the read section to widen the race against concurrent sysfs weight writers, and placement checked against the configured weights. Every retired state reached its callback. Swapping the deferred free for a bare kfree() in the same test reports a use-after-free immediately. Link: https://lore.kernel.org/20260918001203.3389165-1-gourry@gourry.net Link: https://lore.kernel.org/20260918001203.3389165-2-gourry@gourry.net Signed-off-by: Gregory Price (Meta) Signed-off-by: Andrew Morton Suggested-by: Andrew Morton Suggested-by: Matthew Wilcox Acked-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Alistair Popple Cc: Byungchul Park Cc: "Huang, Ying" Cc: Joshua Hahn Cc: Matthew Brost Cc: Rakie Kim Cc: Zi Yan --- mm/mempolicy.c | 70 ++++++++++++++++++++++++-------------------------- 1 file changed, 34 insertions(+), 36 deletions(-) diff --git a/mm/mempolicy.c b/mm/mempolicy.c index 060a0eb2691709..2643915dc96699 100644 --- a/mm/mempolicy.c +++ b/mm/mempolicy.c @@ -112,6 +112,7 @@ #include #include #include +#include #include #include @@ -157,6 +158,7 @@ static const int weightiness = 32; */ struct weighted_interleave_state { bool mode_auto; + struct rcu_head rcu; u8 iw_table[]; }; static struct weighted_interleave_state __rcu *wi_state; @@ -168,6 +170,24 @@ static unsigned int *node_bw_table; */ static DEFINE_MUTEX(wi_state_lock); +/* Readers that sleep while walking iw_table hold this instead */ +DEFINE_STATIC_SRCU_FAST(wi_srcu); + +static void wi_state_free_rcu(struct rcu_head *head) +{ + struct weighted_interleave_state *state = + container_of(head, struct weighted_interleave_state, rcu); + + kfree_rcu(state, rcu); +} + +/* Retire through both flavors: sleeping readers use SRCU, the rest RCU */ +static void wi_state_retire(struct weighted_interleave_state *state) +{ + if (state) + call_srcu(&wi_srcu, &state->rcu, wi_state_free_rcu); +} + static u8 get_il_weight(int node) { struct weighted_interleave_state *state; @@ -266,10 +286,7 @@ int mempolicy_set_node_perf(unsigned int node, struct access_coordinate *coords) rcu_assign_pointer(wi_state, new_wi_state); mutex_unlock(&wi_state_lock); - if (old_wi_state) { - synchronize_rcu(); - kfree(old_wi_state); - } + wi_state_retire(old_wi_state); out: kfree(old_bw); return 0; @@ -2644,7 +2661,8 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, unsigned long nr_allocated = 0; unsigned long rounds; unsigned long node_pages, delta; - u8 *weights, weight; + struct srcu_ctr __percpu *scp; + u8 *table, weight; unsigned int weight_total = 0; unsigned long rem_pages = nr_pages; nodemask_t nodes; @@ -2688,25 +2706,14 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, me->il_weight = 0; prev_node = node; - /* create a local copy of node weights to operate on outside rcu */ - weights = kmalloc(nr_node_ids, gfp & GFP_RECLAIM_MASK); - if (!weights) - return total_allocated; - - rcu_read_lock(); - state = rcu_dereference(wi_state); - if (state) { - memcpy(weights, state->iw_table, nr_node_ids * sizeof(u8)); - rcu_read_unlock(); - } else { - rcu_read_unlock(); - for (i = 0; i < nr_node_ids; i++) - weights[i] = 1; - } + /* The page allocator may sleep, pin the weight table with SRCU */ + scp = srcu_read_lock_fast(&wi_srcu); + state = srcu_dereference(wi_state, &wi_srcu); + table = state ? state->iw_table : NULL; /* calculate total, detect system default usage */ for_each_node_mask(node, nodes) - weight_total += weights[node]; + weight_total += table ? table[node] : 1; /* * Calculate rounds/partial rounds to minimize __alloc_pages_bulk calls. @@ -2718,10 +2725,10 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, rounds = rem_pages / weight_total; delta = rem_pages % weight_total; resume_node = next_node_in(prev_node, nodes); - resume_weight = weights[resume_node]; + resume_weight = table ? table[resume_node] : 1; for (i = 0; i < nnodes; i++) { node = next_node_in(prev_node, nodes); - weight = weights[node]; + weight = table ? table[node] : 1; node_pages = weight * rounds; /* If a delta exists, add this node's portion of the delta */ if (delta > weight) { @@ -2747,7 +2754,7 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, } me->il_prev = resume_node; me->il_weight = resume_weight; - kfree(weights); + srcu_read_unlock_fast(&wi_srcu, scp); return total_allocated; } @@ -3673,10 +3680,7 @@ static ssize_t node_store(struct kobject *kobj, struct kobj_attribute *attr, rcu_assign_pointer(wi_state, new_wi_state); mutex_unlock(&wi_state_lock); - if (old_wi_state) { - synchronize_rcu(); - kfree(old_wi_state); - } + wi_state_retire(old_wi_state); return count; } @@ -3742,10 +3746,7 @@ static ssize_t weighted_interleave_auto_store(struct kobject *kobj, update_wi_state: rcu_assign_pointer(wi_state, new_wi_state); mutex_unlock(&wi_state_lock); - if (old_wi_state) { - synchronize_rcu(); - kfree(old_wi_state); - } + wi_state_retire(old_wi_state); return count; } @@ -3789,10 +3790,7 @@ static void wi_state_free(void) rcu_assign_pointer(wi_state, NULL); mutex_unlock(&wi_state_lock); - if (old_wi_state) { - synchronize_rcu(); - kfree(old_wi_state); - } + wi_state_retire(old_wi_state); } static struct kobj_attribute wi_auto_attr = { From e66dea375f3fff448050b3daebe761ec68e8db97 Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Thu, 17 Sep 2026 20:12:03 -0400 Subject: [PATCH 0615/1352] mm/mempolicy: stop copying the nodemask in the interleave paths The interleave node selectors copy pol->nodes onto the stack so the mask cannot change while they walk it. nodemask_t is 128 bytes at MAX_NUMNODES=1024, and two of the three run per folio fault. The copy only buys consistency between the node count and the walk. Drop the consistency and just bounds check the walk instead. The nodelist access is racy by design, but safe as long as we handle the scenario where a cpuset rebind causes a torn nodemask read to perceive the nodemask as empty (weight_total == 0). If an empty nodelist or weight is perceived, fall back to numa_node_id(), which is what the functions already did when the copy came back empty Otherwise, iterating the nodelist during the actual allocation loop is perfectly safe - a concurrent rebind may simply cause a skew in in the distribution of memory (or fail and fall back the same as any other error condition). weighted_interleave_nid() counts the nodes as we sum the weights. We use that node count to limit the maximum skew a single node can host. interleave_nid() walks with next_node_in() rather than next_node(), so a mask that shrank mid-walk wraps to a node still in the policy. alloc_pages_bulk_weighted_interleave() derives per-node counts from a weight total summed over the mask, so a changing mask can make them exceed the request. Clamp each chunk to the space left in page_array. A cpuset cookie will not work here: two of these take VMA policies, which mpol_rebind_mm() rebinds under mmap_write_lock(), not mems_allowed_seq. Cost is distribution accuracy during a rebind - but the copy never corrected this anyway, it was just a safety mechanism to prevent div/0 and overrunning the alloc request buffer. Remove read_once_policy_nodemask(), now unused. -fstack-usage at MAX_NUMNODES=1024: weighted_interleave_nid 184 -> 56 interleave_nid 168 -> 32 alloc_pages_bulk_mempolicy_noprof 360 -> 136 Link: https://lore.kernel.org/20260918001203.3389165-3-gourry@gourry.net Signed-off-by: Gregory Price (Meta) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Rakie Kim Assisted-by: LLM Cc: Alistair Popple Cc: Byungchul Park Cc: "Huang, Ying" Cc: Joshua Hahn Cc: Matthew Brost Cc: Matthew Wilcox Cc: Zi Yan --- mm/mempolicy.c | 93 ++++++++++++++++++++++++++++++-------------------- 1 file changed, 56 insertions(+), 37 deletions(-) diff --git a/mm/mempolicy.c b/mm/mempolicy.c index 2643915dc96699..fd97fb0289bc98 100644 --- a/mm/mempolicy.c +++ b/mm/mempolicy.c @@ -2197,34 +2197,15 @@ unsigned int mempolicy_slab_node(void) } } -static unsigned int read_once_policy_nodemask(struct mempolicy *pol, - nodemask_t *mask) -{ - /* - * barrier stabilizes the nodemask locally so that it can be iterated - * over safely without concern for changes. Allocators validate node - * selection does not violate mems_allowed, so this is safe. - */ - barrier(); - memcpy(mask, &pol->nodes, sizeof(nodemask_t)); - barrier(); - return nodes_weight(*mask); -} - static unsigned int weighted_interleave_nid(struct mempolicy *pol, pgoff_t ilx) { struct weighted_interleave_state *state; - nodemask_t nodemask; - unsigned int target, nr_nodes; + unsigned int target, nnodes = 0; u8 *table = NULL; unsigned int weight_total = 0; u8 weight; int nid = 0; - nr_nodes = read_once_policy_nodemask(pol, &nodemask); - if (!nr_nodes) - return numa_node_id(); - rcu_read_lock(); state = rcu_dereference(wi_state); @@ -2232,22 +2213,45 @@ static unsigned int weighted_interleave_nid(struct mempolicy *pol, pgoff_t ilx) if (state) table = state->iw_table; - /* calculate the total weight */ - for_each_node_mask(nid, nodemask) + /* calculate the total weight and the node count */ + for_each_node_mask(nid, pol->nodes) { weight_total += table ? table[nid] : 1; + nnodes++; + } + + /* the mask is empty */ + if (!weight_total) { + rcu_read_unlock(); + return numa_node_id(); + } /* Calculate the node offset based on totals */ target = ilx % weight_total; - nid = first_node(nodemask); - while (target) { + nid = first_node(pol->nodes); + + /* + * The target was calculated in a separate loop, and a concurrent + * rebind can change the contents of pol->nodes as we calculate. + * Access is safe, in the worst case we suddenly perceive an empty + * nodemask and return numa_node_id() below - otherwise we may + * simply cause a skew in allocations. + * + * Clamp this loop to a single pass (nnodes) to keep the walk + * bounded by node count. + */ + while (target && nnodes-- && nid < MAX_NUMNODES) { /* detect system default usage */ weight = table ? table[nid] : 1; if (target < weight) break; target -= weight; - nid = next_node_in(nid, nodemask); + nid = next_node_in(nid, pol->nodes); } rcu_read_unlock(); + + /* the mask emptied under the walk */ + if (nid >= MAX_NUMNODES) + return numa_node_id(); return nid; } @@ -2258,18 +2262,23 @@ static unsigned int weighted_interleave_nid(struct mempolicy *pol, pgoff_t ilx) */ static unsigned int interleave_nid(struct mempolicy *pol, pgoff_t ilx) { - nodemask_t nodemask; unsigned int target, nnodes; int i; int nid; - nnodes = read_once_policy_nodemask(pol, &nodemask); + nnodes = nodes_weight(pol->nodes); if (!nnodes) return numa_node_id(); target = ilx % nnodes; - nid = first_node(nodemask); - for (i = 0; i < target; i++) - nid = next_node(nid, nodemask); + nid = first_node(pol->nodes); + + /* A concurrent cpuset rebind may cause us to see an empty nodemask */ + for (i = 0; i < target && nid < MAX_NUMNODES; i++) + nid = next_node_in(nid, pol->nodes); + + /* the mask emptied under the walk */ + if (nid >= MAX_NUMNODES) + return numa_node_id(); return nid; } @@ -2665,7 +2674,6 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, u8 *table, weight; unsigned int weight_total = 0; unsigned long rem_pages = nr_pages; - nodemask_t nodes; int nnodes, node; int resume_node = MAX_NUMNODES - 1; u8 resume_weight = 0; @@ -2675,10 +2683,10 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, if (!nr_pages) return 0; - /* read the nodes onto the stack, retry if done during rebind */ + /* count the nodes, retry if a rebind happened during the read */ do { cpuset_mems_cookie = read_mems_allowed_begin(); - nnodes = read_once_policy_nodemask(pol, &nodes); + nnodes = nodes_weight(pol->nodes); } while (read_mems_allowed_retry(cpuset_mems_cookie)); /* if the nodemask has become invalid, we cannot do anything */ @@ -2688,7 +2696,7 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, /* Continue allocating from most recent node and adjust the nr_pages */ node = me->il_prev; weight = me->il_weight; - if (weight && node_isset(node, nodes)) { + if (weight && node_isset(node, pol->nodes)) { node_pages = min(rem_pages, weight); nr_allocated = __alloc_pages_bulk(gfp, node, NULL, node_pages, page_array); @@ -2712,9 +2720,13 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, table = state ? state->iw_table : NULL; /* calculate total, detect system default usage */ - for_each_node_mask(node, nodes) + for_each_node_mask(node, pol->nodes) weight_total += table ? table[node] : 1; + /* the mask emptied since it was counted */ + if (!weight_total) + goto out; + /* * Calculate rounds/partial rounds to minimize __alloc_pages_bulk calls. * Track which node weighted interleave should resume from. @@ -2724,10 +2736,14 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, */ rounds = rem_pages / weight_total; delta = rem_pages % weight_total; - resume_node = next_node_in(prev_node, nodes); + resume_node = next_node_in(prev_node, pol->nodes); + if (resume_node >= MAX_NUMNODES) + goto out; resume_weight = table ? table[resume_node] : 1; for (i = 0; i < nnodes; i++) { - node = next_node_in(prev_node, nodes); + node = next_node_in(prev_node, pol->nodes); + if (node >= MAX_NUMNODES) + break; weight = table ? table[node] : 1; node_pages = weight * rounds; /* If a delta exists, add this node's portion of the delta */ @@ -2744,6 +2760,8 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, /* node_pages can be 0 if an allocation fails and rounds == 0 */ if (!node_pages) break; + /* a rebind can invalidate the counts: never overrun page_array */ + node_pages = min(node_pages, nr_pages - total_allocated); nr_allocated = __alloc_pages_bulk(gfp, node, NULL, node_pages, page_array); page_array += nr_allocated; @@ -2754,6 +2772,7 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, } me->il_prev = resume_node; me->il_weight = resume_weight; +out: srcu_read_unlock_fast(&wi_srcu, scp); return total_allocated; } From 5930c56808dda3d93da9914ff06884a702e19f5f Mon Sep 17 00:00:00 2001 From: Sang-Heon Jeon Date: Wed, 19 Aug 2026 01:48:30 +0900 Subject: [PATCH 0616/1352] percpu: fix the comment about which sizes share a slot A percpu allocation is at least PCPU_MIN_ALLOC_SIZE bytes, and __pcpu_size_to_slot() returns 1 for sizes below 16 bytes and 2 for sizes from 16 to 31 bytes. So fix the wrong comment. Link: https://lore.kernel.org/20260818164831.3138490-1-ekffu200098@gmail.com Signed-off-by: Sang-Heon Jeon Signed-off-by: Andrew Morton Cc: Dennis Zhou Cc: Tejun Heo Cc: Christoph Lameter --- mm/percpu.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/percpu.c b/mm/percpu.c index a802d72c116fbb..47a903fe3b5124 100644 --- a/mm/percpu.c +++ b/mm/percpu.c @@ -100,7 +100,7 @@ /* * The slots are sorted by the size of the biggest continuous free area. - * 1-31 bytes share the same slot. + * [PCPU_MIN_ALLOC_SIZE..15] bytes share the same slot. */ #define PCPU_SLOT_BASE_SHIFT 5 /* chunks in slots below this are subject to being sidelined on failed alloc */ From 3e65a20ac846a40893e00055061779e00eef5af9 Mon Sep 17 00:00:00 2001 From: Breno Leitao Date: Tue, 18 Aug 2026 03:06:23 -0700 Subject: [PATCH 0617/1352] mm, swap: distinguish a malformed swap entry from a dying device Patch series "mm, swap: don't spin on a bad swap entry", v3. I've seen some machines at Meta fleet that show the following type of problem: 1) It gets some weird warning: BUG: Bad page map in process khugepaged pte:f000eef300000017 pmd:00000067 addr:00007f57c0a01000 vm_flags:20200073 anon_vma:ffff88829af7c340 mapping:0000000000000000 index:7f57c0a01 The corruption is most likely the collapse/PT_RECLAIM race fixed by commit 366a4532d96f ("mm: fix the race between collapse and PT_RECLAIM under per-vma lock"). But this series is not about this one. 2) Then the fault never makes progress. do_swap_page() returns 0 when get_swap_device() fails, so the fault is retried, reads the same entry and faults again. Nothing in the round trip changes the PTE, and the same line comes out on every pass: get_swap_device: Bad swap offset entry 3ffffffc043c5 Patch 1 makes get_swap_device() return ERR_PTR(-EIO) for a malformed entry, keeping NULL for a device swapoff is taking away, and converts the callers. No functional change expected. Patch 2 uses that to return VM_FAULT_SIGBUS instead of retrying. This patch (of 2): get_swap_device() returns NULL for two different things: an entry whose type names no swap device or whose offset is past the end of one, and a device that swapoff is taking away. The first never becomes valid, the second does, and callers cannot tell them apart. Return ERR_PTR(-EIO) for the two malformed cases and keep NULL for swapoff. copy_nonpresent_pte() already reports -EIO for an entry whose type names no device. Callers bail out on failure either way, so switch them to IS_ERR_OR_NULL(), and let the two paths that drop the reference skip an error pointer. No functional change. Link: https://lore.kernel.org/20260818-swap-v3-0-d3fa52598a59@debian.org Link: https://lore.kernel.org/20260818-swap-v3-1-d3fa52598a59@debian.org Signed-off-by: Breno Leitao Signed-off-by: Andrew Morton Reviewed-by: Barry Song Acked-by: Kairui Song Reviewed-by: Nhat Pham Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Baoquan He Cc: Chengming Zhou Cc: Chris Li Cc: Hugh Dickins Cc: Jann Horn Cc: Johannes Weiner Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Pedro Falcato Cc: Peter Xu Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/memory.c | 6 +++--- mm/mincore.c | 2 +- mm/shmem.c | 2 +- mm/swap_state.c | 4 ++-- mm/swapfile.c | 15 ++++++++++----- mm/userfaultfd.c | 4 ++-- mm/zswap.c | 2 +- 7 files changed, 20 insertions(+), 15 deletions(-) diff --git a/mm/memory.c b/mm/memory.c index 27ffe1a99a08b0..e59d6c2a343205 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -4956,9 +4956,9 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) goto out; } - /* Prevent swapoff from happening to us. */ + /* Prevent swapoff from happening to us, and reject a bad entry. */ si = get_swap_device(entry); - if (unlikely(!si)) + if (IS_ERR_OR_NULL(si)) goto out; folio = swap_cache_get_folio(entry); @@ -5268,7 +5268,7 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) if (vmf->pte) pte_unmap_unlock(vmf->pte, vmf->ptl); out: - if (si) + if (!IS_ERR_OR_NULL(si)) put_swap_device(si); return ret; out_nomap: diff --git a/mm/mincore.c b/mm/mincore.c index ff4ac828176837..c086836bc4bcc5 100644 --- a/mm/mincore.c +++ b/mm/mincore.c @@ -71,7 +71,7 @@ static unsigned char mincore_swap(swp_entry_t entry, bool shmem) */ if (shmem) { si = get_swap_device(entry); - if (!si) + if (IS_ERR_OR_NULL(si)) return 0; } folio = swap_cache_get_folio(entry); diff --git a/mm/shmem.c b/mm/shmem.c index ae39966fc7304e..d3f24b5977bd5f 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -2481,7 +2481,7 @@ static int shmem_swapin_folio(struct inode *inode, pgoff_t index, si = get_swap_device(index_entry); order = shmem_confirm_swap(mapping, index, index_entry); - if (unlikely(!si)) { + if (IS_ERR_OR_NULL(si)) { if (order < 0) return -EEXIST; else diff --git a/mm/swap_state.c b/mm/swap_state.c index f3961fdd857dc6..305877e1f4d7bf 100644 --- a/mm/swap_state.c +++ b/mm/swap_state.c @@ -716,7 +716,7 @@ struct folio *read_swap_cache_async(struct swap_io_ctx *ctx, swp_entry_t entry, struct folio *folio; si = get_swap_device(entry); - if (!si) + if (IS_ERR_OR_NULL(si)) return NULL; mpol = get_vma_policy(vma, addr, 0, &ilx); @@ -952,7 +952,7 @@ static struct folio *swap_vma_readahead(swp_entry_t targ_entry, gfp_t gfp_mask, */ if (swp_type(entry) != swp_type(targ_entry)) { si = get_swap_device(entry); - if (!si) + if (IS_ERR_OR_NULL(si)) continue; } folio = swap_cache_read_folio(&ctx, entry, gfp_mask, mpol, ilx, diff --git a/mm/swapfile.c b/mm/swapfile.c index 3b2279a16d1cf8..408f6c72fb5a69 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -1504,7 +1504,7 @@ int swap_retry_table_alloc(swp_entry_t entry, gfp_t gfp) unsigned long offset = swp_offset(entry); si = get_swap_device(entry); - if (!si) + if (IS_ERR_OR_NULL(si)) return 0; ci = __swap_offset_to_cluster(si, offset); @@ -1859,7 +1859,10 @@ void folio_put_swap(struct folio *folio, struct page *page) * Check whether swap entry is valid in the swap device. If so, * return pointer to swap_info_struct, and keep the swap entry valid * via preventing the swap device from being swapoff, until - * put_swap_device() is called. Otherwise return NULL. + * put_swap_device() is called. Return NULL for an empty entry or a + * device that is going away, and ERR_PTR(-EIO) if the entry's type + * names no swap device or its offset is past the end of one. These EIOs + * are preceded by pr_err(). * * Notice that swapoff or swapoff+swapon can still happen before the * percpu_ref_tryget_live() in get_swap_device() or after the @@ -1900,12 +1903,14 @@ struct swap_info_struct *get_swap_device(swp_entry_t entry) return si; bad_nofile: pr_err_ratelimited("%s: %s%08lx\n", __func__, Bad_file, entry.val); + return ERR_PTR(-EIO); + out: return NULL; put_out: pr_err_ratelimited("%s: %s%08lx\n", __func__, Bad_offset, entry.val); percpu_ref_put(&si->users); - return NULL; + return ERR_PTR(-EIO); } /* @@ -2001,7 +2006,7 @@ int swp_swapcount(swp_entry_t entry) int count; si = get_swap_device(entry); - if (!si) + if (IS_ERR_OR_NULL(si)) return 0; ci = swap_cluster_lock(si, swp_offset(entry)); @@ -2127,7 +2132,7 @@ void swap_put_entries_direct(swp_entry_t entry, int nr) struct swap_info_struct *si; si = get_swap_device(entry); - if (WARN_ON_ONCE(!si)) + if (WARN_ON_ONCE(IS_ERR_OR_NULL(si))) return; if (WARN_ON_ONCE(end_offset > si->max)) goto out; diff --git a/mm/userfaultfd.c b/mm/userfaultfd.c index f39f109f17989e..b909ec8ef20bce 100644 --- a/mm/userfaultfd.c +++ b/mm/userfaultfd.c @@ -1701,7 +1701,7 @@ static long move_pages_ptes(struct mm_struct *mm, pmd_t *dst_pmd, pmd_t *src_pmd } si = get_swap_device(entry); - if (unlikely(!si)) { + if (IS_ERR_OR_NULL(si)) { ret = -EAGAIN; goto out; } @@ -1758,7 +1758,7 @@ static long move_pages_ptes(struct mm_struct *mm, pmd_t *dst_pmd, pmd_t *src_pmd if (dst_pte) pte_unmap(dst_pte); mmu_notifier_invalidate_range_end(&range); - if (si) + if (!IS_ERR_OR_NULL(si)) put_swap_device(si); return ret; diff --git a/mm/zswap.c b/mm/zswap.c index fc869d60ef1b48..f3ae3c81e48eac 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -984,7 +984,7 @@ static int zswap_writeback_entry(struct zswap_entry *entry, /* try to allocate swap cache folio */ si = get_swap_device(swpentry); - if (!si) + if (IS_ERR_OR_NULL(si)) return -EEXIST; mpol = get_task_policy(current); From 3d238f9f0243de0b1f1c9dc989ff495b869ea901 Mon Sep 17 00:00:00 2001 From: Breno Leitao Date: Tue, 18 Aug 2026 03:06:24 -0700 Subject: [PATCH 0618/1352] mm: fail the fault on a malformed swap entry instead of retrying it do_swap_page() returns 0 when get_swap_device() fails, which the fault handler reads as "handled". For an entry that can never become valid the retry takes the same fault again, so the thread spins forever, retrying on the same fault. Return VM_FAULT_SIGBUS (Bad access) for a malformed entry (pr_err() was called at get_swap_device()). Link: https://lore.kernel.org/20260818-swap-v3-2-d3fa52598a59@debian.org Signed-off-by: Breno Leitao Signed-off-by: Andrew Morton Acked-by: Kairui Song Reviewed-by: Barry Song Reviewed-by: Nhat Pham Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Baoquan He Cc: Chengming Zhou Cc: Chris Li Cc: Hugh Dickins Cc: Jann Horn Cc: Johannes Weiner Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Pedro Falcato Cc: Peter Xu Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/memory.c | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/mm/memory.c b/mm/memory.c index e59d6c2a343205..347db2acd0f83b 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -4958,8 +4958,11 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) /* Prevent swapoff from happening to us, and reject a bad entry. */ si = get_swap_device(entry); - if (IS_ERR_OR_NULL(si)) + if (IS_ERR_OR_NULL(si)) { + if (IS_ERR(si)) + ret = VM_FAULT_SIGBUS; goto out; + } folio = swap_cache_get_folio(entry); if (folio) From 9a46d797f06c1cb634493e4210134b46acb080a4 Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Mon, 17 Aug 2026 18:08:08 -0400 Subject: [PATCH 0619/1352] mm/huge_memory: skip zone device folios in madvise_free_huge_pmd() Patch series "mm: reject zone device folios in more folio walkers", v2. Several LRU-oriented mm walkers resolve the folio backing a PMD entry (or a physical pfn) and then reclaim, age, migrate, or lazyfree it without ever checking for ZONE_DEVICE memory. This series adds missing folio_is_zone_device() rejections, matching the checks that comparable walkers already perform. - mm/huge_memory, mm/madvise: the !pmd_present branch above these sites only filters device-private entries (which are non-present). A present zone device PMD (e.g. device-coherent) would still reach the folio and be lazyfreed / aged / paged out. Add an explicit check. - mm/mempolicy: queue_folios_pmd() can see a present zone device PMD (e.g. device-coherent) and queue it for migration. No crash reproducer - this is a correctness/hardening cleanup found by inspection. All checks are placed after the folio is resolved and before it is acted upon, on paths that already hold the relevant page-table lock, so no locking or refcount changes are involved. This patch (of 3): madvise_free_huge_pmd() resolves the folio backing a PMD via pmd_folio() and marks it lazyfree without checking for zone device memory. The surrounding guards do not cover every zone device case: - MADV_FREE only operates on anonymous VMAs (DAX mappings are excluded) - !pmd_present() branch rejects device-private and migration entries - present zone device PMD (device coherent THP) is not filtered. Unlike vm_normal_page_pmd(), it performs no special/pfnmap check, and would be marked lazyfree here. Bail out when the folio is a zone device folio. Link: https://lore.kernel.org/20260817220810.1175596-1-gourry@gourry.net Link: https://lore.kernel.org/20260817220810.1175596-2-gourry@gourry.net Fixes: a30b48bf1b24 ("mm/migrate_device: implement THP migration of zone device pages") Signed-off-by: Gregory Price (Meta) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Tested-by: Lance Yang Cc: Alistair Popple Cc: Balbir Singh Cc: Baolin Wang Cc: Barry Song Cc: Byungchul Park Cc: Dev Jain Cc: "Huang, Ying" Cc: Jann Horn Cc: Joshua Hahn Cc: Liam R. Howlett Cc: Matthew Brost Cc: Rakie Kim Cc: Ryan Roberts Cc: Vlastimil Babka Cc: Zi Yan Cc: --- mm/huge_memory.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 1e5d68acf62a52..54494c3fa9835e 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -2423,6 +2423,10 @@ bool madvise_free_huge_pmd(struct mmu_gather *tlb, struct vm_area_struct *vma, } folio = pmd_folio(orig_pmd); + + if (folio_is_zone_device(folio)) + goto out; + /* * If other processes are mapping this folio, we couldn't discard * the folio unless they all do MADV_FREE so let's skip the folio. From aaab94f972858944db6b43a0387acc6f8b98e344 Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Mon, 17 Aug 2026 18:08:09 -0400 Subject: [PATCH 0620/1352] mm/madvise: skip zone device folios in cold/pageout PMD range madvise_cold_or_pageout_pte_range() resolves the folio backing a PMD via pmd_folio() and ages or reclaims it without checking for zone device memory. The surrounding guards do not cover every zone device case: - can_madv_lru_vma() excludes VM_PFNMAP and VM_HUGETLB VMAs (so device DAX is filtered) - !pmd_present() branch above rejects device-private and migration entries, which are non-present. - A present zone device PMD - e.g. a device-coherent THP - is not filtered by any of these, nor by pmd_folio() (unlike vm_normal_page_pmd(), it performs no special/pfnmap check), and would be aged or paged out here. Skip ZONE_DEVICE folios explicitly during MADV_COLD/PAGEOUT. Link: https://lore.kernel.org/20260817220810.1175596-3-gourry@gourry.net Fixes: a30b48bf1b24 ("mm/migrate_device: implement THP migration of zone device pages") Signed-off-by: Gregory Price (Meta) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Balbir Singh Tested-by: Lance Yang Cc: Alistair Popple Cc: Baolin Wang Cc: Barry Song Cc: Byungchul Park Cc: Dev Jain Cc: "Huang, Ying" Cc: Jann Horn Cc: Joshua Hahn Cc: Liam R. Howlett Cc: Matthew Brost Cc: Rakie Kim Cc: Ryan Roberts Cc: Vlastimil Babka Cc: Zi Yan Cc: --- mm/madvise.c | 3 +++ 1 file changed, 3 insertions(+) diff --git a/mm/madvise.c b/mm/madvise.c index eeee82cf2b3f4b..73c2901b9adbf0 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -404,6 +404,9 @@ static int madvise_cold_or_pageout_pte_range(pmd_t *pmd, folio = pmd_folio(orig_pmd); + if (folio_is_zone_device(folio)) + goto huge_unlock; + /* Do not interfere with other mappings of this folio */ if (folio_maybe_mapped_shared(folio)) goto huge_unlock; From e350ba68baa872ac59be21e04fb67623d626e372 Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Mon, 17 Aug 2026 18:08:10 -0400 Subject: [PATCH 0621/1352] mm/mempolicy: skip zone device folios when queueing folios queue_folios_pte_range() already pairs vm_normal_folio() with an explicit folio_is_zone_device() check before adding folios to the migration pagelist. vm_normal_folio() alone does not reject zone device memory (a present device-coherent page in a normal VMA is returned as "normal"). Mirror the explicit check in queue_folios_pmd() as well. queue_folios_pmd() uses pmd_folio() directly and can encounter a present zone device PMD - e.g. a device-coherent THP. This is not filtered by existing checks: !pmd_present() - only rejects non-present device-private and migration entries vma_migratable() - excludes DAX and VM_PFNMAP. The early return also means such a folio is no longer counted in qp->nr_failed under MPOL_MF_STRICT. This is the same pattern used by queue_folios_pte_range() (skipping zone device without failing). Link: https://lore.kernel.org/20260817220810.1175596-4-gourry@gourry.net Fixes: a30b48bf1b24 ("mm/migrate_device: implement THP migration of zone device pages") Signed-off-by: Gregory Price (Meta) Signed-off-by: Andrew Morton Reviewed-by: Balbir Singh Acked-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) Tested-by: Lance Yang Cc: Alistair Popple Cc: Baolin Wang Cc: Barry Song Cc: Byungchul Park Cc: Dev Jain Cc: "Huang, Ying" Cc: Jann Horn Cc: Joshua Hahn Cc: Liam R. Howlett Cc: Matthew Brost Cc: Rakie Kim Cc: Ryan Roberts Cc: Vlastimil Babka Cc: Zi Yan Cc: --- mm/mempolicy.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/mm/mempolicy.c b/mm/mempolicy.c index fd97fb0289bc98..95dba5d919e925 100644 --- a/mm/mempolicy.c +++ b/mm/mempolicy.c @@ -679,6 +679,8 @@ static void queue_folios_pmd(pmd_t *pmd, struct mm_walk *walk) return; } folio = pmd_folio(pmdval); + if (folio_is_zone_device(folio)) + return; if (is_huge_zero_folio(folio)) { walk->action = ACTION_CONTINUE; return; From f4ddef22e483b067c3d522d5b8d0a61163ef83f9 Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Mon, 17 Aug 2026 14:19:55 +0800 Subject: [PATCH 0622/1352] selftests/mm: khugepaged: consolidate error exits via kselftest helpers Replace the perror()+exit(EXIT_FAILURE) pattern with ksft_exit_fail_perror() so failures are reported through the kselftest framework, consistent with the rest of the file. Link: https://lore.kernel.org/20260817061955.45454-1-hongfu.li@linux.dev Signed-off-by: Hongfu Li Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Mike Rapoport (Microsoft) Reviewed-by: Zi Yan Reviewed-by: Lance Yang Cc: Baolin Wang Cc: Barry Song Cc: Dev Jain Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Ryan Roberts Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/mm/khugepaged.c | 20 +++++++------------- 1 file changed, 7 insertions(+), 13 deletions(-) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 83d27d069c4139..f82673f5f6b47e 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -337,21 +337,15 @@ static void *file_setup_area_common(int nr_hpages, enum file_setup_ops setup) ksft_exit_fail_perror("open()"); size = nr_hpages * hpage_pmd_size; - if (ftruncate(fd, size)) { - perror("ftruncate()"); - exit(EXIT_FAILURE); - } + if (ftruncate(fd, size)) + ksft_exit_fail_perror("ftruncate()"); p = mmap(BASE_ADDR, size, PROT_READ | PROT_WRITE, MAP_SHARED, fd, 0); - if (p != BASE_ADDR) { - perror("mmap()"); - exit(EXIT_FAILURE); - } + if (p != BASE_ADDR) + ksft_exit_fail_perror("mmap()"); fill_memory(p, 0, size); - if (msync(p, size, MS_SYNC)) { - perror("msync()"); - exit(EXIT_FAILURE); - } + if (msync(p, size, MS_SYNC)) + ksft_exit_fail_perror("msync()"); close(fd); munmap(p, size); success("OK"); @@ -426,7 +420,7 @@ static bool file_check_huge(void *addr, size_t len, int nr_hpages, case VMA_SHMEM: return check_huge_shmem(addr, len, nr_hpages, hpage_size); default: - exit(EXIT_FAILURE); + ksft_exit_fail_msg("Unknown VMA type\n"); return false; } } From e3be0dbda75e2ee1e12fd72a37a9015b6c6883ee Mon Sep 17 00:00:00 2001 From: Ridong Chen Date: Sun, 30 Aug 2026 08:20:43 +0800 Subject: [PATCH 0623/1352] memcg: acquire peaks_lock when reading memory.peak MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Patch series "mm, memcg: fix memory.peak reset clobbering other fds' watermark", v4. The memory.peak / memory.swap.peak per-fd watermark tracking has two issues. Each open fd is a watcher and reads back max(its own value, the shared local_watermark); both bugs live in that scheme. Worst case for both is the same and is userspace-visible: a reader of memory.peak (or memory.swap.peak) gets a value lower than the true peak, so a tool that sizes or bills a cgroup by its peak usage under-reports it. Patch 1 (read side) fixes the race Sashiko pointed out in the v1 review [1]: peak_show() inspects local_watermark and the per-fd values without holding peaks_lock, so a reader that races an unrelated peak_write() reset briefly observes the lowered value. Transient. It takes peaks_lock in the show path. Patch 2 (write side) fixes peak_write(): on a reset it stores the current usage into the other watchers instead of the old watermark, so once usage has dropped from a peak a reset on one fd drags every other fd's peak down too, even fds that never reset. This patch (of 2): Sashiko reported that a reader can transiently observe a lower peak within a race window [1]. peak_show() returns max(local_watermark, ofp->value), but peak_write() updates those two under peaks_lock while the reader takes no lock. The interleaving is: writer (reset on fd A) reader (fd B) ---------------------- ------------- usage = page_counter_read(pc) WRITE_ONCE(local_watermark, usage) // watermark lowered to usage lw = READ_ONCE(local_watermark) // sees the lowered usage val = READ_ONCE(ofp->value) // B's value not updated yet return max(lw, val) // both low -> low peak WRITE_ONCE(peer_ctx->value, usage) // B updated, but too late Fix it by acquiring peaks_lock when reading the peak, so the reader sees a consistent snapshot of local_watermark and the per-fd values. The same race applies to memory.swap.peak, which shares peaks_lock and the peak_write() path, so take the lock there as well. Link: https://lore.kernel.org/20260830002044.1938621-1-ridong.chen@linux.dev Link: https://lore.kernel.org/20260830002044.1938621-2-ridong.chen@linux.dev Link: https://sashiko.dev/#/patchset/20260730115314.1069089-1-ridong.chen@linux.dev?part=1 [1] Fixes: c6f53ed8f213 ("mm, memcg: cg2 memory{.swap,}.peak write handlers") Signed-off-by: Ridong Chen Signed-off-by: Andrew Morton Acked-by: Johannes Weiner Acked-by: Shakeel Butt Reviewed-by: Muchun Song Assisted-by: Claude:claude-opus-4-8 Cc: David Finkel Cc: Michal Hocko Cc: Michal Koutný Cc: Roman Gushchin Cc: Tejun Heo Cc: Tao Cui --- mm/memcontrol.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 30636b9d96739e..1ee974cb6d3727 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -4745,6 +4745,7 @@ static int memory_peak_show(struct seq_file *sf, void *v) { struct mem_cgroup *memcg = mem_cgroup_from_css(seq_css(sf)); + guard(spinlock)(&memcg->peaks_lock); return peak_show(sf, v, &memcg->memory); } @@ -5888,6 +5889,7 @@ static int swap_peak_show(struct seq_file *sf, void *v) { struct mem_cgroup *memcg = mem_cgroup_from_css(seq_css(sf)); + guard(spinlock)(&memcg->peaks_lock); return peak_show(sf, v, &memcg->swap); } From 075a87d9c0f90639245ecd56b1af5ff05fa0d900 Mon Sep 17 00:00:00 2001 From: Ridong Chen Date: Sun, 30 Aug 2026 08:20:44 +0800 Subject: [PATCH 0624/1352] mm, memcg: fix memory.peak reset clobbering other fds' watermark MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Writing to memory.peak resets the peak for that fd only. Each fd is a watcher and reads back max(its own value, the shared local_watermark). peak_write() resets by lowering local_watermark to the current usage. To keep the other watchers' peaks it then walks the watcher list, but it stores the current usage into them instead of the old watermark. So once usage has dropped from a peak, a reset on one fd wrongly drags every other fd's peak down too, even fds that never reset. Reproduced on 7.2.0-rc5-next under QEMU, two fds A and B on one cgroup: B sees the peak (410624 KB), usage drops, then A resets -- and B's peak collapses to 1060 KB although B never reset. With this patch B keeps reading 410624 KB. Fix: save the old watermark before lowering it and use that to floor the other watchers, so a reset only affects the fd that issued it. Link: https://lore.kernel.org/20260830002044.1938621-3-ridong.chen@linux.dev Fixes: c6f53ed8f213 ("mm, memcg: cg2 memory{.swap,}.peak write handlers") Signed-off-by: Ridong Chen Signed-off-by: Andrew Morton Closes: https://sashiko.dev/#/patchset/20260807090000.1532495-1-ridong.chen@linux.dev Acked-by: Tao Cui Acked-by: Johannes Weiner Acked-by: Shakeel Butt Assisted-by: Claude:claude-opus-4-8 Cc: David Finkel Cc: Michal Hocko Cc: Michal Koutný Cc: Muchun Song Cc: Roman Gushchin Cc: Tejun Heo --- mm/memcontrol.c | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 1ee974cb6d3727..d5ebe83eae3efc 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -4775,7 +4775,7 @@ static ssize_t peak_write(struct kernfs_open_file *of, char *buf, size_t nbytes, loff_t off, struct page_counter *pc, struct list_head *watchers) { - unsigned long usage; + unsigned long usage, old_watermark; struct cgroup_of_peak *peer_ctx; struct mem_cgroup *memcg = mem_cgroup_from_css(of_css(of)); struct cgroup_of_peak *ofp = of_peak(of); @@ -4783,11 +4783,12 @@ static ssize_t peak_write(struct kernfs_open_file *of, char *buf, size_t nbytes, spin_lock(&memcg->peaks_lock); usage = page_counter_read(pc); + old_watermark = READ_ONCE(pc->local_watermark); WRITE_ONCE(pc->local_watermark, usage); list_for_each_entry(peer_ctx, watchers, list) - if (usage > peer_ctx->value) - WRITE_ONCE(peer_ctx->value, usage); + if (peer_ctx != ofp && old_watermark > peer_ctx->value) + WRITE_ONCE(peer_ctx->value, old_watermark); /* initial write, register watcher */ if (ofp->value == OFP_PEAK_UNSET) From 1cfff3326ea8d787517266a7c9c7daa8bf0b48df Mon Sep 17 00:00:00 2001 From: Eamon Sippy Date: Sat, 15 Aug 2026 10:32:45 +0000 Subject: [PATCH 0625/1352] mm: cma: make mm/cma.h self-contained and conditionalize includes mm/cma.h uses types from , , and without explicitly including them, violating the kernel header self-containment guidelines. and are also included unconditionally even though they are only needed under CONFIG_CMA_DEBUGFS and CONFIG_CMA_SYSFS respectively. Move the struct cma_kobject definition and inside the CONFIG_CMA_SYSFS block, and move inside CONFIG_CMA_DEBUGFS. Remove spurious trailing semicolons after the empty inline function bodies in the CONFIG_CMA_SYSFS #else branch. Add so that MAX_CMA_AREAS and CMA_MAX_NAME are always available when this header is included. Link: https://lore.kernel.org/20260815103246.5315-1-eamon112009@gmail.com Signed-off-by: Eamon Sippy Signed-off-by: Andrew Morton Reviewed-by: Barry Song --- mm/cma.h | 23 +++++++++++++++++------ 1 file changed, 17 insertions(+), 6 deletions(-) diff --git a/mm/cma.h b/mm/cma.h index ab6d39898ea52e..68e574b0f95187 100644 --- a/mm/cma.h +++ b/mm/cma.h @@ -2,14 +2,24 @@ #ifndef __MM_CMA_H__ #define __MM_CMA_H__ +#include #include +#include +#include +#include + +#ifdef CONFIG_CMA_DEBUGFS #include +#endif + +#ifdef CONFIG_CMA_SYSFS #include struct cma_kobject { struct kobject kobj; struct cma *cma; }; +#endif /* * Multi-range support. This can be useful if the size of the allocation @@ -38,10 +48,10 @@ struct cma_memrange { #define CMA_MAX_RANGES 8 struct cma { - unsigned long count; - unsigned long available_count; + unsigned long count; + unsigned long available_count; unsigned int order_per_bit; /* Order of pages represented by one bit */ - spinlock_t lock; + spinlock_t lock; struct mutex alloc_mutex; #ifdef CONFIG_CMA_DEBUGFS struct hlist_head mem_head; @@ -87,10 +97,11 @@ void cma_sysfs_account_fail_pages(struct cma *cma, unsigned long nr_pages); void cma_sysfs_account_release_pages(struct cma *cma, unsigned long nr_pages); #else static inline void cma_sysfs_account_success_pages(struct cma *cma, - unsigned long nr_pages) {}; + unsigned long nr_pages) {} static inline void cma_sysfs_account_fail_pages(struct cma *cma, - unsigned long nr_pages) {}; + unsigned long nr_pages) {} static inline void cma_sysfs_account_release_pages(struct cma *cma, - unsigned long nr_pages) {}; + unsigned long nr_pages) {} #endif + #endif From 3dbe3161b68fa799923e5c6dbc2aef7da2641138 Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Thu, 13 Aug 2026 17:38:09 +0800 Subject: [PATCH 0626/1352] mm/oom_kill: remove unreachable __GFP_THISNODE check in constrained_alloc() The __GFP_THISNODE check in constrained_alloc() is dead code: global OOM is never triggered with __GFP_THISNODE (blocked in __alloc_pages_may_oom before out_of_memory() is called), and memcg OOM returns CONSTRAINT_MEMCG at the top of the function before reaching this point. Remove the check, its stale comment, and update the following comment that referenced __GFP_THISNODE. Link: https://lore.kernel.org/20260813093810.573302-1-ye.liu@linux.dev Signed-off-by: Ye Liu Signed-off-by: Andrew Morton Acked-by: Michal Hocko Acked-by: Shakeel Butt Cc: David Rientjes --- mm/oom_kill.c | 13 +++---------- 1 file changed, 3 insertions(+), 10 deletions(-) diff --git a/mm/oom_kill.c b/mm/oom_kill.c index 5f372f6e26fa32..fd3c476846a302 100644 --- a/mm/oom_kill.c +++ b/mm/oom_kill.c @@ -267,18 +267,11 @@ static enum oom_constraint constrained_alloc(struct oom_control *oc) if (!oc->zonelist) return CONSTRAINT_NONE; - /* - * Reach here only when __GFP_NOFAIL is used. So, we should avoid - * to kill current.We have to random task kill in this case. - * Hopefully, CONSTRAINT_THISNODE...but no way to handle it, now. - */ - if (oc->gfp_mask & __GFP_THISNODE) - return CONSTRAINT_NONE; /* - * This is not a __GFP_THISNODE allocation, so a truncated nodemask in - * the page allocator means a mempolicy is in effect. Cpuset policy - * is enforced in get_page_from_freelist(). + * A truncated nodemask in the page allocator means a mempolicy + * is in effect. Cpuset policy is enforced in + * get_page_from_freelist(). */ if (oc->nodemask && !nodes_subset(node_states[N_MEMORY], *oc->nodemask)) { From 242cac053b8de48ba4013e7872fc71efb08bdd93 Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Tue, 11 Aug 2026 11:36:08 +0800 Subject: [PATCH 0627/1352] mm/oom_kill, proc: replace magic number 1000 with OOM_SCORE_ADJ_MAX In oom_badness() and proc_oom_score(), the oom_score_adj normalization uses a hardcoded 1000, which is the value of OOM_SCORE_ADJ_MAX defined in include/uapi/linux/oom.h. Other code in the kernel (e.g. fs/proc/base.c oom_adj handling) already uses OOM_SCORE_ADJ_MAX for the same purpose. Replace the magic number with the macro for consistency and readability. No functional change. Link: https://lore.kernel.org/20260811033609.3992348-1-ye.liu@linux.dev Signed-off-by: Ye Liu Signed-off-by: Andrew Morton Acked-by: Michal Hocko Cc: David Rientjes Cc: Shakeel Butt Cc: Song Hu --- fs/proc/base.c | 3 ++- mm/oom_kill.c | 2 +- 2 files changed, 3 insertions(+), 2 deletions(-) diff --git a/fs/proc/base.c b/fs/proc/base.c index 6a39de424f62a1..58be3894246054 100644 --- a/fs/proc/base.c +++ b/fs/proc/base.c @@ -594,7 +594,8 @@ static int proc_oom_score(struct seq_file *m, struct pid_namespace *ns, * exporting for a long time so userspace might depend on it. */ if (badness != LONG_MIN) - points = (1000 + badness * 1000 / (long)totalpages) * 2 / 3; + points = (OOM_SCORE_ADJ_MAX + + badness * OOM_SCORE_ADJ_MAX / (long)totalpages) * 2 / 3; seq_printf(m, "%lu\n", points); diff --git a/mm/oom_kill.c b/mm/oom_kill.c index fd3c476846a302..5d48bd862c27b2 100644 --- a/mm/oom_kill.c +++ b/mm/oom_kill.c @@ -230,7 +230,7 @@ long oom_badness(struct task_struct *p, unsigned long totalpages) task_unlock(p); /* Normalize to oom_score_adj units */ - adj *= totalpages / 1000; + adj *= totalpages / OOM_SCORE_ADJ_MAX; points += adj; return points; From f8dc80d41e483ad82b2351bd229ef2b8b521aeec Mon Sep 17 00:00:00 2001 From: Pedro Falcato Date: Tue, 11 Aug 2026 18:21:55 +0100 Subject: [PATCH 0628/1352] mm: replace custom bad page map ratelimiting logic Patch series "mm: replace custom ratelimiting logic". The kernel has a perfectly cromulent and mostly-equivalent variant in lib/ratelimit.c that can be used. This patch (of 2): The current logic (allow up to $BURST prints per minute) can be entirely replaced by the generic version in lib/ratelimit.c, used around the kernel. Do so. The only functional difference should be that the new logs will read something like: KERN_WARNING "print_bad_page_map: %d callbacks suppressed\n", ... But that should be fine enough. Link: https://lore.kernel.org/20260811172156.356053-1-pfalcato@suse.de Link: https://lore.kernel.org/20260811172156.356053-2-pfalcato@suse.de Signed-off-by: Pedro Falcato Signed-off-by: Andrew Morton Acked-by: Johannes Weiner Acked-by: Zi Yan Reviewed-by: SJ Park Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: Vlastimil Babka (SUSE) Acked-by: Mike Rapoport (Microsoft) Cc: Brendan Jackman Cc: Liam R. Howlett Cc: Michal Hocko Cc: Suren Baghdasaryan --- mm/memory.c | 30 +++--------------------------- 1 file changed, 3 insertions(+), 27 deletions(-) diff --git a/mm/memory.c b/mm/memory.c index 347db2acd0f83b..09ac784f8b7b39 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -492,32 +492,8 @@ static inline void add_mm_rss_vec(struct mm_struct *mm, int *rss) add_mm_counter(mm, i, rss[i]); } -static bool is_bad_page_map_ratelimited(void) -{ - static unsigned long resume; - static unsigned long nr_shown; - static unsigned long nr_unshown; - - /* - * Allow a burst of 60 reports, then keep quiet for that minute; - * or allow a steady drip of one report per second. - */ - if (nr_shown == 60) { - if (time_before(jiffies, resume)) { - nr_unshown++; - return true; - } - if (nr_unshown) { - pr_alert("BUG: Bad page map: %lu messages suppressed\n", - nr_unshown); - nr_unshown = 0; - } - nr_shown = 0; - } - if (nr_shown++ == 0) - resume = jiffies + 60 * HZ; - return false; -} +/* Allow a burst of 60 bad page map reports per minute. */ +static DEFINE_RATELIMIT_STATE(bad_page_map_ratelimit, 60 * HZ, 60); static void ptval_bytes_to_hex_str(char *buf, size_t buf_size, const void *entry, size_t entry_size) { @@ -633,7 +609,7 @@ static void print_bad_page_map(struct vm_area_struct *vma, char entry_str[PTVAL_STR_MAX]; pgoff_t index, anon_index; - if (is_bad_page_map_ratelimited()) + if (!__ratelimit(&bad_page_map_ratelimit)) return; mapping = vma->vm_file ? vma->vm_file->f_mapping : NULL; From a6c5eda38f0f6be1aa2d5d6dd5839089a83cf3c0 Mon Sep 17 00:00:00 2001 From: Pedro Falcato Date: Tue, 11 Aug 2026 18:21:56 +0100 Subject: [PATCH 0629/1352] mm/page_alloc: replace custom bad page ratelimiting logic The current logic (allow up to $BURST prints per minute) can be entirely replaced by the generic version in lib/ratelimit.c, used around the kernel. Do so. The only functional difference should be that the new logs will read something like: KERN_WARNING "bad_page: %d callbacks suppressed\n", ... But that should be fine enough. Link: https://lore.kernel.org/20260811172156.356053-3-pfalcato@suse.de Signed-off-by: Pedro Falcato Signed-off-by: Andrew Morton Acked-by: Johannes Weiner Acked-by: Zi Yan Reviewed-by: SJ Park Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: Vlastimil Babka (SUSE) Acked-by: Mike Rapoport (Microsoft) Cc: Brendan Jackman Cc: Liam R. Howlett Cc: Michal Hocko Cc: Suren Baghdasaryan --- mm/page_alloc.c | 28 +++++----------------------- 1 file changed, 5 insertions(+), 23 deletions(-) diff --git a/mm/page_alloc.c b/mm/page_alloc.c index 242edcaa915b33..f2eea5e7637cd1 100644 --- a/mm/page_alloc.c +++ b/mm/page_alloc.c @@ -613,31 +613,13 @@ static inline bool __maybe_unused bad_range(struct zone *zone, struct page *page } #endif +/* Allow a burst of 60 reports per minute */ +static DEFINE_RATELIMIT_STATE(bad_page_ratelimit, 60 * HZ, 60); + static void bad_page(struct page *page, const char *reason) { - static unsigned long resume; - static unsigned long nr_shown; - static unsigned long nr_unshown; - - /* - * Allow a burst of 60 reports, then keep quiet for that minute; - * or allow a steady drip of one report per second. - */ - if (nr_shown == 60) { - if (time_before(jiffies, resume)) { - nr_unshown++; - goto out; - } - if (nr_unshown) { - pr_alert( - "BUG: Bad page state: %lu messages suppressed\n", - nr_unshown); - nr_unshown = 0; - } - nr_shown = 0; - } - if (nr_shown++ == 0) - resume = jiffies + 60 * HZ; + if (!__ratelimit(&bad_page_ratelimit)) + goto out; pr_alert("BUG: Bad page state in process %s pfn:%05lx\n", current->comm, page_to_pfn(page)); From eb8689a5e3abb6ee1b5a6e41a08cacb874d45313 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Mon, 31 Aug 2026 14:42:11 +0100 Subject: [PATCH 0630/1352] mm/vmpressure: remove window size TODO There has been a steady stream of patches that have been submitted by newcomers to core mm 'fixing' this TODO, with all but the original having very likely been generated by LLMs. It appears that TODOs to LLMs are like red rags to a bull. In addition, TODOs in code often bitrot and are distracting - those who understand the code know what could be improved in future. Therefore remove the TODO. The work required to actually fix this TODO requires somebody who both has understanding of the code and significant real-world data to back their changes. Such a person doesn't require a TODO prompt to implement this change, so nothing of value is being lost here. Link: https://lore.kernel.org/all/20260831130316.448-1-tahasezer.is@gmail.com/ Link: https://lore.kernel.org/linux-mm/20260724054305.516126-1-cui.tao@linux.dev/ Link: https://lore.kernel.org/linux-mm/20260715143646.15828-1-gaikwad.dcg@gmail.com/ Link: https://lore.kernel.org/all/20260227221555.29969-1-mcq@disroot.org/ Link: https://lore.kernel.org/20260831-remove-vmpressure-todo-v1-1-498515e59cdf@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Vlastimil Babka (SUSE) Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan --- mm/vmpressure.c | 3 --- 1 file changed, 3 deletions(-) diff --git a/mm/vmpressure.c b/mm/vmpressure.c index 9629240d77adc7..3de99fef392894 100644 --- a/mm/vmpressure.c +++ b/mm/vmpressure.c @@ -30,9 +30,6 @@ * * As the vmscan reclaimer logic works with chunks which are multiple of * SWAP_CLUSTER_MAX, it makes sense to use it for the window size as well. - * - * TODO: Make the window size depend on machine size, as we do for vmstat - * thresholds. Currently we set it to 512 pages (2MB for 4KB pages). */ const unsigned long vmpressure_win = SWAP_CLUSTER_MAX * 16; From 8c8c0c76ea31fce8599660c909db519c2a1c05fd Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Mon, 31 Aug 2026 15:18:23 +0100 Subject: [PATCH 0631/1352] tools/testing/selftests/mm: add missing .gitignore entries Commit 2bee308f3adb ("selftests/mm: use pattern matching in .gitignore") switched to a pattern-matching mechanism to reduce churn in .gitignore. It however accidentally excluded the page_frag test's-generated module intermediate C file with .mod.c extension, and also the local_config.h header generated if liburing is available locally. Explicitly fix both the issues, fixing the module-generated C file as a general pattern as these are always intermediate files that should be ignored. Since this is a trivial .gitignore change it doesn't seem necessary to treat it as a hotfix. Link: https://lore.kernel.org/20260831-fix-mm-selftests-gitignore-v1-1-c984bbd4c5e4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Gregory Price (Meta) Reviewed-by: Sarthak Sharma Acked-by: David Hildenbrand (Arm) Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/mm/.gitignore | 2 ++ 1 file changed, 2 insertions(+) diff --git a/tools/testing/selftests/mm/.gitignore b/tools/testing/selftests/mm/.gitignore index fcd892ed21e32c..a306d775478690 100644 --- a/tools/testing/selftests/mm/.gitignore +++ b/tools/testing/selftests/mm/.gitignore @@ -2,7 +2,9 @@ * !/**/ !*.c +*.mod.c !*.h +local_config.h !*.sh !.gitignore !Makefile From cf12a4ae1b14f7ec9ab224e64c8a6e71c50adfe6 Mon Sep 17 00:00:00 2001 From: Dave Hansen Date: Mon, 31 Aug 2026 13:30:52 -0700 Subject: [PATCH 0632/1352] mm: make per-VMA locks available universally MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Patch series "mm: Unconditional per-VMA locks and cleanups", v7. tl;dr: Make per-VMA locks available in all configs. Simplify some of the per-VMA lock users now that they can rely on them being always available. Binder and networking folks: Your code is the target of the cleanups. I'm cc'ing you now on v2 because there's emerging consensus on the mm side that the approach here is sane. I'm not quite sure how this pile would get merged, but ack/review tags would be appreciated if this looks good to you. Longer version: When working on some x86 shadow stack code, it was a real pain to avoid causing recursive locking problems with mmap_lock. One way to avoid those was to avoid mmap_lock and use per-VMA locks instead. They are great, but they are not available in all configs which makes them unusable in generic code, or if you want to completely avoid mmap_lock. Make per-VMA locks available in all configs. Right now, they are only available on select architectures when SMP and MMU are enabled. But all of the primitives that per-VMA locks are built on (RCU, maple trees, refcounts) work just fine without SMP or MMU. The only real downside is that making VMAs a wee bit bigger on !MMU and !SMP builds. The upside is much cleaner code, lower complexity and less #ifdeffery. Clean up a binder VMA locking site now that it can rely on per-VMA locks. Building on top of universally-available per-VMA locks, introduce a new helper. Since the new API does not require callers to have a fallback to mmap_lock, it's much easier to use. Callers can potentially replace this very common kernel idiom: mmap_read_lock(mm); vma = vma_lookup() // fiddle with vma mmap_read_unlock(mm); with: vma = vma_start_read_unlocked(mm, address); // fiddle with vma vma_end_read(vma); Which avoids mmap_lock entirely in the fast path. Use that new API for another binder site and one in the TCP code. This patch (of 7): The per-VMA locks have been around for several years. They've had some bugs worked out of them and have seen quite wide use. However, they are still only available when architectures explicitly enable them. Remove the conditional compilation around the per-VMA locks, making them available on all architectures and configs. The approach up to now seemed to be to add ARCH_SUPPORTS_PER_VMA_LOCK when the architecture started using per-VMA locks in the fault handler. But, contrary to the naming, the Kconfig option does not really indicate whether the architecture supports per-VMA locks or not. It is more of a marker for whether the architecture is likely to benefit from per-VMA locks. To me, the most important thing side-effect of universal availability is letting per-VMA locks be used in SMP=n configs. This lets us use per-VMA locking in all x86 code without fallbacks. Overall, this just generally makes the kernel simpler. Just look at the diffstat. It also opens the door to users that want to use the per-VMA locks in common code. Doing *that* brings additional simplifications. The downside of this is adding some fields to vm_area_struct and mm_struct. There are likely ways to optimize this, especially for things like SMP=n configs. For now, do the simplest thing: use the same implementation everywhere. == Considerations for NOMMU config == NOMMU systems do not write-lock VMAs, therefore read-locking a VMA would always succeed unless VMA is detached. Therefore for NOMMU config we make vma_mark_attached() a NOOP, which keeps VMAs always in detached state. This causes VMA read-locking to always fail and the caller falls back to locking mmap_lock. The following functions will have a different implementation in NOMMU config: - vma_mark_attached(), vma_mark_detached() are made NOOPs, keeping VMAs always in a detached state and preventing assertions and refcount underflows; - vma_start_write(), vma_start_write_killable() are made NOOPs to avoid warnings in __vma_start_write() due to VMAs being detached. These functions are not used in NOMMU code but __vma_start_write() is an exported function, therefore might be used by drivers. - vma_assert_attached() is made NOOP because it's reachable from NOMMU code via split_vma()->vma_iter_store_new()->vma_iter_store_overwrite(); - vma_assert_write_locked() is asserting vma->vm_mm is write-locked, as was done before this change; - vma_assert_locked() is asserting vma->vm_mm is locked, as was done before this change; The following functions work for both MMU and NOMMU configs: - vma_lock_init() performs the same initialization as for MMU config; - mm_lock_seqcount_init(), mm_lock_seqcount_begin(), mm_lock_seqcount_end() are called from mmap_write_{lock|unlock} and update mm_lock_seq correctly. - mmap_lock_speculate_try_begin(), mmap_lock_speculate_retry() work as is because mm_lock_seq is updated correctly; - vma_start_read(), vma_start_read_locked() will always fail because VMAs are always detached; - vma_end_read() will never be called because vma_start_read() never succeeds; - vma_is_attached() always return false because VMAs are always detached; - vma_assert_detached() will never trigger because VMAs are never attached; - vma_start_read_locked() always return false because VMAs are always detached; - lock_vma_under_rcu() will be safe as the attempted read lock will bail; Changes in the following files are not affecting NOMMU config: task_mmu.c - not compiled when CONFIG_MMU=n; pagewalk.c - not compiled when CONFIG_MMU=n; userfaultfd.c - not compiled when CONFIG_MMU=n (CONFIG_USERFAULTFD depends on CONFIG_MMU); The following changes in the BPF code are made to keep NOMMU config working like before: stack_map_lock_vma() - keeps mmap_lock in NOMMU config; bpf_iter_task_vma_new() - bails out in NOMMU config; Link: https://lore.kernel.org/20260831203056.838265-1-surenb@google.com Link: https://lore.kernel.org/20260831203056.838265-2-surenb@google.com Signed-off-by: Dave Hansen Signed-off-by: Suren Baghdasaryan Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: Vlastimil Babka (SUSE) Acked-by: David Hildenbrand (Arm) Cc: Liam R. Howlett Cc: Shakeel Butt Cc: Greg Kroah-Hartman Cc: Todd Kjos Cc: Christian Brauner Cc: Carlos Llamas Cc: Alice Ryhl Cc: David S. Miller Cc: David Ahern Cc: Arve Hjønnevåg --- arch/arm/Kconfig | 1 - arch/arm64/Kconfig | 1 - arch/loongarch/Kconfig | 1 - arch/powerpc/platforms/powernv/Kconfig | 1 - arch/powerpc/platforms/pseries/Kconfig | 1 - arch/riscv/Kconfig | 1 - arch/s390/Kconfig | 1 - arch/x86/Kconfig | 2 - fs/proc/internal.h | 2 - fs/proc/task_mmu.c | 93 -------------------------- include/linux/mm.h | 12 ---- include/linux/mm_types.h | 8 +-- include/linux/mmap_lock.h | 75 +++++++-------------- kernel/bpf/stackmap.c | 17 ++--- kernel/bpf/task_iter.c | 2 +- kernel/fork.c | 2 - mm/Kconfig | 12 ---- mm/Kconfig.debug | 1 - mm/debug.c | 4 -- mm/init-mm.c | 2 - mm/memory.c | 2 - mm/mmap_lock.c | 26 +------ mm/pagewalk.c | 2 - mm/rmap.c | 2 - mm/userfaultfd.c | 55 --------------- rust/kernel/mm.rs | 32 +++------ tools/testing/vma/include/dup.h | 5 +- tools/testing/vma/vma_internal.h | 1 - 28 files changed, 48 insertions(+), 316 deletions(-) diff --git a/arch/arm/Kconfig b/arch/arm/Kconfig index ffbc7f38613151..408aa58a2a5bbc 100644 --- a/arch/arm/Kconfig +++ b/arch/arm/Kconfig @@ -42,7 +42,6 @@ config ARM select ARCH_SUPPORTS_ATOMIC_RMW select ARCH_SUPPORTS_CFI select ARCH_SUPPORTS_HUGETLBFS if ARM_LPAE - select ARCH_SUPPORTS_PER_VMA_LOCK select ARCH_SUPPORTS_RT select ARCH_USE_BUILTIN_BSWAP select ARCH_USE_CMPXCHG_LOCKREF diff --git a/arch/arm64/Kconfig b/arch/arm64/Kconfig index b5a51b0ef9440a..2bbeded33da0da 100644 --- a/arch/arm64/Kconfig +++ b/arch/arm64/Kconfig @@ -81,7 +81,6 @@ config ARM64 select ARCH_HAS_PTE_PROTNONE select ARCH_SUPPORTS_NUMA_BALANCING select ARCH_SUPPORTS_PAGE_TABLE_CHECK - select ARCH_SUPPORTS_PER_VMA_LOCK select ARCH_SUPPORTS_HUGE_PFNMAP if TRANSPARENT_HUGEPAGE select ARCH_SUPPORTS_RT select ARCH_SUPPORTS_SCHED_SMT diff --git a/arch/loongarch/Kconfig b/arch/loongarch/Kconfig index 2067d1f2ad7acb..1d8fb1e456d6d8 100644 --- a/arch/loongarch/Kconfig +++ b/arch/loongarch/Kconfig @@ -69,7 +69,6 @@ config LOONGARCH select ARCH_SUPPORTS_MSEAL_SYSTEM_MAPPINGS select ARCH_HAS_PTE_PROTNONE if 64BIT select ARCH_SUPPORTS_NUMA_BALANCING if NUMA - select ARCH_SUPPORTS_PER_VMA_LOCK select ARCH_SUPPORTS_RT select ARCH_SUPPORTS_SCHED_SMT if SMP select ARCH_SUPPORTS_SCHED_MC if SMP diff --git a/arch/powerpc/platforms/powernv/Kconfig b/arch/powerpc/platforms/powernv/Kconfig index b5ad7c173ef0c1..dd8f6060fb7a2e 100644 --- a/arch/powerpc/platforms/powernv/Kconfig +++ b/arch/powerpc/platforms/powernv/Kconfig @@ -17,7 +17,6 @@ config PPC_POWERNV select PPC_DOORBELL select MMU_NOTIFIER select FORCE_SMP - select ARCH_SUPPORTS_PER_VMA_LOCK select PPC_RADIX_BROADCAST_TLBIE if PPC_RADIX_MMU default y diff --git a/arch/powerpc/platforms/pseries/Kconfig b/arch/powerpc/platforms/pseries/Kconfig index 74910ce3a541c3..7d125e288f6ef7 100644 --- a/arch/powerpc/platforms/pseries/Kconfig +++ b/arch/powerpc/platforms/pseries/Kconfig @@ -23,7 +23,6 @@ config PPC_PSERIES select HOTPLUG_CPU select FORCE_SMP select SWIOTLB - select ARCH_SUPPORTS_PER_VMA_LOCK select PPC_RADIX_BROADCAST_TLBIE if PPC_RADIX_MMU default y diff --git a/arch/riscv/Kconfig b/arch/riscv/Kconfig index d6c2dbf8455ced..5965666194b0f0 100644 --- a/arch/riscv/Kconfig +++ b/arch/riscv/Kconfig @@ -72,7 +72,6 @@ config RISCV select ARCH_SUPPORTS_LTO_CLANG_THIN select ARCH_SUPPORTS_MSEAL_SYSTEM_MAPPINGS if 64BIT && MMU select ARCH_SUPPORTS_PAGE_TABLE_CHECK if MMU - select ARCH_SUPPORTS_PER_VMA_LOCK if MMU select ARCH_HAS_PTE_PROTNONE if MMU select ARCH_SUPPORTS_RT select ARCH_SUPPORTS_SHADOW_CALL_STACK if HAVE_SHADOW_CALL_STACK diff --git a/arch/s390/Kconfig b/arch/s390/Kconfig index 4b51bc6e8948d7..b88b8504213692 100644 --- a/arch/s390/Kconfig +++ b/arch/s390/Kconfig @@ -156,7 +156,6 @@ config S390 select ARCH_HAS_PTE_PROTNONE select ARCH_SUPPORTS_NUMA_BALANCING select ARCH_SUPPORTS_PAGE_TABLE_CHECK - select ARCH_SUPPORTS_PER_VMA_LOCK select ARCH_USES_CFI_GENERIC_LLVM_PASS if CC_IS_CLANG select ARCH_USE_BUILTIN_BSWAP select ARCH_USE_CMPXCHG_LOCKREF diff --git a/arch/x86/Kconfig b/arch/x86/Kconfig index 7aa74bcc72f9db..a8c3b3d31a2761 100644 --- a/arch/x86/Kconfig +++ b/arch/x86/Kconfig @@ -27,7 +27,6 @@ config X86_64 select ARCH_HAS_GIGANTIC_PAGE select ARCH_SUPPORTS_MSEAL_SYSTEM_MAPPINGS select ARCH_SUPPORTS_INT128 if CC_HAS_INT128 - select ARCH_SUPPORTS_PER_VMA_LOCK select ARCH_SUPPORTS_HUGE_PFNMAP if TRANSPARENT_HUGEPAGE select HAVE_ARCH_SOFT_DIRTY select MODULES_USE_ELF_RELA @@ -1848,7 +1847,6 @@ config X86_USER_SHADOW_STACK bool "X86 userspace shadow stack" depends on AS_WRUSS depends on X86_64 - depends on PER_VMA_LOCK select ARCH_USES_HIGH_VMA_FLAGS select ARCH_HAS_USER_SHADOW_STACK select X86_CET diff --git a/fs/proc/internal.h b/fs/proc/internal.h index 04bd6c9e65a722..623bb43ede5509 100644 --- a/fs/proc/internal.h +++ b/fs/proc/internal.h @@ -385,10 +385,8 @@ struct mem_size_stats; struct proc_maps_locking_ctx { struct mm_struct *mm; -#ifdef CONFIG_PER_VMA_LOCK bool mmap_locked; struct vm_area_struct *locked_vma; -#endif }; struct proc_maps_private { diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c index 5c54aebe211824..e671b4fd8dedd9 100644 --- a/fs/proc/task_mmu.c +++ b/fs/proc/task_mmu.c @@ -130,8 +130,6 @@ static void release_task_mempolicy(struct proc_maps_private *priv) } #endif -#ifdef CONFIG_PER_VMA_LOCK - static inline int lock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx) { int ret = mmap_read_lock_killable(lock_ctx->mm); @@ -233,46 +231,6 @@ static inline void reacquire_rcu(struct proc_maps_private *priv) vma_iter_set(&priv->iter, priv->lock_ctx.locked_vma->vm_end); } -#else /* CONFIG_PER_VMA_LOCK */ - -static inline int lock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx) -{ - return mmap_read_lock_killable(lock_ctx->mm); -} - -static inline void unlock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx) -{ - mmap_read_unlock(lock_ctx->mm); -} - -static inline bool lock_vma_range(struct seq_file *m, - struct proc_maps_locking_ctx *lock_ctx) -{ - return lock_ctx_mm(lock_ctx) == 0; -} - -static inline void unlock_vma_range(struct proc_maps_locking_ctx *lock_ctx) -{ - unlock_ctx_mm(lock_ctx); -} - -static struct vm_area_struct *get_next_vma(struct proc_maps_private *priv, - loff_t last_pos) -{ - return vma_next(&priv->iter); -} - -static inline bool fallback_to_mmap_lock(struct proc_maps_private *priv, - loff_t pos) -{ - return false; -} - -static inline void drop_rcu(struct proc_maps_private *priv) {} -static inline void reacquire_rcu(struct proc_maps_private *priv) {} - -#endif /* CONFIG_PER_VMA_LOCK */ - static struct vm_area_struct *proc_get_vma(struct seq_file *m, loff_t *ppos) { struct proc_maps_private *priv = m->private; @@ -560,8 +518,6 @@ static int pid_maps_open(struct inode *inode, struct file *file) PROCMAP_QUERY_VMA_FLAGS \ ) -#ifdef CONFIG_PER_VMA_LOCK - static int query_vma_setup(struct proc_maps_locking_ctx *lock_ctx) { reset_lock_ctx(lock_ctx); @@ -612,26 +568,6 @@ static struct vm_area_struct *query_vma_find_by_addr(struct proc_maps_locking_ct return vma; } -#else /* CONFIG_PER_VMA_LOCK */ - -static int query_vma_setup(struct proc_maps_locking_ctx *lock_ctx) -{ - return mmap_read_lock_killable(lock_ctx->mm); -} - -static void query_vma_teardown(struct proc_maps_locking_ctx *lock_ctx) -{ - mmap_read_unlock(lock_ctx->mm); -} - -static struct vm_area_struct *query_vma_find_by_addr(struct proc_maps_locking_ctx *lock_ctx, - unsigned long addr) -{ - return find_vma(lock_ctx->mm, addr); -} - -#endif /* CONFIG_PER_VMA_LOCK */ - static struct vm_area_struct *query_matching_vma(struct proc_maps_locking_ctx *lock_ctx, unsigned long addr, u32 flags) { @@ -1314,8 +1250,6 @@ static const struct mm_walk_ops smaps_shmem_walk_ops = { .walk_lock = PGWALK_RDLOCK, }; -#ifdef CONFIG_PER_VMA_LOCK - static const struct mm_walk_ops smaps_walk_vma_lock_ops = { .pmd_entry = smaps_pte_range, .hugetlb_entry = smaps_hugetlb_range, @@ -1345,22 +1279,6 @@ get_smaps_shmem_walk_ops(struct proc_maps_private *priv) return &smaps_shmem_walk_vma_lock_ops; } -#else /* CONFIG_PER_VMA_LOCK */ - -static inline const struct mm_walk_ops * -get_smaps_walk_ops(struct proc_maps_private *priv) -{ - return &smaps_walk_ops; -} - -static inline const struct mm_walk_ops * -get_smaps_shmem_walk_ops(struct proc_maps_private *priv) -{ - return &smaps_shmem_walk_ops; -} - -#endif /* CONFIG_PER_VMA_LOCK */ - /* * Gather mem stats from @vma with the indicated beginning * address @start, and keep them in @mss. @@ -3497,7 +3415,6 @@ static const struct mm_walk_ops show_numa_ops = { .walk_lock = PGWALK_RDLOCK, }; -#ifdef CONFIG_PER_VMA_LOCK static const struct mm_walk_ops show_numa_vma_lock_ops = { .hugetlb_entry = gather_hugetlb_stats, .pmd_entry = gather_pte_stats, @@ -3512,16 +3429,6 @@ get_show_numa_ops(struct proc_maps_private *priv) return &show_numa_vma_lock_ops; } -#else /* CONFIG_PER_VMA_LOCK */ - -static inline const struct mm_walk_ops * -get_show_numa_ops(struct proc_maps_private *priv) -{ - return &show_numa_ops; -} - -#endif /* CONFIG_PER_VMA_LOCK */ - /* * Display pages allocated per node and memory policy via /proc. */ diff --git a/include/linux/mm.h b/include/linux/mm.h index b19711b6dbc69a..a9fbe26536f450 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -928,7 +928,6 @@ static inline void vma_numab_state_free(struct vm_area_struct *vma) {} * These must be here rather than mmap_lock.h as dependent on vm_fault type, * declared in this header. */ -#ifdef CONFIG_PER_VMA_LOCK static inline void release_fault_lock(struct vm_fault *vmf) { if (vmf->flags & FAULT_FLAG_VMA_LOCK) @@ -944,17 +943,6 @@ static inline void assert_fault_locked(const struct vm_fault *vmf) else mmap_assert_locked(vmf->vma->vm_mm); } -#else -static inline void release_fault_lock(struct vm_fault *vmf) -{ - mmap_read_unlock(vmf->vma->vm_mm); -} - -static inline void assert_fault_locked(const struct vm_fault *vmf) -{ - mmap_assert_locked(vmf->vma->vm_mm); -} -#endif /* CONFIG_PER_VMA_LOCK */ static inline bool mm_flags_test(int flag, const struct mm_struct *mm) { diff --git a/include/linux/mm_types.h b/include/linux/mm_types.h index f3e5a2fadbe5b7..2a3988178adfdc 100644 --- a/include/linux/mm_types.h +++ b/include/linux/mm_types.h @@ -950,7 +950,6 @@ struct vm_area_struct { vma_flags_t flags; }; -#ifdef CONFIG_PER_VMA_LOCK /* * Can only be written (using WRITE_ONCE()) while holding both: * - mmap_lock (in write mode) @@ -966,7 +965,7 @@ struct vm_area_struct { * slowpath. */ unsigned int vm_lock_seq; -#endif + /* * Low 32-bits of anonymous page offset. * See vma_start_anon_pgoff() comment for details. @@ -1003,7 +1002,6 @@ struct vm_area_struct { #ifdef CONFIG_NUMA_BALANCING struct vma_numab_state *numab_state; /* NUMA Balancing state */ #endif -#ifdef CONFIG_PER_VMA_LOCK /* * Used to keep track of firstly, whether the VMA is attached, secondly, * if attached, how many read locks are taken, and thirdly, if the @@ -1046,7 +1044,6 @@ struct vm_area_struct { #ifdef CONFIG_DEBUG_LOCK_ALLOC struct lockdep_map vmlock_dep_map; #endif -#endif #ifdef CONFIG_64BIT /* * High 32-bits of anonymous page offset. @@ -1254,7 +1251,6 @@ struct mm_struct { * init_mm.mmlist, and are protected * by mmlist_lock */ -#ifdef CONFIG_PER_VMA_LOCK struct rcuwait vma_writer_wait; /* * This field has lock-like semantics, meaning it is sometimes @@ -1274,7 +1270,7 @@ struct mm_struct { * mmap_lock. */ seqcount_t mm_lock_seq; -#endif + struct futex_mm_data futex; unsigned long hiwater_rss; /* High-watermark of RSS usage */ diff --git a/include/linux/mmap_lock.h b/include/linux/mmap_lock.h index b8a13b8d36a45b..6a0a8cf501bdea 100644 --- a/include/linux/mmap_lock.h +++ b/include/linux/mmap_lock.h @@ -76,8 +76,6 @@ static inline void mmap_assert_write_locked(const struct mm_struct *mm) rwsem_assert_held_write(&mm->mmap_lock); } -#ifdef CONFIG_PER_VMA_LOCK - #ifdef CONFIG_LOCKDEP #define __vma_lockdep_map(vma) (&vma->vmlock_dep_map) #else @@ -297,6 +295,9 @@ int __vma_start_write(struct vm_area_struct *vma, int state); */ static inline void vma_start_write(struct vm_area_struct *vma) { + if (!IS_ENABLED(CONFIG_MMU)) + return; + if (__is_vma_write_locked(vma)) return; @@ -319,6 +320,9 @@ static inline void vma_start_write(struct vm_area_struct *vma) static inline __must_check int vma_start_write_killable(struct vm_area_struct *vma) { + if (!IS_ENABLED(CONFIG_MMU)) + return 0; + if (__is_vma_write_locked(vma)) return 0; @@ -331,6 +335,11 @@ int vma_start_write_killable(struct vm_area_struct *vma) */ static inline void vma_assert_write_locked(struct vm_area_struct *vma) { + if (!IS_ENABLED(CONFIG_MMU)) { + mmap_assert_write_locked(vma->vm_mm); + return; + } + VM_WARN_ON_ONCE_VMA(!__is_vma_write_locked(vma), vma); } @@ -343,6 +352,11 @@ static inline void vma_assert_locked(struct vm_area_struct *vma) { unsigned int refcnt; + if (!IS_ENABLED(CONFIG_MMU)) { + mmap_assert_locked(vma->vm_mm); + return; + } + if (IS_ENABLED(CONFIG_LOCKDEP)) { if (!lock_is_held(__vma_lockdep_map(vma))) vma_assert_write_locked(vma); @@ -432,6 +446,9 @@ static inline bool vma_is_attached(struct vm_area_struct *vma) */ static inline void vma_assert_attached(struct vm_area_struct *vma) { + if (!IS_ENABLED(CONFIG_MMU)) + return; + WARN_ON_ONCE(!vma_is_attached(vma)); } @@ -442,6 +459,9 @@ static inline void vma_assert_detached(struct vm_area_struct *vma) static inline void vma_mark_attached(struct vm_area_struct *vma) { + if (!IS_ENABLED(CONFIG_MMU)) + return; + vma_assert_write_locked(vma); vma_assert_detached(vma); refcount_set_release(&vma->vm_refcnt, 1); @@ -451,6 +471,9 @@ void __vma_exclude_readers_for_detach(struct vm_area_struct *vma); static inline void vma_mark_detached(struct vm_area_struct *vma) { + if (!IS_ENABLED(CONFIG_MMU)) + return; + vma_assert_write_locked(vma); vma_assert_attached(vma); @@ -484,54 +507,6 @@ struct vm_area_struct *lock_next_vma(struct mm_struct *mm, struct vma_iterator *iter, unsigned long address); -#else /* CONFIG_PER_VMA_LOCK */ - -static inline void mm_lock_seqcount_init(struct mm_struct *mm) {} -static inline void mm_lock_seqcount_begin(struct mm_struct *mm) {} -static inline void mm_lock_seqcount_end(struct mm_struct *mm) {} - -static inline bool mmap_lock_speculate_try_begin(struct mm_struct *mm, unsigned int *seq) -{ - return false; -} - -static inline bool mmap_lock_speculate_retry(struct mm_struct *mm, unsigned int seq) -{ - return true; -} -static inline void vma_lock_init(struct vm_area_struct *vma, bool reset_refcnt) {} -static inline void vma_end_read(struct vm_area_struct *vma) {} -static inline void vma_start_write(struct vm_area_struct *vma) {} -static inline __must_check -int vma_start_write_killable(struct vm_area_struct *vma) { return 0; } -static inline void vma_assert_write_locked(struct vm_area_struct *vma) - { mmap_assert_write_locked(vma->vm_mm); } -static inline bool vma_is_attached(struct vm_area_struct *vma) - { return true; } -static inline void vma_assert_attached(struct vm_area_struct *vma) {} -static inline void vma_assert_detached(struct vm_area_struct *vma) {} -static inline void vma_mark_attached(struct vm_area_struct *vma) {} -static inline void vma_mark_detached(struct vm_area_struct *vma) {} - -static inline struct vm_area_struct *lock_vma_under_rcu(struct mm_struct *mm, - unsigned long address) -{ - return NULL; -} - -static inline void vma_assert_locked(struct vm_area_struct *vma) -{ - mmap_assert_locked(vma->vm_mm); -} - -static inline void vma_assert_stabilised(struct vm_area_struct *vma) -{ - /* If no VMA locks, then either mmap lock suffices to stabilise. */ - mmap_assert_locked(vma->vm_mm); -} - -#endif /* CONFIG_PER_VMA_LOCK */ - static inline void vma_assert_can_modify(struct vm_area_struct *vma) { if (vma_is_attached(vma)) diff --git a/kernel/bpf/stackmap.c b/kernel/bpf/stackmap.c index d09d4c3fe547c6..f7e8d766d12834 100644 --- a/kernel/bpf/stackmap.c +++ b/kernel/bpf/stackmap.c @@ -272,13 +272,10 @@ struct stack_map_vma_lock { /* * Acquire a stable read-side reference on the VMA covering @ip. * - * With CONFIG_PER_VMA_LOCK=y this returns a VMA with its per-VMA read - * lock held and mmap_lock dropped, so the caller may sleep. - * - * With CONFIG_PER_VMA_LOCK=n it returns a VMA with mmap_lock still - * held; the caller must snapshot any fields it needs and pin vm_file - * with get_file() before stack_map_unlock_vma() drops mmap_lock, as - * the VMA may be split, merged, or freed after that. + * On NOMMU configurations, returns with the mmap_lock held. If the MMU + * is enabled, the per-VMA lock will be held instead. The lock + * should be released with stack_map_unlock_vma() which will release the + * appropriate lock. Once the lock is released, the VMA may be freed. * * Returns NULL on failure, in which case no lock is held. */ @@ -288,7 +285,6 @@ stack_map_lock_vma(struct stack_map_vma_lock *lock, unsigned long ip) struct mm_struct *mm = lock->mm; struct vm_area_struct *vma; - /* noop under !CONFIG_PER_VMA_LOCK */ vma = lock_vma_under_rcu(mm, ip); if (vma) { lock->vma = vma; @@ -308,21 +304,20 @@ stack_map_lock_vma(struct stack_map_vma_lock *lock, unsigned long ip) return NULL; } -#ifdef CONFIG_PER_VMA_LOCK +#ifdef CONFIG_MMU if (!vma_start_read_locked(vma)) { mmap_read_unlock(mm); return NULL; } mmap_read_unlock(mm); #endif - lock->vma = vma; return vma; } static void stack_map_unlock_vma(struct stack_map_vma_lock *lock) { -#ifdef CONFIG_PER_VMA_LOCK +#ifdef CONFIG_MMU vma_end_read(lock->vma); #else mmap_read_unlock(lock->mm); diff --git a/kernel/bpf/task_iter.c b/kernel/bpf/task_iter.c index 13e1aabe6f8868..c65ba1dcd86672 100644 --- a/kernel/bpf/task_iter.c +++ b/kernel/bpf/task_iter.c @@ -869,7 +869,7 @@ __bpf_kfunc int bpf_iter_task_vma_new(struct bpf_iter_task_vma *it, BUILD_BUG_ON(sizeof(struct bpf_iter_task_vma_kern) != sizeof(struct bpf_iter_task_vma)); BUILD_BUG_ON(__alignof__(struct bpf_iter_task_vma_kern) != __alignof__(struct bpf_iter_task_vma)); - if (!IS_ENABLED(CONFIG_PER_VMA_LOCK)) { + if (!IS_ENABLED(CONFIG_MMU)) { kit->data = NULL; return -EOPNOTSUPP; } diff --git a/kernel/fork.c b/kernel/fork.c index 10f2d05d816a5f..22eaf5fb844d0a 100644 --- a/kernel/fork.c +++ b/kernel/fork.c @@ -1083,9 +1083,7 @@ static void mmap_init_lock(struct mm_struct *mm) { init_rwsem(&mm->mmap_lock); mm_lock_seqcount_init(mm); -#ifdef CONFIG_PER_VMA_LOCK rcuwait_init(&mm->vma_writer_wait); -#endif } static struct mm_struct *mm_init(struct mm_struct *mm, struct task_struct *p) diff --git a/mm/Kconfig b/mm/Kconfig index 2c385f8b29445e..c1ddf59c0d71a8 100644 --- a/mm/Kconfig +++ b/mm/Kconfig @@ -1425,18 +1425,6 @@ config LRU_GEN_WALKS_MMU depends on LRU_GEN && ARCH_HAS_HW_PTE_YOUNG # } -config ARCH_SUPPORTS_PER_VMA_LOCK - def_bool n - -config PER_VMA_LOCK - def_bool y - depends on ARCH_SUPPORTS_PER_VMA_LOCK && MMU && SMP - help - Allow per-vma locking during page fault handling. - - This feature allows locking each virtual memory area separately when - handling page faults instead of taking mmap_lock. - config LOCK_MM_AND_FIND_VMA bool depends on !STACK_GROWSUP diff --git a/mm/Kconfig.debug b/mm/Kconfig.debug index 15dca19dd07da9..9eaa25d1cf2340 100644 --- a/mm/Kconfig.debug +++ b/mm/Kconfig.debug @@ -310,7 +310,6 @@ config DEBUG_KMEMLEAK_VERBOSE config PER_VMA_LOCK_STATS bool "Statistics for per-vma locks" - depends on PER_VMA_LOCK help Say Y here to enable success, retry and failure counters of page faults handled under protection of per-vma locks. When enabled, the diff --git a/mm/debug.c b/mm/debug.c index 9a0297b3988d89..655e6bcc0e8d91 100644 --- a/mm/debug.c +++ b/mm/debug.c @@ -157,17 +157,13 @@ void dump_vma(const struct vm_area_struct *vma) pr_emerg("vma %px start %px end %px mm %px\n" "prot %lx anon_vma %px vm_ops %px\n" "pgoff %lx file %px private_data %px\n" -#ifdef CONFIG_PER_VMA_LOCK "refcnt %x\n" -#endif "flags: %#lx(%pGv)\n", vma, (void *)vma->vm_start, (void *)vma->vm_end, vma->vm_mm, (unsigned long)pgprot_val(vma->vm_page_prot), vma->anon_vma, vma->vm_ops, vma_start_pgoff(vma), vma->vm_file, vma->vm_private_data, -#ifdef CONFIG_PER_VMA_LOCK refcount_read(&vma->vm_refcnt), -#endif vma->vm_flags, &vma->vm_flags); } EXPORT_SYMBOL(dump_vma); diff --git a/mm/init-mm.c b/mm/init-mm.c index 3e792aad762616..a1bb2c2d0284a1 100644 --- a/mm/init-mm.c +++ b/mm/init-mm.c @@ -39,10 +39,8 @@ struct mm_struct init_mm = { .page_table_lock = __SPIN_LOCK_UNLOCKED(init_mm.page_table_lock), .arg_lock = __SPIN_LOCK_UNLOCKED(init_mm.arg_lock), .mmlist = LIST_HEAD_INIT(init_mm.mmlist), -#ifdef CONFIG_PER_VMA_LOCK .vma_writer_wait = __RCUWAIT_INITIALIZER(init_mm.vma_writer_wait), .mm_lock_seq = SEQCNT_ZERO(init_mm.mm_lock_seq), -#endif #ifdef CONFIG_SCHED_MM_CID .mm_cid.lock = __RAW_SPIN_LOCK_UNLOCKED(init_mm.mm_cid.lock), #endif diff --git a/mm/memory.c b/mm/memory.c index 09ac784f8b7b39..bc14cae3c49d72 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -6799,7 +6799,6 @@ static vm_fault_t sanitize_fault_flags(struct vm_area_struct *vma, !vma_is_cow_mapping(vma))) return VM_FAULT_SIGSEGV; } -#ifdef CONFIG_PER_VMA_LOCK /* * Per-VMA locks can't be used with FAULT_FLAG_RETRY_NOWAIT because of * the assumption that lock is dropped on VM_FAULT_RETRY. @@ -6808,7 +6807,6 @@ static vm_fault_t sanitize_fault_flags(struct vm_area_struct *vma, (FAULT_FLAG_VMA_LOCK | FAULT_FLAG_RETRY_NOWAIT)) == (FAULT_FLAG_VMA_LOCK | FAULT_FLAG_RETRY_NOWAIT))) return VM_FAULT_SIGSEGV; -#endif return 0; } diff --git a/mm/mmap_lock.c b/mm/mmap_lock.c index 898c2ef1e95803..272f9ac762b91e 100644 --- a/mm/mmap_lock.c +++ b/mm/mmap_lock.c @@ -43,9 +43,6 @@ void __mmap_lock_do_trace_released(struct mm_struct *mm, bool write) EXPORT_SYMBOL(__mmap_lock_do_trace_released); #endif /* CONFIG_TRACING */ -#ifdef CONFIG_MMU -#ifdef CONFIG_PER_VMA_LOCK - /* State shared across __vma_[start, end]_exclude_readers. */ struct vma_exclude_readers_state { /* Input parameters. */ @@ -299,6 +296,8 @@ struct vm_area_struct *lock_vma_under_rcu(struct mm_struct *mm, MA_STATE(mas, &mm->mm_mt, address, address); struct vm_area_struct *vma; + if (!IS_ENABLED(CONFIG_MMU)) + return NULL; retry: rcu_read_lock(); vma = mas_walk(&mas); @@ -431,7 +430,6 @@ struct vm_area_struct *lock_next_vma(struct mm_struct *mm, return vma; } -#endif /* CONFIG_PER_VMA_LOCK */ #ifdef CONFIG_LOCK_MM_AND_FIND_VMA #include @@ -548,23 +546,3 @@ struct vm_area_struct *lock_mm_and_find_vma(struct mm_struct *mm, return NULL; } #endif /* CONFIG_LOCK_MM_AND_FIND_VMA */ - -#else /* CONFIG_MMU */ - -/* - * At least xtensa ends up having protection faults even with no - * MMU.. No stack expansion, at least. - */ -struct vm_area_struct *lock_mm_and_find_vma(struct mm_struct *mm, - unsigned long addr, struct pt_regs *regs) -{ - struct vm_area_struct *vma; - - mmap_read_lock(mm); - vma = vma_lookup(mm, addr); - if (!vma) - mmap_read_unlock(mm); - return vma; -} - -#endif /* CONFIG_MMU */ diff --git a/mm/pagewalk.c b/mm/pagewalk.c index cc07fcf50e87b3..7411702a37f58d 100644 --- a/mm/pagewalk.c +++ b/mm/pagewalk.c @@ -444,7 +444,6 @@ static inline void process_mm_walk_lock(struct mm_struct *mm, static inline void process_vma_walk_lock(struct vm_area_struct *vma, enum page_walk_lock walk_lock) { -#ifdef CONFIG_PER_VMA_LOCK switch (walk_lock) { case PGWALK_WRLOCK: vma_start_write(vma); @@ -459,7 +458,6 @@ static inline void process_vma_walk_lock(struct vm_area_struct *vma, /* PGWALK_RDLOCK is handled by process_mm_walk_lock */ break; } -#endif } /* diff --git a/mm/rmap.c b/mm/rmap.c index f3b21aaa34ee98..3c67ad0e95620f 100644 --- a/mm/rmap.c +++ b/mm/rmap.c @@ -264,11 +264,9 @@ static void check_anon_vma_clone(struct vm_area_struct *dst, /* For the anon_vma to be compatible, it can only be singular. */ VM_WARN_ON_ONCE(operation == VMA_OP_MERGE_UNFAULTED && !list_is_singular(&src->anon_vma_chain)); -#ifdef CONFIG_PER_VMA_LOCK /* Only merging an unfaulted VMA leaves the destination attached. */ VM_WARN_ON_ONCE(operation != VMA_OP_MERGE_UNFAULTED && vma_is_attached(dst)); -#endif } static void maybe_reuse_anon_vma(struct vm_area_struct *dst, diff --git a/mm/userfaultfd.c b/mm/userfaultfd.c index b909ec8ef20bce..cf9c6ad3b3ad2c 100644 --- a/mm/userfaultfd.c +++ b/mm/userfaultfd.c @@ -122,7 +122,6 @@ struct vm_area_struct *find_vma_and_prepare_anon(struct mm_struct *mm, return vma; } -#ifdef CONFIG_PER_VMA_LOCK /* * uffd_lock_vma() - Lookup and lock vma corresponding to @address. * @mm: mm to search vma in. @@ -182,34 +181,6 @@ static void uffd_mfill_unlock(struct vm_area_struct *vma) vma_end_read(vma); } -#else - -static struct vm_area_struct *uffd_mfill_lock(struct mm_struct *dst_mm, - unsigned long dst_start, - unsigned long len) -{ - struct vm_area_struct *dst_vma; - - mmap_read_lock(dst_mm); - dst_vma = find_vma_and_prepare_anon(dst_mm, dst_start); - if (IS_ERR(dst_vma)) - goto out_unlock; - - if (validate_dst_vma(dst_vma, dst_start + len)) - return dst_vma; - - dst_vma = ERR_PTR(-ENOENT); -out_unlock: - mmap_read_unlock(dst_mm); - return dst_vma; -} - -static void uffd_mfill_unlock(struct vm_area_struct *vma) -{ - mmap_read_unlock(vma->vm_mm); -} -#endif - static void mfill_put_vma(struct mfill_state *state) { if (!state->vma) @@ -1851,7 +1822,6 @@ int find_vmas_mm_locked(struct mm_struct *mm, return 0; } -#ifdef CONFIG_PER_VMA_LOCK static int uffd_move_lock(struct mm_struct *mm, unsigned long dst_start, unsigned long src_start, @@ -1926,31 +1896,6 @@ static void uffd_move_unlock(struct vm_area_struct *dst_vma, vma_end_read(dst_vma); } -#else - -static int uffd_move_lock(struct mm_struct *mm, - unsigned long dst_start, - unsigned long src_start, - struct vm_area_struct **dst_vmap, - struct vm_area_struct **src_vmap) -{ - int err; - - mmap_read_lock(mm); - err = find_vmas_mm_locked(mm, dst_start, src_start, dst_vmap, src_vmap); - if (err) - mmap_read_unlock(mm); - return err; -} - -static void uffd_move_unlock(struct vm_area_struct *dst_vma, - struct vm_area_struct *src_vma) -{ - mmap_assert_locked(src_vma->vm_mm); - mmap_read_unlock(dst_vma->vm_mm); -} -#endif - /** * move_pages - move arbitrary anonymous pages of an existing vma * @ctx: pointer to the userfaultfd context diff --git a/rust/kernel/mm.rs b/rust/kernel/mm.rs index 4764d7b68f2a7f..f4fa54616085f2 100644 --- a/rust/kernel/mm.rs +++ b/rust/kernel/mm.rs @@ -170,30 +170,20 @@ impl MmWithUser { /// /// This is an optimistic trylock operation, so it may fail if there is contention. In that /// case, you should fall back to taking the mmap read lock. - /// - /// When per-vma locks are disabled, this always returns `None`. #[inline] pub fn lock_vma_under_rcu(&self, vma_addr: usize) -> Option> { - #[cfg(CONFIG_PER_VMA_LOCK)] - { - // SAFETY: Calling `bindings::lock_vma_under_rcu` is always okay given an mm where - // `mm_users` is non-zero. - let vma = unsafe { bindings::lock_vma_under_rcu(self.as_raw(), vma_addr) }; - if !vma.is_null() { - return Some(VmaReadGuard { - // SAFETY: If `lock_vma_under_rcu` returns a non-null ptr, then it points at a - // valid vma. The vma is stable for as long as the vma read lock is held. - vma: unsafe { VmaRef::from_raw(vma) }, - _nts: NotThreadSafe, - }); - } + // SAFETY: Calling `bindings::lock_vma_under_rcu` is always okay given an mm where + // `mm_users` is non-zero. + let vma = unsafe { bindings::lock_vma_under_rcu(self.as_raw(), vma_addr) }; + if vma.is_null() { + return None; } - - // Silence warnings about unused variables. - #[cfg(not(CONFIG_PER_VMA_LOCK))] - let _ = vma_addr; - - None + Some(VmaReadGuard { + // SAFETY: If `lock_vma_under_rcu` returns a non-null ptr, then it points at a + // valid vma. The vma is stable for as long as the vma read lock is held. + vma: unsafe { VmaRef::from_raw(vma) }, + _nts: NotThreadSafe, + }) } /// Lock the mmap read lock. diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index 4c58487b764e9d..57046d8ac81d80 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -560,7 +560,6 @@ struct vm_area_struct { vma_flags_t flags; }; -#ifdef CONFIG_PER_VMA_LOCK /* * Can only be written (using WRITE_ONCE()) while holding both: * - mmap_lock (in write mode) @@ -576,7 +575,7 @@ struct vm_area_struct { * slowpath. */ unsigned int vm_lock_seq; -#endif + unsigned int __vm_anon_pgoff_lo; /* @@ -610,10 +609,8 @@ struct vm_area_struct { #ifdef CONFIG_NUMA_BALANCING struct vma_numab_state *numab_state; /* NUMA Balancing state */ #endif -#ifdef CONFIG_PER_VMA_LOCK /* Unstable RCU readers are allowed to read this. */ refcount_t vm_refcnt; -#endif #ifdef CONFIG_64BIT unsigned int __vm_anon_pgoff_hi; #endif diff --git a/tools/testing/vma/vma_internal.h b/tools/testing/vma/vma_internal.h index 8a48b231aa7abf..54d5c3360aa26b 100644 --- a/tools/testing/vma/vma_internal.h +++ b/tools/testing/vma/vma_internal.h @@ -15,7 +15,6 @@ #include #define CONFIG_MMU 1 -#define CONFIG_PER_VMA_LOCK 1 #ifdef __CONCAT #undef __CONCAT From 465104f8aea2f3fdde564ebda6756770b5520aec Mon Sep 17 00:00:00 2001 From: Dave Hansen Date: Mon, 31 Aug 2026 13:30:53 -0700 Subject: [PATCH 0633/1352] binder: make shrinker rely solely on per-VMA lock MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit tl;dr: lock_vma_under_rcu() is already a trylock. No need to do both it and mmap_read_trylock(). Long Version: == Background == Historically, binder used an mmap_read_trylock() in its shrinker code. This ensures that reclaim is not blocked on an mmap_lock. Commit 95bc2d4a9020 ("binder: use per-vma lock in page reclaiming") added support for the per-VMA lock, but left mmap_read_trylock() as a fallback. This was presumably because the per-VMA locking can fail for several reasons and most (all?) lock_vma_under_rcu() callers have a fallback to mmap_read_trylock(). == Problem == The fallback is not worth the complexity here. lock_vma_under_rcu() is essentially already a non-blocking trylock. The main reason it fails is also the reason mmap_read_trylock() fails: something is holding mmap_write_lock(). The only remedy for a collision with mmap_write_lock() is to wait, which this code can not do. So the "fallback" after lock_vma_under_rcu() failure is not really a fallback: it is really likely to just be retrying in vain. That retry in an of itself isn't horrible. But it adds complexity. == Solution == Now that per-VMA locks are universally available, lock_vma_under_rcu() will not persistently fail. Rely on it alone and simplify the code. The removal of the fallback does not affect NOMMU case because binder driver depends on CONFIG_MMU. While at it we also make the handling of the cases where the original binder VMA is gone consistent. There are two cases to consider when Binder VMA is gone: 1. there is no VMA at that location anymore. 2. there is now another unrelated VMA at that location. Before this change we handle case 1 by having the shrinker proceed to free the page, and just skip the zap_vma_range() call. And we handle case 2 by having the shrinker return LRU_SKIP. While either behavior is acceptable, we need to handle them in a consistent way. Handle both cases by freeing the page without touching the VMA (skipping the zap_vma_range()). Full disclosure: I originally tried to do this with lock_vma_under_rcu_wait(), but it did not fit well with the mmap_lock trylock semantics. Claude caught this in a review and suggested the approach in this path. It seemed sane to me. So, Suggesed-by: Claude, I guess. Link: https://lore.kernel.org/20260831203056.838265-3-surenb@google.com Signed-off-by: Dave Hansen Signed-off-by: Suren Baghdasaryan Signed-off-by: Andrew Morton Reviewed-by: Alice Ryhl Acked-by: Lorenzo Stoakes (ARM) Acked-by: Carlos Llamas Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Shakeel Butt Cc: Greg Kroah-Hartman Cc: Todd Kjos Cc: Christian Brauner Cc: David S. Miller Cc: David Ahern Cc: Arve Hjønnevåg Cc: David Hildenbrand (Arm) --- drivers/android/binder_alloc.c | 46 ++++++++++++++++------------------ 1 file changed, 21 insertions(+), 25 deletions(-) diff --git a/drivers/android/binder_alloc.c b/drivers/android/binder_alloc.c index e4488ad86a6557..fcb744088e77c6 100644 --- a/drivers/android/binder_alloc.c +++ b/drivers/android/binder_alloc.c @@ -1142,7 +1142,6 @@ enum lru_status binder_alloc_free_page(struct list_head *item, struct vm_area_struct *vma; struct page *page_to_free; unsigned long page_addr; - int mm_locked = 0; size_t index; if (!mmget_not_zero(mm)) @@ -1151,27 +1150,25 @@ enum lru_status binder_alloc_free_page(struct list_head *item, index = mdata->page_index; page_addr = alloc->vm_start + index * PAGE_SIZE; - /* attempt per-vma lock first */ + /* + * Attempt per-vma lock. This is essentially a + * "trylock". It can fail even if the VMA exists + * for 'page_addr'. + */ vma = lock_vma_under_rcu(mm, page_addr); if (!vma) { - /* fall back to mmap_lock */ - if (!mmap_read_trylock(mm)) - goto err_mmap_read_lock_failed; - mm_locked = 1; - vma = vma_lookup(mm, page_addr); + /* + * If the vma exists, we can't continue because we cannot + * remove the page from the vma. However, if the vma was + * unmapped, it's okay to continue. + */ + if (binder_alloc_is_mapped(alloc)) + goto err_vma_lock_failed; } if (!mutex_trylock(&alloc->mutex)) goto err_get_alloc_mutex_failed; - /* - * Since a binder_alloc can only be mapped once, we ensure - * the vma corresponds to this mapping by checking whether - * the binder_alloc is still mapped. - */ - if (vma && !binder_alloc_is_mapped(alloc)) - goto err_invalid_vma; - trace_binder_unmap_kernel_start(alloc, index); page_to_free = alloc->pages[index]; @@ -1182,7 +1179,12 @@ enum lru_status binder_alloc_free_page(struct list_head *item, list_lru_isolate(lru, item); spin_unlock(&lru->lock); - if (vma) { + /* + * Since a binder_alloc can only be mapped once, we ensure + * the vma corresponds to this mapping by checking whether + * the binder_alloc is still mapped. + */ + if (vma && binder_alloc_is_mapped(alloc)) { trace_binder_unmap_user_start(alloc, index); zap_vma_range(vma, page_addr, PAGE_SIZE); @@ -1191,23 +1193,17 @@ enum lru_status binder_alloc_free_page(struct list_head *item, } mutex_unlock(&alloc->mutex); - if (mm_locked) - mmap_read_unlock(mm); - else + if (vma) vma_end_read(vma); mmput_async(mm); binder_free_page(page_to_free); return LRU_REMOVED_RETRY; -err_invalid_vma: - mutex_unlock(&alloc->mutex); err_get_alloc_mutex_failed: - if (mm_locked) - mmap_read_unlock(mm); - else + if (vma) vma_end_read(vma); -err_mmap_read_lock_failed: +err_vma_lock_failed: mmput_async(mm); err_mmget: return LRU_SKIP; From db5614a21535ca3c47d156d21057bd1b1e19348b Mon Sep 17 00:00:00 2001 From: Dave Hansen Date: Mon, 31 Aug 2026 13:30:54 -0700 Subject: [PATCH 0634/1352] mm: add RCU-based VMA lookup helper that waits for writers MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit There are basically two parallel ways to look up a VMA: the traditional way, which is protected by mmap_read_lock, and the RCU-based per-VMA lock way which is based on RCU and refcounts. However, per-VMA locks will fail if the lock is help by a writer and therefore never waits. In a number of places we need to wait for the lock and it's done by falling back to mmap_read_lock, locking the VMA and releasing the mmap_lock once VMA is locked. Add vma_start_read_unlocked() - a variant of the RCU-based lookup that waits for writers. This is basically the same as the existing RCU-based lookup, but on a failure to lock it temporarily takes mmap_lock for read and waits for writers to finish before locking the VMA, dropping the mmap_read_lock and returning the locked VMA. This has some advantages: 1. Callers do not need to have a fallback path for when they collide with writers. 2. Its fast path does not require taking mmap_lock for read. Basically, when applied correctly, this approach results in faster *and* simpler code. While at it, fix the comments for vma_start_read_locked(), vma_start_read_locked_nested(), and uffd_lock_vma(). Link: https://lore.kernel.org/20260831203056.838265-4-surenb@google.com Signed-off-by: Dave Hansen Signed-off-by: Suren Baghdasaryan Signed-off-by: Andrew Morton Suggested-by: Lorenzo Stoakes (ARM) Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: Vlastimil Babka (SUSE) Cc: Liam R. Howlett Cc: Shakeel Butt Cc: Greg Kroah-Hartman Cc: Todd Kjos Cc: Christian Brauner Cc: Carlos Llamas Cc: Alice Ryhl Cc: David S. Miller Cc: David Ahern Cc: Arve Hjønnevåg Cc: David Hildenbrand (Arm) --- include/linux/mmap_lock.h | 19 +++++++++++++++---- mm/mmap_lock.c | 35 +++++++++++++++++++++++++++++++++++ mm/userfaultfd.c | 6 ++++-- 3 files changed, 54 insertions(+), 6 deletions(-) diff --git a/include/linux/mmap_lock.h b/include/linux/mmap_lock.h index 6a0a8cf501bdea..28e3696ce9ff37 100644 --- a/include/linux/mmap_lock.h +++ b/include/linux/mmap_lock.h @@ -228,10 +228,14 @@ static inline void vma_refcount_put(struct vm_area_struct *vma) } /* - * Use only while holding mmap read lock which guarantees that locking will not - * fail (nobody can concurrently write-lock the vma). vma_start_read() should + * Use only while holding mmap read lock which guarantees that vma lock is not + * contended (nobody can concurrently write-lock the vma). vma_start_read() should * not be used in such cases because it might fail due to mm_lock_seq overflow. * This functionality is used to obtain vma read lock and drop the mmap read lock. + * + * VMA can't be detached while we are holding mmap lock, therefore in practice this + * function can fail only when there are so many readers that vm_refcnt overflows. + * The failure case is very unlikely and is already annotated as such internally. */ static inline bool vma_start_read_locked_nested(struct vm_area_struct *vma, int subclass) { @@ -247,16 +251,23 @@ static inline bool vma_start_read_locked_nested(struct vm_area_struct *vma, int } /* - * Use only while holding mmap read lock which guarantees that locking will not - * fail (nobody can concurrently write-lock the vma). vma_start_read() should + * Use only while holding mmap read lock which guarantees that vma lock is not + * contended (nobody can concurrently write-lock the vma). vma_start_read() should * not be used in such cases because it might fail due to mm_lock_seq overflow. * This functionality is used to obtain vma read lock and drop the mmap read lock. + * + * VMA can't be detached while we are holding mmap lock, therefore in practice this + * function can fail only when there are so many readers that vm_refcnt overflows. + * The failure case is very unlikely and is already annotated as such internally. */ static inline bool vma_start_read_locked(struct vm_area_struct *vma) { return vma_start_read_locked_nested(vma, 0); } +struct vm_area_struct *vma_start_read_unlocked(struct mm_struct *mm, + unsigned long address); + static inline void vma_end_read(struct vm_area_struct *vma) { vma_refcount_put(vma); diff --git a/mm/mmap_lock.c b/mm/mmap_lock.c index 272f9ac762b91e..2f94ee0fdee2a2 100644 --- a/mm/mmap_lock.c +++ b/mm/mmap_lock.c @@ -340,6 +340,41 @@ struct vm_area_struct *lock_vma_under_rcu(struct mm_struct *mm, return NULL; } +/** + * vma_start_read_unlocked() - Find the VMA covering 'address' and read-lock it. + * @mm: the mm_struct of the address space to search + * @address: address that the vma should contain + * + * The fast path does not take mmap_lock. Waits for writers to finish if the + * VMA is being modified by taking mmap_lock. + * Use when mmap_lock is not held, otherwise use vma_start_read_locked(). + * Nothing prevents VMAs being unmapped/mapped before or after the VMA is + * looked up, if a stronger guarantee is required, take an mmap_lock. + * + * Return: If a VMA exists which spans @address, return that VMA, read-locked. + * If no VMA is mapped there or, very unlikely, a reference count overflow + * occurred, return NULL. + */ +struct vm_area_struct *vma_start_read_unlocked(struct mm_struct *mm, + unsigned long address) +{ + struct vm_area_struct *vma; + + /* Fast path: return stable VMA covering 'address': */ + vma = lock_vma_under_rcu(mm, address); + if (vma) + return vma; + + /* Slow path: preclude VMA writers by temporarily getting mmap read lock. */ + mmap_read_lock(mm); + vma = vma_lookup(mm, address); + if (vma && !vma_start_read_locked(vma)) + vma = NULL; + mmap_read_unlock(mm); + + return vma; +} + static struct vm_area_struct *lock_next_vma_under_mmap_lock(struct mm_struct *mm, struct vma_iterator *vmi, unsigned long from_addr) diff --git a/mm/userfaultfd.c b/mm/userfaultfd.c index cf9c6ad3b3ad2c..bf50bff3838aa4 100644 --- a/mm/userfaultfd.c +++ b/mm/userfaultfd.c @@ -129,8 +129,10 @@ struct vm_area_struct *find_vma_and_prepare_anon(struct mm_struct *mm, * * Should be called without holding mmap_lock. * - * Return: A locked vma containing @address, -ENOENT if no vma is found, or - * -ENOMEM if anon_vma couldn't be allocated. + * Return: A locked vma containing @address, -ENOENT if no vma is found, + * -ENOMEM if anon_vma couldn't be allocated, or -EAGAIN if vma refcount + * overflow happened due to high number of readers and the caller should + * retry later. */ static struct vm_area_struct *uffd_lock_vma(struct mm_struct *mm, unsigned long address) From 77da1f1b92e8ae147ff28a9ef3a5c2b635e73d9c Mon Sep 17 00:00:00 2001 From: Dave Hansen Date: Mon, 31 Aug 2026 13:30:55 -0700 Subject: [PATCH 0635/1352] binder: remove mmap_lock fallback MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Previously, the per-VMA locking could fail in the face of writers which necessitate a fallback to mmap_lock. The new vma_start_read_unlocked() will wait for writers instead of failing. Use the new helper. Wait for writers. Remove the fallback to mmap_lock. Link: https://lore.kernel.org/20260831203056.838265-5-surenb@google.com Signed-off-by: Dave Hansen Signed-off-by: Suren Baghdasaryan Signed-off-by: Andrew Morton Reviewed-by: Alice Ryhl Acked-by: Lorenzo Stoakes (ARM) Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Shakeel Butt Cc: Greg Kroah-Hartman Cc: Todd Kjos Cc: Christian Brauner Cc: Carlos Llamas Cc: David S. Miller Cc: David Ahern Cc: Arve Hjønnevåg Cc: David Hildenbrand (Arm) --- drivers/android/binder/page_range.rs | 19 +++---------------- drivers/android/binder_alloc.c | 17 +++++------------ rust/kernel/mm.rs | 28 ++++++++++++++++++++++++++++ 3 files changed, 36 insertions(+), 28 deletions(-) diff --git a/drivers/android/binder/page_range.rs b/drivers/android/binder/page_range.rs index 52ffbf3504e7f7..71febd3d5b0730 100644 --- a/drivers/android/binder/page_range.rs +++ b/drivers/android/binder/page_range.rs @@ -439,22 +439,9 @@ impl ShrinkablePageRange { // workqueue. let mm = MmWithUser::into_mmput_async(self.mm.mmget_not_zero().ok_or(ESRCH)?); { - let vma_read; - let mmap_read; - let vma = if let Some(ret) = mm.lock_vma_under_rcu(vma_addr) { - vma_read = ret; - check_vma(&vma_read, self) - } else { - mmap_read = mm.mmap_read_lock(); - mmap_read - .vma_lookup(vma_addr) - .and_then(|vma| check_vma(vma, self)) - }; - - match vma { - Some(vma) => vma.vm_insert_page(user_page_addr, &new_page)?, - None => return Err(ESRCH), - } + let vma_read_guard = mm.vma_start_read_unlocked(vma_addr).ok_or(ESRCH)?; + let vma = check_vma(&vma_read_guard, self).ok_or(ESRCH)?; + vma.vm_insert_page(user_page_addr, &new_page)?; } let inner = self.lock.lock(); diff --git a/drivers/android/binder_alloc.c b/drivers/android/binder_alloc.c index fcb744088e77c6..d6eae0aa708541 100644 --- a/drivers/android/binder_alloc.c +++ b/drivers/android/binder_alloc.c @@ -259,21 +259,14 @@ static int binder_page_insert(struct binder_alloc *alloc, struct vm_area_struct *vma; int ret = -ESRCH; - /* attempt per-vma lock first */ - vma = lock_vma_under_rcu(mm, addr); - if (vma) { - if (binder_alloc_is_mapped(alloc)) - ret = vm_insert_page(vma, addr, page); - vma_end_read(vma); + vma = vma_start_read_unlocked(mm, addr); + if (!vma) return ret; - } - /* fall back to mmap_lock */ - mmap_read_lock(mm); - vma = vma_lookup(mm, addr); - if (vma && binder_alloc_is_mapped(alloc)) + if (binder_alloc_is_mapped(alloc)) ret = vm_insert_page(vma, addr, page); - mmap_read_unlock(mm); + + vma_end_read(vma); return ret; } diff --git a/rust/kernel/mm.rs b/rust/kernel/mm.rs index f4fa54616085f2..58bc1793fdaf5e 100644 --- a/rust/kernel/mm.rs +++ b/rust/kernel/mm.rs @@ -186,6 +186,34 @@ impl MmWithUser { }) } + /// Find the VMA covering 'address' and read-lock it. + /// + /// The fast path does not take mmap_lock. Waits for writers to finish if the + /// VMA is being modified by taking mmap_lock. + /// Use when mmap_lock is not held, otherwise use vma_start_read_locked(). + /// Nothing prevents VMAs being unmapped/mapped before or after the VMA is + /// looked up, if a stronger guarantee is required, take an mmap_lock. + /// + /// Return: If a VMA exists which spans @address, return that VMA, read-locked. + /// If no VMA is mapped there or, very unlikely, a reference count overflow + /// occurred, return NULL. + #[inline] + pub fn vma_start_read_unlocked(&self, vma_addr: usize) -> Option> { + // SAFETY: We may invoke `vma_start_read_unlocked` because we know this `mm` has non-zero + // `mm_users`. + let vma = unsafe { bindings::vma_start_read_unlocked(self.as_raw(), vma_addr) }; + if vma.is_null() { + return None; + } + // INVARIANT: We just acquired the VMA read lock. + Some(VmaReadGuard { + // SAFETY: If `vma_start_read_unlocked` returns a non-null ptr, then it points at a + // valid vma. The vma is stable for as long as the vma read lock is held. + vma: unsafe { VmaRef::from_raw(vma) }, + _nts: NotThreadSafe, + }) + } + /// Lock the mmap read lock. #[inline] pub fn mmap_read_lock(&self) -> MmapReadGuard<'_> { From 1a76886510b41a4670fea3144139beb9a4a54402 Mon Sep 17 00:00:00 2001 From: Dave Hansen Date: Mon, 31 Aug 2026 13:30:56 -0700 Subject: [PATCH 0636/1352] tcp: remove mmap_lock fallback path MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Previously, the per-VMA locking could fail in the face of writers which necessitates a fallback to mmap_lock. The new vma_start_read_unlocked() will wait for writers instead of failing. Use the new helper. Wait for writers. Remove the fallback to mmap_lock. The fallback removal does not affect NOMMU case because TCP_ZEROCOPY is gated on CONFIG_MMU. This really is a nice cleanup. It removes the need to pass the lock state back and forth to find_tcp_vma(). Link: https://lore.kernel.org/20260831203056.838265-6-surenb@google.com Signed-off-by: Dave Hansen Signed-off-by: Suren Baghdasaryan Signed-off-by: Andrew Morton Acked-by: Lorenzo Stoakes Acked-by: Vlastimil Babka (SUSE) Tested-by: syzbot@syzkaller.appspotmail.com Cc: Liam R. Howlett Cc: Shakeel Butt Cc: Greg Kroah-Hartman Cc: Arve Hjønnevåg Cc: Todd Kjos Cc: Christian Brauner Cc: Carlos Llamas Cc: Alice Ryhl Cc: David S. Miller Cc: David Ahern Cc: David Hildenbrand (Arm) --- net/ipv4/tcp.c | 31 +++++++++---------------------- 1 file changed, 9 insertions(+), 22 deletions(-) diff --git a/net/ipv4/tcp.c b/net/ipv4/tcp.c index 562752352afe4d..5588310bc64891 100644 --- a/net/ipv4/tcp.c +++ b/net/ipv4/tcp.c @@ -2167,27 +2167,18 @@ static void tcp_zc_finalize_rx_tstamp(struct sock *sk, } static struct vm_area_struct *find_tcp_vma(struct mm_struct *mm, - unsigned long address, - bool *mmap_locked) + unsigned long address) { - struct vm_area_struct *vma = lock_vma_under_rcu(mm, address); + struct vm_area_struct *vma = vma_start_read_unlocked(mm, address); - if (vma) { - if (vma->vm_ops != &tcp_vm_ops) { - vma_end_read(vma); - return NULL; - } - *mmap_locked = false; - return vma; - } + if (!vma) + return NULL; - mmap_read_lock(mm); - vma = vma_lookup(mm, address); - if (!vma || vma->vm_ops != &tcp_vm_ops) { - mmap_read_unlock(mm); + if (vma->vm_ops != &tcp_vm_ops) { + vma_end_read(vma); return NULL; } - *mmap_locked = true; + return vma; } @@ -2208,7 +2199,6 @@ static int tcp_zerocopy_receive(struct sock *sk, u32 seq = tp->copied_seq; u32 total_bytes_to_map; int inq = tcp_inq(sk); - bool mmap_locked; int ret; zc->copybuf_len = 0; @@ -2233,7 +2223,7 @@ static int tcp_zerocopy_receive(struct sock *sk, return 0; } - vma = find_tcp_vma(current->mm, address, &mmap_locked); + vma = find_tcp_vma(current->mm, address); if (!vma) return -EINVAL; @@ -2315,10 +2305,7 @@ static int tcp_zerocopy_receive(struct sock *sk, zc, total_bytes_to_map); } out: - if (mmap_locked) - mmap_read_unlock(current->mm); - else - vma_end_read(vma); + vma_end_read(vma); /* Try to copy straggler data. */ if (!ret) copylen = tcp_zc_handle_leftover(zc, sk, skb, &seq, copybuf_len, tss); From 801f3de31f5dfedf7a103d863fd8870cf6b22cd5 Mon Sep 17 00:00:00 2001 From: Joe Damato Date: Mon, 31 Aug 2026 10:48:35 -0700 Subject: [PATCH 0637/1352] mm: memcontrol: raise MEMCG_MAX for charges that fail without reclaiming Charges that exceed memory.max and return through the nomem label can raise no event and simply return -ENOMEM. A non-blocking charge can hit the limit, get rejected, but is not visible in memory.events. This was noticed in a production setting where bpf_mem_alloc() attempted to refill its per-cpu freelists, which triggered a non-blocking charge while at the limit. Commit d6e103a757fa ("mm: memcontrol: do not miss MEMCG_MAX events for enforced allocations") added raised_max_event to cover charges that are force charged without ever reaching reclaim, but charges that are rejected outright were left out. Getting an allocation failure without the corresponding MEMCG_MAX event is unexpected and makes debugging and monitoring harder. Raise the event on the way out for rejected charges as well, by routing the -ENOMEM return through the same exit path that already covers forced charges. The existing behavior of raising a MEMCG_MAX event on every charge/reclaim/retry iteration is left unchanged. Tested with a module that performs accounted GFP_NOWAIT page allocations from a task in a cgroup at its memory.max, and measures the resulting memory.events:max delta. Without this patch the rejected charges raise no event at all; with it the delta matches the number of rejected charges exactly. A GFP_KERNEL|__GFP_NORETRY control, which reaches reclaim, raises the same two events per failed charge before and after, confirming the existing charge/reclaim/retry accounting is unchanged. Link: https://lore.kernel.org/20260831174836.3102406-1-joe@dama.to Fixes: d6e103a757fa ("mm: memcontrol: do not miss MEMCG_MAX events for enforced allocations") Signed-off-by: Joe Damato Signed-off-by: Andrew Morton Suggested-by: Shakeel Butt Acked-by: Shakeel Butt Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Cc: --- mm/memcontrol.c | 28 ++++++++++++++++------------ 1 file changed, 16 insertions(+), 12 deletions(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index d5ebe83eae3efc..aeaa09e01d70ea 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -2685,10 +2685,11 @@ static int try_charge_memcg(struct mem_cgroup *memcg, gfp_t gfp_mask, bool raised_max_event = false; unsigned long pflags; bool allow_spinning = gfpflags_allow_spinning(gfp_mask); + int ret = 0; retry: if (consume_stock(memcg, nr_pages)) - return 0; + return ret; if (!allow_spinning) /* Avoid the refill and flush of the older stock */ @@ -2799,16 +2800,11 @@ static int try_charge_memcg(struct mem_cgroup *memcg, gfp_t gfp_mask, * put the burden of reclaim on regular allocation requests * and let these go through as privileged allocations. */ - if (!(gfp_mask & (__GFP_NOFAIL | __GFP_HIGH))) - return -ENOMEM; + if (!(gfp_mask & (__GFP_NOFAIL | __GFP_HIGH))) { + ret = -ENOMEM; + goto out; + } force: - /* - * If the allocation has to be enforced, don't forget to raise - * a MEMCG_MAX event. - */ - if (!raised_max_event) - __memcg_memory_event(mem_over_limit, MEMCG_MAX, allow_spinning); - /* * The allocation either can't fail or will lead to more memory * being freed very soon. Allow memory usage go over the limit @@ -2818,7 +2814,15 @@ static int try_charge_memcg(struct mem_cgroup *memcg, gfp_t gfp_mask, if (do_memsw_account()) page_counter_charge(&memcg->memsw, nr_pages); - return 0; +out: + /* + * Don't forget to raise a MEMCG_MAX event for forced or rejected + * requests. + */ + if (!raised_max_event) + __memcg_memory_event(mem_over_limit, MEMCG_MAX, allow_spinning); + + return ret; done_restock: if (batch > nr_pages) @@ -2877,7 +2881,7 @@ static int try_charge_memcg(struct mem_cgroup *memcg, gfp_t gfp_mask, !(current->flags & PF_MEMALLOC) && gfpflags_allow_blocking(gfp_mask)) __mem_cgroup_handle_over_high(gfp_mask); - return 0; + return ret; } static inline int try_charge(struct mem_cgroup *memcg, gfp_t gfp_mask, From c3d448baac28dd7eca2384107b123e6c815e2b55 Mon Sep 17 00:00:00 2001 From: Jason Angelov Date: Mon, 31 Aug 2026 08:06:48 -0700 Subject: [PATCH 0638/1352] mm/damon/core-kunit: test probe_hits handling at region split and merge Patch series "mm/damon: add kunit tests for probe_hits handling and probe params validation", v2. DAMON recently introduced probes and probe weights. Add kunit tests for the propagation of probe_hits at region split and merge, and the rejection of invalid probe parameters by damon_valid_probe_params(). This patch (of 2): damon_split_region_at() copies probe_hits[] and last_probe_hits[] to the new split region. damon_merge_two_regions() sets probe_hits[] to the size-weighted average of the merged regions. Extend damon_test_split_at() and damon_test_merge_two() tests to cover those fields. Link: https://lore.kernel.org/20260831150650.84829-1-sj@kernel.org Link: https://lore.kernel.org/20260831150650.84829-2-sj@kernel.org Signed-off-by: Jason Angelov Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: David Gow Cc: Brendan Higgins --- mm/damon/tests/core-kunit.h | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index 4a536d41cdb2d0..2db94d49c9bae4 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -152,6 +152,8 @@ static void damon_test_split_at(struct kunit *test) } r->nr_accesses = 42; r->last_nr_accesses = 15; + r->probe_hits[0] = 7; + r->last_probe_hits[0] = 3; r->age = 10; damon_add_region(r, t); damon_split_region_at(t, r, 25); @@ -168,6 +170,8 @@ static void damon_test_split_at(struct kunit *test) KUNIT_EXPECT_EQ(test, r->nr_accesses, r_new->nr_accesses); KUNIT_EXPECT_EQ(test, r->last_nr_accesses, r_new->last_nr_accesses); + KUNIT_EXPECT_EQ(test, r->probe_hits[0], r_new->probe_hits[0]); + KUNIT_EXPECT_EQ(test, r->last_probe_hits[0], r_new->last_probe_hits[0]); KUNIT_EXPECT_EQ(test, r->age, r_new->age); out: @@ -189,6 +193,7 @@ static void damon_test_merge_two(struct kunit *test) kunit_skip(test, "region alloc fail"); } r->nr_accesses = 10; + r->probe_hits[0] = 6; r->age = 9; damon_add_region(r, t); r2 = damon_new_region(100, 300); @@ -197,6 +202,7 @@ static void damon_test_merge_two(struct kunit *test) kunit_skip(test, "second region alloc fail"); } r2->nr_accesses = 20; + r2->probe_hits[0] = 14; r2->age = 21; damon_add_region(r2, t); @@ -204,6 +210,7 @@ static void damon_test_merge_two(struct kunit *test) KUNIT_EXPECT_EQ(test, r->ar.start, 0ul); KUNIT_EXPECT_EQ(test, r->ar.end, 300ul); KUNIT_EXPECT_EQ(test, r->nr_accesses, 16u); + KUNIT_EXPECT_EQ(test, r->probe_hits[0], 11); KUNIT_EXPECT_EQ(test, r->age, 17u); i = 0; From 8bf6b3dff0312c0cff1fc939966cd30b606cf14a Mon Sep 17 00:00:00 2001 From: Jason Angelov Date: Mon, 31 Aug 2026 08:06:49 -0700 Subject: [PATCH 0639/1352] mm/damon/core-kunit: test damon_valid_probe_params() damon_valid_probe_params() makes damon_commit_ctx() reject probe configurations that could overflow a probe_hits counter, a single (weight * probe_hits) product, or the sum of those products. Add a kunit test covering each rejection at its boundary: - samples per aggregation interval: U8_MAX is allowed, one more could overflow a probe_hits counter - single weight: the largest whose product fits in unsigned int is allowed, one larger is rejected - multiple probes: each product fits, but their sum overflows - no weight set: the validation is skipped Link: https://lore.kernel.org/20260831150650.84829-3-sj@kernel.org Signed-off-by: Jason Angelov Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Brendan Higgins Cc: David Gow --- mm/damon/tests/core-kunit.h | 57 +++++++++++++++++++++++++++++++++++++ 1 file changed, 57 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index 2db94d49c9bae4..b1ca4c8e03f091 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -1338,6 +1338,62 @@ static void damon_test_commit_ctx(struct kunit *test) damon_destroy_ctx(dst); } +static void damon_test_valid_probe_params(struct kunit *test) +{ + struct damon_ctx *ctx; + struct damon_probe *probe, *probe2; + + ctx = damon_new_ctx(); + if (!ctx) + kunit_skip(test, "ctx alloc fail"); + probe = damon_new_probe(); + if (!probe) { + damon_destroy_ctx(ctx); + kunit_skip(test, "probe alloc fail"); + } + damon_add_probe(ctx, probe); + + /* Parameters are validated only if any probe weight is set. */ + ctx->attrs.sample_interval = 1; + ctx->attrs.aggr_interval = 1000000; + KUNIT_EXPECT_TRUE(test, damon_valid_probe_params(ctx)); + + /* Up to U8_MAX samples per aggregation interval are allowed. */ + probe->weight = 100; + ctx->attrs.aggr_interval = 255; + KUNIT_EXPECT_TRUE(test, damon_valid_probe_params(ctx)); + + /* More samples could overflow the probe_hits counters. */ + ctx->attrs.aggr_interval = 256; + KUNIT_EXPECT_FALSE(test, damon_valid_probe_params(ctx)); + + /* The largest weight whose weighted hit count fits in unsigned int. */ + ctx->attrs.aggr_interval = 255; + probe->weight = UINT_MAX / 255; + KUNIT_EXPECT_TRUE(test, damon_valid_probe_params(ctx)); + + /* Any larger weight could overflow its weighted hit count. */ + probe->weight = UINT_MAX / 255 + 1; + KUNIT_EXPECT_FALSE(test, damon_valid_probe_params(ctx)); + + /* With one sample per aggregation, even the largest weight fits. */ + ctx->attrs.aggr_interval = 1; + probe->weight = UINT_MAX; + KUNIT_EXPECT_TRUE(test, damon_valid_probe_params(ctx)); + + /* The sum of all probes' weighted hit counts could also overflow. */ + probe2 = damon_new_probe(); + if (!probe2) { + damon_destroy_ctx(ctx); + kunit_skip(test, "probe2 alloc fail"); + } + probe2->weight = 1; + damon_add_probe(ctx, probe2); + KUNIT_EXPECT_FALSE(test, damon_valid_probe_params(ctx)); + + damon_destroy_ctx(ctx); +} + static void damos_test_filter_out(struct kunit *test) { struct damon_target *t; @@ -1664,6 +1720,7 @@ static struct kunit_case damon_test_cases[] = { KUNIT_CASE(damos_test_commit_migrate_hot), KUNIT_CASE(damon_test_commit_target_regions), KUNIT_CASE(damon_test_commit_ctx), + KUNIT_CASE(damon_test_valid_probe_params), KUNIT_CASE(damos_test_filter_out), KUNIT_CASE(damon_test_feed_loop_next_input), KUNIT_CASE(damon_test_set_filters_default_reject), From bd8bac2bdc02b1c0cd7d034be1a5c3c38e63fb4a Mon Sep 17 00:00:00 2001 From: Liew Rui Yan Date: Mon, 31 Aug 2026 08:02:24 -0700 Subject: [PATCH 0640/1352] docs/mm/damon/design: accurate semantics of nr_snapshots Patch series "docs/mm/damon/design: add explanation of nr_snapshots", v3. Add an explanation of nr_snapshots to avoid misunderstandings. This patch (of 3): Change "tried to be applied" -> "completely tried to be applied" to maintain consistency between the documentation and the code. Link: https://lore.kernel.org/20260831150227.83416-1-sj@kernel.org Link: https://lore.kernel.org/20260831150227.83416-2-sj@kernel.org Signed-off-by: Liew Rui Yan Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/mm/damon/design.rst | 4 ++-- include/linux/damon.h | 3 ++- 2 files changed, 4 insertions(+), 3 deletions(-) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index aed6cb1cf48310..1739aeec6eb95f 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -846,8 +846,8 @@ scheme's execution. - ``nr_applied``: Total number of regions that the scheme is applied. - ``sz_applied``: Total size of regions that the scheme is applied. - ``qt_exceeds``: Total number of times the quota of the scheme has exceeded. -- ``nr_snapshots``: Total number of DAMON snapshots that the scheme is tried to - be applied. +- ``nr_snapshots``: Total number of DAMON snapshots that the scheme is + completely tried to be applied. - ``max_nr_snapshots``: Upper limit of ``nr_snapshots``. "A scheme is tried to be applied to a region" means DAMOS core logic determined diff --git a/include/linux/damon.h b/include/linux/damon.h index 0c8b7ddef9abb3..cbdf5f77978e72 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -357,7 +357,8 @@ struct damos_watermarks { * Total bytes that passed ops layer-handled DAMOS filters. * @qt_exceeds: Total number of times the quota of the scheme has exceeded. * @nr_snapshots: - * Total number of DAMON snapshots that the scheme has tried. + * Total number of DAMON snapshots that the scheme is completely + * tried to be applied. * * "Tried an action to a region" in this context means the DAMOS core logic * determined the region as eligible to apply the action. The access pattern From c4c7e40bc8c2387f928ad57821285bc0cb2b21a8 Mon Sep 17 00:00:00 2001 From: Liew Rui Yan Date: Mon, 31 Aug 2026 08:02:25 -0700 Subject: [PATCH 0641/1352] docs/mm/damon/design: difference between watermarks and nr_snapshots Explain the difference between nr_snapshots reaches max_nr_snapshots and watermarks. Link: https://lore.kernel.org/20260831150227.83416-3-sj@kernel.org Signed-off-by: Liew Rui Yan Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/mm/damon/design.rst | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index 1739aeec6eb95f..e7977f005ac06e 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -872,7 +872,8 @@ the action to the region will fail. Unlike normal stats, ``max_nr_snapshots`` is set by users. If it is set as non-zero and ``nr_snapshots`` be same to or greater than ``nr_snapshots``, the -scheme is deactivated. +scheme is deactivated. Note that, unlike watermarks, even if a scheme's +``nr_snapshots`` reaches ``max_nr_snapshots``, monitoring will not stop. To know how user-space can read the stats via :ref:`DAMON sysfs interface `, refer to :ref:s`stats ` part of the From 936007de2275d52a0ee974ff07bab0ca72e605fd Mon Sep 17 00:00:00 2001 From: Liew Rui Yan Date: Mon, 31 Aug 2026 08:02:26 -0700 Subject: [PATCH 0642/1352] docs/mm/damon/design: fix typo of max_nr_snapshots Fix a typo (nr_snapshots -> max_nr_snapshots) and corrects a grammar error. Link: https://lore.kernel.org/20260831150227.83416-4-sj@kernel.org Signed-off-by: Liew Rui Yan Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/mm/damon/design.rst | 4 ++-- include/linux/damon.h | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index e7977f005ac06e..d1dd9050ebf40b 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -871,8 +871,8 @@ action is ``pageout`` while all pages of the region are unreclaimable, applying the action to the region will fail. Unlike normal stats, ``max_nr_snapshots`` is set by users. If it is set as -non-zero and ``nr_snapshots`` be same to or greater than ``nr_snapshots``, the -scheme is deactivated. Note that, unlike watermarks, even if a scheme's +non-zero and ``nr_snapshots`` equals or is greater than ``max_nr_snapshots``, +the scheme is deactivated. Note that, unlike watermarks, even if a scheme's ``nr_snapshots`` reaches ``max_nr_snapshots``, monitoring will not stop. To know how user-space can read the stats via :ref:`DAMON sysfs interface diff --git a/include/linux/damon.h b/include/linux/damon.h index cbdf5f77978e72..4b0d2d2e4ea4eb 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -549,7 +549,7 @@ struct damos_migrate_dests { * * After applying the &action to each region, &stat is updated. * - * If &max_nr_snapshots is set as non-zero and &stat.nr_snapshots be same to or + * If &max_nr_snapshots is set as non-zero and &stat.nr_snapshots equals or is * greater than it, the scheme is deactivated. */ struct damos { From 8d895512229d4ff90afcce31fdb6618d5bf7811b Mon Sep 17 00:00:00 2001 From: Cheng-Han Wu Date: Mon, 31 Aug 2026 07:57:21 -0700 Subject: [PATCH 0643/1352] mm/damon/core: remove unused damon_targets_empty() Patch series "mm/damon/core: remove unused helper functions", v2. Both damon_targets_empty() and damon_nr_running_ctxs() have had no in-tree users since commit 5ec4333b1967 ("mm/damon: remove DAMON debugfs interface") removed their remaining callers. Remove the unused declarations and definitions. This patch (of 2): damon_targets_empty() has had no in-tree users since commit 5ec4333b1967 ("mm/damon: remove DAMON debugfs interface") removed its last caller. Remove the unused declaration and definition. Link: https://lore.kernel.org/20260831145724.82387-1-sj@kernel.org Link: https://lore.kernel.org/20260831145724.82387-2-sj@kernel.org Signed-off-by: Cheng-Han Wu Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park --- include/linux/damon.h | 1 - mm/damon/core.c | 5 ----- 2 files changed, 6 deletions(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index 4b0d2d2e4ea4eb..89a41dea1d23f4 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -1055,7 +1055,6 @@ int damos_commit_quota_goals(struct damos_quota *dst, struct damos_quota *src); struct damon_target *damon_new_target(void); void damon_add_target(struct damon_ctx *ctx, struct damon_target *t); -bool damon_targets_empty(struct damon_ctx *ctx); void damon_free_target(struct damon_target *t); void damon_destroy_target(struct damon_target *t, struct damon_ctx *ctx); unsigned int damon_nr_regions(struct damon_target *t); diff --git a/mm/damon/core.c b/mm/damon/core.c index ab3c4d75496449..dc772ceae26ffc 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -795,11 +795,6 @@ void damon_add_target(struct damon_ctx *ctx, struct damon_target *t) list_add_tail(&t->list, &ctx->adaptive_targets); } -bool damon_targets_empty(struct damon_ctx *ctx) -{ - return list_empty(&ctx->adaptive_targets); -} - static void damon_del_target(struct damon_target *t) { list_del(&t->list); From bd312e60521a1f9c2398ca751f83d96c6c4b85ba Mon Sep 17 00:00:00 2001 From: Cheng-Han Wu Date: Mon, 31 Aug 2026 07:57:22 -0700 Subject: [PATCH 0644/1352] mm/damon/core: remove unused damon_nr_running_ctxs() damon_nr_running_ctxs() has had no in-tree users since commit 5ec4333b1967 ("mm/damon: remove DAMON debugfs interface") removed all of its callers. Remove the unused declaration and definition. Link: https://lore.kernel.org/20260831145724.82387-3-sj@kernel.org Signed-off-by: Cheng-Han Wu Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park --- include/linux/damon.h | 1 - mm/damon/core.c | 14 -------------- 2 files changed, 15 deletions(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index 89a41dea1d23f4..6993dca0f355cc 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -1065,7 +1065,6 @@ int damon_set_attrs(struct damon_ctx *ctx, struct damon_attrs *attrs); void damon_set_schemes(struct damon_ctx *ctx, struct damos **schemes, ssize_t nr_schemes); int damon_commit_ctx(struct damon_ctx *old_ctx, struct damon_ctx *new_ctx); -int damon_nr_running_ctxs(void); bool damon_is_registered_ops(enum damon_ops_id id); int damon_register_ops(struct damon_operations *ops); int damon_select_ops(struct damon_ctx *ctx, enum damon_ops_id id); diff --git a/mm/damon/core.c b/mm/damon/core.c index dc772ceae26ffc..d832a527bcf623 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -1847,20 +1847,6 @@ int damon_commit_ctx(struct damon_ctx *dst, struct damon_ctx *src) return err; } -/** - * damon_nr_running_ctxs() - Return number of currently running contexts. - */ -int damon_nr_running_ctxs(void) -{ - int nr_ctxs; - - mutex_lock(&damon_lock); - nr_ctxs = nr_running_ctxs; - mutex_unlock(&damon_lock); - - return nr_ctxs; -} - /* Returns the size upper limit for each monitoring region */ static unsigned long damon_region_sz_limit(struct damon_ctx *ctx) { From 231c710f700a9841a235ce8781474a7f6b45f7b4 Mon Sep 17 00:00:00 2001 From: Asier Gutierrez Date: Mon, 31 Aug 2026 07:47:28 -0700 Subject: [PATCH 0645/1352] mm/damon: introduce DAMOS_QUOTA_HUGEPAGE auto tuning Patch series "mm/damon: Introduce a huge page collapsing mechanism using auto tuning", v4. Overview ======== This patchset introduces a new autotuning which allows to collapse hot regions into hugepages. Motivation ========== Since TLB is a bottleneck for many systems[1], a way to optimize TLB misses (or hits) is to use huge pages. Unfortunately, using "always" in THP leads to memory fragmentation and memory waste. For this reason, most application guides and system administrators suggest to disable THP. Selective huge page collapse per process is possible using prctl and a launcher. However, this does not solve the issue with hot region detection. Additionally, it the sysadmin should create a launcher that uses PRCTL to enable THP for a particular process. We can use the DAMON support for DAMOS_HUGEPAGE and DAMOS_COLLAPSE, to target a certain process. DAMOS_COLLAPSE can also target the hot regions in that process. Still, there is an issue with the amount of huge page consumption. Since huge pages can lead to memory fragmentation and waste, there should be a way to limit the amount of huge page consumption. There is hugetlbfs, but it requires changes to the application code or the use of libhugetlbfs. DAMON has now a way to autotune some of the variables and adjust quotas automatically, so that DAMON is fired only under the right circumstances. It would be nice to have something similar, but for huge pages. Solution ======== A new autotuning quota goal[2], damos_hugepage_mem_bp, is introduced, which checks the huge page consumption to total memory consumption. This new quota mechanism reuses current autotuning architecture. In order to test this new mechanism, a sample module[3] was created, but not included in this patch series. To demonstrate the tool, damo user space tool was modified[4], which sets up huge pages collapse autotuning. Benchmarks ========== Setup: physical server with arm64 processor with 4 NUMA nodes, 1 TB RAM and running mariaDB 10.5.29. Sysbench was used for the benchmark, with 20 tables and 3 million rows per table. The database was pinned to one of the nodes, and the benchmark framework to a different node. No network traffic involved in the benchmark. Damo user space tool was forked and hugepage_mem_bp support added[4]. DAMON was lauched using this command line: sudo ./damo start $(pidof mariadbd) \ --monitoring_nr_regions_range 10 1000 \ --monitoring_intervals 5000 100000 60000000 \ --damos_quota_time 0 --damos_quota_space 128000000 \ --damos_quota_interval 1000 \ --damos_quota_weights 0 1 1 \ --damos_quota_goal hugepage_mem_bp \ --damos_quota_goal_tuner temporal \ --damos_apply_interval 50000 \ --damos_access_rate 0 max --damos_age 50 max \ --damos_action collapse --debug_damon was 1000 to taget 10% hugepage to total memory ratio, or 2500 to target 25%. Tuner was also tested with consistent and temporal. Results ======= After the last timestamp, there was no change in huge page use, and the total huge page to memory consumption ratio barely moved. hugepage_mem_bp: 1000 goal tuner: temporal +-----------+----------------+----------------+----------------------+ | timestamp | total mem used | huge page used | percentage hugepage | +-----------+----------------+----------------+----------------------+ | 0 | 16945.04297 | 0 | 0 | | 7 | 17008.69531 | 74 | 0.435071583 | | 8 | 17036.40234 | 194 | 1.138738074 | | 9 | 17017.01563 | 314 | 1.845211916 | | 10 | 17029.67969 | 434 | 2.548491856 | | 61 | 17111.30859 | 584 | 3.412947623 | | 120 | 17071.05859 | 694 | 4.065360072 | | 180 | 17133.88281 | 804 | 4.692456513 | | 203 | 17088.16406 | 916 | 5.360435426 | | 204 | 17126.34766 | 1046 | 6.107548562 | | 205 | 17093.84375 | 1176 | 6.879669764 | | 206 | 17142.77734 | 1298 | 7.571701913 | | 209 | 17149.17969 | 1686 | 9.831374041 | | 210 | 17097.30859 | 1754 | 10.25892462 | +-----------+----------------+----------------+----------------------+ hugepage_mem_bp: 1000 goal tuner: consistent +-----------+----------------+----------------+----------------------+ | timestamp | total mem used | huge page used | percentage hugepage | +-----------+----------------+----------------+----------------------+ | 0 | 16955.24609 | 0 | 0 | | 34 | 17039.71875 | 106 | 0.622075995 | | 78 | 17009.47656 | 554 | 3.257007927 | | 90 | 17048.92188 | 596 | 3.495822225 | | 150 | 17092.90625 | 706 | 4.130368409 | | 180 | 17053.08984 | 764 | 4.480126517 | | 233 | 17100.50391 | 1496 | 8.748280216 | | 239 | 17098.89063 | 2216 | 12.95990511 | | 240 | 17135.44531 | 2334 | 13.62088908 | | 245 | 17132.55078 | 2932 | 17.11362212 | | 246 | 17117.95313 | 3052 | 17.82923448 | | 250 | 17163.12109 | 3532 | 20.57900763 | +-----------+----------------+----------------+----------------------+ hugepage_mem_bp: 2500 goal tuner: temporal +-----------+----------------+----------------+----------------------+ | timestamp | total mem used | huge page used | percentage hugepage | +-----------+----------------+----------------+----------------------+ | 0 | 17010.31641 | 0 | 0 | | 9 | 17063.6875 | 50 | 0.2930199 | | 10 | 17051.75781 | 170 | 0.996964664 | | 60 | 17133.85547 | 572 | 3.338419663 | | 90 | 17192.07813 | 626 | 3.641211932 | | 120 | 17221.44531 | 682 | 3.960178647 | | 181 | 17199.76172 | 790 | 4.593086886 | | 208 | 17222.77734 | 1206 | 7.002354939 | | 214 | 17245.17969 | 1904 | 11.04076637 | | 215 | 17240.45703 | 2024 | 11.73982799 | | 220 | 17234.79688 | 2624 | 15.22501262 | | 228 | 17222.83594 | 3584 | 20.80958103 | | 231 | 17247.55469 | 3944 | 22.86700968 | | 235 | 17229.37109 | 4424 | 25.67708349 | +-----------+----------------+----------------+----------------------+ hugepage_mem_bp: 1000 goal tuner: consist +-----------+----------------+----------------+----------------------+ | timestamp | total mem used | huge page used | percentage hugepage | +-----------+----------------+----------------+----------------------+ | 0 | 17125.85156 | 0 | 0 | | 38 | 17081.23438 | 76 | 0.444932716 | | 39 | 17133.11719 | 196 | 1.143983304 | | 40 | 17119.83984 | 316 | 1.84581166 | | 60 | 17109.72656 | 554 | 3.237924335 | | 90 | 17164.11328 | 628 | 3.65879664 | | 180 | 17177.66016 | 792 | 4.610639591 | | 220 | 17180.86719 | 1378 | 8.020549749 | | 226 | 17187.82031 | 1980 | 11.51978531 | | 233 | 17143.48438 | 2818 | 16.4377319 | | 240 | 17137.38281 | 3656 | 21.33347921 | | 250 | 17175.5 | 4856 | 28.27283049 | | 260 | 17199.66406 | 6056 | 35.20999002 | | 270 | 17203.98438 | 7254 | 42.16465118 | | 275 | 17207.21875 | 7762 | 45.10897498 | +-----------+----------------+----------------+----------------------+ More detailed tables are provided here[5] From this, we can conclude that the huge page autotuner works fine, achieving the target. When using consistent autotuner, it actually over-achieves the target, which is expected, since quota esz_bp is not set to 0 to cap the DAMOS policy. Patches Sequence ================ Patch 1 -> Introduce DAMOS_QUOTA_HUGEPAGE_MEM_BP and autotuning Patch 2 -> sysfs support for the new quota goal Patch 3 -> Document hugepage_mem_bp parameter This patch (of 3): Introduce DAMOS_QUOTA_HUGEPAGE_MEM_BP auto tuning. Add a new DAMOS quota goal metric to measure the amount of huge page consumption to total memory consumption ratio. Vmstat may lag, which in some cases may lead to NR_FREE_PAGES being greater than or equal to the amount of RAM in the system. A guard is added to avoid the extremely unlikely case [6]. In the case, return 100% (10000 bp). Link: https://lore.kernel.org/20260831144732.80910-1-sj@kernel.org Link: https://lore.kernel.org/20260831144732.80910-2-sj@kernel.org Link: https://dl.acm.org/doi/pdf/10.1145/3307650.3322227 [1] Link: https://lore.kernel.org/e67f05ad-dbb9-45e6-ba30-b167a99ac67d@huawei-partners.com [2] Link: https://lore.kernel.org/20260616150316.580819-3-gutierrez.asier@huawei-partners.com [3] Link: https://github.com/asierHuawei/damo/commit/79ae1a4ab1c012a7161db85a000d14f08fa36736 [4] Link: https://lore.kernel.org/all/03f678dd-9ef3-4b97-b753-c2e4554c5159@huawei-partners.com/ [5] Link: https://lore.kernel.org/all/20260715151615.99767-1-sj@kernel.org/ [6] Signed-off-by: Asier Gutierrez Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/damon.h | 2 ++ mm/damon/core.c | 19 +++++++++++++++++++ 2 files changed, 21 insertions(+) diff --git a/include/linux/damon.h b/include/linux/damon.h index 6993dca0f355cc..955b9f614e5bcd 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -153,6 +153,7 @@ enum damos_action { * @DAMOS_QUOTA_INACTIVE_MEM_BP: Inactive to total LRU memory ratio. * @DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP: Scheme-eligible memory ratio of a * node in basis points (0-10000). + * @DAMOS_QUOTA_HUGEPAGE_MEM_BP: Huge page to total used memory ratio. * @NR_DAMOS_QUOTA_GOAL_METRICS: Number of DAMOS quota goal metrics. * * Metrics equal to larger than @NR_DAMOS_QUOTA_GOAL_METRICS are unsupported. @@ -167,6 +168,7 @@ enum damos_quota_goal_metric { DAMOS_QUOTA_ACTIVE_MEM_BP, DAMOS_QUOTA_INACTIVE_MEM_BP, DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP, + DAMOS_QUOTA_HUGEPAGE_MEM_BP, NR_DAMOS_QUOTA_GOAL_METRICS, }; diff --git a/mm/damon/core.c b/mm/damon/core.c index d832a527bcf623..153425e416f55e 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2991,6 +2991,22 @@ static unsigned int damos_get_in_active_mem_bp(bool active_ratio) return mult_frac(inactive, 10000, total); } +static unsigned int damos_hugepage_mem_bp(void) +{ + unsigned long thp, total_pages, free_pages; + + total_pages = totalram_pages(); + free_pages = global_zone_page_state(NR_FREE_PAGES); + + if (total_pages <= free_pages) + return 10000; + + thp = global_node_page_state(NR_ANON_THPS) + + global_node_page_state(NR_SHMEM_THPS) + + global_node_page_state(NR_FILE_THPS); + return mult_frac(thp, 10000, total_pages - free_pages); +} + static void damos_set_quota_goal_current_value(struct damon_ctx *c, struct damos *s, struct damos_quota_goal *goal) { @@ -3022,6 +3038,9 @@ static void damos_set_quota_goal_current_value(struct damon_ctx *c, goal->current_value = damos_get_node_eligible_mem_bp(c, s, goal->nid); break; + case DAMOS_QUOTA_HUGEPAGE_MEM_BP: + goal->current_value = damos_hugepage_mem_bp(); + break; default: break; } From 113927bf8eede9a0250fe2d9329429eb5854b193 Mon Sep 17 00:00:00 2001 From: Asier Gutierrez Date: Mon, 31 Aug 2026 07:47:29 -0700 Subject: [PATCH 0646/1352] mm/damon/sysfs: support hugepage_mem_bp quota goal metric DAMOS has a new autotune policy metric: DAMOS_QUOTA_HUGEPAGE_MEM_BP. This patch exposes DAMOS_QUOTA_HUGEPAGE_MEM_BP through sysfs. Add the "hugepage_mem_bp" to the sysfs-schemes interface. Link: https://lore.kernel.org/20260831144732.80910-3-sj@kernel.org Signed-off-by: Asier Gutierrez Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs-schemes.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/mm/damon/sysfs-schemes.c b/mm/damon/sysfs-schemes.c index 32f495a96b17a8..d9b81d7b5910ed 100644 --- a/mm/damon/sysfs-schemes.c +++ b/mm/damon/sysfs-schemes.c @@ -1269,6 +1269,10 @@ struct damos_sysfs_qgoal_metric_name damos_sysfs_qgoal_metric_names[] = { .metric = DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP, .name = "node_eligible_mem_bp", }, + { + .metric = DAMOS_QUOTA_HUGEPAGE_MEM_BP, + .name = "hugepage_mem_bp", + }, }; static ssize_t target_metric_show(struct kobject *kobj, From c6956dbf5aba66b1e7ab06823b505cd7020601ce Mon Sep 17 00:00:00 2001 From: Asier Gutierrez Date: Mon, 31 Aug 2026 07:47:30 -0700 Subject: [PATCH 0647/1352] Docs/mm/damon/design: cocument hugepage_mem_bp target metric Document hugepage_mem_bp metric exposed by sysfs. Link: https://lore.kernel.org/20260831144732.80910-4-sj@kernel.org Signed-off-by: Asier Gutierrez Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/mm/damon/design.rst | 2 ++ 1 file changed, 2 insertions(+) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index d1dd9050ebf40b..63cbb7b536da20 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -713,6 +713,8 @@ mechanism tries to make ``current_value`` of ``target_metric`` be same to bp (1/10,000). - ``node_eligible_mem_bp``: Scheme target access pattern-eligible memory ratio of a node in bp (1/10,000). +- ``hugepage_mem_bp``: Total huge page to total used memory ratio in bp + (1/10,000). ``nid`` is optionally required for ``node_mem_used_bp``, ``node_mem_free_bp``, ``node_memcg_used_bp``, ``node_memcg_free_bp`` and ``node_eligible_mem_bp`` to From 06b0dce60af1a1f3c4381c85a52e56db7ea51187 Mon Sep 17 00:00:00 2001 From: "Zenghui Yu (Huawei)" Date: Mon, 31 Aug 2026 07:26:03 -0700 Subject: [PATCH 0648/1352] mm/damon/core: remove declaration of __damon_commit_ctx() Patch series "mm/damon: misc cleanups". Cleanup the code, tests and samples for clarifications and readability. The patches are individually sent by the authors. I'm reposting those as one series for convenience of handling. For this reason, changelog is on each patch's commentary section. This patch (of 7): __damon_commit_ctx() was added by commit b1471afe4d10 ("mm/damon/core: do parameter testing commit on damon_start()") but is actually not needed. Remove it. Link: https://lore.kernel.org/20260831142611.77572-1-sj@kernel.org Link: https://lore.kernel.org/20260831142611.77572-2-sj@kernel.org Signed-off-by: Zenghui Yu (Huawei) Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Greg Kroah-Hartman Cc: Shuah Khan Cc: Enze Li Cc: Hari Mishal Cc: Jaeyeon Lee Cc: Li Youhong Cc: zhaozhengzhuo --- mm/damon/core.c | 2 -- 1 file changed, 2 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 153425e416f55e..fb52d99661bb57 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -1935,8 +1935,6 @@ static int __damon_start(struct damon_ctx *ctx) return err; } -static int __damon_commit_ctx(struct damon_ctx *dst, struct damon_ctx *src); - /** * damon_start() - Starts the monitorings for a given group of contexts. * @ctxs: an array of the pointers for contexts to start monitoring From 7bf9aeb7e63ce8a6387bc452e693c4fbbb580c01 Mon Sep 17 00:00:00 2001 From: Enze Li Date: Mon, 31 Aug 2026 07:26:04 -0700 Subject: [PATCH 0649/1352] mm/damon/core: introduce damon_set_target_pid() The logic that finds the struct pid for a given pid number and assigns it to a damon_target is duplicated in multiple places. Including damon_sysfs_add_target() of mm/damon/sysfs.c and the start functions of the two sample modules, samples/damon/wsse.c and samples/damon/prcl.c. Add a function that does the work, and replace the duplicated code in the places with calls to the function. Link: https://lore.kernel.org/20260831142611.77572-3-sj@kernel.org Signed-off-by: Enze Li Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Greg Kroah-Hartman Cc: Hari Mishal Cc: Jaeyeon Lee Cc: Li Youhong Cc: Shuah Khan Cc: "Zenghui Yu (Huawei)" Cc: zhaozhengzhuo --- include/linux/damon.h | 1 + mm/damon/core.c | 12 ++++++++++++ mm/damon/sysfs.c | 6 ++---- samples/damon/prcl.c | 5 +---- samples/damon/wsse.c | 5 +---- 5 files changed, 17 insertions(+), 12 deletions(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index 955b9f614e5bcd..7b1b6050a8286f 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -1057,6 +1057,7 @@ int damos_commit_quota_goals(struct damos_quota *dst, struct damos_quota *src); struct damon_target *damon_new_target(void); void damon_add_target(struct damon_ctx *ctx, struct damon_target *t); +int damon_set_target_pid(struct damon_target *t, int pid); void damon_free_target(struct damon_target *t); void damon_destroy_target(struct damon_target *t, struct damon_ctx *ctx); unsigned int damon_nr_regions(struct damon_target *t); diff --git a/mm/damon/core.c b/mm/damon/core.c index fb52d99661bb57..d63d4c6fd3ffdd 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -10,6 +10,7 @@ #include #include #include +#include #include #include #include @@ -795,6 +796,17 @@ void damon_add_target(struct damon_ctx *ctx, struct damon_target *t) list_add_tail(&t->list, &ctx->adaptive_targets); } +/* + * Assign the struct pid of the given pid number to the given target. + */ +int damon_set_target_pid(struct damon_target *t, int pid) +{ + t->pid = find_get_pid(pid); + if (!t->pid) + return -EINVAL; + return 0; +} + static void damon_del_target(struct damon_target *t) { list_del(&t->list); diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index e3858ffab4b227..3c81b4c91ac0dd 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -3,7 +3,6 @@ * DAMON sysfs Interface */ -#include #include #include @@ -2035,9 +2034,8 @@ static int damon_sysfs_add_target(struct damon_sysfs_target *sys_target, return -ENOMEM; damon_add_target(ctx, t); if (damon_target_has_pid(ctx)) { - t->pid = find_get_pid(sys_target->pid); - if (!t->pid) - /* caller will destroy targets */ + /* caller will destroy targets */ + if (damon_set_target_pid(t, sys_target->pid)) return -EINVAL; } t->obsolete = sys_target->obsolete; diff --git a/samples/damon/prcl.c b/samples/damon/prcl.c index 842099bd622861..83ddf12811d57f 100644 --- a/samples/damon/prcl.c +++ b/samples/damon/prcl.c @@ -32,7 +32,6 @@ module_param_cb(enabled, &enabled_param_ops, &enabled, 0600); MODULE_PARM_DESC(enabled, "Enable or disable DAMON_SAMPLE_PRCL"); static struct damon_ctx *ctx; -static struct pid *target_pidp; static int damon_sample_prcl_repeat_call_fn(void *data) { @@ -79,12 +78,10 @@ static int damon_sample_prcl_start(void) return -ENOMEM; } damon_add_target(ctx, target); - target_pidp = find_get_pid(target_pid); - if (!target_pidp) { + if (damon_set_target_pid(target, target_pid)) { damon_destroy_ctx(ctx); return -EINVAL; } - target->pid = target_pidp; scheme = damon_new_scheme( &(struct damos_access_pattern) { diff --git a/samples/damon/wsse.c b/samples/damon/wsse.c index 37fd5da2015885..53944aea8428ea 100644 --- a/samples/damon/wsse.c +++ b/samples/damon/wsse.c @@ -33,7 +33,6 @@ module_param_cb(enabled, &enabled_param_ops, &enabled, 0600); MODULE_PARM_DESC(enabled, "Enable or disable DAMON_SAMPLE_WSSE"); static struct damon_ctx *ctx; -static struct pid *target_pidp; static int damon_sample_wsse_repeat_call_fn(void *data) { @@ -79,12 +78,10 @@ static int damon_sample_wsse_start(void) return -ENOMEM; } damon_add_target(ctx, target); - target_pidp = find_get_pid(target_pid); - if (!target_pidp) { + if (damon_set_target_pid(target, target_pid)) { damon_destroy_ctx(ctx); return -EINVAL; } - target->pid = target_pidp; err = damon_start(&ctx, 1, true); if (err) { From f7bb99636599d5b3c6d358eb45bc0f4e9199cfb4 Mon Sep 17 00:00:00 2001 From: Li Youhong Date: Mon, 31 Aug 2026 07:26:05 -0700 Subject: [PATCH 0650/1352] mm/damon/ops-common: factor out damon_putback_folio_list() The putback loop is duplicated in damon_migrate_folio_list() and on the invalid-nid path of damon_migrate_pages(). Factor it into a small helper for readability. No functional change. Link: https://lore.kernel.org/20260831142611.77572-4-sj@kernel.org Signed-off-by: Li Youhong Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Enze Li Cc: Greg Kroah-Hartman Cc: Hari Mishal Cc: Jaeyeon Lee Cc: Shuah Khan Cc: "Zenghui Yu (Huawei)" Cc: zhaozhengzhuo --- mm/damon/ops-common.c | 30 +++++++++++++++--------------- 1 file changed, 15 insertions(+), 15 deletions(-) diff --git a/mm/damon/ops-common.c b/mm/damon/ops-common.c index 8fc61d06d35859..7219c608b1952b 100644 --- a/mm/damon/ops-common.c +++ b/mm/damon/ops-common.c @@ -335,6 +335,19 @@ static unsigned int __damon_migrate_folio_list( return nr_succeeded; } +static void damon_putback_folio_list(struct list_head *folio_list) +{ + struct folio *folio; + + while (!list_empty(folio_list)) { + folio = lru_to_folio(folio_list); + list_del(&folio->lru); + node_stat_sub_folio(folio, NR_ISOLATED_ANON + + folio_is_file_lru(folio)); + folio_putback_lru(folio); + } +} + static unsigned int damon_migrate_folio_list(struct list_head *folio_list, struct pglist_data *pgdat, int target_nid) @@ -376,13 +389,7 @@ static unsigned int damon_migrate_folio_list(struct list_head *folio_list, list_splice(&ret_folios, folio_list); - while (!list_empty(folio_list)) { - folio = lru_to_folio(folio_list); - list_del(&folio->lru); - node_stat_sub_folio(folio, NR_ISOLATED_ANON + - folio_is_file_lru(folio)); - folio_putback_lru(folio); - } + damon_putback_folio_list(folio_list); return nr_migrated; } @@ -399,14 +406,7 @@ unsigned long damon_migrate_pages(struct list_head *folio_list, int target_nid) if (target_nid < 0 || target_nid >= MAX_NUMNODES || !node_state(target_nid, N_MEMORY)) { - while (!list_empty(folio_list)) { - struct folio *folio = lru_to_folio(folio_list); - - list_del(&folio->lru); - node_stat_sub_folio(folio, NR_ISOLATED_ANON + - folio_is_file_lru(folio)); - folio_putback_lru(folio); - } + damon_putback_folio_list(folio_list); return nr_migrated; } From 83e83f54fc555f9dcc9a45037639d9e41487becf Mon Sep 17 00:00:00 2001 From: Hari Mishal Date: Mon, 31 Aug 2026 07:26:06 -0700 Subject: [PATCH 0651/1352] selftests/damon/sysfs.py: clean up sh processes used for obsolete_target test The obsolete_target test spawns three sh processes and uses their pids as DAMON monitoring targets. These processes are never terminated or waited on, so they are left running (or become zombies) as orphaned children after the test program exits. Terminate each process and communicate() with it after the targets are no longer needed, so it exits and gets reaped instead of being leaked. Link: https://lore.kernel.org/20260831142611.77572-5-sj@kernel.org Signed-off-by: Hari Mishal Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Greg Kroah-Hartman Cc: Enze Li Cc: Jaeyeon Lee Cc: Li Youhong Cc: Shuah Khan Cc: "Zenghui Yu (Huawei)" Cc: zhaozhengzhuo --- tools/testing/selftests/damon/sysfs.py | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/tools/testing/selftests/damon/sysfs.py b/tools/testing/selftests/damon/sysfs.py index 3ffa054b63867d..88a26422ff44cf 100755 --- a/tools/testing/selftests/damon/sysfs.py +++ b/tools/testing/selftests/damon/sysfs.py @@ -385,6 +385,10 @@ def main(): assert_ctxs_committed(kdamonds) kdamonds.stop() + for proc in (proc1, proc2, proc3): + proc.terminate() + proc.communicate() + test_memcg_filter_memcg_path_staging() if __name__ == '__main__': From 02ed38cc8ac52d1776ea914cf973252e164b0428 Mon Sep 17 00:00:00 2001 From: Jaeyeon Lee Date: Mon, 31 Aug 2026 07:26:07 -0700 Subject: [PATCH 0652/1352] mm/damon/tests: use scoped_guard() for damon_test_ops_registration Replace manual mutex_lock() and mutex_unlock() calls with the scoped_guard() macro. This simplifies the code, improves readability, and ensures that the lock is automatically released when the scope ends, preventing potential lock leaks in the future. Link: https://lore.kernel.org/20260831142611.77572-6-sj@kernel.org Signed-off-by: Jaeyeon Lee Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Enze Li Cc: Greg Kroah-Hartman Cc: Hari Mishal Cc: Li Youhong Cc: Shuah Khan Cc: "Zenghui Yu (Huawei)" Cc: zhaozhengzhuo --- mm/damon/tests/core-kunit.h | 21 ++++++++++----------- 1 file changed, 10 insertions(+), 11 deletions(-) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index b1ca4c8e03f091..7071ec277b0072 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -441,17 +441,17 @@ static void damon_test_ops_registration(struct kunit *test) KUNIT_EXPECT_EQ(test, damon_select_ops(c, NR_DAMON_OPS), -EINVAL); /* Registration should success after unregistration */ - mutex_lock(&damon_ops_lock); - bak = damon_registered_ops[DAMON_OPS_VADDR]; - damon_registered_ops[DAMON_OPS_VADDR] = (struct damon_operations){}; - mutex_unlock(&damon_ops_lock); + scoped_guard(mutex, &damon_ops_lock) { + bak = damon_registered_ops[DAMON_OPS_VADDR]; + damon_registered_ops[DAMON_OPS_VADDR] = + (struct damon_operations){}; + } ops.id = DAMON_OPS_VADDR; KUNIT_EXPECT_EQ(test, damon_register_ops(&ops), 0); - mutex_lock(&damon_ops_lock); - damon_registered_ops[DAMON_OPS_VADDR] = bak; - mutex_unlock(&damon_ops_lock); + scoped_guard(mutex, &damon_ops_lock) + damon_registered_ops[DAMON_OPS_VADDR] = bak; /* Check double-registration failure again */ KUNIT_EXPECT_EQ(test, damon_register_ops(&ops), -EINVAL); @@ -459,10 +459,9 @@ static void damon_test_ops_registration(struct kunit *test) damon_destroy_ctx(c); if (need_cleanup) { - mutex_lock(&damon_ops_lock); - damon_registered_ops[DAMON_OPS_VADDR] = - (struct damon_operations){}; - mutex_unlock(&damon_ops_lock); + scoped_guard(mutex, &damon_ops_lock) + damon_registered_ops[DAMON_OPS_VADDR] = + (struct damon_operations){}; } } From bc5f8fd06a4c335f30da7ff8297278e5d6399a5a Mon Sep 17 00:00:00 2001 From: zhaozhengzhuo Date: Mon, 31 Aug 2026 07:26:08 -0700 Subject: [PATCH 0653/1352] selftests/damon: prevent remaining cross-object state pollution _damon_sysfs.py defines constructors with mutable default arguments, including DamosAccessPattern(), DamosQuota(), DamosWatermarks(), DamosDests(), IntervalsGoal(), and empty lists. Default arguments are evaluated once at function definition time. Damos() instances created without explicit arguments therefore share the same DamosQuota(), and the other default-constructed sub-objects and lists are shared in the same way. The sub-objects keep back-pointers to their owner scheme, so constructing the second Damos() rebinds the shared quota's scheme pointer to the second object. An item appended to one object's default contexts or filters list is also visible from other default-constructed objects. The shared state can corrupt test configurations. DamosQuota.sysfs_dir() derives the sysfs directory from its scheme pointer, so operating on the first scheme's default quota may write to the second scheme's directory. The wrong values often match the defaults, so tests still pass, but the behavior depends on object creation order. Commit 8319dadcbd81 ("selftests/damon: prevent cross-context state pollution in DamonCtx") fixed the same pattern in DamonCtx only. Fix the remaining constructors by defaulting to None and creating fresh objects or lists inside each constructor. Explicit arguments keep their previous behavior. Link: https://lore.kernel.org/20260831142611.77572-7-sj@kernel.org Signed-off-by: zhaozhengzhuo Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Enze Li Cc: Greg Kroah-Hartman Cc: Hari Mishal Cc: Jaeyeon Lee Cc: Li Youhong Cc: Shuah Khan Cc: "Zenghui Yu (Huawei)" --- tools/testing/selftests/damon/_damon_sysfs.py | 38 ++++++++++++++----- 1 file changed, 28 insertions(+), 10 deletions(-) diff --git a/tools/testing/selftests/damon/_damon_sysfs.py b/tools/testing/selftests/damon/_damon_sysfs.py index e6a2265d721e8b..f604b7d6530b3c 100644 --- a/tools/testing/selftests/damon/_damon_sysfs.py +++ b/tools/testing/selftests/damon/_damon_sysfs.py @@ -321,8 +321,10 @@ class DamosFilters: filters = None scheme = None # owner scheme - def __init__(self, name, filters=[]): + def __init__(self, name, filters=None): self.name = name + if filters is None: + filters = [] self.filters = filters for idx, filter_ in enumerate(self.filters): filter_.idx = idx @@ -368,7 +370,9 @@ class DamosDests: dests = None scheme = None # owner scheme - def __init__(self, dests=[]): + def __init__(self, dests=None): + if dests is None: + dests = [] self.dests = dests for idx, dest in enumerate(self.dests): dest.idx = idx @@ -426,15 +430,21 @@ class Damos: stats = None tried_regions = None - def __init__(self, action='stat', access_pattern=DamosAccessPattern(), - quota=DamosQuota(), watermarks=DamosWatermarks(), - core_filters=[], ops_filters=[], filters=[], target_nid=0, - dests=DamosDests(), apply_interval_us=0): + def __init__(self, action='stat', access_pattern=None, quota=None, + watermarks=None, core_filters=None, ops_filters=None, + filters=None, target_nid=0, dests=None, + apply_interval_us=0): self.action = action + if access_pattern is None: + access_pattern = DamosAccessPattern() self.access_pattern = access_pattern self.access_pattern.scheme = self + if quota is None: + quota = DamosQuota() self.quota = quota self.quota.scheme = self + if watermarks is None: + watermarks = DamosWatermarks() self.watermarks = watermarks self.watermarks.scheme = self @@ -448,6 +458,8 @@ def __init__(self, action='stat', access_pattern=DamosAccessPattern(), self.filters.scheme = self self.target_nid = target_nid + if dests is None: + dests = DamosDests() self.dests = dests self.dests.scheme = self @@ -568,10 +580,12 @@ class DamonAttrs: context = None def __init__(self, sample_us=5000, aggr_us=100000, - intervals_goal=IntervalsGoal(), update_us=1000000, - min_nr_regions=10, max_nr_regions=1000): + intervals_goal=None, update_us=1000000, min_nr_regions=10, + max_nr_regions=1000): self.sample_us = sample_us self.aggr_us = aggr_us + if intervals_goal is None: + intervals_goal = IntervalsGoal() self.intervals_goal = intervals_goal self.intervals_goal.attrs = self self.update_us = update_us @@ -703,7 +717,9 @@ class Kdamond: idx = None # index of this kdamond between siblings kdamonds = None # parent - def __init__(self, contexts=[], refresh_ms=None): + def __init__(self, contexts=None, refresh_ms=None): + if contexts is None: + contexts = [] self.contexts = contexts self.refresh_ms = refresh_ms for idx, context in enumerate(self.contexts): @@ -853,7 +869,9 @@ def commit_schemes_quota_goals(self): class Kdamonds: kdamonds = [] - def __init__(self, kdamonds=[]): + def __init__(self, kdamonds=None): + if kdamonds is None: + kdamonds = [] self.kdamonds = kdamonds for idx, kdamond in enumerate(self.kdamonds): kdamond.idx = idx From 97bff5bfedc0274a71befbbceeaf79b2df41fb15 Mon Sep 17 00:00:00 2001 From: Enze Li Date: Mon, 31 Aug 2026 07:26:09 -0700 Subject: [PATCH 0654/1352] samples/damon/mtier: add comment for struct region_range The mtier sample defines a local struct region_range using phys_addr_t instead of damon_addr_range which uses unsigned long. Add a comment explaining the rationale: on 32-bit systems with more than 4GiB memory, phys_addr_t will be 64-bit while unsigned long is 32-bit. Link: https://lore.kernel.org/20260831142611.77572-8-sj@kernel.org Signed-off-by: Enze Li Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Greg Kroah-Hartman Cc: Hari Mishal Cc: Jaeyeon Lee Cc: Li Youhong Cc: Shuah Khan Cc: "Zenghui Yu (Huawei)" Cc: zhaozhengzhuo --- samples/damon/mtier.c | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/samples/damon/mtier.c b/samples/damon/mtier.c index d1123ebbfab906..bea45c87cc9bed 100644 --- a/samples/damon/mtier.c +++ b/samples/damon/mtier.c @@ -52,6 +52,11 @@ module_param(detect_node_addresses, bool, 0600); static struct damon_ctx *ctxs[2]; +/* + * Use phys_addr_t instead of damon_addr_range (unsigned long) for physical + * addresses. On 32-bit systems with more than 4GB memory, phys_addr_t will + * be 64-bit while unsigned long is 32-bit. + */ struct region_range { phys_addr_t start; phys_addr_t end; From f81c859beb09b38c94f1a48a5c598ab36d188062 Mon Sep 17 00:00:00 2001 From: Li Zhe Date: Mon, 31 Aug 2026 19:16:32 +0800 Subject: [PATCH 0655/1352] mm: fix stale ZONE_DEVICE refcount comment Patch series "mm: optimize zone-device memmap initialization", v11. memmap_init_zone_device() can take a noticeable amount of time when large pmem namespaces are bound or rebound, because it initializes nearly identical struct page descriptors one PFN at a time. This series reduces that ZONE_DEVICE memmap initialization overhead by reusing prepared struct page templates and, on x86, using memcpy_nontemporal() for the template copy path. The main target is large fsdax/devdax pmem configurations, where the cost of initializing the memmap shows up directly in nd_pmem/dax_pmem bind and rebind latency. This matters because the cost is paid in the synchronous probe/bind path for large DAX/PMEM ZONE_DEVICE mappings. Userspace workflows such as provisioning or reconfiguring nd_pmem/dax_pmem namespaces, bringing hot-added PMEM-backed capacity online, and recovering or rebinding a device after driver or device changes all wait for this initialization to finish. Reducing this cost will yield benefits as lower user-visible provisioning, hot-add, recovery, and rebind latency for large DAX/PMEM devices. Patches 1-2 are preparatory cleanups and helper extraction. Patches 3-4 add the template-copy path for head pages and compound tails. Patch 5 introduces memcpy_nontemporal(). Patch 6 switches the ZONE_DEVICE template-copy path over to memcpy_nontemporal(). Patch 7 extends the x86 fixed-size memcpy_flushcache() inline cases used by the x86 memcpy_nontemporal() backend for struct page sized copies. Architectures without a specialized memcpy_nontemporal() backend fall back to memcpy(), so the generic template-copy optimization remains available without arch-specific support. On x86, memcpy_nontemporal() maps to the existing memcpy_flushcache() backend and can use the fixed-size MOVNTI paths added by this series for struct page sized copies. memcpy_nontemporal() is only a copy primitive. It does not imply a drain or a publication barrier. Callers that use it before a producer-consumer or device-visible handoff must provide the required ordering. The ZONE_DEVICE template-copy path uses it only while initializing struct page metadata, so the copy primitive itself does not grow a separate drain contract. The numbers below measure the time spent in memmap_init_zone_device() during driver bind/rebind. They are not measurements of the full nd_pmem or dax_pmem bind/rebind operation. Tested in an x86_64 QEMU/KVM VM with a 100 GB fsdax namespace device configured with map=dev and a 100 GB devdax namespace (align=2097152) on Intel Ice Lake server. Test procedure: Rebind the nd_pmem and dax_pmem drivers 30 times and collect the memmap initialization time from the pr_debug() output of memmap_init_zone_device(). Base(v7.3-rc1): Average of nd_pmem rebinds: 221.07 ms Average of dax_pmem rebinds: 191.20 ms With this series applied: Average of nd_pmem rebinds: 71.93 ms Average of dax_pmem rebinds: 87.37 ms This reduces the average memmap initialization time measured during rebind by about 67.5% for nd_pmem and 54.3% for dax_pmem. As an additional x86_64 data point, I also ran measurements on the same physical host with a 100 GB PMEM region created via the memmap= kernel command line, configured as fsdax and devdax namespaces with map=dev and 2 MiB alignment. For brevity, the individual patches keep only the VM results rather than including a second set of physical-host measurements throughout the series. The physical-host numbers below are included only as supplemental evidence that the same optimization also provides a similar benefit on a non-virtualized system. Test procedure: Reconfigure the namespace mode, rebind the nd_pmem or dax_pmem driver 30 times, and collect the memmap initialization time from the pr_debug() output of memmap_init_zone_device(). Base (v7.3-rc1): nd_pmem / fsdax: 205.90 ms dax_pmem / devdax: 225.43 ms With this series applied: nd_pmem / fsdax: 69.13 ms dax_pmem / devdax: 90.67 ms This reduces the measured memmap initialization time during rebind by about 66.4% for nd_pmem and 59.8% for dax_pmem on that setup, which is broadly consistent with the VM results above. As another supplemental data point, I measured the test_hmm.ko module on the same physical x86_64 host, using the test_hmm.ko setup from the previous discussion that times ten 64 GB memremap_pages()/memunmap_pages() iterations during module insertion[1]. By default, module insertion initializes two DEVICE_PRIVATE dmirror devices, so two avg memremap values are reported; each value is the average for one 64 GB chunk. This is not the primary target workload of the series, but it exercises the same large ZONE_DEVICE memmap initialization path and shows the same direction of improvement. Base (v7.3-rc1): avg memremap reported during module insertion: 116500596 ns, 116438028 ns With this series applied: avg memremap reported during module insertion: 46953088 ns, 46428399 ns This corresponds to about a 59.9% reduction based on the mean of the reported values, which is again consistent with the pmem bind/rebind results above. I also include an arm64 data point for the generic template-copy part. It was measured on an arm64 QEMU virt VM with 64 KB pages and a 100 GB ACPI NVDIMM sparse backend. This setup does not use the x86 MOVNTI fast paths, so it exercises the architecture-independent part of the optimization. For devdax, 2 MiB alignment is rejected in this 64 KB page setup, so the devdax namespace was tested with the supported default 512 MiB alignment. Base (v7.3-rc1): Average of rebinds for nd_pmem driver: 27.93 ms Average of rebinds for dax_pmem driver: 27.87 ms With this series applied: Average of rebinds for nd_pmem driver: 14.53 ms Average of rebinds for dax_pmem driver: 16.27 ms This reduces the average memmap initialization time measured during rebind by about 48.0% for nd_pmem and 41.6% for dax_pmem on that arm64 VM setup. Since this arm64 setup does not use the x86 MOVNTI fast paths, the result also suggests that the generic template-copy optimization can benefit architectures without an architecture-specific memcpy_nontemporal() backend. This patch (of 7): The comment in __init_zone_device_page() still uses the old MEMORY_TYPE_* names and implies that FS_DAX pages regain a refcount of 1 in the free path. That no longer matches the code. Update the comment to describe the current policy correctly: MEMORY_DEVICE_GENERIC pages regain a refcount of 1 in the free path, while the remaining ZONE_DEVICE types start from 0 here and raise the count again when the allocator or driver hands the page out. No functional change intended. Link: https://lore.kernel.org/20260831111638.76012-1-lizhe.67@bytedance.com Link: https://lore.kernel.org/20260831111638.76012-2-lizhe.67@bytedance.com Link: https://lore.kernel.org/all/aiEoByaQdRR3xtM5@nvdebian.thelocal/ [1] Signed-off-by: Li Zhe Signed-off-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) Reviewed-by: Alistair Popple Reviewed-by: Muchun Song Reviewed-by: Mike Rapoport (Microsoft) Cc: Arnd Bergmann Cc: Balbir Singh Cc: "Borislav Petkov (AMD)" Cc: Dave Hansen Cc: Ingo Molnar Cc: Kees Cook --- mm/mm_init.c | 10 +++------- 1 file changed, 3 insertions(+), 7 deletions(-) diff --git a/mm/mm_init.c b/mm/mm_init.c index 304c88da5cce21..951e6fc17f581b 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -1012,13 +1012,9 @@ static void __ref __init_zone_device_page(struct page *page, unsigned long pfn, page->zone_device_data = NULL; /* - * ZONE_DEVICE pages other than MEMORY_TYPE_GENERIC are released - * directly to the driver page allocator which will set the page count - * to 1 when allocating the page. - * - * MEMORY_TYPE_GENERIC and MEMORY_TYPE_FS_DAX pages automatically have - * their refcount reset to one whenever they are freed (ie. after - * their refcount drops to 0). + * MEMORY_DEVICE_GENERIC pages regain a refcount of 1 in the free + * path. The remaining ZONE_DEVICE types start from 0 here and raise + * the count again when the allocator or driver hands the page out. */ switch (pgmap->type) { case MEMORY_DEVICE_FS_DAX: From faa34466856958d8c0a4af86f0b633230cf9daea Mon Sep 17 00:00:00 2001 From: Li Zhe Date: Mon, 31 Aug 2026 19:16:33 +0800 Subject: [PATCH 0656/1352] mm: add a set_page_section_from_pfn() helper Callers that want to update section bits from a PFN currently need to open-code: set_page_section(page, pfn_to_section_nr(pfn)); and guard that sequence with #ifdef SECTION_IN_PAGE_FLAGS. Add set_page_section_from_pfn() to wrap that update in one place. When section bits are stored in page flags, the helper derives the section number from the PFN and updates the page flags. Otherwise keep it as a no-op so callers can use one helper without open-coding SECTION_IN_PAGE_FLAGS. Convert set_page_links() to use the new helper so later ZONE_DEVICE fast-path patches can also update section bits without open-coding SECTION_IN_PAGE_FLAGS at each callsite. This keeps the PFN-to-section translation local to the configurations that actually store section bits in struct page flags, and avoids exposing that detail to generic callers. No functional change intended. Link: https://lore.kernel.org/20260831111638.76012-3-lizhe.67@bytedance.com Signed-off-by: Li Zhe Signed-off-by: Andrew Morton Reviewed-by: Mike Rapoport (Microsoft) Acked-by: Muchun Song Reviewed-by: Balbir Singh Cc: Alistair Popple Cc: Arnd Bergmann Cc: "Borislav Petkov (AMD)" Cc: Dave Hansen Cc: David Hildenbrand (Arm) Cc: Ingo Molnar Cc: Kees Cook --- include/linux/mm.h | 15 ++++++++++++--- 1 file changed, 12 insertions(+), 3 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index a9fbe26536f450..1b28e6fc8d5dd1 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -2632,12 +2632,23 @@ static inline void set_page_section(struct page *page, unsigned long section) page->flags.f |= (section & SECTIONS_MASK) << SECTIONS_PGSHIFT; } +static inline void set_page_section_from_pfn(struct page *page, + unsigned long pfn) +{ + set_page_section(page, pfn_to_section_nr(pfn)); +} + static inline unsigned long memdesc_section(const memdesc_flags_t *mdf) { ASSERT_EXCLUSIVE_BITS(mdf->f, SECTIONS_MASK << SECTIONS_PGSHIFT); return (mdf->f >> SECTIONS_PGSHIFT) & SECTIONS_MASK; } #else /* !SECTION_IN_PAGE_FLAGS */ +static inline void set_page_section_from_pfn(struct page *page, + unsigned long pfn) +{ +} + static inline unsigned long memdesc_section(const memdesc_flags_t *mdf) { return 0; @@ -2860,9 +2871,7 @@ static inline void set_page_links(struct page *page, enum zone_type zone, { set_page_zone(page, zone); set_page_node(page, node); -#ifdef SECTION_IN_PAGE_FLAGS - set_page_section(page, pfn_to_section_nr(pfn)); -#endif + set_page_section_from_pfn(page, pfn); } /** From 42e6078553717848e8a5c3f2d7970d097f4b3690 Mon Sep 17 00:00:00 2001 From: Li Zhe Date: Mon, 31 Aug 2026 19:16:34 +0800 Subject: [PATCH 0657/1352] mm: add a template-based fast path for zone-device page init memmap_init_zone_device() repeats nearly identical head-page initialization for each PFN. Initialize the first real ZONE_DEVICE head page through the existing path, copy that final state into a reusable template, refresh the PFN-dependent fields in that template before each copy, and copy it into the remaining destination pages. Use the template path unconditionally. The page_ref_set tracepoint is primarily a debugging aid, while this code is still initializing struct pages before they are handed out. From the perspective of users of those pages, the initialization-time refcount transitions are not part of the observable page lifetime. This means page_ref_set will no longer observe every initialization-time refcount assignment for copied ZONE_DEVICE head pages. The impact is controlled because the final initialized struct page state is unchanged, and keeping a separate non-template path only for this local tracepoint observability would add complexity to the common path. This patch accelerates head-page initialization. The pfns_per_compound == 1 case gets the full benefit here, compound tails are handled in the next patch. Tested in a VM with a 100 GB fsdax namespace device configured with map=dev on Intel Ice Lake server. This test exercises the nd_pmem rebind path (pfns_per_compound == 1). Test procedure: Rebind the nd_pmem driver 30 times and collect the memmap initialization time from the pr_debug() output of memmap_init_zone_device(). Base(v7.3-rc1): Average of rebinds for nd_pmem driver: 221.07 ms With this patch and its prerequisites applied: Average of rebinds for nd_pmem driver: 155.00 ms This reduces the average memmap initialization time measured during rebind from 221.07 ms to 155.00 ms, or about 29.9%. Link: https://lore.kernel.org/20260831111638.76012-4-lizhe.67@bytedance.com Signed-off-by: Li Zhe Signed-off-by: Andrew Morton Reviewed-by: Mike Rapoport (Microsoft) Cc: Alistair Popple Cc: Arnd Bergmann Cc: Balbir Singh Cc: "Borislav Petkov (AMD)" Cc: Dave Hansen Cc: David Hildenbrand (Arm) Cc: Ingo Molnar Cc: Kees Cook Cc: Muchun Song --- mm/mm_init.c | 38 +++++++++++++++++++++++++++++++++++--- 1 file changed, 35 insertions(+), 3 deletions(-) diff --git a/mm/mm_init.c b/mm/mm_init.c index 951e6fc17f581b..0494a795f5fa58 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -1029,6 +1029,17 @@ static void __ref __init_zone_device_page(struct page *page, unsigned long pfn, } } +static void zone_device_page_init_from_template(struct page *page, + unsigned long pfn, struct page *template) +{ + set_page_section_from_pfn(template, pfn); +#ifdef WANT_PAGE_VIRTUAL + if (!is_highmem_idx(ZONE_DEVICE)) + set_page_address(template, __va(pfn << PAGE_SHIFT)); +#endif + memcpy(page, template, sizeof(*page)); +} + /* * With compound page geometry and when struct pages are stored in ram most * tail pages are reused. Consequently, the amount of unique struct pages to @@ -1091,6 +1102,8 @@ void __ref memmap_init_zone_device(struct zone *zone, unsigned long zone_idx = zone_idx(zone); unsigned long start = jiffies; int nid = pgdat->node_id; + struct page template; + struct page *page; if (WARN_ON_ONCE(!pgmap || zone_idx != ZONE_DEVICE)) return; @@ -1105,10 +1118,29 @@ void __ref memmap_init_zone_device(struct zone *zone, nr_pages = end_pfn - start_pfn; } - for (pfn = start_pfn; pfn < end_pfn; pfn += pfns_per_compound) { - struct page *page = pfn_to_page(pfn); + if (!nr_pages) + return; - __init_zone_device_page(page, pfn, zone_idx, nid, pgmap); + /* + * Seed the reusable head-page template from the first real struct + * page. The normal page-init and refcount helpers must operate on + * a real memmap entry rather than a stack object. + */ + pfn = start_pfn; + page = pfn_to_page(pfn); + __init_zone_device_page(page, pfn, zone_idx, nid, pgmap); + memcpy(&template, page, sizeof(*page)); + if (pfns_per_compound != 1) + memmap_init_compound(page, pfn, zone_idx, nid, pgmap, + compound_nr_pages(pfn, altmap, pgmap)); + pfn += pfns_per_compound; + + /* Initialize the remaining head pages from template. */ + for (; pfn < end_pfn; pfn += pfns_per_compound) { + page = pfn_to_page(pfn); + + zone_device_page_init_from_template(page, pfn, + &template); if (IS_ALIGNED(pfn, PAGES_PER_SECTION)) cond_resched(); From f3f1d95c77676a5383a8b23a37d829a2156cddf4 Mon Sep 17 00:00:00 2001 From: Li Zhe Date: Thu, 3 Sep 2026 10:58:06 +0800 Subject: [PATCH 0658/1352] mm-add-a-template-based-fast-path-for-zone-device-page-init-fix whitespace fix, per Mike Link: https://lore.kernel.org/20260903025806.70825-1-lizhe.67@bytedance.com Signed-off-by: Li Zhe Signed-off-by: Andrew Morton Cc: Mike Rapoport (Microsoft) --- mm/mm_init.c | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/mm/mm_init.c b/mm/mm_init.c index 0494a795f5fa58..af11885f17ef05 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -1139,8 +1139,7 @@ void __ref memmap_init_zone_device(struct zone *zone, for (; pfn < end_pfn; pfn += pfns_per_compound) { page = pfn_to_page(pfn); - zone_device_page_init_from_template(page, pfn, - &template); + zone_device_page_init_from_template(page, pfn, &template); if (IS_ALIGNED(pfn, PAGES_PER_SECTION)) cond_resched(); From 30b9ad4d4ea606ce608a8355b041478a8a0b6ed9 Mon Sep 17 00:00:00 2001 From: Li Zhe Date: Mon, 31 Aug 2026 19:16:35 +0800 Subject: [PATCH 0659/1352] mm: extend the template fast path to zone-device compound tails The template fast path from the previous patch only accelerates head pages. Compound tails in memmap_init_compound() still go through the normal initialization path one by one. Build separate head and tail templates and reuse one prepared tail template across the tail pages in a compound range. Head pages preserve the existing refcount policy, while compound tails always start with a refcount of 0 after prep_compound_tail(). This extends the template-copy fast path to pfns_per_compound > 1. Tail-page PFN-dependent fields are refreshed in the reusable tail template before each copy. Do not keep a separate non-template fallback for compound tails either. These pages are still under memmap initialization, and the initialization-time refcount updates are not part of the observable lifetime of pages handed out later. The impact is controlled for the same reason as for head pages. The first tail page still seeds the reusable tail template through the normal tail initialization sequence, and the copied tail pages have the same final initialized state except for the PFN-dependent fields refreshed before each copy. Tested in a VM with a 100 GB devdax namespace (align=2097152) on Intel Ice Lake server. This test exercises the dax_pmem rebind path and measures memmap initialization latency. Test procedure: Unbind and rebind the dax_pmem driver 30 times, collect memmap initialization time from the pr_debug() output of memmap_init_zone_device(). Base(v7.3-rc1): Average of rebinds for dax_pmem driver: 191.20 ms With this patch and its prerequisites applied: Average of rebinds for dax_pmem driver: 176.87 ms This reduces the average memmap initialization time measured during rebind from 191.20 ms to 176.87 ms, or about 7.5%. Link: https://lore.kernel.org/20260831111638.76012-5-lizhe.67@bytedance.com Signed-off-by: Li Zhe Signed-off-by: Andrew Morton Reviewed-by: Mike Rapoport (Microsoft) Cc: Alistair Popple Cc: Arnd Bergmann Cc: Balbir Singh Cc: "Borislav Petkov (AMD)" Cc: Dave Hansen Cc: David Hildenbrand (Arm) Cc: Ingo Molnar Cc: Kees Cook Cc: Muchun Song --- mm/mm_init.c | 24 ++++++++++++++++++------ 1 file changed, 18 insertions(+), 6 deletions(-) diff --git a/mm/mm_init.c b/mm/mm_init.c index af11885f17ef05..e2b0952023b3b5 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -1072,6 +1072,8 @@ static void __ref memmap_init_compound(struct page *head, { unsigned long pfn, end_pfn = head_pfn + nr_pages; unsigned int order = pgmap->vmemmap_shift; + struct page template; + struct page *page; /* * We have to initialize the pages, including setting up page links. @@ -1080,13 +1082,23 @@ static void __ref memmap_init_compound(struct page *head, * the pages in the same go. */ __SetPageHead(head); - for (pfn = head_pfn + 1; pfn < end_pfn; pfn++) { - struct page *page = pfn_to_page(pfn); - __init_zone_device_page(page, pfn, zone_idx, nid, pgmap); - prep_compound_tail(page, head, order); - set_page_count(page, 0); - } + /* + * All tails of the same compound page share the state established by + * prep_compound_tail(). Reuse one tail template for the whole range and + * refresh only the PFN-dependent fields in that template before each copy. + */ + pfn = head_pfn + 1; + page = pfn_to_page(pfn); + __init_zone_device_page(page, pfn, zone_idx, nid, pgmap); + prep_compound_tail(page, head, order); + set_page_count(page, 0); + memcpy(&template, page, sizeof(*page)); + + /* Initialize the remaining tail pages from template. */ + for (pfn = head_pfn + 2; pfn < end_pfn; pfn++) + zone_device_page_init_from_template(pfn_to_page(pfn), pfn, + &template); prep_compound_head(head, order); } From c8a8056d755ebc14e1f7c599668fec5690847c53 Mon Sep 17 00:00:00 2001 From: Li Zhe Date: Mon, 31 Aug 2026 19:16:36 +0800 Subject: [PATCH 0660/1352] string: introduce memcpy_nontemporal() Introduce memcpy_nontemporal() for write-once copy sites that want a named non-temporal copy primitive. On x86_64, override the helper in arch/x86/include/asm/string_64.h using the usual self-macro pattern, next to the existing memcpy_flushcache() backend that memcpy_nontemporal() wraps. include/linux/string.h provides the generic memcpy_nontemporal() fallback as #define memcpy_nontemporal(dst, src, len) \ ((void)memcpy(dst, src, len)) instead of an inline wrapper, so architectures without a specialized backend keep the usual memcpy() FORTIFY coverage when the compiler can still see object sizes at the original call site. It also makes the memcpy_nontemporal() API uniformly void, matching memcpy_flushcache() and the x86 backend, so callers cannot accidentally depend on a return value on fallback architectures. memcpy_nontemporal() is only a copy primitive. It does not imply a drain or a publication barrier. Callers that use it before a producer-consumer or device-visible handoff must provide the required ordering at that handoff point. The immediate user is the ZONE_DEVICE template-copy path. It populates struct page descriptors in a write-once pattern, so a regular cached memcpy() can incur avoidable write-allocate traffic and cache pollution for data with little near-term reuse. Link: https://lore.kernel.org/20260831111638.76012-6-lizhe.67@bytedance.com Signed-off-by: Li Zhe Signed-off-by: Andrew Morton Cc: Alistair Popple Cc: Arnd Bergmann Cc: Balbir Singh Cc: "Borislav Petkov (AMD)" Cc: Dave Hansen Cc: David Hildenbrand (Arm) Cc: Ingo Molnar Cc: Kees Cook Cc: Mike Rapoport (Microsoft) Cc: Muchun Song --- arch/x86/include/asm/string_64.h | 12 ++++++++++++ include/linux/string.h | 13 +++++++++++++ 2 files changed, 25 insertions(+) diff --git a/arch/x86/include/asm/string_64.h b/arch/x86/include/asm/string_64.h index 4635616863f53d..21ae515ae35a3d 100644 --- a/arch/x86/include/asm/string_64.h +++ b/arch/x86/include/asm/string_64.h @@ -100,6 +100,18 @@ static __always_inline void memcpy_flushcache(void *dst, const void *src, size_t } __memcpy_flushcache(dst, src, cnt); } + +#define memcpy_nontemporal memcpy_nontemporal +/* + * Reuse the existing x86 flushcache backend as the non-temporal copy + * primitive. + */ +static __always_inline void memcpy_nontemporal(void *dst, const void *src, + size_t cnt) +{ + memcpy_flushcache(dst, src, cnt); +} + #endif #endif /* __KERNEL__ */ diff --git a/include/linux/string.h b/include/linux/string.h index 5702daca4326b7..6cb5cdd01158b2 100644 --- a/include/linux/string.h +++ b/include/linux/string.h @@ -278,6 +278,19 @@ static inline void memcpy_flushcache(void *dst, const void *src, size_t cnt) } #endif +#ifndef memcpy_nontemporal +/* + * memcpy_nontemporal() requests a non-temporal copy when the + * architecture has a suitable backend. Architectures without a + * specialized backend fall back to memcpy(). Keep this as a + * function-like macro so the compiler can still see the original + * memcpy() call site and preserve the usual FORTIFY coverage when + * object sizes remain visible there, while keeping the API void. + */ +#define memcpy_nontemporal(dst, src, len) \ + ((void)memcpy(dst, src, len)) +#endif + void *memchr_inv(const void *s, int c, size_t n); char *strreplace(char *str, char old, char new); From d695df591122c19ce5f08f2d31454f09e1944873 Mon Sep 17 00:00:00 2001 From: Li Zhe Date: Mon, 31 Aug 2026 19:16:37 +0800 Subject: [PATCH 0661/1352] mm: use memcpy_nontemporal() in zone-device template copies The template fast path currently uses memcpy() for the actual struct page copy. Switch zone_device_page_init_from_template() to memcpy_nontemporal(). ZONE_DEVICE memmap initialization is largely write-once: each struct page is populated once, and most destination cachelines are not expected to be reused immediately afterwards. On x86, a regular cached memcpy() can therefore incur write-allocate traffic by pulling destination cachelines into the cache before writeback, and can populate the cache with data that has little near-term reuse. Using memcpy_nontemporal() lets this path request nontemporal stores for that copy pattern, which can reduce cache pollution and avoid part of the associated write-allocate overhead, while architectures without a specialized backend still fall back to memcpy(). Do not add a KASAN/KMSAN-specific fallback around this call site. As Muchun pointed out, special KASAN handling for memcpy_flushcache() or memcpy_nontemporal(), if needed, belongs in the low-level helper rather than in this ZONE_DEVICE caller. No separate drain is added here. memcpy_nontemporal() is used only as the copy primitive while memmap_init_zone_device() is still initializing the struct page array. The ordinary stores that follow in this path, such as compound-page setup, are part of the same CPU's initialization sequence; they are not used as a publication store that tells another CPU or device to consume data written by the non-temporal copy. Therefore this call site does not need a helper-level drain for correctness. Callers that use memcpy_nontemporal() as part of a producer-consumer or device-visible handoff must add the required ordering themselves. Tested in a VM with a 100 GB fsdax namespace device configured with map=dev and a 100 GB devdax namespace (align=2097152) on Intel Ice Lake server. Test procedure: Rebind the nd_pmem and dax_pmem driver 30 times and collect the memmap initialization time from the pr_debug() output of memmap_init_zone_device(). Base(v7.3-rc1): Average of rebinds for nd_pmem driver: 221.07 ms Average of rebinds for dax_pmem driver: 191.20 ms With this patch and its prerequisites applied: Average of rebinds for nd_pmem driver: 150.40 ms Average of rebinds for dax_pmem driver: 161.83 ms This reduces the average memmap initialization time measured during rebind by about 32.0% for nd_pmem and 15.4% for dax_pmem. Link: https://lore.kernel.org/20260831111638.76012-7-lizhe.67@bytedance.com Signed-off-by: Li Zhe Signed-off-by: Andrew Morton Cc: Alistair Popple Cc: Arnd Bergmann Cc: Balbir Singh Cc: "Borislav Petkov (AMD)" Cc: Dave Hansen Cc: David Hildenbrand (Arm) Cc: Ingo Molnar Cc: Kees Cook Cc: Mike Rapoport (Microsoft) Cc: Muchun Song --- mm/mm_init.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/mm_init.c b/mm/mm_init.c index e2b0952023b3b5..97e0158d2aca5b 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -1037,7 +1037,7 @@ static void zone_device_page_init_from_template(struct page *page, if (!is_highmem_idx(ZONE_DEVICE)) set_page_address(template, __va(pfn << PAGE_SHIFT)); #endif - memcpy(page, template, sizeof(*page)); + memcpy_nontemporal(page, template, sizeof(*page)); } /* From a080be1b0a9f2ed05dcf08a61d327d640aae090c Mon Sep 17 00:00:00 2001 From: Li Zhe Date: Mon, 31 Aug 2026 19:16:38 +0800 Subject: [PATCH 0662/1352] x86/string: extend memcpy_flushcache() fixed-size fastpaths The x86 memcpy_nontemporal() helper maps to memcpy_flushcache(), and the ZONE_DEVICE template-copy path uses it to copy one struct page at a time. The relevant copy size is sizeof(struct page). On x86_64, the base struct page layout is 64 bytes. Adding either the KMSAN metadata pointers or an out-of-flags last_cpupid field can make it 80 bytes after alignment, and enabling both can make it 96 bytes. memcpy_flushcache() currently only has inline fixed-size cases for 4, 8, and 16 bytes. As a result, these constant-sized struct page copies fall through to __memcpy_flushcache() even though the compiler knows the copy size at the call site. Add fixed-size MOVNTI cases up to 96 bytes so the ZONE_DEVICE template-copy path can keep these struct page copies in the inline memcpy_flushcache() path. This matters for ZONE_DEVICE memmap initialization because the copy happens once per initialized struct page. For a 100 GB fsdax namespace with map=dev, this is about 25 million struct page copies during nd_pmem binding or rebinding. Tested in a VM with a 100 GB fsdax namespace device configured with map=dev and a 100 GB devdax namespace (align=2097152) on Intel Ice Lake server. Test procedure: Rebind the nd_pmem and dax_pmem drivers 30 times and collect the memmap initialization time from the pr_debug() output of memmap_init_zone_device(). With memcpy_nontemporal() used by the ZONE_DEVICE template-copy path: Average of rebinds for nd_pmem driver: 150.40 ms Average of rebinds for dax_pmem driver: 161.83 ms With this x86 fixed-size fastpath patch applied: Average of rebinds for nd_pmem driver: 71.93 ms Average of rebinds for dax_pmem driver: 87.37 ms This further reduces the average memmap initialization time measured during rebind by about 52.2% for nd_pmem and 46.0% for dax_pmem. Link: https://lore.kernel.org/20260831111638.76012-8-lizhe.67@bytedance.com Signed-off-by: Li Zhe Signed-off-by: Andrew Morton Suggested-by: Borislav Petkov Acked-by: Borislav Petkov (AMD) Acked-by: Dave Hansen Cc: Alistair Popple Cc: Arnd Bergmann Cc: Balbir Singh Cc: David Hildenbrand (Arm) Cc: Ingo Molnar Cc: Kees Cook Cc: Mike Rapoport (Microsoft) Cc: Muchun Song --- arch/x86/include/asm/string_64.h | 71 +++++++++++++++++++++++++------- 1 file changed, 56 insertions(+), 15 deletions(-) diff --git a/arch/x86/include/asm/string_64.h b/arch/x86/include/asm/string_64.h index 21ae515ae35a3d..831d3dda3b380e 100644 --- a/arch/x86/include/asm/string_64.h +++ b/arch/x86/include/asm/string_64.h @@ -82,23 +82,64 @@ int strcmp(const char *cs, const char *ct); #ifdef CONFIG_ARCH_HAS_UACCESS_FLUSHCACHE #define __HAVE_ARCH_MEMCPY_FLUSHCACHE 1 void __memcpy_flushcache(void *dst, const void *src, size_t cnt); -static __always_inline void memcpy_flushcache(void *dst, const void *src, size_t cnt) + +static __always_inline void movnti_4(void *dst, const void *src) +{ + asm volatile("movntil %1, %0" + : "=m"(*(u32 *)dst) + : "r"(*(const u32 *)src) + : "memory"); +} + +static __always_inline void movnti_8(void *dst, const void *src) +{ + asm volatile("movntiq %1, %0" + : "=m"(*(u64 *)dst) + : "r"(*(const u64 *)src) + : "memory"); +} + +static __always_inline void movnti_16(void *dst, const void *src) +{ + movnti_8(dst, src); + movnti_8(dst + 8, src + 8); +} + +static __always_inline void movnti_32(void *dst, const void *src) +{ + movnti_16(dst, src); + movnti_16(dst + 16, src + 16); +} + +static __always_inline void movnti_64(void *dst, const void *src) +{ + movnti_32(dst, src); + movnti_32(dst + 32, src + 32); +} + +static __always_inline void memcpy_flushcache(void *dst, const void *src, + size_t cnt) { - if (__builtin_constant_p(cnt)) { - switch (cnt) { - case 4: - asm ("movntil %1, %0" : "=m"(*(u32 *)dst) : "r"(*(u32 *)src)); - return; - case 8: - asm ("movntiq %1, %0" : "=m"(*(u64 *)dst) : "r"(*(u64 *)src)); - return; - case 16: - asm ("movntiq %1, %0" : "=m"(*(u64 *)dst) : "r"(*(u64 *)src)); - asm ("movntiq %1, %0" : "=m"(*(u64 *)(dst + 8)) : "r"(*(u64 *)(src + 8))); - return; - } + if (!__builtin_constant_p(cnt)) + return __memcpy_flushcache(dst, src, cnt); + + /* + * The relevant fixed-size copies here are the x86_64 struct page sizes: + * 64, 80, and 96 bytes. Keep 32-byte and 48-byte copies inline as well + * instead of sending those nearby fixed-size cases back to + * __memcpy_flushcache(). + */ + switch (cnt) { + case 4: movnti_4(dst, src); break; + case 8: movnti_8(dst, src); break; + case 16: movnti_16(dst, src); break; + case 32: movnti_32(dst, src); break; + case 48: movnti_32(dst, src); movnti_16(dst + 32, src + 32); break; + case 64: movnti_64(dst, src); break; + case 80: movnti_64(dst, src); movnti_16(dst + 64, src + 64); break; + case 96: movnti_64(dst, src); movnti_32(dst + 64, src + 64); break; + default: __memcpy_flushcache(dst, src, cnt); break; } - __memcpy_flushcache(dst, src, cnt); } #define memcpy_nontemporal memcpy_nontemporal From 2804cfed5771f4d772ff245f6e0f32bbccbf335f Mon Sep 17 00:00:00 2001 From: Dev Jain Date: Mon, 31 Aug 2026 08:28:47 +0000 Subject: [PATCH 0663/1352] mm/rmap: remove stale hugetlb check in try_to_unmap_one Post commit d4ec5572825a ("mm/rmap: add try_to_unmap_poisoned_hugetlb_one") try_to_unmap_one() cannot be called with a hugetlb folio. Therefore remove the folio_test_hugetlb() check. Link: https://lore.kernel.org/20260831082849.3573957-1-dev.jain@arm.com Signed-off-by: Dev Jain Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Lance Yang Reviewed-by: Kunwu Chan Acked-by: David Hildenbrand (Arm) Cc: Harry Yoo Cc: Jann Horn Cc: Liam R. Howlett Cc: Rik van Riel Cc: Vlastimil Babka --- mm/rmap.c | 7 ++----- 1 file changed, 2 insertions(+), 5 deletions(-) diff --git a/mm/rmap.c b/mm/rmap.c index 3c67ad0e95620f..0a3952706faf5c 100644 --- a/mm/rmap.c +++ b/mm/rmap.c @@ -2301,11 +2301,8 @@ static bool try_to_unmap_one(struct folio *folio, struct vm_area_struct *vma, VM_BUG_ON_FOLIO(!pvmw.pte, folio); address = pvmw.address; - if (folio_test_hugetlb(folio)) { - pteval = huge_ptep_get(mm, address, pvmw.pte); - } else { - pteval = ptep_get(pvmw.pte); - } + pteval = ptep_get(pvmw.pte); + if (likely(pte_present(pteval))) { pfn = pte_pfn(pteval); } else { From 2232199095d1602864048b43b2a083616dced15f Mon Sep 17 00:00:00 2001 From: Sarthak Sharma Date: Mon, 31 Aug 2026 15:43:04 +0530 Subject: [PATCH 0664/1352] mm/gup_test: report actual pinned bytes __gup_test_ioctl() advances addr to the end of the current batch before checking if GUP pinned the entire requested batch. If GUP pins more than 0 pages but less than the requested batch size, addr still advances by the requested batch size. The next iteration detects the partial pinning and breaks out of the loop. Again gup->size is calculated using addr - gup->addr, so it also includes the unpinned pages of the requested batch. Calculate gup->size using the actual number of pages pinned multiplied by PAGE_SIZE. Link: https://lore.kernel.org/20260831101304.162867-1-sarthak.sharma@arm.com Fixes: 64c349f4ae78 ("mm: add infrastructure for get_user_pages_fast() benchmarking") Signed-off-by: Sarthak Sharma Signed-off-by: Andrew Morton Reviewed-by: Kiryl Shutsemau (Meta) Acked-by: David Hildenbrand (Arm) Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu --- mm/gup_test.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/gup_test.c b/mm/gup_test.c index 44c1cdfb9c3717..185ba3bb8ed10b 100644 --- a/mm/gup_test.c +++ b/mm/gup_test.c @@ -188,7 +188,7 @@ static int __gup_test_ioctl(unsigned int cmd, nr_pages = i; gup->get_delta_usec = ktime_us_delta(end_time, start_time); - gup->size = addr - gup->addr; + gup->size = nr_pages * PAGE_SIZE; /* * Take an un-benchmark-timed moment to verify DMA pinned From dd8332301be00071fbf3003724dec80619a0844f Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 31 Aug 2026 10:15:13 +0100 Subject: [PATCH 0665/1352] mm/huge_memory: do not touch frozen folios in deferred_split_isolate() Patch series "Fix deferred_split_isolate() and drop the split workaround", v2. deferred_split_isolate() probes each queued folio with folio_try_get(). folio_try_get() failure is treated as a lost race with folio_put(). It leads to wrong results when !folio_try_get() was not caused by folio_put(): for a frozen folio, PG_partially_mapped gets wrongfully cleared and the folio dropped from the queue. It came up in the review of my collapse RFC series: https://lore.kernel.org/all/20260824131224.73344-1-lance.yang@linux.dev/ The bug is inert in upstream code: - __folio_split() works around it; - __folio_migrate_mapping() freezes a folio it is about to replace; - reclaim freezes only what try_to_unmap() already unmapped. No cc:stable needed. But my collapse rework steps on it, so it is worth fixing. The branch the first patch removes also hid an inert, pre-existing bug in the zone device path: https://lore.kernel.org/all/20260827163838.1813081-1-usama.arif@linux.dev/ The first patch fixes deferred_split_isolate(). The second patch removes the workaround for this deferred_split_isolate() behaviour from __folio_freeze_and_split_unmapped(). Tested in a VM: split_huge_page_test, folio_split_race_test and cow pass. Also ran a test that leaves 16 partially mapped THPs on the deferred split queue and drives thp-deferred_split through debugfs, checking nr_anon_partially_mapped. This patch (of 2): deferred_split_isolate() probes each queued folio with folio_try_get(). folio_try_get() failure is treated as a lost race with folio_put(): clear PG_partially_mapped, correct MTHP_STAT_NR_ANON_PARTIALLY_MAPPED, take the folio off the queue. The folio_put() race is the most common case for !folio_try_get(), but it is not the only option. Another scenario is folio_ref_freeze(). A zero refcount in such cases does not mean the folio is going away. It means "don't touch me" and current deferred_split_isolate() doesn't respect it. It can lead to unqueueing folios from the deferred list for no reason: CPU 0 CPU 1 --------------------------- ------------------------------ freeze a mapped folio deferred_split_scan() folio_ref_freeze() folio_try_get() fails folio_clear_partially_mapped() NR_ANON_PARTIALLY_MAPPED-- folio off the queue give up, put it back folio_ref_unfreeze() The folio is still partially mapped, but it is no longer a split candidate. Nothing queues it again until part of it is unmapped once more. Skip the folio instead: whoever freezes the folio, owns it and owner is responsible for its fate. It also covers the folio_put() case: __folio_put() unqueues the folio via folio_unqueue_deferred_split(). Nothing is lost by skipping. Everything that frees a queued folio unqueues it first, and folio_unqueue_deferred_split() clears PG_partially_mapped and brings MTHP_STAT_NR_ANON_PARTIALLY_MAPPED down on the way: __folio_put(), folios_put_refs() mm/folio.c __folio_migrate_mapping() mm/migrate.c shrink_folio_list() mm/vmscan.c __folio_freeze_and_split_unmapped() does the same by hand, under the list_lru lock it holds across the freeze. A freeze that ends in folio_ref_unfreeze() leaves a folio that is still partially mapped and still belongs on the queue. Link: https://lore.kernel.org/20260831091514.1879786-1-kirill@shutemov.name Link: https://lore.kernel.org/20260831091514.1879786-2-kirill@shutemov.name Fixes: 8422acdc97ed ("mm: introduce a pageflag for partially mapped folios") Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Reported-by: Lance Yang Closes: https://lore.kernel.org/all/20260824131224.73344-1-lance.yang@linux.dev/ Reviewed-by: Zi Yan Reviewed-by: Johannes Weiner Reviewed-by: Lance Yang Acked-by: Usama Arif Reviewed-by: Baolin Wang Acked-by: David Hildenbrand (Arm) Assisted-by: Claude-Code:claude-opus-5 Cc: Balbir Singh Cc: Barry Song Cc: Dev Jain Cc: Hugh Dickins Cc: Kairui Song Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Ryan Roberts --- mm/huge_memory.c | 19 ++++--------------- 1 file changed, 4 insertions(+), 15 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 54494c3fa9835e..779c02e0bf6cc3 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -4638,22 +4638,11 @@ static enum lru_status deferred_split_isolate(struct list_head *item, struct folio *folio = container_of(item, struct folio, _deferred_list); struct list_head *freeable = cb_arg; - if (folio_try_get(folio)) { - list_lru_isolate_move(lru, item, freeable); - return LRU_REMOVED; - } + /* Lost race to folio_put() or the folio is under folio_ref_freeze() */ + if (!folio_try_get(folio)) + return LRU_SKIP; - /* - * We lost race with folio_put(). Read folio state before the - * isolate: folio_unqueue_deferred_split() checks list_empty() - * locklessly, so once removed the folio can be freed any time. - */ - if (folio_test_partially_mapped(folio)) { - folio_clear_partially_mapped(folio); - mod_mthp_stat(folio_order(folio), - MTHP_STAT_NR_ANON_PARTIALLY_MAPPED, -1); - } - list_lru_isolate(lru, item); + list_lru_isolate_move(lru, item, freeable); return LRU_REMOVED; } From 4c6f070bea5a07c166a7ccca86d7ea0266683a09 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 31 Aug 2026 10:15:14 +0100 Subject: [PATCH 0666/1352] mm/huge_memory: dequeue the deferred split after the split freeze __folio_freeze_and_split_unmapped() takes the deferred split list_lru lock across the freeze. It is only there to stop deferred_split_scan() from touching the folio under split. With deferred_split_isolate() fixed, the workaround can be dropped. Unqueue the folio after folio_ref_freeze(), the way __folio_migrate_mapping() does: folio_unqueue_deferred_split() needs a zero refcount and a memcg still set, and both hold there. If the split is called from deferred_split_scan(), the unqueue is a no-op -- the folio is already removed from the list. But PG_partially_mapped is still set, so it has to be cleared here or MTHP_STAT_NR_ANON_PARTIALLY_MAPPED never comes back down. Link: https://lore.kernel.org/20260831091514.1879786-3-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Johannes Weiner Acked-by: David Hildenbrand (Arm) Reviewed-by: Lance Yang Assisted-by: Claude-Code:claude-opus-5 Cc: Balbir Singh Cc: Baolin Wang Cc: Barry Song Cc: Dev Jain Cc: Hugh Dickins Cc: Kairui Song Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Ryan Roberts Cc: Usama Arif --- mm/huge_memory.c | 46 ++++++++++++++-------------------------------- 1 file changed, 14 insertions(+), 32 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 779c02e0bf6cc3..c5d11147b69aec 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -3979,41 +3979,27 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n struct folio *end_folio = folio_next(folio); struct folio *new_folio, *next; int old_order = folio_order(folio); - struct list_lru_one *lru; - bool dequeue_deferred; int ret = 0; VM_WARN_ON_ONCE(!mapping && end); - /* - * If this folio can be on the deferred split queue, lock out - * the shrinker before freezing the ref. If the shrinker sees - * a 0-ref folio, it assumes it beat folio_put() to the list - * lock and must clean up the LRU state - the same dequeue we - * will do below as part of the split. - */ - dequeue_deferred = folio_test_anon(folio) && old_order > 1; - if (dequeue_deferred) { - struct mem_cgroup *memcg; - - rcu_read_lock(); - memcg = folio_memcg(folio); - lru = list_lru_lock(&deferred_split_lru, - folio_nid(folio), &memcg); - } + if (folio_ref_freeze(folio, folio_cache_ref_count(folio) + 1)) { struct swap_cluster_info *ci = NULL; struct lruvec *lruvec; - if (dequeue_deferred) { - __list_lru_del(&deferred_split_lru, lru, - &folio->_deferred_list, folio_nid(folio)); - if (folio_test_partially_mapped(folio)) { - folio_clear_partially_mapped(folio); - mod_mthp_stat(old_order, - MTHP_STAT_NR_ANON_PARTIALLY_MAPPED, -1); - } - list_lru_unlock(lru); - rcu_read_unlock(); + /* Take off the deferred split queue while frozen and memcg set */ + folio_unqueue_deferred_split(folio); + + /* + * deferred_split_scan() takes the folio off the queue before it + * splits it, so the unqueue above finds an empty list and + * leaves PG_partially_mapped set. + * Clear it here: the flag does not survive the split. + */ + if (folio_test_partially_mapped(folio)) { + folio_clear_partially_mapped(folio); + mod_mthp_stat(old_order, + MTHP_STAT_NR_ANON_PARTIALLY_MAPPED, -1); } if (mapping) { @@ -4115,10 +4101,6 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n if (ci) swap_cluster_unlock(ci); } else { - if (dequeue_deferred) { - list_lru_unlock(lru); - rcu_read_unlock(); - } return -EAGAIN; } From 53e5b7c17aebe40b15e675c951bcb2ff1ebb8130 Mon Sep 17 00:00:00 2001 From: Longlong Xia Date: Mon, 31 Aug 2026 21:35:18 +0800 Subject: [PATCH 0667/1352] mm/hugetlb: preserve source surplus accounting during demotion Patch series "mm/hugetlb: fix surplus accounting and availability checks during demotion", v2. Fix surplus accounting and availability checks in the hugetlb demote path. Patch 1 fixes source hstate accounting when the free folio selected for demotion accounts for a surplus page. Patch 2 prevents demotion from removing free huge pages that back reservations. Both fixes were tested with x86_64 QEMU guests. The commands below use: hstate=/sys/kernel/mm/hugepages/hugepages-1048576kB Patch 1: surplus accounting A vmemmap restoration failure is difficult to trigger deterministically. For this test only, add a one-shot fault injection that makes the first attempt to restore the vmemmap of an optimized 1 GiB folio fail: /* TEST ONLY: fail the first optimized 1G folio restore. */ static atomic_t fail_next_1g_restore = ATOMIC_INIT(1); /* In __hugetlb_vmemmap_restore_folio(). */ if (huge_page_size(h) == SZ_1G && atomic_cmpxchg(&fail_next_1g_restore, 1, 0) == 1) { pr_info("TEST ONLY: forcing one 1G vmemmap " "restore failure\n"); return -ENOMEM; } The injection does not modify the demotion or accounting code. It is one-shot so that the later restore performed during demotion can succeed. 1. Boot QEMU with: hugepagesz=1G hugepages=0 hugetlb_cma=1G hugetlb_free_vmemmap=on 2. Enable overcommit: echo 1 > "$hstate/nr_overcommit_hugepages" 3. Allocate one 1 GiB huge page: nr=1 surplus=1 free=0 resv=0 4. Unmap it. The forced restoration failure leaves the folio on the freelist while it is still accounted as surplus: nr=1 surplus=1 free=1 resv=0 5. Demote one page: echo 1 > "$hstate/demote" Before this fix: nr=0 surplus=1 free=0 resv=0 surplus > nr After this fix: nr=0 surplus=0 free=0 resv=0 Patch 2: cap demotion This reproducer requires no kernel instrumentation. 1. Boot QEMU with: hugepagesz=1G hugepages=2 nr=2 surplus=0 free=2 resv=0 2. Reserve one 1 GiB huge page with an untouched hugetlbfs mapping: nr=2 surplus=0 free=2 resv=1 3. Request demotion of two pages: echo 2 > "$hstate/demote" Before this fix: nr=0 surplus=0 free=0 resv=1 resv > free After this fix: nr=1 surplus=0 free=1 resv=1 resv == free 4. Touch the reserved page and let the process exit. Before this fix, the access fails with SIGBUS and leaves: nr=0 surplus=0 free=0 resv=0 After this fix, the access succeeds and leaves: nr=1 surplus=0 free=1 resv=0 This patch (of 2): demote_pool_huge_page() currently removes every source folio as a persistent folio. A free folio can instead account for one of the source hstate's surplus pages, for example after a vmemmap restoration failure. Removing such a folio without adjusting surplus_huge_pages makes the persistent count underflow, and later subtracting it from max_huge_pages can underflow that counter as well. Classify selected folios against the node's surplus count while holding hugetlb_lock, and preserve that classification on rollback. Track the number of successfully demoted persistent folios separately so only those folios reduce the source max_huge_pages target. All successfully demoted folios still increase the destination target because the new destination folios are added as persistent pages. Testing: Tested on an x86_64 QEMU guest booted with: hugepagesz=1G hugepages=0 hugetlb_cma=1G hugetlb_free_vmemmap=on For testing only, add a one-shot fault injection that makes the first call to __hugetlb_vmemmap_restore_folio() for an optimized 1 GiB folio return -ENOMEM. Set nr_overcommit_hugepages to 1, then allocate one 1 GiB huge page: nr=1 surplus=1 free=0 resv=0 Unmap it. The failed restoration leaves the folio on the freelist while it is still accounted as surplus: nr=1 surplus=1 free=1 resv=0 Demote one page. Before this fix, the result is: nr=0 surplus=1 free=0 resv=0 After this fix, the result is: nr=0 surplus=0 free=0 resv=0 The fault injection is one-shot, so the restore performed during demotion can succeed. Link: https://lore.kernel.org/20260831133519.2505020-2-xialonglong2025@163.com Fixes: 8531fc6f52f5 ("hugetlb: add hugetlb demote page support") Signed-off-by: Longlong Xia Signed-off-by: Andrew Morton Assisted-by: Codex:gpt-5.6-sol Cc: David Hildenbrand Cc: Muchun Song Cc: Oscar Salvador Cc: Yu Zhao --- mm/hugetlb.c | 35 +++++++++++++++++++++++++++++++---- 1 file changed, 31 insertions(+), 4 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 03182cc28a7dbc..ea79d31e6160ab 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -4011,6 +4011,7 @@ long demote_pool_huge_page(struct hstate *src, nodemask_t *nodes_allowed, struct hstate *dst; long rc = 0; long nr_demoted = 0; + long nr_persistent = 0; lockdep_assert_held(&hugetlb_lock); @@ -4023,22 +4024,40 @@ long demote_pool_huge_page(struct hstate *src, nodemask_t *nodes_allowed, for_each_node_mask_to_free(src, nr_nodes, node, nodes_allowed) { LIST_HEAD(list); + LIST_HEAD(surplus_list); struct folio *folio, *next; list_for_each_entry_safe(folio, next, &src->hugepage_freelists[node], lru) { + bool adjust_surplus; + if (folio_test_hwpoison(folio)) continue; - remove_hugetlb_folio(src, folio, false); - list_add(&folio->lru, &list); + /* Surplus accounting is maintained per node, not per folio. */ + adjust_surplus = src->surplus_huge_pages_node[node] > 0; + remove_hugetlb_folio(src, folio, adjust_surplus); + list_add(&folio->lru, adjust_surplus ? &surplus_list : &list); + if (!adjust_surplus) + nr_persistent++; if (++nr_demoted == nr_to_demote) break; } + if (list_empty(&list) && list_empty(&surplus_list)) + continue; + spin_unlock_irq(&hugetlb_lock); - rc = demote_free_hugetlb_folios(src, dst, &list); + if (!list_empty(&list)) + rc = demote_free_hugetlb_folios(src, dst, &list); + if (!list_empty(&surplus_list)) { + long tmp_rc; + + tmp_rc = demote_free_hugetlb_folios(src, dst, &surplus_list); + if (rc >= 0) + rc = tmp_rc; + } spin_lock_irq(&hugetlb_lock); @@ -4046,6 +4065,14 @@ long demote_pool_huge_page(struct hstate *src, nodemask_t *nodes_allowed, list_del(&folio->lru); add_hugetlb_folio(src, folio, false); + nr_demoted--; + nr_persistent--; + } + + list_for_each_entry_safe(folio, next, &surplus_list, lru) { + list_del(&folio->lru); + add_hugetlb_folio(src, folio, true); + nr_demoted--; } @@ -4057,7 +4084,7 @@ long demote_pool_huge_page(struct hstate *src, nodemask_t *nodes_allowed, * Not absolutely necessary, but for consistency update max_huge_pages * based on pool changes for the demoted page. */ - src->max_huge_pages -= nr_demoted; + src->max_huge_pages -= nr_persistent; dst->max_huge_pages += nr_demoted << (huge_page_order(src) - huge_page_order(dst)); if (rc < 0) From 72550b508a86b6014fc3f61084e4cc3abc113628 Mon Sep 17 00:00:00 2001 From: Longlong Xia Date: Mon, 31 Aug 2026 21:35:19 +0800 Subject: [PATCH 0668/1352] mm/hugetlb: cap demotion at currently available free pages Demotion must not remove free huge pages that back existing reservations. The sysfs path checks whether any page is available, but passes the entire request to demote_pool_huge_page(). For example, with two free pages and one reservation, a request for two pages removes both and leaves the reservation without a backing page. Cap the sysfs request by both global availability and the selected node's free pages. Recheck global availability in demote_pool_huge_page() before each node batch because that function drops hugetlb_lock while restoring vmemmap and reservations can change before the next batch. Testing: Tested on an x86_64 QEMU guest booted with: hugepagesz=1G hugepages=2 Reserve one 1 GiB huge page with an untouched hugetlbfs mapping: nr=2 surplus=0 free=2 resv=1 Request demotion of two pages. Before this fix, both free pages are demoted: nr=0 surplus=0 free=0 resv=1 Touching the reserved mapping then fails with SIGBUS. After this fix, the request is capped at the single available page: nr=1 surplus=0 free=1 resv=1 Touching the reserved mapping succeeds. After the process exits, the counters are: nr=1 surplus=0 free=1 resv=0 Link: https://lore.kernel.org/20260831133519.2505020-3-xialonglong2025@163.com Fixes: c0f398c3b2cf ("mm/hugetlb_vmemmap: batch HVO work when demoting") Signed-off-by: Longlong Xia Signed-off-by: Andrew Morton Assisted-by: Codex:gpt-5.6-sol Cc: David Hildenbrand Cc: Muchun Song Cc: Oscar Salvador Cc: Yu Zhao --- mm/hugetlb.c | 22 +++++++++++++++++++++- mm/hugetlb_sysfs.c | 10 +++++----- 2 files changed, 26 insertions(+), 6 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index ea79d31e6160ab..7edc2a860a4007 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -4026,6 +4026,26 @@ long demote_pool_huge_page(struct hstate *src, nodemask_t *nodes_allowed, LIST_HEAD(list); LIST_HEAD(surplus_list); struct folio *folio, *next; + unsigned long nr_available, nr_target; + + /* + * Re-check available each node batch: the previous + * batch released hugetlb_lock for vmemmap restore/split, + * and a new reservation could have been added in that + * window, shrinking the budget. available is global + * (resv is not per-node), so 0 means no node can + * contribute -- stop the whole scan. + */ + nr_available = available_huge_pages(src); + if (!nr_available) + break; + + /* + * Cap this batch at the current budget; expressed as a + * cumulative stop point because nr_demoted is running. + */ + nr_target = nr_demoted + min_t(unsigned long, + nr_to_demote - nr_demoted, nr_available); list_for_each_entry_safe(folio, next, &src->hugepage_freelists[node], lru) { bool adjust_surplus; @@ -4040,7 +4060,7 @@ long demote_pool_huge_page(struct hstate *src, nodemask_t *nodes_allowed, if (!adjust_surplus) nr_persistent++; - if (++nr_demoted == nr_to_demote) + if (++nr_demoted == nr_target) break; } diff --git a/mm/hugetlb_sysfs.c b/mm/hugetlb_sysfs.c index 79ece91406bfa4..326a54b4d991c9 100644 --- a/mm/hugetlb_sysfs.c +++ b/mm/hugetlb_sysfs.c @@ -211,15 +211,15 @@ static ssize_t demote_store(struct kobject *kobj, * Check for available pages to demote each time thorough the * loop as demote_pool_huge_page will drop hugetlb_lock. */ + nr_available = h->free_huge_pages - h->resv_huge_pages; if (nid != NUMA_NO_NODE) - nr_available = h->free_huge_pages_node[nid]; - else - nr_available = h->free_huge_pages; - nr_available -= h->resv_huge_pages; + nr_available = min(nr_available, + h->free_huge_pages_node[nid]); if (!nr_available) break; - rc = demote_pool_huge_page(h, n_mask, nr_demote); + rc = demote_pool_huge_page(h, n_mask, + min(nr_demote, nr_available)); if (rc < 0) { err = rc; break; From 695eb9b0e572e3e3eefc0028b648ef113fde2bad Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Mon, 31 Aug 2026 11:13:23 +0530 Subject: [PATCH 0669/1352] mm: make ptval_to_str() generally available Patch series "mm: Drop pxd_ERROR()". pxd_ERROR() macros have been provided by all platforms, which are very much identical and can be dropped off completely if these pgtable printing could be moved to callers in generic MM aka all pxd_clear_bad(). But first cleanups and re-organizations are required in some platforms that are using these macros internally. Afterwards [pte|pmd|pud|p4d|pgd]_ERROR() macros have been completely dropped from the entire tree. This patch (of 8): Move ptval_to_str() inside a header thus making the helper more generally available for new users which are being added later. While here, also move another related string size macro PTVAL_STR_MAX inside the header as well. Link: https://lore.kernel.org/20260831054331.625505-1-anshuman.khandual@arm.com Link: https://lore.kernel.org/20260831054331.625505-2-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: Lorenzo Stoakes Cc: Helge Deller Cc: Huacai Chen Cc: James Bottomley Cc: John Paul Adrian Glaubitz Cc: Lorenzo Stoakes Cc: Rich Felker Cc: Samuel Holland Cc: WANG Xuerui Cc: Yoshinori Sato Cc: Geert Uytterhoeven --- include/linux/pgtable.h | 14 ++++++++++++++ mm/memory.c | 15 +-------------- 2 files changed, 15 insertions(+), 14 deletions(-) diff --git a/include/linux/pgtable.h b/include/linux/pgtable.h index 8c093c119e5a82..e3c8ab96941c5e 100644 --- a/include/linux/pgtable.h +++ b/include/linux/pgtable.h @@ -2313,6 +2313,20 @@ static inline const char *pgtable_level_to_str(enum pgtable_level level) } } +void ptval_bytes_to_hex_str(char *buf, size_t buf_size, const void *entry, size_t entry_size); + +#define ptval_to_str(buf, val) \ + do { \ + auto __val = (val); \ + \ + ptval_bytes_to_hex_str((buf), sizeof(buf), &__val, sizeof(__val)); \ + } while (0) + +#if defined(__SIZEOF_INT128__) +#define PTVAL_STR_MAX (32 + 1) /* Max 128-bit value in hex + NUL */ +#else +#define PTVAL_STR_MAX (16 + 1) /* Max 64-bit value in hex + NUL */ +#endif #endif /* !__ASSEMBLER__ */ #if !defined(MAX_POSSIBLE_PHYSMEM_BITS) && !defined(CONFIG_64BIT) diff --git a/mm/memory.c b/mm/memory.c index bc14cae3c49d72..ec63dd6212ac5a 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -495,7 +495,7 @@ static inline void add_mm_rss_vec(struct mm_struct *mm, int *rss) /* Allow a burst of 60 bad page map reports per minute. */ static DEFINE_RATELIMIT_STATE(bad_page_map_ratelimit, 60 * HZ, 60); -static void ptval_bytes_to_hex_str(char *buf, size_t buf_size, const void *entry, size_t entry_size) +void ptval_bytes_to_hex_str(char *buf, size_t buf_size, const void *entry, size_t entry_size) { if (WARN_ON_ONCE(buf_size < entry_size * 2 + 1)) { snprintf(buf, buf_size, "overflow"); @@ -522,19 +522,6 @@ static void ptval_bytes_to_hex_str(char *buf, size_t buf_size, const void *entry } } -#define ptval_to_str(buf, val) \ - do { \ - auto __val = (val); \ - \ - ptval_bytes_to_hex_str((buf), sizeof(buf), &__val, sizeof(__val)); \ - } while (0) - -#if defined(__SIZEOF_INT128__) -#define PTVAL_STR_MAX (32 + 1) /* Max 128-bit value in hex + NUL */ -#else -#define PTVAL_STR_MAX (16 + 1) /* Max 64-bit value in hex + NUL */ -#endif - static void __print_bad_page_map_pgtable(struct mm_struct *mm, unsigned long addr) { char pgd_str[PTVAL_STR_MAX]; From afccb9a825f68a13cf7f73c330faa10f45d02be4 Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Mon, 31 Aug 2026 11:13:24 +0530 Subject: [PATCH 0670/1352] mm: stop using pxd_ERROR() pxd_ERROR() has been used in generic mm just to print the page table entry in pxd_clear_bad() before clearing those out with pxd_clear() later. These pxd_ERROR() macros have been provided by all platforms which basically did the same thing. Make pxd_clear_bad() use recently added ptval_to_str() instead for printing page table entries thus completely dropping dependency on platform provided pxd_ERROR() macros which can then be dropped off later on. Link: https://lore.kernel.org/20260831054331.625505-3-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: Lorenzo Stoakes Cc: Geert Uytterhoeven Cc: Helge Deller Cc: Huacai Chen Cc: James Bottomley Cc: John Paul Adrian Glaubitz Cc: Rich Felker Cc: Samuel Holland Cc: WANG Xuerui Cc: Yoshinori Sato --- mm/pgtable-generic.c | 20 ++++++++++++++++---- 1 file changed, 16 insertions(+), 4 deletions(-) diff --git a/mm/pgtable-generic.c b/mm/pgtable-generic.c index cd227fc05d2d8f..b45e891d1193fd 100644 --- a/mm/pgtable-generic.c +++ b/mm/pgtable-generic.c @@ -26,14 +26,20 @@ void pgd_clear_bad(pgd_t *pgd) { - pgd_ERROR(*pgd); + char str[PTVAL_STR_MAX]; + + ptval_to_str(str, pgd_val(*pgd)); + pr_err("bad pgd %s.\n", str); pgd_clear(pgd); } #ifndef __PAGETABLE_P4D_FOLDED void p4d_clear_bad(p4d_t *p4d) { - p4d_ERROR(*p4d); + char str[PTVAL_STR_MAX]; + + ptval_to_str(str, p4d_val(*p4d)); + pr_err("bad p4d %s.\n", str); p4d_clear(p4d); } #endif @@ -41,7 +47,10 @@ void p4d_clear_bad(p4d_t *p4d) #ifndef __PAGETABLE_PUD_FOLDED void pud_clear_bad(pud_t *pud) { - pud_ERROR(*pud); + char str[PTVAL_STR_MAX]; + + ptval_to_str(str, pud_val(*pud)); + pr_err("bad pud %s.\n", str); pud_clear(pud); } #endif @@ -53,7 +62,10 @@ void pud_clear_bad(pud_t *pud) */ void pmd_clear_bad(pmd_t *pmd) { - pmd_ERROR(*pmd); + char str[PTVAL_STR_MAX]; + + ptval_to_str(str, pmd_val(*pmd)); + pr_err("bad pmd %s.\n", str); pmd_clear(pmd); } From 07be469883b8a6c0099f019bf4b6aee3adf36703 Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Mon, 31 Aug 2026 11:13:25 +0530 Subject: [PATCH 0671/1352] loongarch/mm: stop using pte_ERROR() Directly use pr_err() in __set_fixmap() and drop pte_ERROR() which helps in eventually dropping pte_ERROR() macro across the tree. In this new printing __FILE__ and __LINE__ has been dropped because they are always the same and don't really add any value. The new ptval_to_str() helper is being used for converting pgtable entry value into a string. The error message itself has been cleaned up as well. Link: https://lore.kernel.org/20260831054331.625505-4-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: Huacai Chen Cc: WANG Xuerui Cc: Geert Uytterhoeven Cc: Helge Deller Cc: James Bottomley Cc: John Paul Adrian Glaubitz Cc: Lorenzo Stoakes Cc: Rich Felker Cc: Samuel Holland Cc: Yoshinori Sato --- arch/loongarch/include/asm/pgtable.h | 2 -- arch/loongarch/mm/init.c | 4 +++- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/arch/loongarch/include/asm/pgtable.h b/arch/loongarch/include/asm/pgtable.h index a05f6a4928dc68..eddd8906b77dfc 100644 --- a/arch/loongarch/include/asm/pgtable.h +++ b/arch/loongarch/include/asm/pgtable.h @@ -135,8 +135,6 @@ struct vm_area_struct; #define ptep_get(ptep) READ_ONCE(*(ptep)) #define pmdp_get(pmdp) READ_ONCE(*(pmdp)) -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %016lx.\n", __FILE__, __LINE__, pte_val(e)) #ifndef __PAGETABLE_PMD_FOLDED #define pmd_ERROR(e) \ pr_err("%s:%d: bad pmd %016lx.\n", __FILE__, __LINE__, pmd_val(e)) diff --git a/arch/loongarch/mm/init.c b/arch/loongarch/mm/init.c index 4b46c5d30708d8..f801f7097f0379 100644 --- a/arch/loongarch/mm/init.c +++ b/arch/loongarch/mm/init.c @@ -197,13 +197,15 @@ void __init __set_fixmap(enum fixed_addresses idx, phys_addr_t phys, pgprot_t flags) { unsigned long addr = __fix_to_virt(idx); + char str[PTVAL_STR_MAX]; pte_t *ptep; BUG_ON(idx <= FIX_HOLE || idx >= __end_of_fixed_addresses); ptep = populate_kernel_pte(addr); if (!pte_none(ptep_get(ptep))) { - pte_ERROR(*ptep); + ptval_to_str(str, pte_val(*ptep)); + pr_err("unexpected set PTE at %lx in %s: %s\n", addr, __func__, str); return; } From 6565bc8ad1b92a27a51af676344573136b0c0c04 Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Mon, 31 Aug 2026 11:13:26 +0530 Subject: [PATCH 0672/1352] parisc/mm: directly use generic [pmd|pgd]_clear_bad() Drop [pmd|pgd]_ERROR() followed by [pmd|pgd]_clear() instances. But instead directly use semantically equivalent generic helpers [pmd|pgd]_clear_bad() in unmap_uncached_[pte|pmd]() which helps in dropping their corresponding [pmd|pgd]_ERROR() macros across the tree. Link: https://lore.kernel.org/20260831054331.625505-5-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Signed-off-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: James E.J. Bottomley Cc: Helge Deller Cc: Geert Uytterhoeven Cc: Huacai Chen Cc: John Paul Adrian Glaubitz Cc: Lorenzo Stoakes Cc: Rich Felker Cc: Samuel Holland Cc: WANG Xuerui Cc: Yoshinori Sato --- arch/parisc/kernel/pci-dma.c | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/arch/parisc/kernel/pci-dma.c b/arch/parisc/kernel/pci-dma.c index bf9f192c826ebe..84e7309a826696 100644 --- a/arch/parisc/kernel/pci-dma.c +++ b/arch/parisc/kernel/pci-dma.c @@ -160,8 +160,7 @@ static inline void unmap_uncached_pte(pmd_t * pmd, unsigned long vaddr, if (pmd_none(*pmd)) return; if (pmd_bad(*pmd)) { - pmd_ERROR(*pmd); - pmd_clear(pmd); + pmd_clear_bad(pmd); return; } pte = pte_offset_kernel(pmd, vaddr); @@ -196,8 +195,7 @@ static inline void unmap_uncached_pmd(pgd_t * dir, unsigned long vaddr, if (pgd_none(*dir)) return; if (pgd_bad(*dir)) { - pgd_ERROR(*dir); - pgd_clear(dir); + pgd_clear_bad(dir); return; } pmd = pmd_offset(pud_offset(p4d_offset(dir, vaddr), vaddr), vaddr); From 0976c0a1ccab216978849e1627ae7f4a9b5eb1a5 Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Mon, 31 Aug 2026 11:13:27 +0530 Subject: [PATCH 0673/1352] sh/mm: stop using pte_ERROR() Directly use pr_err() in set_pte_phys() and drop pte_ERROR() which helps in eventually dropping pte_ERROR() macro across the tree. In this new printing __FILE__ and __LINE__ has been dropped because they are always the same and don't really add any value. Besides ptrval_to_str() has been able to handle different PTE representation with and without CONFIG_X2TLB, which helped in unifying error message printing. Link: https://lore.kernel.org/20260831054331.625505-6-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: Yoshinori Sato Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Geert Uytterhoeven Cc: Helge Deller Cc: Huacai Chen Cc: James Bottomley Cc: Lorenzo Stoakes Cc: Samuel Holland Cc: WANG Xuerui --- arch/sh/include/asm/pgtable_32.h | 5 ----- arch/sh/mm/init.c | 6 +++++- 2 files changed, 5 insertions(+), 6 deletions(-) diff --git a/arch/sh/include/asm/pgtable_32.h b/arch/sh/include/asm/pgtable_32.h index 5f51af18997b57..c8eb9a7a4c4c78 100644 --- a/arch/sh/include/asm/pgtable_32.h +++ b/arch/sh/include/asm/pgtable_32.h @@ -401,14 +401,9 @@ static inline unsigned long pmd_page_vaddr(pmd_t pmd) #define pmd_page(pmd) (virt_to_page(pmd_val(pmd))) #ifdef CONFIG_X2TLB -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %p(%08lx%08lx).\n", __FILE__, __LINE__, \ - &(e), (e).pte_high, (e).pte_low) #define pgd_ERROR(e) \ printk("%s:%d: bad pgd %016llx.\n", __FILE__, __LINE__, pgd_val(e)) #else -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %08lx.\n", __FILE__, __LINE__, pte_val(e)) #define pgd_ERROR(e) \ printk("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) #endif diff --git a/arch/sh/mm/init.c b/arch/sh/mm/init.c index 110308bdef01d0..9466ae6f9f164d 100644 --- a/arch/sh/mm/init.c +++ b/arch/sh/mm/init.c @@ -84,7 +84,11 @@ static void set_pte_phys(unsigned long addr, unsigned long phys, pgprot_t prot) pte = __get_pte_phys(addr); if (!pte_none(*pte)) { - pte_ERROR(*pte); + char str[PTVAL_STR_MAX]; + + ptval_to_str(str, pte_val(*pte)); + pr_err("unexpected set PTE at %lx in %s: bad pte %p(%s).\n", + addr, __func__, pte, str); return; } From 094b0f831e7df804dd1eec1ac8607345dbbb5bdc Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Mon, 31 Aug 2026 11:13:28 +0530 Subject: [PATCH 0674/1352] sh/mm: stop using [p4d|pud|pmd]_ERROR() Stop using [p4d|pud|pmd]_ERROR() in __get_pte_phys() as the pgtable entries are known to be NULL and hence could not really be accessed. Link: https://lore.kernel.org/20260831054331.625505-7-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Signed-off-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: Yoshinori Sato Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Geert Uytterhoeven Cc: Helge Deller Cc: Huacai Chen Cc: James Bottomley Cc: Lorenzo Stoakes Cc: Samuel Holland Cc: WANG Xuerui --- arch/sh/mm/init.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/arch/sh/mm/init.c b/arch/sh/mm/init.c index 9466ae6f9f164d..8d65e60688dcbd 100644 --- a/arch/sh/mm/init.c +++ b/arch/sh/mm/init.c @@ -59,19 +59,19 @@ static pte_t *__get_pte_phys(unsigned long addr) p4d = p4d_alloc(NULL, pgd, addr); if (unlikely(!p4d)) { - p4d_ERROR(*p4d); + pr_err("allocating p4d table failed\n"); return NULL; } pud = pud_alloc(NULL, p4d, addr); if (unlikely(!pud)) { - pud_ERROR(*pud); + pr_err("allocating pud table failed\n"); return NULL; } pmd = pmd_alloc(NULL, pud, addr); if (unlikely(!pmd)) { - pmd_ERROR(*pmd); + pr_err("allocating pmd table failed\n"); return NULL; } From 7df03decfb9667246295dbb53839a5b6a188fe43 Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Mon, 31 Aug 2026 11:13:29 +0530 Subject: [PATCH 0675/1352] sh/mm: stop using pgd_ERROR() Stop using pgd_ERROR() in __get_pte_phys() when page table entry is already known to be empty. Link: https://lore.kernel.org/20260831054331.625505-8-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Signed-off-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: Yoshinori Sato Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Geert Uytterhoeven Cc: Helge Deller Cc: Huacai Chen Cc: James Bottomley Cc: Lorenzo Stoakes Cc: Samuel Holland Cc: WANG Xuerui --- arch/sh/mm/init.c | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/arch/sh/mm/init.c b/arch/sh/mm/init.c index 8d65e60688dcbd..93921109f4e64b 100644 --- a/arch/sh/mm/init.c +++ b/arch/sh/mm/init.c @@ -52,10 +52,8 @@ static pte_t *__get_pte_phys(unsigned long addr) pmd_t *pmd; pgd = pgd_offset_k(addr); - if (pgd_none(*pgd)) { - pgd_ERROR(*pgd); + if (pgd_none(*pgd)) return NULL; - } p4d = p4d_alloc(NULL, pgd, addr); if (unlikely(!p4d)) { From 63ad32b526509f95608bf892492240bd5b1d11d2 Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Mon, 31 Aug 2026 11:13:30 +0530 Subject: [PATCH 0676/1352] mm: drop pxd_ERROR() There are no more users left for any pxd_ERROR() either in generic MM or in the platform MM. Hence all these platform macros along with their generic fallback could be dropped across the tree. Link: https://lore.kernel.org/20260831054331.625505-9-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Signed-off-by: Andrew Morton Acked-by: Geert Uytterhoeven # m68k Acked-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: Helge Deller Cc: Huacai Chen Cc: James Bottomley Cc: John Paul Adrian Glaubitz Cc: Lorenzo Stoakes Cc: Rich Felker Cc: Samuel Holland Cc: WANG Xuerui Cc: Yoshinori Sato --- arch/alpha/include/asm/pgtable.h | 7 ------- arch/arc/include/asm/pgtable-levels.h | 11 ----------- arch/arm/include/asm/pgtable.h | 7 ------- arch/arm/kernel/traps.c | 17 ----------------- arch/arm64/include/asm/pgtable.h | 15 --------------- arch/csky/include/asm/pgtable.h | 4 ---- arch/hexagon/include/asm/pgtable.h | 3 --- arch/loongarch/include/asm/pgtable.h | 11 ----------- arch/m68k/include/asm/mcf_pgtable.h | 6 ------ arch/m68k/include/asm/motorola_pgtable.h | 8 -------- arch/m68k/include/asm/sun3_pgtable.h | 7 ------- arch/microblaze/include/asm/pgtable.h | 7 ------- arch/mips/include/asm/pgtable-32.h | 10 ---------- arch/mips/include/asm/pgtable-64.h | 13 ------------- arch/nios2/include/asm/pgtable.h | 7 ------- arch/openrisc/include/asm/pgtable.h | 7 ------- arch/parisc/include/asm/pgtable.h | 9 --------- arch/powerpc/include/asm/book3s/32/pgtable.h | 2 -- arch/powerpc/include/asm/book3s/64/pgtable.h | 7 ------- arch/powerpc/include/asm/nohash/32/pgtable.h | 2 -- .../powerpc/include/asm/nohash/64/pgtable-4k.h | 3 --- arch/powerpc/include/asm/nohash/64/pgtable.h | 5 ----- arch/riscv/include/asm/page.h | 6 ------ arch/riscv/include/asm/pgtable-64.h | 9 --------- arch/riscv/include/asm/pgtable.h | 4 ---- arch/s390/include/asm/pgtable.h | 11 ----------- arch/sh/include/asm/pgtable-3level.h | 3 --- arch/sh/include/asm/pgtable_32.h | 8 -------- arch/sparc/include/asm/pgtable_32.h | 3 --- arch/sparc/include/asm/pgtable_64.h | 10 ---------- arch/um/include/asm/pgtable-2level.h | 7 ------- arch/um/include/asm/pgtable-4level.h | 13 ------------- arch/x86/include/asm/pgtable-2level.h | 5 ----- arch/x86/include/asm/pgtable-3level.h | 11 ----------- arch/x86/include/asm/pgtable_64.h | 18 ------------------ arch/xtensa/include/asm/pgtable.h | 4 ---- include/asm-generic/pgtable-nop4d.h | 1 - include/asm-generic/pgtable-nopmd.h | 1 - include/asm-generic/pgtable-nopud.h | 1 - 39 files changed, 283 deletions(-) diff --git a/arch/alpha/include/asm/pgtable.h b/arch/alpha/include/asm/pgtable.h index 8e00cf9dc39dea..7cac8241ee674c 100644 --- a/arch/alpha/include/asm/pgtable.h +++ b/arch/alpha/include/asm/pgtable.h @@ -357,13 +357,6 @@ static inline pte_t pte_swp_clear_exclusive(pte_t pte) return pte; } -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %016lx.\n", __FILE__, __LINE__, pte_val(e)) -#define pmd_ERROR(e) \ - printk("%s:%d: bad pmd %016lx.\n", __FILE__, __LINE__, pmd_val(e)) -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %016lx.\n", __FILE__, __LINE__, pgd_val(e)) - extern void paging_init(void); /* We have our own get_unmapped_area */ diff --git a/arch/arc/include/asm/pgtable-levels.h b/arch/arc/include/asm/pgtable-levels.h index c8f9273372c073..167b82fcfafe36 100644 --- a/arch/arc/include/asm/pgtable-levels.h +++ b/arch/arc/include/asm/pgtable-levels.h @@ -98,8 +98,6 @@ /* * 1st level paging: pgd */ -#define pgd_ERROR(e) \ - pr_crit("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) #if CONFIG_PGTABLE_LEVELS > 3 @@ -115,9 +113,6 @@ /* * 2nd level paging: pud */ -#define pud_ERROR(e) \ - pr_crit("%s:%d: bad pud %08lx.\n", __FILE__, __LINE__, pud_val(e)) - #endif #if CONFIG_PGTABLE_LEVELS > 2 @@ -137,9 +132,6 @@ /* * 3rd level paging: pmd */ -#define pmd_ERROR(e) \ - pr_crit("%s:%d: bad pmd %08lx.\n", __FILE__, __LINE__, pmd_val(e)) - #define pmd_pfn(pmd) ((pmd_val(pmd) & PMD_MASK) >> PAGE_SHIFT) #define pfn_pmd(pfn,prot) __pmd(((pfn) << PAGE_SHIFT) | pgprot_val(prot)) @@ -165,9 +157,6 @@ /* * 4th level paging: pte */ -#define pte_ERROR(e) \ - pr_crit("%s:%d: bad pte %08lx.\n", __FILE__, __LINE__, pte_val(e)) - #define PFN_PTE_SHIFT PAGE_SHIFT #define pte_none(x) (!pte_val(x)) #define pte_present(x) (pte_val(x) & _PAGE_PRESENT) diff --git a/arch/arm/include/asm/pgtable.h b/arch/arm/include/asm/pgtable.h index 982795cf45637e..8dd17d20faa33f 100644 --- a/arch/arm/include/asm/pgtable.h +++ b/arch/arm/include/asm/pgtable.h @@ -44,13 +44,6 @@ #define LIBRARY_TEXT_START 0x0c000000 #ifndef __ASSEMBLY__ -extern void __pte_error(const char *file, int line, pte_t); -extern void __pmd_error(const char *file, int line, pmd_t); -extern void __pgd_error(const char *file, int line, pgd_t); - -#define pte_ERROR(pte) __pte_error(__FILE__, __LINE__, pte) -#define pmd_ERROR(pmd) __pmd_error(__FILE__, __LINE__, pmd) -#define pgd_ERROR(pgd) __pgd_error(__FILE__, __LINE__, pgd) /* * This is the lowest virtual address we can permit any user space diff --git a/arch/arm/kernel/traps.c b/arch/arm/kernel/traps.c index afbd2ebe5c39dc..ad04c806cc9d8f 100644 --- a/arch/arm/kernel/traps.c +++ b/arch/arm/kernel/traps.c @@ -753,23 +753,6 @@ void __readwrite_bug(const char *fn) } EXPORT_SYMBOL(__readwrite_bug); -#ifdef CONFIG_MMU -void __pte_error(const char *file, int line, pte_t pte) -{ - pr_err("%s:%d: bad pte %08llx.\n", file, line, (long long)pte_val(pte)); -} - -void __pmd_error(const char *file, int line, pmd_t pmd) -{ - pr_err("%s:%d: bad pmd %08llx.\n", file, line, (long long)pmd_val(pmd)); -} - -void __pgd_error(const char *file, int line, pgd_t pgd) -{ - pr_err("%s:%d: bad pgd %08llx.\n", file, line, (long long)pgd_val(pgd)); -} -#endif - asmlinkage void __div0(void) { pr_err("Division by zero in kernel.\n"); diff --git a/arch/arm64/include/asm/pgtable.h b/arch/arm64/include/asm/pgtable.h index 6000905a2e865e..e89ec5f4787b49 100644 --- a/arch/arm64/include/asm/pgtable.h +++ b/arch/arm64/include/asm/pgtable.h @@ -107,9 +107,6 @@ static inline void arch_leave_lazy_mmu_mode(void) __flush_tlb_range(vma, address, address + PMD_SIZE, PMD_SIZE, 2, \ TLBF_NOBROADCAST | TLBF_NONOTIFY | TLBF_NOWALKCACHE) -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %016llx.\n", __FILE__, __LINE__, pte_val(e)) - #ifdef CONFIG_ARM64_PA_BITS_52 static inline phys_addr_t __pte_to_phys(pte_t pte) { @@ -866,9 +863,6 @@ static inline unsigned long pmd_page_vaddr(pmd_t pmd) #if CONFIG_PGTABLE_LEVELS > 2 -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %016llx.\n", __FILE__, __LINE__, pmd_val(e)) - #define pud_none(pud) (!pud_val(pud)) #define pud_bad(pud) ((pud_val(pud) & PUD_TYPE_MASK) != \ PUD_TYPE_TABLE) @@ -960,9 +954,6 @@ static inline bool mm_pud_folded(const struct mm_struct *mm) } #define mm_pud_folded mm_pud_folded -#define pud_ERROR(e) \ - pr_err("%s:%d: bad pud %016llx.\n", __FILE__, __LINE__, pud_val(e)) - #define p4d_none(p4d) (pgtable_l4_enabled() && !p4d_val(p4d)) #define p4d_bad(p4d) (pgtable_l4_enabled() && \ ((p4d_val(p4d) & P4D_TYPE_MASK) != \ @@ -1088,9 +1079,6 @@ static inline bool mm_p4d_folded(const struct mm_struct *mm) } #define mm_p4d_folded mm_p4d_folded -#define p4d_ERROR(e) \ - pr_err("%s:%d: bad p4d %016llx.\n", __FILE__, __LINE__, p4d_val(e)) - #define pgd_none(pgd) (pgtable_l5_enabled() && !pgd_val(pgd)) #define pgd_bad(pgd) (pgtable_l5_enabled() && \ ((pgd_val(pgd) & PGD_TYPE_MASK) != \ @@ -1217,9 +1205,6 @@ p4d_t *p4d_offset_lockless_folded(pgd_t *pgdp, pgd_t pgd, unsigned long addr) #endif /* CONFIG_PGTABLE_LEVELS > 4 */ -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %016llx.\n", __FILE__, __LINE__, pgd_val(e)) - #define pgd_set_fixmap(addr) ((pgd_t *)set_fixmap_offset(FIX_PGD, addr)) #define pgd_clear_fixmap() clear_fixmap(FIX_PGD) diff --git a/arch/csky/include/asm/pgtable.h b/arch/csky/include/asm/pgtable.h index bafcd5823531a5..5ca77ff89ef939 100644 --- a/arch/csky/include/asm/pgtable.h +++ b/arch/csky/include/asm/pgtable.h @@ -23,10 +23,6 @@ #define PTRS_PER_PMD 1 #define PTRS_PER_PTE (PAGE_SIZE / sizeof(pte_t)) -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %08lx.\n", __FILE__, __LINE__, (e).pte_low) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) #define PFN_PTE_SHIFT PAGE_SHIFT #define pmd_pfn(pmd) (pmd_phys(pmd) >> PAGE_SHIFT) diff --git a/arch/hexagon/include/asm/pgtable.h b/arch/hexagon/include/asm/pgtable.h index 27b269e2870d37..2fdb27afe70329 100644 --- a/arch/hexagon/include/asm/pgtable.h +++ b/arch/hexagon/include/asm/pgtable.h @@ -94,9 +94,6 @@ #endif /* Any bigger and the PTE disappears. */ -#define pgd_ERROR(e) \ - printk(KERN_ERR "%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__,\ - pgd_val(e)) /* * Page Protection Constants. Includes (in this variant) cache attributes. diff --git a/arch/loongarch/include/asm/pgtable.h b/arch/loongarch/include/asm/pgtable.h index eddd8906b77dfc..cf29a4c8ac593a 100644 --- a/arch/loongarch/include/asm/pgtable.h +++ b/arch/loongarch/include/asm/pgtable.h @@ -135,17 +135,6 @@ struct vm_area_struct; #define ptep_get(ptep) READ_ONCE(*(ptep)) #define pmdp_get(pmdp) READ_ONCE(*(pmdp)) -#ifndef __PAGETABLE_PMD_FOLDED -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %016lx.\n", __FILE__, __LINE__, pmd_val(e)) -#endif -#ifndef __PAGETABLE_PUD_FOLDED -#define pud_ERROR(e) \ - pr_err("%s:%d: bad pud %016lx.\n", __FILE__, __LINE__, pud_val(e)) -#endif -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %016lx.\n", __FILE__, __LINE__, pgd_val(e)) - extern pte_t invalid_pte_table[PTRS_PER_PTE]; #ifndef __PAGETABLE_PUD_FOLDED diff --git a/arch/m68k/include/asm/mcf_pgtable.h b/arch/m68k/include/asm/mcf_pgtable.h index 189bb7b1e6630f..f45a882238dbb6 100644 --- a/arch/m68k/include/asm/mcf_pgtable.h +++ b/arch/m68k/include/asm/mcf_pgtable.h @@ -137,12 +137,6 @@ static inline int pmd_bad2(pmd_t *pmd) { return 0; } #define pmd_present(pmd) (!pmd_none2(&(pmd))) static inline void pmd_clear(pmd_t *pmdp) { pmd_val(*pmdp) = 0; } -#define pte_ERROR(e) \ - printk(KERN_ERR "%s:%d: bad pte %08lx.\n", \ - __FILE__, __LINE__, pte_val(e)) -#define pgd_ERROR(e) \ - printk(KERN_ERR "%s:%d: bad pgd %08lx.\n", \ - __FILE__, __LINE__, pgd_val(e)) /* * The following only work if pte_present() is true. diff --git a/arch/m68k/include/asm/motorola_pgtable.h b/arch/m68k/include/asm/motorola_pgtable.h index dcf6829b3eab97..d9393b310add6a 100644 --- a/arch/m68k/include/asm/motorola_pgtable.h +++ b/arch/m68k/include/asm/motorola_pgtable.h @@ -131,14 +131,6 @@ static inline void pud_set(pud_t *pudp, pmd_t *pmdp) #define pud_clear(pudp) ({ pud_val(*pudp) = 0; }) #define pud_page(pud) (mem_map + ((unsigned long)(__va(pud_val(pud)) - PAGE_OFFSET) >> PAGE_SHIFT)) -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %08lx.\n", __FILE__, __LINE__, pte_val(e)) -#define pmd_ERROR(e) \ - printk("%s:%d: bad pmd %08lx.\n", __FILE__, __LINE__, pmd_val(e)) -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) - - /* * The following only work if pte_present() is true. * Undefined behaviour if not.. diff --git a/arch/m68k/include/asm/sun3_pgtable.h b/arch/m68k/include/asm/sun3_pgtable.h index 80ca185a18a193..704442a391fd74 100644 --- a/arch/m68k/include/asm/sun3_pgtable.h +++ b/arch/m68k/include/asm/sun3_pgtable.h @@ -119,13 +119,6 @@ static inline int pmd_present2 (pmd_t *pmd) { return pmd_val (*pmd) & SUN3_PMD_V #define pmd_present(pmd) (!pmd_none2(&(pmd))) static inline void pmd_clear (pmd_t *pmdp) { pmd_val (*pmdp) = 0; } - -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %08lx.\n", __FILE__, __LINE__, pte_val(e)) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) - - /* * The following only work if pte_present() is true. * Undefined behaviour if not... diff --git a/arch/microblaze/include/asm/pgtable.h b/arch/microblaze/include/asm/pgtable.h index 7678c040a2fd36..72708f9af1c0b8 100644 --- a/arch/microblaze/include/asm/pgtable.h +++ b/arch/microblaze/include/asm/pgtable.h @@ -103,13 +103,6 @@ extern pte_t *va_to_pte(unsigned long address); #define USER_PGD_PTRS (PAGE_OFFSET >> PGDIR_SHIFT) #define KERNEL_PGD_PTRS (PTRS_PER_PGD-USER_PGD_PTRS) -#define pte_ERROR(e) \ - printk(KERN_ERR "%s:%d: bad pte "PTE_FMT".\n", \ - __FILE__, __LINE__, pte_val(e)) -#define pgd_ERROR(e) \ - printk(KERN_ERR "%s:%d: bad pgd %08lx.\n", \ - __FILE__, __LINE__, pgd_val(e)) - /* * Bits in a linux-style PTE. These match the bits in the * (hardware-defined) PTE as closely as possible. diff --git a/arch/mips/include/asm/pgtable-32.h b/arch/mips/include/asm/pgtable-32.h index 92b7591aac2acd..ef1001ab09c5be 100644 --- a/arch/mips/include/asm/pgtable-32.h +++ b/arch/mips/include/asm/pgtable-32.h @@ -104,16 +104,6 @@ extern int add_temporary_entry(unsigned long entrylo0, unsigned long entrylo1, # define VMALLOC_END (FIXADDR_START-2*PAGE_SIZE) #endif -#ifdef CONFIG_PHYS_ADDR_T_64BIT -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %016Lx.\n", __FILE__, __LINE__, pte_val(e)) -#else -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %08lx.\n", __FILE__, __LINE__, pte_val(e)) -#endif -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) - extern void load_pgd(unsigned long pg_dir); extern pte_t invalid_pte_table[PTRS_PER_PTE]; diff --git a/arch/mips/include/asm/pgtable-64.h b/arch/mips/include/asm/pgtable-64.h index 6e854bb11f37de..785fc37bab9417 100644 --- a/arch/mips/include/asm/pgtable-64.h +++ b/arch/mips/include/asm/pgtable-64.h @@ -151,19 +151,6 @@ #define MODULES_END (FIXADDR_START-2*PAGE_SIZE) #endif -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %016lx.\n", __FILE__, __LINE__, pte_val(e)) -#ifndef __PAGETABLE_PMD_FOLDED -#define pmd_ERROR(e) \ - printk("%s:%d: bad pmd %016lx.\n", __FILE__, __LINE__, pmd_val(e)) -#endif -#ifndef __PAGETABLE_PUD_FOLDED -#define pud_ERROR(e) \ - printk("%s:%d: bad pud %016lx.\n", __FILE__, __LINE__, pud_val(e)) -#endif -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %016lx.\n", __FILE__, __LINE__, pgd_val(e)) - extern pte_t invalid_pte_table[PTRS_PER_PTE]; #ifndef __PAGETABLE_PUD_FOLDED diff --git a/arch/nios2/include/asm/pgtable.h b/arch/nios2/include/asm/pgtable.h index d389aa9ca57ce0..272707d48f1bb1 100644 --- a/arch/nios2/include/asm/pgtable.h +++ b/arch/nios2/include/asm/pgtable.h @@ -223,13 +223,6 @@ static inline unsigned long pmd_page_vaddr(pmd_t pmd) return pmd_val(pmd); } -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %08lx.\n", \ - __FILE__, __LINE__, pte_val(e)) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %08lx.\n", \ - __FILE__, __LINE__, pgd_val(e)) - /* * Encode/decode swap entries and swap PTEs. Swap PTEs are all PTEs that * are !pte_none() && !pte_present(). diff --git a/arch/openrisc/include/asm/pgtable.h b/arch/openrisc/include/asm/pgtable.h index 6b89996d0b628e..13afcc0bd8631b 100644 --- a/arch/openrisc/include/asm/pgtable.h +++ b/arch/openrisc/include/asm/pgtable.h @@ -338,13 +338,6 @@ static inline unsigned long pmd_page_vaddr(pmd_t pmd) #define pte_pfn(x) ((unsigned long)(((x).pte)) >> PAGE_SHIFT) #define pfn_pte(pfn, prot) __pte((((pfn) << PAGE_SHIFT)) | pgprot_val(prot)) -#define pte_ERROR(e) \ - printk(KERN_ERR "%s:%d: bad pte %p(%08lx).\n", \ - __FILE__, __LINE__, &(e), pte_val(e)) -#define pgd_ERROR(e) \ - printk(KERN_ERR "%s:%d: bad pgd %p(%08lx).\n", \ - __FILE__, __LINE__, &(e), pgd_val(e)) - extern pgd_t swapper_pg_dir[PTRS_PER_PGD]; /* defined in head.S */ struct vm_area_struct; diff --git a/arch/parisc/include/asm/pgtable.h b/arch/parisc/include/asm/pgtable.h index 467b8547ac8bfa..f6899375cb4393 100644 --- a/arch/parisc/include/asm/pgtable.h +++ b/arch/parisc/include/asm/pgtable.h @@ -75,15 +75,6 @@ extern void __update_cache(pte_t pte); #endif /* !__ASSEMBLER__ */ -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %08lx.\n", __FILE__, __LINE__, pte_val(e)) -#if CONFIG_PGTABLE_LEVELS == 3 -#define pmd_ERROR(e) \ - printk("%s:%d: bad pmd %08lx.\n", __FILE__, __LINE__, (unsigned long)pmd_val(e)) -#endif -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, (unsigned long)pgd_val(e)) - /* This is the size of the initially mapped kernel memory */ #if defined(CONFIG_64BIT) || defined(CONFIG_KALLSYMS) #define KERNEL_INITIAL_ORDER 26 /* 1<<26 = 64MB */ diff --git a/arch/powerpc/include/asm/book3s/32/pgtable.h b/arch/powerpc/include/asm/book3s/32/pgtable.h index e18a4fa282a1b6..835e84caee13fa 100644 --- a/arch/powerpc/include/asm/book3s/32/pgtable.h +++ b/arch/powerpc/include/asm/book3s/32/pgtable.h @@ -203,8 +203,6 @@ void unmap_kernel_page(unsigned long va); /* Bits to mask out from a PGD to get to the PUD page */ #define PGD_MASKED_BITS 0 -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) /* * Bits in a linux-style PTE. These match the bits in the * (hardware-defined) PowerPC PTE as closely as possible. diff --git a/arch/powerpc/include/asm/book3s/64/pgtable.h b/arch/powerpc/include/asm/book3s/64/pgtable.h index f4db7d7fbd5c62..dff8790a047db5 100644 --- a/arch/powerpc/include/asm/book3s/64/pgtable.h +++ b/arch/powerpc/include/asm/book3s/64/pgtable.h @@ -991,13 +991,6 @@ static inline pmd_t *pud_pgtable(pud_t pud) return (pmd_t *)__va(pud_val(pud) & ~PUD_MASKED_BITS); } -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %08lx.\n", __FILE__, __LINE__, pmd_val(e)) -#define pud_ERROR(e) \ - pr_err("%s:%d: bad pud %08lx.\n", __FILE__, __LINE__, pud_val(e)) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) - static inline int map_kernel_page(unsigned long ea, unsigned long pa, pgprot_t prot) { if (radix_enabled()) { diff --git a/arch/powerpc/include/asm/nohash/32/pgtable.h b/arch/powerpc/include/asm/nohash/32/pgtable.h index 496ecc65ac255a..f17afde89fa13d 100644 --- a/arch/powerpc/include/asm/nohash/32/pgtable.h +++ b/arch/powerpc/include/asm/nohash/32/pgtable.h @@ -51,8 +51,6 @@ #define USER_PTRS_PER_PGD (TASK_SIZE / PGDIR_SIZE) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %08llx.\n", __FILE__, __LINE__, (unsigned long long)pgd_val(e)) /* * This is the bottom of the PKMAP area with HIGHMEM or an arbitrary diff --git a/arch/powerpc/include/asm/nohash/64/pgtable-4k.h b/arch/powerpc/include/asm/nohash/64/pgtable-4k.h index fb6fa1d4e0749a..75cf3c331b9292 100644 --- a/arch/powerpc/include/asm/nohash/64/pgtable-4k.h +++ b/arch/powerpc/include/asm/nohash/64/pgtable-4k.h @@ -82,9 +82,6 @@ extern struct page *p4d_page(p4d_t p4d); #endif /* !__ASSEMBLER__ */ -#define pud_ERROR(e) \ - pr_err("%s:%d: bad pud %08lx.\n", __FILE__, __LINE__, pud_val(e)) - /* * On all 4K setups, remap_4k_pfn() equates to remap_pfn_range() */ #define remap_4k_pfn(vma, addr, pfn, prot) \ diff --git a/arch/powerpc/include/asm/nohash/64/pgtable.h b/arch/powerpc/include/asm/nohash/64/pgtable.h index 661eb3820d1291..446dde8b6ead54 100644 --- a/arch/powerpc/include/asm/nohash/64/pgtable.h +++ b/arch/powerpc/include/asm/nohash/64/pgtable.h @@ -159,11 +159,6 @@ static inline void huge_ptep_set_wrprotect(struct mm_struct *mm, __young; \ }) -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %08lx.\n", __FILE__, __LINE__, pmd_val(e)) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) - /* * Encode/decode swap entries and swap PTEs. Swap PTEs are all PTEs that * are !pte_none() && !pte_present(). diff --git a/arch/riscv/include/asm/page.h b/arch/riscv/include/asm/page.h index 709a36fb432343..b4bbae55e93111 100644 --- a/arch/riscv/include/asm/page.h +++ b/arch/riscv/include/asm/page.h @@ -76,12 +76,6 @@ typedef struct page *pgtable_t; #define __pgd(x) ((pgd_t) { (x) }) #define __pgprot(x) ((pgprot_t) { (x) }) -#ifdef CONFIG_64BIT -#define PTE_FMT "%016lx" -#else -#define PTE_FMT "%08lx" -#endif - #if defined(CONFIG_64BIT) && defined(CONFIG_MMU) /* * We override this value as its generic definition uses __pa too early in diff --git a/arch/riscv/include/asm/pgtable-64.h b/arch/riscv/include/asm/pgtable-64.h index 6e789fa58514c7..ae23182b572cdd 100644 --- a/arch/riscv/include/asm/pgtable-64.h +++ b/arch/riscv/include/asm/pgtable-64.h @@ -264,15 +264,6 @@ static inline unsigned long _pmd_pfn(pmd_t pmd) return __page_val_to_pfn(pmd_val(pmd)); } -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %016lx.\n", __FILE__, __LINE__, pmd_val(e)) - -#define pud_ERROR(e) \ - pr_err("%s:%d: bad pud %016lx.\n", __FILE__, __LINE__, pud_val(e)) - -#define p4d_ERROR(e) \ - pr_err("%s:%d: bad p4d %016lx.\n", __FILE__, __LINE__, p4d_val(e)) - static inline void set_p4d(p4d_t *p4dp, p4d_t p4d) { if (pgtable_l4_enabled) diff --git a/arch/riscv/include/asm/pgtable.h b/arch/riscv/include/asm/pgtable.h index 40b1ed4f3ea893..4c8fc684550311 100644 --- a/arch/riscv/include/asm/pgtable.h +++ b/arch/riscv/include/asm/pgtable.h @@ -556,10 +556,6 @@ static inline pte_t pte_modify(pte_t pte, pgprot_t newprot) return __pte((pte_val(pte) & _PAGE_CHG_MASK) | newprot_val); } -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd " PTE_FMT ".\n", __FILE__, __LINE__, pgd_val(e)) - - /* Commit new configuration to MMU hardware */ static inline void update_mmu_cache_range(struct vm_fault *vmf, struct vm_area_struct *vma, unsigned long address, diff --git a/arch/s390/include/asm/pgtable.h b/arch/s390/include/asm/pgtable.h index e882663a58e776..2d5c2ab06de988 100644 --- a/arch/s390/include/asm/pgtable.h +++ b/arch/s390/include/asm/pgtable.h @@ -68,17 +68,6 @@ extern unsigned long zero_page_mask; /* TODO: s390 cannot support io_remap_pfn_range... */ -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %016lx.\n", __FILE__, __LINE__, pte_val(e)) -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %016lx.\n", __FILE__, __LINE__, pmd_val(e)) -#define pud_ERROR(e) \ - pr_err("%s:%d: bad pud %016lx.\n", __FILE__, __LINE__, pud_val(e)) -#define p4d_ERROR(e) \ - pr_err("%s:%d: bad p4d %016lx.\n", __FILE__, __LINE__, p4d_val(e)) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %016lx.\n", __FILE__, __LINE__, pgd_val(e)) - /* * The vmalloc and module area will always be on the topmost area of the * kernel mapping. 512GB are reserved for vmalloc by default. diff --git a/arch/sh/include/asm/pgtable-3level.h b/arch/sh/include/asm/pgtable-3level.h index d1ce73f3bd85ef..3f4d747f30f40b 100644 --- a/arch/sh/include/asm/pgtable-3level.h +++ b/arch/sh/include/asm/pgtable-3level.h @@ -25,9 +25,6 @@ #define PTRS_PER_PMD ((1 << PGDIR_SHIFT) / PMD_SIZE) -#define pmd_ERROR(e) \ - printk("%s:%d: bad pmd %016llx.\n", __FILE__, __LINE__, pmd_val(e)) - typedef union { struct { unsigned long pmd_low; diff --git a/arch/sh/include/asm/pgtable_32.h b/arch/sh/include/asm/pgtable_32.h index c8eb9a7a4c4c78..cde1bf0c67342b 100644 --- a/arch/sh/include/asm/pgtable_32.h +++ b/arch/sh/include/asm/pgtable_32.h @@ -400,14 +400,6 @@ static inline unsigned long pmd_page_vaddr(pmd_t pmd) #define pmd_pfn(pmd) (__pa(pmd_val(pmd)) >> PAGE_SHIFT) #define pmd_page(pmd) (virt_to_page(pmd_val(pmd))) -#ifdef CONFIG_X2TLB -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %016llx.\n", __FILE__, __LINE__, pgd_val(e)) -#else -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) -#endif - /* * Encode/decode swap entries and swap PTEs. Swap PTEs are all PTEs that * are !pte_none() && !pte_present(). diff --git a/arch/sparc/include/asm/pgtable_32.h b/arch/sparc/include/asm/pgtable_32.h index f89b1250661dff..5a5f54a090f5b2 100644 --- a/arch/sparc/include/asm/pgtable_32.h +++ b/arch/sparc/include/asm/pgtable_32.h @@ -40,9 +40,6 @@ void load_mmu(void); unsigned long calc_highpages(void); unsigned long __init bootmem_init(unsigned long *pages_avail); -#define pte_ERROR(e) __builtin_trap() -#define pmd_ERROR(e) __builtin_trap() -#define pgd_ERROR(e) __builtin_trap() #define PTRS_PER_PTE 64 #define PTRS_PER_PMD 64 diff --git a/arch/sparc/include/asm/pgtable_64.h b/arch/sparc/include/asm/pgtable_64.h index 0837ebbc5dce63..44d1333065a6ae 100644 --- a/arch/sparc/include/asm/pgtable_64.h +++ b/arch/sparc/include/asm/pgtable_64.h @@ -96,16 +96,6 @@ bool kern_addr_valid(unsigned long addr); #define PTRS_PER_PUD (1UL << PUD_BITS) #define PTRS_PER_PGD (1UL << PGDIR_BITS) -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %p(%016lx) seen at (%pS)\n", \ - __FILE__, __LINE__, &(e), pmd_val(e), __builtin_return_address(0)) -#define pud_ERROR(e) \ - pr_err("%s:%d: bad pud %p(%016lx) seen at (%pS)\n", \ - __FILE__, __LINE__, &(e), pud_val(e), __builtin_return_address(0)) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %p(%016lx) seen at (%pS)\n", \ - __FILE__, __LINE__, &(e), pgd_val(e), __builtin_return_address(0)) - #endif /* !(__ASSEMBLER__) */ /* PTE bits which are the same in SUN4U and SUN4V format. */ diff --git a/arch/um/include/asm/pgtable-2level.h b/arch/um/include/asm/pgtable-2level.h index 14ec16f92ce408..fa625f5b5ef750 100644 --- a/arch/um/include/asm/pgtable-2level.h +++ b/arch/um/include/asm/pgtable-2level.h @@ -24,13 +24,6 @@ #define USER_PTRS_PER_PGD ((TASK_SIZE + (PGDIR_SIZE - 1)) / PGDIR_SIZE) #define PTRS_PER_PGD 1024 -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %p(%08lx).\n", __FILE__, __LINE__, &(e), \ - pte_val(e)) -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %p(%08lx).\n", __FILE__, __LINE__, &(e), \ - pgd_val(e)) - static inline int pgd_needsync(pgd_t pgd) { return 0; } static inline void pgd_mkuptodate(pgd_t pgd) { } diff --git a/arch/um/include/asm/pgtable-4level.h b/arch/um/include/asm/pgtable-4level.h index 7a271b7b83d2bd..ff82f99c80fa10 100644 --- a/arch/um/include/asm/pgtable-4level.h +++ b/arch/um/include/asm/pgtable-4level.h @@ -42,19 +42,6 @@ #define USER_PTRS_PER_PGD ((TASK_SIZE + (PGDIR_SIZE - 1)) / PGDIR_SIZE) -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %p(%016lx).\n", __FILE__, __LINE__, &(e), \ - pte_val(e)) -#define pmd_ERROR(e) \ - printk("%s:%d: bad pmd %p(%016lx).\n", __FILE__, __LINE__, &(e), \ - pmd_val(e)) -#define pud_ERROR(e) \ - printk("%s:%d: bad pud %p(%016lx).\n", __FILE__, __LINE__, &(e), \ - pud_val(e)) -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %p(%016lx).\n", __FILE__, __LINE__, &(e), \ - pgd_val(e)) - #define pud_none(x) (!(pud_val(x) & ~_PAGE_NEEDSYNC)) #define pud_bad(x) ((pud_val(x) & (~PAGE_MASK & ~_PAGE_USER)) != _KERNPG_TABLE) #define pud_present(x) (pud_val(x) & _PAGE_PRESENT) diff --git a/arch/x86/include/asm/pgtable-2level.h b/arch/x86/include/asm/pgtable-2level.h index e9482a11ac52d6..83427765cfbfd2 100644 --- a/arch/x86/include/asm/pgtable-2level.h +++ b/arch/x86/include/asm/pgtable-2level.h @@ -2,11 +2,6 @@ #ifndef _ASM_X86_PGTABLE_2LEVEL_H #define _ASM_X86_PGTABLE_2LEVEL_H -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %08lx\n", __FILE__, __LINE__, (e).pte_low) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %08lx\n", __FILE__, __LINE__, pgd_val(e)) - /* * Certain architectures need to do special things when PTEs * within a page table are directly modified. Thus, the following diff --git a/arch/x86/include/asm/pgtable-3level.h b/arch/x86/include/asm/pgtable-3level.h index dabafba957ea6f..d6729911e09a28 100644 --- a/arch/x86/include/asm/pgtable-3level.h +++ b/arch/x86/include/asm/pgtable-3level.h @@ -8,17 +8,6 @@ * * Copyright (C) 1999 Ingo Molnar */ - -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %p(%08lx%08lx)\n", \ - __FILE__, __LINE__, &(e), (e).pte_high, (e).pte_low) -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %p(%016Lx)\n", \ - __FILE__, __LINE__, &(e), pmd_val(e)) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %p(%016Lx)\n", \ - __FILE__, __LINE__, &(e), pgd_val(e)) - #define pxx_xchg64(_pxx, _ptr, _val) ({ \ _pxx##val_t *_p = (_pxx##val_t *)_ptr; \ _pxx##val_t _o = *_p; \ diff --git a/arch/x86/include/asm/pgtable_64.h b/arch/x86/include/asm/pgtable_64.h index ce45882ccd071b..c861f3832bed2e 100644 --- a/arch/x86/include/asm/pgtable_64.h +++ b/arch/x86/include/asm/pgtable_64.h @@ -29,24 +29,6 @@ extern pgd_t init_top_pgt[]; extern void paging_init(void); static inline void sync_initial_page_table(void) { } -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %p(%016lx)\n", \ - __FILE__, __LINE__, &(e), pte_val(e)) -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %p(%016lx)\n", \ - __FILE__, __LINE__, &(e), pmd_val(e)) -#define pud_ERROR(e) \ - pr_err("%s:%d: bad pud %p(%016lx)\n", \ - __FILE__, __LINE__, &(e), pud_val(e)) - -#define p4d_ERROR(e) \ - pr_err("%s:%d: bad p4d %p(%016lx)\n", \ - __FILE__, __LINE__, &(e), p4d_val(e)) - -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %p(%016lx)\n", \ - __FILE__, __LINE__, &(e), pgd_val(e)) - struct mm_struct; #define mm_p4d_folded mm_p4d_folded diff --git a/arch/xtensa/include/asm/pgtable.h b/arch/xtensa/include/asm/pgtable.h index f00a879dc298a5..60fb67a9972907 100644 --- a/arch/xtensa/include/asm/pgtable.h +++ b/arch/xtensa/include/asm/pgtable.h @@ -204,10 +204,6 @@ */ #ifndef __ASSEMBLER__ -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %08lx.\n", __FILE__, __LINE__, pte_val(e)) -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd entry %08lx.\n", __FILE__, __LINE__, pgd_val(e)) #ifdef CONFIG_MMU extern pgd_t swapper_pg_dir[PAGE_SIZE/sizeof(pgd_t)]; diff --git a/include/asm-generic/pgtable-nop4d.h b/include/asm-generic/pgtable-nop4d.h index 89c21f84cffbe2..1cf739ee38aa15 100644 --- a/include/asm-generic/pgtable-nop4d.h +++ b/include/asm-generic/pgtable-nop4d.h @@ -22,7 +22,6 @@ static inline int pgd_none(pgd_t pgd) { return 0; } static inline int pgd_bad(pgd_t pgd) { return 0; } static inline int pgd_present(pgd_t pgd) { return 1; } static inline void pgd_clear(pgd_t *pgd) { } -#define p4d_ERROR(p4d) (pgd_ERROR((p4d).pgd)) #define pgd_populate(mm, pgd, p4d) do { } while (0) #define pgd_populate_safe(mm, pgd, p4d) do { } while (0) diff --git a/include/asm-generic/pgtable-nopmd.h b/include/asm-generic/pgtable-nopmd.h index 36b6490ed18081..ff4235cf84d772 100644 --- a/include/asm-generic/pgtable-nopmd.h +++ b/include/asm-generic/pgtable-nopmd.h @@ -33,7 +33,6 @@ static inline int pud_present(pud_t pud) { return 1; } static inline int pud_user(pud_t pud) { return 0; } static inline int pud_leaf(pud_t pud) { return 0; } static inline void pud_clear(pud_t *pud) { } -#define pmd_ERROR(pmd) (pud_ERROR((pmd).pud)) #define pud_populate(mm, pmd, pte) do { } while (0) diff --git a/include/asm-generic/pgtable-nopud.h b/include/asm-generic/pgtable-nopud.h index 356cbfbaab2476..eedee8e3ad68fd 100644 --- a/include/asm-generic/pgtable-nopud.h +++ b/include/asm-generic/pgtable-nopud.h @@ -29,7 +29,6 @@ static inline int p4d_none(p4d_t p4d) { return 0; } static inline int p4d_bad(p4d_t p4d) { return 0; } static inline int p4d_present(p4d_t p4d) { return 1; } static inline void p4d_clear(p4d_t *p4d) { } -#define pud_ERROR(pud) (p4d_ERROR((pud).p4d)) #define p4d_populate(mm, p4d, pud) do { } while (0) #define p4d_populate_safe(mm, p4d, pud) do { } while (0) From 366032b4859d294fee967411fe9112d269015099 Mon Sep 17 00:00:00 2001 From: Johannes Weiner Date: Sun, 30 Aug 2026 12:29:17 +0800 Subject: [PATCH 0677/1352] mm: add page_counter_margin() Patch series "mm: avoid large folio splits when swap is unavailable", v7. This is v7 of Barry's original RFC patch, "mm: Avoiding split large folios if swap has no space": https://lore.kernel.org/r/20260618221720.71768-1-baohua@kernel.org Barry's RFC showed the no-swap case with MADV_PAGEOUT on 16KB mTHP: the large-folio split counter increased by 1024 even though no swapout progress was possible. Skipping the split in that case kept the counter at 0. This series makes folio_alloc_swap() classify failures according to whether splitting a large folio might allow swapout to make progress. Callers can then avoid destroying the large folio when neither global swap availability nor the folio's memcg swap hierarchy has capacity for even a smaller folio. Patch #1 adds page_counter_margin(), a small helper that computes the minimum remaining chargeable space across a page_counter hierarchy. Patch #2 establishes the folio_alloc_swap() return-value contract: - -E2BIG: splitting may let smaller folios make progress - -ENOSPC: no global swap space is available - -ENOMEM: splitting is not expected to help, including when the folio's memcg swap hierarchy has no remaining capacity Patch #3 makes vmscan split a large folio only when folio_alloc_swap() returns -E2BIG. Other failures keep the existing activation path and avoid destroying the large folio when no smaller part can be backed by swap either. Patch #4 applies the same contract to shmem_writeout(), which currently splits a large folio on every folio_alloc_swap() failure. It now enters the split fallback only on -E2BIG; other failures redirty and reactivate the folio as before. Testing: With a 1GB anonymous mapping backed by 16KB mTHPs and memory.swap.max=0, the patch reduced the median latency of 30 process_madvise(MADV_PAGEOUT) runs from 743.8 ms to 181.7 ms, while the number of large-folio splits per run dropped from 65536 to 0. Neither kernel swapped out any pages. I also ran DaCapo h2 under swap pressure and found no statistically significant change in wall time or CPU time. The overall benefit appears minor and workload-dependent. This patch (of 4): mem_cgroup_get_nr_swap_pages() open-codes the remaining capacity across the memcg swap counter hierarchy. Add page_counter_margin() to return the minimum usable space from a page counter to the root, and use it in mem_cgroup_get_nr_swap_pages(). This is a pure refactoring with no intended behavior change. Link: https://lore.kernel.org/20260830042920.2280454-1-xueyuan.chen21@gmail.com Link: https://lore.kernel.org/20260830042920.2280454-2-xueyuan.chen21@gmail.com Signed-off-by: Johannes Weiner Signed-off-by: Xueyuan Chen Signed-off-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) Reviewed-by: Barry Song Cc: Baolin Wang Cc: Baoquan He Cc: Chris Li Cc: Hugh Dickins Cc: Kairui Song Cc: Kemeng Shi Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Nhat Pham Cc: Roman Gushchin Cc: Shakeel Butt Cc: Nanzhe Zhao Cc: Youngjun Park --- include/linux/page_counter.h | 1 + mm/memcontrol.c | 9 +++------ mm/page_counter.c | 20 ++++++++++++++++++++ 3 files changed, 24 insertions(+), 6 deletions(-) diff --git a/include/linux/page_counter.h b/include/linux/page_counter.h index d649b6bbbc871b..07b7cb12249c7c 100644 --- a/include/linux/page_counter.h +++ b/include/linux/page_counter.h @@ -68,6 +68,7 @@ static inline unsigned long page_counter_read(struct page_counter *counter) return atomic_long_read(&counter->usage); } +long page_counter_margin(struct page_counter *counter); void page_counter_cancel(struct page_counter *counter, unsigned long nr_pages); void page_counter_charge(struct page_counter *counter, unsigned long nr_pages); bool page_counter_try_charge(struct page_counter *counter, diff --git a/mm/memcontrol.c b/mm/memcontrol.c index aeaa09e01d70ea..7f63bf9e8ef140 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -5832,12 +5832,9 @@ long mem_cgroup_get_nr_swap_pages(struct mem_cgroup *memcg) { long nr_swap_pages = get_nr_swap_pages(); - if (mem_cgroup_disabled() || do_memsw_account()) - return nr_swap_pages; - for (; !mem_cgroup_is_root(memcg); memcg = parent_mem_cgroup(memcg)) - nr_swap_pages = min_t(long, nr_swap_pages, - READ_ONCE(memcg->swap.max) - - page_counter_read(&memcg->swap)); + if (!mem_cgroup_disabled() && !do_memsw_account()) + nr_swap_pages = min(nr_swap_pages, page_counter_margin(&memcg->swap)); + return nr_swap_pages; } diff --git a/mm/page_counter.c b/mm/page_counter.c index 661e0f2a5127a5..450543f4b318b6 100644 --- a/mm/page_counter.c +++ b/mm/page_counter.c @@ -46,6 +46,26 @@ static void propagate_protected_usage(struct page_counter *c, } } +/** + * page_counter_margin - remaining usable space within hierarchical limits + * @counter: counter + * + * Return: The minimum value of max minus usage across @counter and all of + * its ancestors. The value may be negative during a concurrent charge. + */ +long page_counter_margin(struct page_counter *counter) +{ + long margin = PAGE_COUNTER_MAX; + + do { + long m = READ_ONCE(counter->max) - page_counter_read(counter); + + margin = min(margin, m); + } while ((counter = counter->parent)); + + return margin; +} + /** * page_counter_cancel - take pages out of the local counter * @counter: counter From 1cff837d9b40cf7039c9556bb1783e9fb5ec0854 Mon Sep 17 00:00:00 2001 From: Xueyuan Chen Date: Sun, 30 Aug 2026 12:29:18 +0800 Subject: [PATCH 0678/1352] mm: distinguish large folio swap allocation failures folio_alloc_swap() reports most failures with generic negative error codes. Reclaim callers consequently cannot tell whether splitting a large folio could make progress, or whether no swap space is available for even a single page. Classify failures using both the global free swap count and the remaining capacity in the folio's memcg swap hierarchy. Return -ENOSPC when global swap space is exhausted, -ENOMEM when splitting cannot overcome the failure, and -E2BIG for a large folio when allocating or charging a smaller folio might still succeed. Use this classification for all folio_alloc_swap() failure paths, including capability rejection, swap slot allocation failure, and memcg swap charge failure. Callers are updated separately to split large folios only on -E2BIG. Link: https://lore.kernel.org/20260830042920.2280454-3-xueyuan.chen21@gmail.com Signed-off-by: Xueyuan Chen Signed-off-by: Andrew Morton Suggested-by: Kairui Song Suggested-by: Barry Song Suggested-by: Youngjun Park Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Baoquan He Cc: Chris Li Cc: Hugh Dickins Cc: Johannes Weiner Cc: Kemeng Shi Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Nanzhe Zhao Cc: Nhat Pham Cc: Roman Gushchin Cc: Shakeel Butt --- include/linux/swap.h | 6 ++++++ mm/memcontrol.c | 23 +++++++++++++++++++++++ mm/swapfile.c | 26 +++++++++++++++++++------- 3 files changed, 48 insertions(+), 7 deletions(-) diff --git a/include/linux/swap.h b/include/linux/swap.h index 5658a1634b85ea..7a43409879caed 100644 --- a/include/linux/swap.h +++ b/include/linux/swap.h @@ -508,6 +508,7 @@ static inline void mem_cgroup_uncharge_swap(unsigned short id, unsigned int nr_p __mem_cgroup_uncharge_swap(id, nr_pages); } +long mem_cgroup_get_folio_swap_margin(struct folio *folio); extern long mem_cgroup_get_nr_swap_pages(struct mem_cgroup *memcg); extern bool mem_cgroup_swap_full(struct folio *folio); #else @@ -521,6 +522,11 @@ static inline void mem_cgroup_uncharge_swap(unsigned short id, { } +static inline long mem_cgroup_get_folio_swap_margin(struct folio *folio) +{ + return PAGE_COUNTER_MAX; +} + static inline long mem_cgroup_get_nr_swap_pages(struct mem_cgroup *memcg) { return get_nr_swap_pages(); diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 7f63bf9e8ef140..bd1e7e15442659 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -5838,6 +5838,29 @@ long mem_cgroup_get_nr_swap_pages(struct mem_cgroup *memcg) return nr_swap_pages; } +/** + * mem_cgroup_get_folio_swap_margin - get a folio's memcg swap margin + * @folio: folio whose memcg margin is queried + * + * Return: Remaining chargeable pages in the folio's memcg hierarchy. + */ +long mem_cgroup_get_folio_swap_margin(struct folio *folio) +{ + struct mem_cgroup *memcg; + long margin; + + if (mem_cgroup_disabled() || do_memsw_account() || + !folio_memcg_charged(folio)) + return PAGE_COUNTER_MAX; + + rcu_read_lock(); + memcg = folio_memcg(folio); + margin = page_counter_margin(&memcg->swap); + rcu_read_unlock(); + + return margin; +} + bool mem_cgroup_swap_full(struct folio *folio) { struct mem_cgroup *memcg; diff --git a/mm/swapfile.c b/mm/swapfile.c index 408f6c72fb5a69..01e7b6b046b67d 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -1735,7 +1735,9 @@ static int swap_dup_entries_cluster(struct swap_info_struct *si, * swap cache. * * Context: Caller needs to hold the folio lock. - * Return: Whether the folio was added to the swap cache. + * Return: %0 on success, %-E2BIG if splitting the folio might allow swapout, + * %-ENOSPC if no global swap space is available, or %-ENOMEM if splitting + * would not help. */ int folio_alloc_swap(struct folio *folio) { @@ -1747,11 +1749,11 @@ int folio_alloc_swap(struct folio *folio) if (order) { /* - * Reject large allocation when THP_SWAP is disabled, - * the caller should split the folio and try again. + * Reject large allocation when THP_SWAP is disabled. Check below + * whether splitting and retrying can make progress. */ if (!IS_ENABLED(CONFIG_THP_SWAP)) - return -EAGAIN; + goto failed; /* * Allocation size should never exceed cluster size @@ -1759,7 +1761,7 @@ int folio_alloc_swap(struct folio *folio) */ if (size > SWAPFILE_CLUSTER) { VM_WARN_ON_ONCE(1); - return -EINVAL; + goto failed; } } @@ -1775,13 +1777,23 @@ int folio_alloc_swap(struct folio *folio) } /* Need to call this even if allocation failed, for MEMCG_SWAP_FAIL. */ - if (unlikely(mem_cgroup_try_charge_swap(folio))) + if (unlikely(mem_cgroup_try_charge_swap(folio))) { swap_cache_del_folio(folio); + goto failed; + } if (unlikely(!folio_test_swapcache(folio))) - return -ENOMEM; + goto failed; return 0; + +failed: + if (get_nr_swap_pages() <= 0) + return -ENOSPC; + if (mem_cgroup_get_folio_swap_margin(folio) <= 0) + return -ENOMEM; + + return order ? -E2BIG : -ENOMEM; } /** From 14a2a2c98651b1b814df94b5d4f7815f69aab83f Mon Sep 17 00:00:00 2001 From: "Barry Song (Xiaomi)" Date: Sun, 30 Aug 2026 12:29:19 +0800 Subject: [PATCH 0679/1352] mm/vmscan: avoid pointless large folio splits without swap When swap is disabled, exhausted, or unavailable due to memcg swap limits, splitting a large anonymous folio cannot make swapout progress. The fallback only destroys the large folio and inflates split statistics. Use -E2BIG from folio_alloc_swap() as the explicit signal that splitting the folio might allow swapout of smaller pieces. For other allocation failures, keep the existing activation path and avoid the split. This preserves the split fallback for fragmented or partially available swap, while avoiding it when there is no backing space for any part of the folio. Link: https://lore.kernel.org/20260830042920.2280454-4-xueyuan.chen21@gmail.com Signed-off-by: Barry Song (Xiaomi) Signed-off-by: Xueyuan Chen Signed-off-by: Andrew Morton Reported-by: Nanzhe Zhao Acked-by: David Hildenbrand (Arm) Reviewed-by: Baolin Wang Cc: Baoquan He Cc: Chris Li Cc: Hugh Dickins Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Nhat Pham Cc: Roman Gushchin Cc: Shakeel Butt Cc: Youngjun Park --- mm/vmscan.c | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 42fcdcd3d2e49d..ce3bab78af3cd1 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -1259,6 +1259,8 @@ static unsigned int shrink_folio_list(struct list_head *folio_list, */ if (folio_test_anon(folio) && folio_test_swapbacked(folio) && !folio_test_swapcache(folio)) { + int ret; + if (!(sc->gfp_mask & __GFP_IO)) goto keep_locked; if (folio_maybe_dma_pinned(folio)) @@ -1277,11 +1279,14 @@ static unsigned int shrink_folio_list(struct list_head *folio_list, split_folio_to_list(folio, folio_list)) goto activate_locked; } - if (folio_alloc_swap(folio)) { + ret = folio_alloc_swap(folio); + if (ret) { int __maybe_unused order = folio_order(folio); if (!folio_test_large(folio)) goto activate_locked_split; + if (ret != -E2BIG) + goto activate_locked; /* Fallback to swap normal pages */ if (split_folio_to_list(folio, folio_list)) goto activate_locked; From 130fe5079719aaa29a2b4c8187038449ce5574fb Mon Sep 17 00:00:00 2001 From: Xueyuan Chen Date: Sun, 30 Aug 2026 12:29:20 +0800 Subject: [PATCH 0680/1352] mm/shmem: split large folios only on -E2BIG shmem_writeout() currently splits a large folio on every folio_alloc_swap() failure. With the refined return-value contract, only -E2BIG indicates that splitting might allow smaller folios to be swapped out. Enter the split fallback only for -E2BIG. For -ENOSPC and -ENOMEM, redirty and reactivate the folio as before. Link: https://lore.kernel.org/20260830042920.2280454-5-xueyuan.chen21@gmail.com Signed-off-by: Xueyuan Chen Signed-off-by: Andrew Morton Suggested-by: Baolin Wang Reviewed-by: Baolin Wang Acked-by: David Hildenbrand (Arm) Reviewed-by: Barry Song Cc: Baoquan He Cc: Chris Li Cc: Hugh Dickins Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Nanzhe Zhao Cc: Nhat Pham Cc: Roman Gushchin Cc: Shakeel Butt Cc: Youngjun Park --- mm/shmem.c | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/mm/shmem.c b/mm/shmem.c index d3f24b5977bd5f..84f0a2eb85fecd 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -1813,7 +1813,7 @@ int shmem_writeout(struct swap_io_ctx *ctx, struct folio *folio, struct shmem_inode_info *info = SHMEM_I(inode); struct shmem_sb_info *sbinfo = SHMEM_SB(inode->i_sb); pgoff_t index; - int nr_pages; + int nr_pages, ret; bool split = false; if ((info->flags & SHMEM_F_LOCKED) || sbinfo->noswap) @@ -1894,7 +1894,8 @@ int shmem_writeout(struct swap_io_ctx *ctx, struct folio *folio, folio_mark_uptodate(folio); } - if (!folio_alloc_swap(folio)) { + ret = folio_alloc_swap(folio); + if (!ret) { bool first_swapped = shmem_recalc_inode(inode, 0, nr_pages); int error; @@ -1947,7 +1948,7 @@ int shmem_writeout(struct swap_io_ctx *ctx, struct folio *folio, swap_cache_del_folio(folio); goto redirty; } - if (nr_pages > 1) + if (nr_pages > 1 && ret == -E2BIG) goto try_split; redirty: folio_mark_dirty(folio); From 928c8159db3f660f82ce1bf5d7bf5e72e62a4329 Mon Sep 17 00:00:00 2001 From: Aristeu Rozanski Date: Thu, 27 Aug 2026 21:55:43 -0400 Subject: [PATCH 0681/1352] mm: gup: move pmd_protnone() into gup_fast_pmd_leaf() Patch series "mm: gup: cleanup gup_fast call chain", v3. These two patches implement the refactor in gup_fast call chain David Hildenbrand mentioned in [1]. This patch (of 2): Make pmd handling match pud handling by calling pmd_protnone() inside gup_fast_pmd_leaf(). Link: https://lore.kernel.org/20260828015542.125576330@ruivo.org Link: https://lore.kernel.org/20260828015542.245315718@ruivo.org Link: https://lore.kernel.org/all/85e760cf-b994-40db-8d13-221feee55c60@redhat.com/T/#u [1] Signed-off-by: Aristeu Rozanski Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand Link: https://lore.kernel.org/all/85e760cf-b994-40db-8d13-221feee55c60@redhat.com/T/#u Acked-by: David Hildenbrand (Arm) Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu --- mm/gup.c | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/mm/gup.c b/mm/gup.c index eb898ea1ee22e5..bd981ab3975143 100644 --- a/mm/gup.c +++ b/mm/gup.c @@ -2929,6 +2929,10 @@ static int gup_fast_pmd_leaf(pmd_t orig, pmd_t *pmdp, unsigned long addr, struct folio *folio; int refs; + /* See gup_fast_pte_range() */ + if (pmd_protnone(orig)) + return 0; + if (!pmd_access_permitted(orig, flags & FOLL_WRITE)) return 0; @@ -3024,10 +3028,6 @@ static int gup_fast_pmd_range(pud_t *pudp, pud_t pud, unsigned long addr, return 0; if (unlikely(pmd_leaf(pmd))) { - /* See gup_fast_pte_range() */ - if (pmd_protnone(pmd)) - return 0; - if (!gup_fast_pmd_leaf(pmd, pmdp, addr, next, flags, pages, nr)) return 0; From e14b14aa35099899d18688fbf34174143da288e6 Mon Sep 17 00:00:00 2001 From: Aristeu Rozanski Date: Thu, 27 Aug 2026 21:55:44 -0400 Subject: [PATCH 0682/1352] mm: gup: cleanup the gup_fast_*() call chain Refactor gup_fast functions so each step of the way returns the number of pages pinned. Because the previous step of the chain knows what the number it should be, less indicates an error. This way there's no need to pass *nr along. Link: https://lore.kernel.org/20260828015542.334186653@ruivo.org Signed-off-by: Aristeu Rozanski Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand Link: https://lore.kernel.org/all/85e760cf-b994-40db-8d13-221feee55c60@redhat.com/T/#u Acked-by: David Hildenbrand (Arm) Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu --- mm/gup.c | 179 +++++++++++++++++++++++++++++-------------------------- 1 file changed, 94 insertions(+), 85 deletions(-) diff --git a/mm/gup.c b/mm/gup.c index bd981ab3975143..a4036c02e2137f 100644 --- a/mm/gup.c +++ b/mm/gup.c @@ -2826,11 +2826,11 @@ static bool gup_fast_folio_allowed(struct folio *folio, unsigned int flags) * also check pmd here to make sure pmd doesn't change (corresponds to * pmdp_collapse_flush() in the THP collapse code path). */ -static int gup_fast_pte_range(pmd_t pmd, pmd_t *pmdp, unsigned long addr, - unsigned long end, unsigned int flags, struct page **pages, - int *nr) +static unsigned long gup_fast_pte_range(pmd_t pmd, pmd_t *pmdp, + unsigned long addr, unsigned long end, + unsigned int flags, struct page **pages) { - int ret = 0; + unsigned long nr_pages = 0; pte_t *ptep, *ptem; ptem = ptep = pte_offset_map(&pmd, addr); @@ -2892,15 +2892,13 @@ static int gup_fast_pte_range(pmd_t pmd, pmd_t *pmdp, unsigned long addr, goto pte_unmap; } folio_set_referenced(folio); - pages[*nr] = page; - (*nr)++; + pages[nr_pages] = page; + nr_pages++; } while (ptep++, addr += PAGE_SIZE, addr != end); - ret = 1; - pte_unmap: pte_unmap(ptem); - return ret; + return nr_pages; } #else @@ -2913,21 +2911,21 @@ static int gup_fast_pte_range(pmd_t pmd, pmd_t *pmdp, unsigned long addr, * get_user_pages_fast_only implementation that can pin pages. Thus it's still * useful to have gup_fast_pmd_leaf even if we can't operate on ptes. */ -static int gup_fast_pte_range(pmd_t pmd, pmd_t *pmdp, unsigned long addr, - unsigned long end, unsigned int flags, struct page **pages, - int *nr) +static unsigned long gup_fast_pte_range(pmd_t pmd, pmd_t *pmdp, + unsigned long addr, unsigned long end, + unsigned int flags, struct page **pages) { return 0; } #endif /* CONFIG_ARCH_HAS_PTE_SPECIAL */ -static int gup_fast_pmd_leaf(pmd_t orig, pmd_t *pmdp, unsigned long addr, - unsigned long end, unsigned int flags, struct page **pages, - int *nr) +static unsigned long gup_fast_pmd_leaf(pmd_t orig, pmd_t *pmdp, + unsigned long addr, unsigned long end, + unsigned int flags, struct page **pages) { struct page *page; struct folio *folio; - int refs; + unsigned long nr_pages, i; /* See gup_fast_pte_range() */ if (pmd_protnone(orig)) @@ -2939,42 +2937,40 @@ static int gup_fast_pmd_leaf(pmd_t orig, pmd_t *pmdp, unsigned long addr, if (pmd_special(orig)) return 0; - refs = (end - addr) >> PAGE_SHIFT; + nr_pages = (end - addr) >> PAGE_SHIFT; page = pmd_page(orig) + ((addr & ~PMD_MASK) >> PAGE_SHIFT); - folio = try_grab_folio_fast(page, refs, flags); + folio = try_grab_folio_fast(page, nr_pages, flags); if (!folio) return 0; if (unlikely(pmd_val(orig) != pmd_val(pmdp_get_lockless(pmdp)))) { - gup_put_folio(folio, refs, flags); + gup_put_folio(folio, nr_pages, flags); return 0; } if (!gup_fast_folio_allowed(folio, flags)) { - gup_put_folio(folio, refs, flags); + gup_put_folio(folio, nr_pages, flags); return 0; } if (!pmd_write(orig) && gup_must_unshare(NULL, flags, &folio->page)) { - gup_put_folio(folio, refs, flags); + gup_put_folio(folio, nr_pages, flags); return 0; } - pages += *nr; - *nr += refs; - for (; refs; refs--) + for (i = 0; i < nr_pages; i++) *(pages++) = page++; folio_set_referenced(folio); - return 1; + return nr_pages; } -static int gup_fast_pud_leaf(pud_t orig, pud_t *pudp, unsigned long addr, - unsigned long end, unsigned int flags, struct page **pages, - int *nr) +static unsigned long gup_fast_pud_leaf(pud_t orig, pud_t *pudp, + unsigned long addr, unsigned long end, + unsigned int flags, struct page **pages) { struct page *page; struct folio *folio; - int refs; + unsigned long nr_pages, i; if (!pud_access_permitted(orig, flags & FOLL_WRITE)) return 0; @@ -2982,41 +2978,39 @@ static int gup_fast_pud_leaf(pud_t orig, pud_t *pudp, unsigned long addr, if (pud_special(orig)) return 0; - refs = (end - addr) >> PAGE_SHIFT; + nr_pages = (end - addr) >> PAGE_SHIFT; page = pud_page(orig) + ((addr & ~PUD_MASK) >> PAGE_SHIFT); - folio = try_grab_folio_fast(page, refs, flags); + folio = try_grab_folio_fast(page, nr_pages, flags); if (!folio) return 0; if (unlikely(pud_val(orig) != pud_val(pudp_get(pudp)))) { - gup_put_folio(folio, refs, flags); + gup_put_folio(folio, nr_pages, flags); return 0; } if (!gup_fast_folio_allowed(folio, flags)) { - gup_put_folio(folio, refs, flags); + gup_put_folio(folio, nr_pages, flags); return 0; } if (!pud_write(orig) && gup_must_unshare(NULL, flags, &folio->page)) { - gup_put_folio(folio, refs, flags); + gup_put_folio(folio, nr_pages, flags); return 0; } - pages += *nr; - *nr += refs; - for (; refs; refs--) + for (i = 0; i < nr_pages; i++) *(pages++) = page++; folio_set_referenced(folio); - return 1; + return nr_pages; } -static int gup_fast_pmd_range(pud_t *pudp, pud_t pud, unsigned long addr, - unsigned long end, unsigned int flags, struct page **pages, - int *nr) +static unsigned long gup_fast_pmd_range(pud_t *pudp, pud_t pud, + unsigned long addr, unsigned long end, + unsigned int flags, struct page **pages) { - unsigned long next; + unsigned long next, nr_pages = 0, chunk_nr_pages; pmd_t *pmdp; pmdp = pmd_offset_lockless(pudp, pud, addr); @@ -3025,26 +3019,30 @@ static int gup_fast_pmd_range(pud_t *pudp, pud_t pud, unsigned long addr, next = pmd_addr_end(addr, end); if (!pmd_present(pmd)) - return 0; + break; if (unlikely(pmd_leaf(pmd))) { - if (!gup_fast_pmd_leaf(pmd, pmdp, addr, next, flags, - pages, nr)) - return 0; - - } else if (!gup_fast_pte_range(pmd, pmdp, addr, next, flags, - pages, nr)) - return 0; + chunk_nr_pages = gup_fast_pmd_leaf(pmd, pmdp, addr, + next, flags, + &pages[nr_pages]); + + } else + chunk_nr_pages = gup_fast_pte_range(pmd, pmdp, addr, + next, flags, + &pages[nr_pages]); + nr_pages += chunk_nr_pages; + if (chunk_nr_pages != (next - addr) >> PAGE_SHIFT) + break; } while (pmdp++, addr = next, addr != end); - return 1; + return nr_pages; } -static int gup_fast_pud_range(p4d_t *p4dp, p4d_t p4d, unsigned long addr, - unsigned long end, unsigned int flags, struct page **pages, - int *nr) +static unsigned long gup_fast_pud_range(p4d_t *p4dp, p4d_t p4d, + unsigned long addr, unsigned long end, + unsigned int flags, struct page **pages) { - unsigned long next; + unsigned long next, nr_pages = 0, chunk_nr_pages; pud_t *pudp; pudp = pud_offset_lockless(p4dp, p4d, addr); @@ -3053,24 +3051,27 @@ static int gup_fast_pud_range(p4d_t *p4dp, p4d_t p4d, unsigned long addr, next = pud_addr_end(addr, end); if (unlikely(!pud_present(pud))) - return 0; - if (unlikely(pud_leaf(pud))) { - if (!gup_fast_pud_leaf(pud, pudp, addr, next, flags, - pages, nr)) - return 0; - } else if (!gup_fast_pmd_range(pudp, pud, addr, next, flags, - pages, nr)) - return 0; + break; + if (unlikely(pud_leaf(pud))) + chunk_nr_pages = gup_fast_pud_leaf(pud, pudp, addr, + next, flags, + &pages[nr_pages]); + else + chunk_nr_pages = gup_fast_pmd_range(pudp, pud, addr, + next, flags, + &pages[nr_pages]); + nr_pages += chunk_nr_pages; + if (chunk_nr_pages != (next - addr) >> PAGE_SHIFT) + break; } while (pudp++, addr = next, addr != end); - return 1; + return nr_pages; } -static int gup_fast_p4d_range(pgd_t *pgdp, pgd_t pgd, unsigned long addr, - unsigned long end, unsigned int flags, struct page **pages, - int *nr) +static unsigned long gup_fast_p4d_range(pgd_t *pgdp, pgd_t pgd, unsigned long addr, + unsigned long end, unsigned int flags, struct page **pages) { - unsigned long next; + unsigned long next, nr_pages = 0, chunk_nr_pages; p4d_t *p4dp; p4dp = p4d_offset_lockless(pgdp, pgd, addr); @@ -3079,20 +3080,23 @@ static int gup_fast_p4d_range(pgd_t *pgdp, pgd_t pgd, unsigned long addr, next = p4d_addr_end(addr, end); if (!p4d_present(p4d)) - return 0; + break; BUILD_BUG_ON(p4d_leaf(p4d)); - if (!gup_fast_pud_range(p4dp, p4d, addr, next, flags, - pages, nr)) - return 0; + chunk_nr_pages = gup_fast_pud_range(p4dp, p4d, addr, next, + flags, &pages[nr_pages]); + nr_pages += chunk_nr_pages; + if (chunk_nr_pages != (next - addr) >> PAGE_SHIFT) + break; } while (p4dp++, addr = next, addr != end); - return 1; + return nr_pages; } -static void gup_fast_pgd_range(unsigned long addr, unsigned long end, - unsigned int flags, struct page **pages, int *nr) +static unsigned long gup_fast_pgd_range(unsigned long addr, + unsigned long end, unsigned int flags, + struct page **pages) { - unsigned long next; + unsigned long next, nr_pages = 0, chunk_nr_pages; pgd_t *pgdp; pgdp = pgd_offset(current->mm, addr); @@ -3101,17 +3105,23 @@ static void gup_fast_pgd_range(unsigned long addr, unsigned long end, next = pgd_addr_end(addr, end); if (pgd_none(pgd)) - return; + break; BUILD_BUG_ON(pgd_leaf(pgd)); - if (!gup_fast_p4d_range(pgdp, pgd, addr, next, flags, - pages, nr)) - return; + chunk_nr_pages = gup_fast_p4d_range(pgdp, pgd, addr, next, + flags, &pages[nr_pages]); + nr_pages += chunk_nr_pages; + if (chunk_nr_pages != (next - addr) >> PAGE_SHIFT) + break; } while (pgdp++, addr = next, addr != end); + + return nr_pages; } #else -static inline void gup_fast_pgd_range(unsigned long addr, unsigned long end, - unsigned int flags, struct page **pages, int *nr) +static inline unsigned long gup_fast_pgd_range(unsigned long addr, + unsigned long end, unsigned int flags, + struct page **pages) { + return 0; } #endif /* CONFIG_HAVE_GUP_FAST */ @@ -3129,8 +3139,7 @@ static bool gup_fast_permitted(unsigned long start, unsigned long end) static unsigned long gup_fast(unsigned long start, unsigned long end, unsigned int gup_flags, struct page **pages) { - unsigned long flags; - int nr_pinned = 0; + unsigned long flags, nr_pinned; unsigned seq; if (!IS_ENABLED(CONFIG_HAVE_GUP_FAST) || @@ -3154,7 +3163,7 @@ static unsigned long gup_fast(unsigned long start, unsigned long end, * that come from callers of tlb_remove_table_sync_one(). */ local_irq_save(flags); - gup_fast_pgd_range(start, end, gup_flags, pages, &nr_pinned); + nr_pinned = gup_fast_pgd_range(start, end, gup_flags, pages); local_irq_restore(flags); /* From ddb1e5e20c0edc5fb957276ecac3f6c2ca78b663 Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Wed, 19 Aug 2026 10:55:15 +0800 Subject: [PATCH 0683/1352] mm/page_table_check: add explicit pmd_none check in pte_clear_range In __page_table_check_pte_clear_range(), the condition to determine whether to iterate over PTEs only checked pmd_bad() and pmd_leaf(). This relies on the implicit assumption that pmd_none() is always a subset of pmd_bad() on all architectures supporting PAGE_TABLE_CHECK. While this assumption currently holds for x86_64, arm64, s390, riscv, and powerpc, it is an architecture-dependent behavior that may not hold for future architectures. Add an explicit pmd_none() check to make the intent clear and avoid calling pte_offset_map() on an empty PMD, which could lead to undefined behavior. Link: https://lore.kernel.org/20260819025516.2967199-1-ye.liu@linux.dev Signed-off-by: Ye Liu Signed-off-by: Andrew Morton Cc: Albert Ou Cc: Alexandre Ghiti Cc: Palmer Dabbelt Cc: Pasha Tatashin --- mm/page_table_check.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/page_table_check.c b/mm/page_table_check.c index 6ffc536359cd06..ed3e1a76f26643 100644 --- a/mm/page_table_check.c +++ b/mm/page_table_check.c @@ -278,7 +278,7 @@ void __page_table_check_pte_clear_range(struct mm_struct *mm, if (&init_mm == mm) return; - if (!pmd_bad(pmd) && !pmd_leaf(pmd)) { + if (!pmd_none(pmd) && !pmd_bad(pmd) && !pmd_leaf(pmd)) { pte_t *ptep = pte_offset_map(&pmd, addr); unsigned long i; From 9b7c2d34a727f59d8f10b35561929553e808d483 Mon Sep 17 00:00:00 2001 From: Zhiling Zou Date: Tue, 21 Jul 2026 23:56:38 +0800 Subject: [PATCH 0684/1352] mm/page_table_check: skip zero pages page_table_check_set() accounts pages by whether they are PageAnon(). The shared zero page is a special page, not an ordinary file-backed page. Read faults on private anonymous mappings can install many read-only PTEs that point at the zero page, but page_table_check currently accounts them in file_map_count. That lets an unprivileged process populate enough zero-page mappings to overflow file_map_count and trip the BUG_ON() in page_table_check_set(). Skip zero pages in page_table_check accounting. They do not need the anonymous/file mapping conflict checks that page_table_check performs for ordinary pages, and this keeps the existing counter size and page_ext layout unchanged. Link: https://lore.kernel.org/1f8848512d2e3ded944f8d595c29faee8fdaeab0.1784645969.git.roxy520tt@gmail.com Fixes: df4e817b7108 ("mm: page table check") Signed-off-by: Zhiling Zou Signed-off-by: Ren Wei Signed-off-by: Andrew Morton Reported-by: Vega Assisted-by: Codex:gpt-5.4 Cc: Ye Liu Cc: --- mm/page_table_check.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/mm/page_table_check.c b/mm/page_table_check.c index ed3e1a76f26643..143a918c0bde22 100644 --- a/mm/page_table_check.c +++ b/mm/page_table_check.c @@ -67,7 +67,7 @@ static void page_table_check_clear(unsigned long pfn, unsigned long pgcnt) struct page *page; bool anon; - if (!pfn_valid(pfn)) + if (!pfn_valid(pfn) || is_zero_pfn(pfn) || is_huge_zero_pfn(pfn)) return; page = pfn_to_page(pfn); @@ -102,7 +102,7 @@ static void page_table_check_set(unsigned long pfn, unsigned long pgcnt, struct page *page; bool anon; - if (!pfn_valid(pfn)) + if (!pfn_valid(pfn) || is_zero_pfn(pfn) || is_huge_zero_pfn(pfn)) return; page = pfn_to_page(pfn); From dcfdb1915f33828c9d5eb3989130a8fa1c0a6514 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:18:42 -0700 Subject: [PATCH 0685/1352] mm/damon/core: skip applying scheme if region split for quota fails Patch series "mm/damon: fix DAMOS bugs in core, paddr and vaddr". Fix misc bugs of DAMOS. Patch 1 makes DAMOS less stress memory allocator under extreme situation. Patches 2 and 3 fix wrong folios walking in DAMON_PADDR. Patches 4 and 5 fix wrong folios walking in DAMON_VADDR. Patches 6-8 handle extreme and unlikely memory situations that can cause divide by zero and underflow. All bugs are discovered by Sashiko. This patch (of 8): damos_apply_scheme() splits a region and apply the action to the subregion if it is needed for not violating the quota. The split operation (damon_split_region_at()) could fail for allocation failure. In the case, the quota could be violated. From the user's perspective, DAMOS becomes more aggressive than expected under the extreme situation. Handle the failure. The user impact is not critical. The failure of damon_split_region_at() is unlikely since it is arguably too small to fail. Also DAMOS being aggressive is limited to the single region. Users can set min_nr_regions to set the maximum size of each region. If it is reasonably set, the transient overhead shouldn't be critical. The issue was discovered [1] by Sashiko. Link: https://lore.kernel.org/20260901131850.98037-1-sj@kernel.org Link: https://lore.kernel.org/20260901131850.98037-2-sj@kernel.org Link: https://lore.kernel.org/20260718171523.87547-1-sj@kernel.org [1] Fixes: 2b8a248d5873 ("mm/damon/schemes: implement size quota for schemes application speed control") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Usama Arif Cc: # 5.16.x --- mm/damon/core.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index d63d4c6fd3ffdd..53952d16ebd882 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2607,7 +2607,8 @@ static void damos_apply_scheme(struct damon_ctx *c, struct damon_target *t, c->min_region_sz); if (!sz) goto update_stat; - damon_split_region_at(t, r, sz); + if (damon_split_region_at(t, r, sz)) + goto update_stat; } if (damos_core_filter_out(c, t, r, s)) return; From b40b0e51821e29371027f768b0051be62b9b01ad Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:18:43 -0700 Subject: [PATCH 0686/1352] mm/damon/paddr: respect folio end for DAMOS_STAT The function for applying DAMOS_STAT in DAMON physical address space operation set (paddr), namely damon_pa_stat(), applies DAMOS filters to folios of the given region. For that, it gets folios of addresses in the region. It starts from the region start address and advances the address by the size of the folio of the address until it goes out of the region. If the start address is in the middle of a large folio, and if the next folios are small, some of the next folios could be skipped. Fix the issue by advancing the address to exactly the start address of the next folio. The user impact is that the DAMOS_STAT-based page level monitoring results become inaccurate. Since the page level monitoring is supposed to provide relatively high precision, this is definitely a problem. It is arguably not critical since it is only monitoring quality degradation. Link: https://lore.kernel.org/20260901131850.98037-3-sj@kernel.org Fixes: bdbe1d7bc325 ("mm/damon/paddr: increment pa_stat damon address range by folio size") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Usama Arif Cc: # 6.14.x --- mm/damon/paddr.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c index 5c6c3a597fd0bf..2ab7b3842701ed 100644 --- a/mm/damon/paddr.c +++ b/mm/damon/paddr.c @@ -379,7 +379,7 @@ static unsigned long damon_pa_stat(struct damon_region *r, if (!damos_pa_filter_out(s, folio)) *sz_filter_passed += folio_size(folio) / addr_unit; - addr += folio_size(folio); + addr = PFN_PHYS(folio_pfn(folio)) + folio_size(folio); folio_put(folio); } s->last_applied = folio; From e60f236f792584c6ebf92c2a18d8879a052744fa Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:18:44 -0700 Subject: [PATCH 0687/1352] mm/damon/paddr: respect folio end for DAMOS actions except STAT A few functions for applying DAMOS actions including pageout, lru_[de]prio and migrate_{hot,cold} in DAMON physical address space operation set (paddr) collect folios of the given region by getting the folios of region-internal addresses. Then, those functions apply the action to the collected folios at once. The collection starts from the region start address and advances the address by the size of the folio of the address until it goes out of the region. If the start address is in the middle of a large folio, and if the next folios are small, some of the next folios could be skipped. Fix the issue by advancing the address to exactly the start address of the next folio. The user impact is that DAMOS action is applied to less than expected amount of memory. Given the best effort nature of DAMON, it is no big problem, but it is clearly a bug that is better to be fixed. The issue was discovered [1] by Sashiko. Link: https://lore.kernel.org/20260901131850.98037-4-sj@kernel.org Link: https://lore.kernel.org/20260517234112.89245-1-sj@kernel.org [1] Fixes: 3a06696305e7 ("mm/damon/ops: have damon_get_folio return folio even for tail pages") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Usama Arif Cc: # 6.15.x --- mm/damon/paddr.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c index 2ab7b3842701ed..9ddd1ec8202b7f 100644 --- a/mm/damon/paddr.c +++ b/mm/damon/paddr.c @@ -264,7 +264,7 @@ static unsigned long damon_pa_pageout(struct damon_region *r, else list_add(&folio->lru, &folio_list); put_folio: - addr += folio_size(folio); + addr = PFN_PHYS(folio_pfn(folio)) + folio_size(folio); folio_put(folio); } if (install_young_filter) @@ -302,7 +302,7 @@ static inline unsigned long damon_pa_de_activate( folio_deactivate(folio); applied += folio_nr_pages(folio); put_folio: - addr += folio_size(folio); + addr = PFN_PHYS(folio_pfn(folio)) + folio_size(folio); folio_put(folio); } s->last_applied = folio; @@ -350,7 +350,7 @@ static unsigned long damon_pa_migrate(struct damon_region *r, folio_is_file_lru(folio)); list_add(&folio->lru, &folio_list); put_folio: - addr += folio_size(folio); + addr = PFN_PHYS(folio_pfn(folio)) + folio_size(folio); folio_put(folio); } applied = damon_migrate_pages(&folio_list, s->target_nid); From 84278209a4b393181ad3b149908d39cae5f12a66 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:18:45 -0700 Subject: [PATCH 0688/1352] mm/damon/vaddr: respect folio end for DAMOS_STAT For applying DAMOS_STAT action to a region, DAMON virtual address space operation set (vaddr) calls walk_page_range[_vma]() for the region. The pmd walk entry function, namely damon_va_stat_pmd_entry(), applies DAMOS filters to folios of addresses of the region in the pmd. It starts from the walking address and advances the address by the size of the folio of the address until it goes out of the pmd or the region. Let's suppose it is for the first pmd of the region, and the region start address is in the middle of a large folio. Also, the next folios are small. Then, some of the next folios could be skipped. Fix the issue by advancing the address to exactly the start address of the next folio. The user impact is that the DAMOS_STAT-based page level monitoring results become inaccurate. Since the page level monitoring is supposed to provide relatively high precision, this is definitely a problem. It is arguably not critical since it is only monitoring quality degradation. The issue was discovered [1] by Sashiko. Link: https://lore.kernel.org/20260901131850.98037-5-sj@kernel.org Link: https://lore.kernel.org/20260514015053.149396-1-sj@kernel.org [1] Fixes: 63f39737d1e3 ("mm/damon/vaddr: support stat-purpose DAMOS filters") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Usama Arif Cc: # 6.18.x --- mm/damon/vaddr.c | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/mm/damon/vaddr.c b/mm/damon/vaddr.c index 04ee2a2c6a4d65..b1bed5d19a34b8 100644 --- a/mm/damon/vaddr.c +++ b/mm/damon/vaddr.c @@ -838,6 +838,8 @@ static int damos_va_stat_pmd_entry(pmd_t *pmd, unsigned long addr, return 0; for (; addr < next; pte += nr, addr += nr * PAGE_SIZE) { + unsigned long page_idx; + nr = 1; ptent = ptep_get(pte); @@ -851,7 +853,8 @@ static int damos_va_stat_pmd_entry(pmd_t *pmd, unsigned long addr, if (!damos_va_filter_out(s, folio, vma, addr, pte, NULL)) *sz_filter_passed += folio_size(folio); - nr = folio_nr_pages(folio); + page_idx = folio_page_idx(folio, pte_page(ptent)); + nr = folio_nr_pages(folio) - page_idx; s->last_applied = folio; } pte_unmap_unlock(start_pte, ptl); From a24b6c86333bcf459526fb65b2b84705e01d7fa8 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:18:46 -0700 Subject: [PATCH 0689/1352] mm/damon/vaddr: respect folio end for DAMOS_MIGRATE_{HOT,COLD} For applying DAMOS_MIGRATE_{HOT,COLD} actions to a region, DAMON virtual address space operation set (vaddr) calls walk_page_range[_vma]() for the region. The pmd walk entry function, namely damon_va_migrate_pmd_entry(), collects folios of addresses of the region in the pmd. It starts from the walking address and advances the address by the size of the folio of the address until it goes out of the pmd or the region. Let's suppose it is for the first pmd of the region, and the region start address is in the middle of a large folio. Also, the next folios are small. Then, some of the next folios could be skipped. Fix the issue by advancing the address to exactly the start address of the next folio. The user impact is that DAMOS action is applied to less than expected amount of memory. Given the best effort nature of DAMON, it is no big problem, but it is clearly a bug that is better to be fixed. The issue was discovered [1] by Sashiko. Link: https://lore.kernel.org/20260901131850.98037-6-sj@kernel.org Link: https://lore.kernel.org/20260514015053.149396-1-sj@kernel.org [1] Fixes: 09efc56a3b1c ("mm/damon/vaddr: consistently use only pmd_entry for damos_migrate") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Usama Arif Cc: # 6.19.x --- mm/damon/vaddr.c | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/mm/damon/vaddr.c b/mm/damon/vaddr.c index b1bed5d19a34b8..5b4d16c8db6285 100644 --- a/mm/damon/vaddr.c +++ b/mm/damon/vaddr.c @@ -676,6 +676,8 @@ static int damos_va_migrate_pmd_entry(pmd_t *pmd, unsigned long addr, return 0; for (; addr < next; pte += nr, addr += nr * PAGE_SIZE) { + unsigned long page_idx; + nr = 1; ptent = ptep_get(pte); @@ -688,7 +690,8 @@ static int damos_va_migrate_pmd_entry(pmd_t *pmd, unsigned long addr, continue; damos_va_migrate_dests_add(folio, walk->vma, addr, dests, migration_lists); - nr = folio_nr_pages(folio); + page_idx = folio_page_idx(folio, pte_page(ptent)); + nr = folio_nr_pages(folio) - page_idx; } pte_unmap_unlock(start_pte, ptl); return 0; From 437f8f72c1bbf8b28780550906ee5d4350c1f9a9 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:18:47 -0700 Subject: [PATCH 0690/1352] mm/damon/core: handle extreme memory state in damon_get_node_mem_bp() In an extreme and unlikely situation, si_meminfo_node() might let the caller show zero total ram. That could cause a divide by zero in damon_get_node_mem_bp(). It could also show free memory larger than the total memory. This could cause underflow and make DAMOS temporarily make unexpected behavior. Thanks to safety guards in the auto-tuning feedback loop, that should not be a real problem, though. Fix the problems by respectively returning 100% and 0% for used and free memory queries in the corner cases. The issue was discovered [1] by Sashiko. Link: https://lore.kernel.org/20260901131850.98037-7-sj@kernel.org Link: https://lore.kernel.org/20260328133216.9697-1-sj@kernel.org [1] Fixes: 0e1c773b501f ("mm/damon/core: introduce damos quota goal metrics for memory node utilization") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Usama Arif Cc: # 6.16.x --- mm/damon/core.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/mm/damon/core.c b/mm/damon/core.c index 53952d16ebd882..2942f23a240b95 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2810,6 +2810,13 @@ static __kernel_ulong_t damos_get_node_mem_bp( } si_meminfo_node(&i, goal->nid); + if (!i.totalram || i.totalram < i.freeram) { + if (goal->metric == DAMOS_QUOTA_NODE_MEM_USED_BP) + return 10000; + else /* DAMOS_QUOTA_NODE_MEM_FREE_BP */ + return 0; + } + if (goal->metric == DAMOS_QUOTA_NODE_MEM_USED_BP) numerator = i.totalram - i.freeram; else /* DAMOS_QUOTA_NODE_MEM_FREE_BP */ From de8ee307aea33b74ceec8ddf02750697dae985c8 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:18:48 -0700 Subject: [PATCH 0691/1352] mm/damon/core: handle extreme memory state in get_node_memcg_used_bp() In extreme unlikely situations, total memory might be zero. In less extreme but still very unlikely situations, lruvec_page_state() calls might let the caller show used memory larger than total memory. In the two cases, damos_get_node_memcg_used_bp() could cause division by zero, or return underflowed value, respectively. Handle the cases by respectively returning 100% and 0% for used and free memory queries in the corner cases. This issue was discovered [1] by Sashiko. Link: https://lore.kernel.org/20260901131850.98037-8-sj@kernel.org Link: https://lore.kernel.org/20260329154813.47382-1-sj@kernel.org [1] Fixes: b74a120bcf50 ("mm/damon/core: implement DAMOS_QUOTA_NODE_MEMCG_USED_BP") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Usama Arif Cc: # 6.19.x --- mm/damon/core.c | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/mm/damon/core.c b/mm/damon/core.c index 2942f23a240b95..d00cf1fae23909 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2857,6 +2857,12 @@ static unsigned long damos_get_node_memcg_used_bp( mem_cgroup_put(memcg); si_meminfo_node(&i, goal->nid); + if (!i.totalram || i.totalram < used_pages) { + if (goal->metric == DAMOS_QUOTA_NODE_MEMCG_USED_BP) + return 10000; + else /* DAMOS_QUOTA_NODE_MEMCG_FREE_BP */ + return 0; + } if (goal->metric == DAMOS_QUOTA_NODE_MEMCG_USED_BP) numerator = used_pages; else /* DAMOS_QUOTA_NODE_MEMCG_FREE_BP */ From 69cac8d5d387d3eb61e3aec1ccdb26f8fc74ac5c Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:18:49 -0700 Subject: [PATCH 0692/1352] mm/damon/core: handle extreme memory state in get_in_active_mem_bp() damos_get_in_active_mem_bp() uses the sum of the active and inactive memory amount as a denominator. In an extreme and unlikely environment, active and inactive memory might be zero. In this case, hence, it results in a divide by zero problem. Avoid it by changing the denominator to one if it is zero, before it is being used. The issue was discovered [1] by Sashiko. Link: https://lore.kernel.org/20260901131850.98037-9-sj@kernel.org Link: https://lore.kernel.org/20260721034756.147011-1-sj@kernel.org [1] Fixes: 4835e2871321 ("mm/damon/core: introduce [in]active memory ratio damos quota goal metric") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Usama Arif Cc: # 7.0.x --- mm/damon/core.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index d00cf1fae23909..f6a9d2da0cd721 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -3009,7 +3009,7 @@ static unsigned int damos_get_in_active_mem_bp(bool active_ratio) global_node_page_state(NR_LRU_BASE + LRU_ACTIVE_FILE); inactive = global_node_page_state(NR_LRU_BASE + LRU_INACTIVE_ANON) + global_node_page_state(NR_LRU_BASE + LRU_INACTIVE_FILE); - total = active + inactive; + total = max(active + inactive, 1); if (active_ratio) return mult_frac(active, 10000, total); return mult_frac(inactive, 10000, total); From 21319731fea3cbc92b67625e3bd3f103dce0f639 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:48 -0700 Subject: [PATCH 0693/1352] mm/damon/core: introduce DAMON_FILTER_TYPE_PGIDLE_UNSET Patch series "mm/damon: introduce data access-as-a-data attribute", v1.1. TL;DR: extend DAMON's data attributes monitoring system to support page table accessed bit and PG_idle based access monitoring. DAMON was initially introduced as a data access monitor. Users found data access pattern becomes more useful when it is combined with other data attributes such as belonging cgroups and backing page types. For such cases, DAMON has extended to support such data attributes monitoring in addition to the original data access monitoring. The DAMON probes system was introduced for this purpose. Users set probes for filtering data attributes of their interest. For use cases where the primary interests are the attributes but the access pattern, probe weights system has been introduced. When it is used, DAMON applies its adaptive regions adjustment based on the monitored data attributes. However, DAMON stops access monitoring when the probe weights are used. DAMON cannot optimally help users who have interests in both data attributes and access patterns. Data access can also be thought of as another data attribute, though. Extend the probe system to support data access as a data attribute. Introduce a new probe filter type, pgidle_unset. It shows if the page is not set as idle. Specifically, it shows the page table accessed bit and the PG_idle flag. It can inform if the region is ever accessed. But it cannot say when it is accessed. To answer the second question, introduce a new probe feature, prep actions. Using the features, Users can specify what preparation actions should be made to each region for each probe. DAMON executes the preparation actions for each sampling interval, like it is doing the preparation for access check in the access monitoring mode. To help 'pgidle_unset' probe action use case, 'set_pgidle' preparation action is introduced together. The action does exactly what the access monitoring was doing: clearing the page table accessed bits and setting the PG_idle flags. Tests ===== I compared the access pattern monitoring results from the classic way and the probe based way. As expected, the probe based way shows the results similar to that of the classic way. More detailed test methods and results are below. First, do the access monitoring using the DAMON user-space tool [1] in the classic way. The system is idle. It shows no access as expected. $ sudo ./damo/damo start $ sudo ./damo/damo report access heatmap: 00000000000000000000000000000000000000008999999811111110000000000000000000000000 # min/max temperatures: -640,000,000, -100,000,000, column size: 99.800 MiB intervals: sample 5 ms aggr 100 ms (max access hz 200) 0 addr 4.000 KiB size 3.898 GiB access 0 hz age 6.400 s 1 addr 3.898 GiB size 787.301 MiB access 0 hz age 1 s 2 addr 4.667 GiB size 773.457 MiB access 0 hz age 5.800 s 3 addr 5.423 GiB size 2.374 GiB access 0 hz age 6.400 s memory bw estimate: 0 B per second total size: 7.797 GiB record DAMON intervals: sample 5 ms, aggr 100 ms Start an artificial memory access generator (masim) [2] in another window. $ ./masim/masim.py run --config_file ./masim/configs/zigzag.cfg Show the monitoring results. As expected, it captures accesses. $ sudo ./damo/damo report access heatmap: 00000000000000000000000000000000000000011111111111111118888888833378988888888888 # min/max temperatures: -1,470,000,000, -12,482,536, column size: 99.800 MiB intervals: sample 5 ms aggr 100 ms (max access hz 200) 0 addr 4.000 KiB size 1.542 GiB access 0 hz age 14.700 s 1 addr 1.542 GiB size 788.324 MiB access 0 hz age 14.600 s 2 addr 2.311 GiB size 772.844 MiB access 0 hz age 14.200 s [...] 50 addr 6.839 GiB size 1.742 MiB access 0 hz age 1.300 s 51 addr 6.840 GiB size 8.000 KiB access 100 hz age 0 ns 52 addr 6.840 GiB size 1.496 MiB access 160 hz age 0 ns [...] 97 addr 7.162 GiB size 1.199 MiB access 40 hz age 400 ms 98 addr 7.163 GiB size 1.199 MiB access 20 hz age 400 ms 99 addr 7.164 GiB size 2.004 MiB access 160 hz age 0 ns 100 addr 7.166 GiB size 646.004 MiB access 0 hz age 400 ms memory bw estimate: 26.148 GiB per second total size: 7.797 GiB record DAMON intervals: sample 5 ms, aggr 100 ms After the artificial memory access generator (masim) is terminated, restart DAMON with the probe-based access monitoring. As expected, it shows no access since the system is idle again. $ sudo ./damo/damo stop $ sudo ./damo/damo start --probe_prep set_pgidle \ --probe_filter allow pgidle_unset --probe_weight 1 $ sudo ./damo/damo report attrs heatmap: 00000000000000000000000000000000000000008999999711111100000000000000000000000000 # min/max temperatures: -600,000,000, -430,000,000, column size: 99.800 MiB probe prep: set_pgidle, filter: allow pgidle_unset (weight: 1) intervals: sample 5 ms aggr 100 ms (max probe hits 20) # addr size age probe_hits 0 4.000 KiB 3.898 GiB 6 s 0 1 5.285 GiB 2.512 GiB 6 s 0 2 4.659 GiB 641.816 MiB 5.700 s 0 3 3.898 GiB 778.375 MiB 4.300 s 0 memory bw estimate: 0 B per second total size: 7.797 GiB record DAMON intervals: sample 5 ms, aggr 100 ms Start the artificial memory access generator [2] again. $ ./masim/masim.py run --config_file ./masim/configs/zigzag.cfg Show the monitoring results. As expected, it captures accesses similar to the classic monitoring mode. $ sudo ./damo/damo report attrs heatmap: 00000000000000000000000000000000000000011111110000000177777777777777878798887777 # min/max temperatures: -1,330,000,000, 84,847,514, column size: 99.800 MiB probe prep: set_pgidle, filter: allow pgidle_unset (weight: 1) intervals: sample 5 ms aggr 100 ms (max probe hits 20) # addr size age probe_hits 0 4.000 KiB 1.508 GiB 13.300 s 0 1 1.508 GiB 794.684 MiB 13.200 s 0 2 2.284 GiB 764.555 MiB 13 s 0 [...] 50 6.711 GiB 4.625 MiB 0 ns 17 51 6.716 GiB 2.504 MiB 0 ns 17 52 6.732 GiB 796.000 KiB 0 ns 17 [...] 90 7.162 GiB 512.000 KiB 2.200 s 2 91 7.061 GiB 700.000 KiB 2.300 s 1 92 6.935 GiB 316.000 KiB 2.300 s 4 memory bw estimate: 0 B per second total size: 7.797 GiB record DAMON intervals: sample 5 ms, aggr 100 ms Patches Sequence ================ First four patches (patches 1-4) introduce the new probe filter type for knowing if a region is accessed. Patch 1 defines the new type in the core. Patch 2 implements the execution of the new filter in the physical address space DAMON operation set. Patch 3 implements a user interface on DAMON sysfs interface. Patch 4 updates the documentation for the new filter type. Following 13 patches (patches 5-17) introduce the probe preparation actions feature. Patch 5 defines the data structure for specifying the preparation actions. Patch 6 completes setup of the API parameter for the prep. Patch 7 extends the DAMON operation set callback list to connect the parameter with the underlying operation set. Patch 8 implements the execution of the prep in the physical address space DAMON operation set. Following five patches (patches 9-13) extends DAMON sysfs interface for the new prep feature. Patch 14 adds simple selftest for basic file operations of the new sysfs files. Final three patches (patches 15-17) respectively update design, usage and ABI documents for the new feature and its interface. This patch (of 17): Introduce a new DAMON filter type, pgidle_unset. It will match pages that have their PG_Idle flag unset, or the page table accessed bit set. In other words, it says if the page is accessed. Link: https://lore.kernel.org/20260901132506.99243-1-sj@kernel.org Link: https://lore.kernel.org/20260901132506.99243-2-sj@kernel.org Link: https://github.com/damonitor/damo [1] Link: https://github.com/sjp38/masim [2] Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/damon.h | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index 7b1b6050a8286f..7a42cbe791845d 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -745,12 +745,14 @@ struct damon_intervals_goal { /** * enum damon_filter_type - Type of &struct damon_filter * - * @DAMON_FILTER_TYPE_ANON: Anonymous pages. - * @DAMON_FILTER_TYPE_MEMCG: Specific memcg's pages. + * @DAMON_FILTER_TYPE_ANON: Anonymous pages. + * @DAMON_FILTER_TYPE_MEMCG: Specific memcg's pages. + * @DAMON_FILTER_TYPE_PGIDLE_UNSET: Pgidle is unset. */ enum damon_filter_type { DAMON_FILTER_TYPE_ANON, DAMON_FILTER_TYPE_MEMCG, + DAMON_FILTER_TYPE_PGIDLE_UNSET, }; /** From 3864315b3521b42025634b639c95414d8b28bf41 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:49 -0700 Subject: [PATCH 0694/1352] mm/damon/paddr: support PGIDLE_UNSET probe filter type Implement support of DAMON_FILTER_TYPE_PGIDLE_UNSET in the physical address space DAMON operations set. It reuses damon_folio_young(), which was being used for access monitoring. Link: https://lore.kernel.org/20260901132506.99243-3-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/paddr.c | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c index 9ddd1ec8202b7f..6f756f84938948 100644 --- a/mm/damon/paddr.c +++ b/mm/damon/paddr.c @@ -132,6 +132,12 @@ static bool damon_pa_filter_match(struct damon_filter *filter, matched = filter->memcg_id == mem_cgroup_id(memcg); rcu_read_unlock(); break; + case DAMON_FILTER_TYPE_PGIDLE_UNSET: + if (!folio) + matched = false; + else + matched = damon_folio_young(folio); + break; default: break; } From cdce3e5649b90b0c66a8f8e9461bfc373edd6f55 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:50 -0700 Subject: [PATCH 0695/1352] mm/damon/sysfs: support pgidle_unset probe filter type Extend DAMON sysfs interface to allow users to set DAMON_FILTER_TYPE_PGIDLE_UNSET by writing 'pgidle_unset' to the probe filter type file. Link: https://lore.kernel.org/20260901132506.99243-4-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index 3c81b4c91ac0dd..c1ff739dab27a8 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -781,6 +781,10 @@ damon_sysfs_filter_type_names[] = { .type = DAMON_FILTER_TYPE_MEMCG, .name = "memcg", }, + { + .type = DAMON_FILTER_TYPE_PGIDLE_UNSET, + .name = "pgidle_unset", + }, }; static ssize_t type_show(struct kobject *kobj, From e966539231b077d21c7d7b10d349a794af420c92 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:51 -0700 Subject: [PATCH 0696/1352] Docs/mm/damon/design: document pgidle_unset probe filter type Update DAMON design document for the newly added pgidle_unset probe filter type. Also use a list for the types, as it becomes not very easy to read the whole types in a simple sentence. Link: https://lore.kernel.org/20260901132506.99243-5-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/mm/damon/design.rst | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index 63cbb7b536da20..947d91ae24a392 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -293,8 +293,12 @@ registration is made by specifying a probe per attribute. Each of the probe specifies a rule to determine if a given memory region has the related attribute. The rule is constructed with multiple filters. The filters work same to :ref:`DAMOS filters ` except the supported -filter types. Currently only ``anon`` and ``memcg`` filter types are supported -for data attributes monitoring. +filter types. Currently below filter types are supported. + +- ``anon``: Same to that for DAMOS filters. +- ``memcg``: Same to that for DAMOS filters. +- ``pgidle_unset``: Matches if the page for the memory is marked as not + access-idle. If such probes are registered, DAMON executes the probes for each region's sampling memory when it does the access :ref:`sampling From 6ccf10b7e1e0c87f12cf8e5403a450891b6b1d16 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:52 -0700 Subject: [PATCH 0697/1352] mm/damon/core: introduce damon_prep struct Some DAMON probe filter types require preparatory actions. For example, pgilde_unset probe filter can say if the page was accessed but when. To answer the second question, the PG_Idle flag should be set at a specific time. It can make life much easier if DAMON can do such preparatory actions. Introduce a new data type called damon_prep. It specifies each of the preparation actions for each probe. DAMON will execute the action for each region per sampling interval, like it clears page table accessed bits and unsets PG_Idle flag for access monitoring. Also introduce DAMON_PREP_SET_PGIDLE as the initial prep action. As the name says, it will do exactly what DAMON was doing as the preparation action for the access monitoring. Link: https://lore.kernel.org/20260901132506.99243-6-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/damon.h | 32 ++++++++++++++++++++++++++++++++ mm/damon/core.c | 26 ++++++++++++++++++++++++++ 2 files changed, 58 insertions(+) diff --git a/include/linux/damon.h b/include/linux/damon.h index 7a42cbe791845d..1780c14942e634 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -742,6 +742,27 @@ struct damon_intervals_goal { unsigned long max_sample_us; }; +/** + * enum damon_prep_action - DAMON probing preparation action. + * + * @DAMON_PREP_SET_PGIDLE: Set the probing memory as idle page. + */ +enum damon_prep_action { + DAMON_PREP_SET_PGIDLE, +}; + +/** + * struct damon_prep - DAMON probing preparation request. + * + * @action: Action to do to the probing memory for the preparation. + */ +struct damon_prep { + enum damon_prep_action action; +/* private: */ + /* siblings list. */ + struct list_head list; +}; + /** * enum damon_filter_type - Type of &struct damon_filter * @@ -783,6 +804,8 @@ struct damon_filter { struct damon_probe { unsigned int weight; /* private: */ + /* Preparation actions to apply to each probing memory. */ + struct list_head preps; /* Filters for assessing if a given region is for this probe. */ struct list_head filters; /* Siblings list. */ @@ -962,6 +985,12 @@ static inline unsigned long damon_sz_region(struct damon_region *r) return r->ar.end - r->ar.start; } +#define damon_for_each_prep(p, probe) \ + list_for_each_entry(p, &(probe)->preps, list) + +#define damon_for_each_prep_safe(p, next, probe) \ + list_for_each_entry_safe(p, next, &(probe)->preps, list) + #define damon_for_each_filter(f, p) \ list_for_each_entry(f, &(p)->filters, list) @@ -1015,6 +1044,9 @@ static inline unsigned long damon_sz_region(struct damon_region *r) #ifdef CONFIG_DAMON +struct damon_prep *damon_new_prep(enum damon_prep_action action); +void damon_add_prep(struct damon_probe *p, struct damon_prep *prep); + struct damon_filter *damon_new_filter(enum damon_filter_type type, bool matching, bool allow); void damon_add_filter(struct damon_probe *probe, struct damon_filter *f); diff --git a/mm/damon/core.c b/mm/damon/core.c index f6a9d2da0cd721..a0a61ebdf5bb50 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -112,6 +112,28 @@ int damon_select_ops(struct damon_ctx *ctx, enum damon_ops_id id) return err; } +struct damon_prep *damon_new_prep(enum damon_prep_action action) +{ + struct damon_prep *prep; + + prep = kmalloc_obj(*prep); + if (!prep) + return NULL; + prep->action = action; + INIT_LIST_HEAD(&prep->list); + return prep; +} + +void damon_add_prep(struct damon_probe *p, struct damon_prep *prep) +{ + list_add_tail(&prep->list, &p->preps); +} + +static void damon_free_prep(struct damon_prep *p) +{ + kfree(p); +} + struct damon_filter *damon_new_filter(enum damon_filter_type type, bool matching, bool allow) { @@ -168,6 +190,7 @@ struct damon_probe *damon_new_probe(void) if (!p) return NULL; p->weight = 0; + INIT_LIST_HEAD(&p->preps); INIT_LIST_HEAD(&p->filters); INIT_LIST_HEAD(&p->list); return p; @@ -185,8 +208,11 @@ static void damon_del_probe(struct damon_probe *p) static void damon_free_probe(struct damon_probe *p) { + struct damon_prep *prep, *prep_next; struct damon_filter *f, *next; + damon_for_each_prep_safe(prep, prep_next, p) + damon_free_prep(prep); damon_for_each_filter_safe(f, next, p) damon_free_filter(f); kfree(p); From 2c7e5b80a1d6a593e64637e3ecb24abac019e202 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:53 -0700 Subject: [PATCH 0698/1352] mm/damon/core: commit preps damon_commit_probes() is ignoring damon_prep. Commit the prep actions, too. Link: https://lore.kernel.org/20260901132506.99243-7-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/core.c | 59 +++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 59 insertions(+) diff --git a/mm/damon/core.c b/mm/damon/core.c index a0a61ebdf5bb50..c631b532437777 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -129,11 +129,34 @@ void damon_add_prep(struct damon_probe *p, struct damon_prep *prep) list_add_tail(&prep->list, &p->preps); } +static void damon_del_prep(struct damon_prep *p) +{ + list_del(&p->list); +} + static void damon_free_prep(struct damon_prep *p) { kfree(p); } +static void damon_destroy_prep(struct damon_prep *p) +{ + damon_del_prep(p); + damon_free_prep(p); +} + +static struct damon_prep *damon_nth_prep(int n, struct damon_probe *p) +{ + struct damon_prep *prep; + int i = 0; + + damon_for_each_prep(prep, p) { + if (i++ == n) + return prep; + } + return NULL; +} + struct damon_filter *damon_new_filter(enum damon_filter_type type, bool matching, bool allow) { @@ -1699,6 +1722,36 @@ static int damon_commit_targets( return err; } +static void damon_commit_prep(struct damon_prep *dst, struct damon_prep *src) +{ + dst->action = src->action; +} + +static int damon_commit_preps(struct damon_probe *dst, struct damon_probe *src) +{ + struct damon_prep *dst_prep, *next, *src_prep, *new_prep; + int i = 0, j = 0; + + damon_for_each_prep_safe(dst_prep, next, dst) { + src_prep = damon_nth_prep(i++, src); + if (src_prep) + damon_commit_prep(dst_prep, src_prep); + else + damon_destroy_prep(dst_prep); + } + + damon_for_each_prep_safe(src_prep, next, src) { + if (j++ < i) + continue; + + new_prep = damon_new_prep(src_prep->action); + if (!new_prep) + return -ENOMEM; + damon_add_prep(dst, new_prep); + } + return 0; +} + static void damon_commit_filter(struct damon_filter *dst, struct damon_filter *src) { @@ -1757,6 +1810,9 @@ static int damon_commit_probes(struct damon_ctx *dst, struct damon_ctx *src) src_probe = damon_nth_probe(i++, src); if (src_probe) { dst_probe->weight = src_probe->weight; + err = damon_commit_preps(dst_probe, src_probe); + if (err) + return err; err = damon_commit_filters(dst_probe, src_probe); if (err) return err; @@ -1774,6 +1830,9 @@ static int damon_commit_probes(struct damon_ctx *dst, struct damon_ctx *src) return -ENOMEM; damon_add_probe(dst, new_probe); new_probe->weight = src_probe->weight; + err = damon_commit_preps(new_probe, src_probe); + if (err) + return err; err = damon_commit_filters(new_probe, src_probe); if (err) return err; From c34d8eb77c690cedc64b9429c2866d41aa760763 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:54 -0700 Subject: [PATCH 0699/1352] mm/damon/core: introduce damon_operations->prep_probes() damon_prep needs to be executed by the underlying DAMON operation set. Extend the operation set callback list for the execution of damon_prep actions. If the underlying operation set implements the callback, DAMON core executes it in the monitoring preparation time. Link: https://lore.kernel.org/20260901132506.99243-8-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/damon.h | 5 +++++ mm/damon/core.c | 20 +++++++++++++++++++- 2 files changed, 24 insertions(+), 1 deletion(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index 1780c14942e634..871d26adf6ae57 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -630,6 +630,7 @@ enum damon_ops_id { * @update: Update operations-related data structures. * @prepare_access_checks: Prepare next access check of target regions. * @check_accesses: Check the accesses to target regions. + * @prep_probes: Prepare applying probes for each region. * @apply_probes: Apply probes for each region. * @get_scheme_score: Get the score of a region for a scheme. * @apply_scheme: Apply a DAMON-based operation scheme. @@ -657,6 +658,9 @@ enum damon_ops_id { * last preparation and update the number of observed accesses of each region. * It should also return max number of observed accesses that made as a result * of its update. The value will be used for regions adjustment threshold. + * @prep_probes should execute required &struct damon_prep for next &struct + * damon_probe applications to each region. It should also set + * &damon_region->sampling_addr of each region if ``set_samples`` is true. * @apply_probes should apply the data attribute probes to each region and * accordingly update the probe hits counter of the region. It should also * set &damon_region->sampling_addr of each region if ``set_samples`` is true. @@ -679,6 +683,7 @@ struct damon_operations { void (*update)(struct damon_ctx *context); void (*prepare_access_checks)(struct damon_ctx *context); unsigned int (*check_accesses)(struct damon_ctx *context); + void (*prep_probes)(struct damon_ctx *context, bool set_samples); unsigned int (*apply_probes)(struct damon_ctx *context, bool set_samples, bool return_max_wsum); int (*get_scheme_score)(struct damon_ctx *context, diff --git a/mm/damon/core.c b/mm/damon/core.c index c631b532437777..5ee2d0448a8091 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -157,6 +157,18 @@ static struct damon_prep *damon_nth_prep(int n, struct damon_probe *p) return NULL; } +static bool damon_has_prep(struct damon_ctx *c) +{ + struct damon_prep *prep; + struct damon_probe *probe; + + damon_for_each_probe(probe, c) { + damon_for_each_prep(prep, probe) + return true; + } + return false; +} + struct damon_filter *damon_new_filter(enum damon_filter_type type, bool matching, bool allow) { @@ -3905,14 +3917,19 @@ static int kdamond_fn(void *data) unsigned long next_ops_update_sis = ctx->next_ops_update_sis; unsigned long sample_interval = ctx->attrs.sample_interval; bool access_check_disabled = damon_has_probe_weights(ctx); + bool do_prep; unsigned int max_merge_score = 0, max_wsum; bool get_max_wsum; if (kdamond_wait_activation(ctx)) break; + do_prep = ctx->ops.prep_probes && damon_has_prep(ctx); + if (!access_check_disabled && ctx->ops.prepare_access_checks) ctx->ops.prepare_access_checks(ctx); + if (do_prep) + ctx->ops.prep_probes(ctx, access_check_disabled); kdamond_usleep(sample_interval); ctx->passed_sample_intervals++; @@ -3927,7 +3944,8 @@ static int kdamond_fn(void *data) else get_max_wsum = false; max_wsum = ctx->ops.apply_probes(ctx, - access_check_disabled, get_max_wsum); + access_check_disabled && !do_prep, + get_max_wsum); if (get_max_wsum) max_merge_score = max_wsum; } From e692871073bbee3336008ca5e627224ae44c3071 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:55 -0700 Subject: [PATCH 0700/1352] mm/damon/paddr: support damon_prep Implement damon_operations->prep_probes() callback. Support the only existing prep action, DAMON_PREP_SET_PGIDLE in a way similar to what it was doing for the access check preparation: unset page table accessed bits and set PG_Idle flag. Reuse the function for the access check preparation. Link: https://lore.kernel.org/20260901132506.99243-9-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/paddr.c | 35 +++++++++++++++++++++++++++++++++++ 1 file changed, 35 insertions(+) diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c index 6f756f84938948..c1e7d7a4f40df3 100644 --- a/mm/damon/paddr.c +++ b/mm/damon/paddr.c @@ -105,6 +105,40 @@ static unsigned int damon_pa_check_accesses(struct damon_ctx *ctx) return max_nr_accesses; } +static void damon_pa_prep_probes_region(struct damon_region *r, + struct damon_probe *probe, struct damon_ctx *ctx) +{ + struct damon_prep *p; + + damon_for_each_prep(p, probe) { + switch (p->action) { + case DAMON_PREP_SET_PGIDLE: + damon_pa_mkold(damon_pa_phys_addr(r->sampling_addr, + ctx->addr_unit)); + break; + default: + break; + } + } +} + +static void damon_pa_prep_probes(struct damon_ctx *ctx, bool set_samples) +{ + struct damon_target *t; + struct damon_region *r; + struct damon_probe *p; + + damon_for_each_target(t, ctx) { + damon_for_each_region(r, t) { + if (set_samples) + r->sampling_addr = damon_rand(ctx, r->ar.start, + r->ar.end); + damon_for_each_probe(p, ctx) + damon_pa_prep_probes_region(r, p, ctx); + } + } +} + static bool damon_pa_filter_match(struct damon_filter *filter, struct folio *folio) { @@ -448,6 +482,7 @@ static int __init damon_pa_initcall(void) .update = NULL, .prepare_access_checks = damon_pa_prepare_access_checks, .check_accesses = damon_pa_check_accesses, + .prep_probes = damon_pa_prep_probes, .apply_probes = damon_pa_apply_probes, .target_valid = NULL, .apply_scheme = damon_pa_apply_scheme, From 2385ce6b25f01ca0501719f1ff69f9f16fb09887 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:56 -0700 Subject: [PATCH 0701/1352] mm/damon/sysfs: implement preps directory Implement a sysfs directory named 'preps' under the probe directory. It will be evolved to be used for specifying probe preps. Link: https://lore.kernel.org/20260901132506.99243-10-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs.c | 65 +++++++++++++++++++++++++++++++++++++++++++----- 1 file changed, 59 insertions(+), 6 deletions(-) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index c1ff739dab27a8..35fb308039b612 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -749,6 +749,35 @@ static const struct kobj_type damon_sysfs_intervals_ktype = { .default_groups = damon_sysfs_intervals_groups, }; +/* + * preps directory + */ + +struct damon_sysfs_preps { + struct kobject kobj; +}; + +static struct damon_sysfs_preps *damon_sysfs_preps_alloc(void) +{ + return kzalloc_obj(struct damon_sysfs_preps); +} + +static void damon_sysfs_preps_release(struct kobject *kobj) +{ + kfree(container_of(kobj, struct damon_sysfs_preps, kobj)); +} + +static struct attribute *damon_sysfs_preps_attrs[] = { + NULL, +}; +ATTRIBUTE_GROUPS(damon_sysfs_preps); + +static const struct kobj_type damon_sysfs_preps_ktype = { + .release = damon_sysfs_preps_release, + .sysfs_ops = &kobj_sysfs_ops, + .default_groups = damon_sysfs_preps_groups, +}; + /* * filter directory */ @@ -1069,6 +1098,7 @@ static const struct kobj_type damon_sysfs_filters_ktype = { struct damon_sysfs_probe { struct kobject kobj; unsigned int weight; + struct damon_sysfs_preps *preps; struct damon_sysfs_filters *filters; }; @@ -1079,25 +1109,48 @@ static struct damon_sysfs_probe *damon_sysfs_probe_alloc(void) static int damon_sysfs_probe_add_dirs(struct damon_sysfs_probe *probe) { + struct damon_sysfs_preps *preps; struct damon_sysfs_filters *filters; int err; - filters = damon_sysfs_filters_alloc(); - if (!filters) + preps = damon_sysfs_preps_alloc(); + if (!preps) return -ENOMEM; + probe->preps = preps; + + err = kobject_init_and_add(&preps->kobj, &damon_sysfs_preps_ktype, + &probe->kobj, "preps"); + if (err) + goto put_preps_out; + + filters = damon_sysfs_filters_alloc(); + if (!filters) { + err = -ENOMEM; + goto del_preps_out; + } probe->filters = filters; err = kobject_init_and_add(&filters->kobj, &damon_sysfs_filters_ktype, &probe->kobj, "filters"); - if (err) { - kobject_put(&filters->kobj); - probe->filters = NULL; - } + if (err) + goto put_filters_out; + return err; + +put_filters_out: + kobject_put(&filters->kobj); + probe->filters = NULL; +del_preps_out: + kobject_del(&preps->kobj); +put_preps_out: + kobject_put(&preps->kobj); + probe->preps = NULL; return err; } static void damon_sysfs_probe_rm_dirs(struct damon_sysfs_probe *probe) { + if (probe->preps) + kobject_put(&probe->preps->kobj); if (probe->filters) { damon_sysfs_filters_rm_dirs(probe->filters); kobject_put(&probe->filters->kobj); From 87a7f5ea5abbf96c70c1bf490d8e2175d76dded6 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:57 -0700 Subject: [PATCH 0702/1352] mm/damon/sysfs: implement preps/nr_preps file Implement nr_preps file under the preps directory. It will be evolved to be used for generating sub directories that will represent each probe preparation action. Link: https://lore.kernel.org/20260901132506.99243-11-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs.c | 53 +++++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 52 insertions(+), 1 deletion(-) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index 35fb308039b612..b9722ebffd6f17 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -755,6 +755,7 @@ static const struct kobj_type damon_sysfs_intervals_ktype = { struct damon_sysfs_preps { struct kobject kobj; + int nr; }; static struct damon_sysfs_preps *damon_sysfs_preps_alloc(void) @@ -762,12 +763,60 @@ static struct damon_sysfs_preps *damon_sysfs_preps_alloc(void) return kzalloc_obj(struct damon_sysfs_preps); } +static void damon_sysfs_preps_rm_dirs(struct damon_sysfs_preps *preps) +{ + preps->nr = 0; +} + +static int damon_sysfs_preps_add_dirs(struct damon_sysfs_preps *preps, + int nr_preps) +{ + preps->nr = nr_preps; + return 0; +} + +static ssize_t nr_preps_show(struct kobject *kobj, struct kobj_attribute *attr, + char *buf) +{ + struct damon_sysfs_preps *preps = container_of(kobj, + struct damon_sysfs_preps, kobj); + + return sysfs_emit(buf, "%d\n", preps->nr); +} + +static ssize_t nr_preps_store(struct kobject *kobj, + struct kobj_attribute *attr, const char *buf, size_t count) +{ + struct damon_sysfs_preps *preps; + int nr, err = kstrtoint(buf, 0, &nr); + + if (err) + return err; + if (nr < 0) + return -EINVAL; + + preps = container_of(kobj, struct damon_sysfs_preps, kobj); + + if (!mutex_trylock(&damon_sysfs_lock)) + return -EBUSY; + err = damon_sysfs_preps_add_dirs(preps, nr); + mutex_unlock(&damon_sysfs_lock); + if (err) + return err; + + return count; +} + static void damon_sysfs_preps_release(struct kobject *kobj) { kfree(container_of(kobj, struct damon_sysfs_preps, kobj)); } +static struct kobj_attribute damon_sysfs_preps_nr_attr = + __ATTR_RW_MODE(nr_preps, 0600); + static struct attribute *damon_sysfs_preps_attrs[] = { + &damon_sysfs_preps_nr_attr.attr, NULL, }; ATTRIBUTE_GROUPS(damon_sysfs_preps); @@ -1149,8 +1198,10 @@ static int damon_sysfs_probe_add_dirs(struct damon_sysfs_probe *probe) static void damon_sysfs_probe_rm_dirs(struct damon_sysfs_probe *probe) { - if (probe->preps) + if (probe->preps) { + damon_sysfs_preps_rm_dirs(probe->preps); kobject_put(&probe->preps->kobj); + } if (probe->filters) { damon_sysfs_filters_rm_dirs(probe->filters); kobject_put(&probe->filters->kobj); From 8ea7e05262a2f54bdae682593103320d0ba8625d Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:58 -0700 Subject: [PATCH 0703/1352] mm/damon/sysfs: create directories for nr_preps writes Implement nr_preps write action to actually create subdirectories of the number. Each of the directory will be evolved to represent each probe preparation action. Link: https://lore.kernel.org/20260901132506.99243-12-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs.c | 80 +++++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 79 insertions(+), 1 deletion(-) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index b9722ebffd6f17..4ff473b6895355 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -749,12 +749,50 @@ static const struct kobj_type damon_sysfs_intervals_ktype = { .default_groups = damon_sysfs_intervals_groups, }; +/* + * prep directory + */ + +struct damon_sysfs_prep { + struct kobject kobj; +}; + +static struct damon_sysfs_prep *damon_sysfs_prep_alloc(void) +{ + struct damon_sysfs_prep *prep; + + prep = kzalloc_obj(struct damon_sysfs_prep); + if (!prep) + return prep; + return prep; +} + +static void damon_sysfs_prep_release(struct kobject *kobj) +{ + struct damon_sysfs_prep *prep = container_of(kobj, + struct damon_sysfs_prep, kobj); + + kfree(prep); +} + +static struct attribute *damon_sysfs_prep_attrs[] = { + NULL, +}; +ATTRIBUTE_GROUPS(damon_sysfs_prep); + +static const struct kobj_type damon_sysfs_prep_ktype = { + .release = damon_sysfs_prep_release, + .sysfs_ops = &kobj_sysfs_ops, + .default_groups = damon_sysfs_prep_groups, +}; + /* * preps directory */ struct damon_sysfs_preps { struct kobject kobj; + struct damon_sysfs_prep **preps_arr; int nr; }; @@ -765,13 +803,53 @@ static struct damon_sysfs_preps *damon_sysfs_preps_alloc(void) static void damon_sysfs_preps_rm_dirs(struct damon_sysfs_preps *preps) { + struct damon_sysfs_prep **preps_arr = preps->preps_arr; + int i; + + for (i = 0; i < preps->nr; i++) { + kobject_del(&preps_arr[i]->kobj); + kobject_put(&preps_arr[i]->kobj); + } preps->nr = 0; + kfree(preps_arr); + preps->preps_arr = NULL; } static int damon_sysfs_preps_add_dirs(struct damon_sysfs_preps *preps, int nr_preps) { - preps->nr = nr_preps; + struct damon_sysfs_prep **preps_arr, *prep; + int err, i; + + damon_sysfs_preps_rm_dirs(preps); + if (!nr_preps) + return 0; + + preps_arr = kmalloc_objs(*preps_arr, nr_preps, + GFP_KERNEL | __GFP_NOWARN); + if (!preps_arr) + return -ENOMEM; + preps->preps_arr = preps_arr; + + for (i = 0; i < nr_preps; i++) { + prep = damon_sysfs_prep_alloc(); + if (!prep) { + damon_sysfs_preps_rm_dirs(preps); + return -ENOMEM; + } + + err = kobject_init_and_add(&prep->kobj, + &damon_sysfs_prep_ktype, &preps->kobj, "%d", + i); + if (err) { + kobject_put(&prep->kobj); + damon_sysfs_preps_rm_dirs(preps); + return err; + } + + preps_arr[i] = prep; + preps->nr++; + } return 0; } From e3f6cec6ee98f54705c5bede055b4c1cddc5ab9c Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:59 -0700 Subject: [PATCH 0704/1352] mm/damon/sysfs: implement prep_action file Add a file named prep_action under the prep directory. It represents the corresponding probe preparation action. Link: https://lore.kernel.org/20260901132506.99243-13-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs.c | 57 ++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 57 insertions(+) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index 4ff473b6895355..2118089dd9d8dd 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -755,6 +755,7 @@ static const struct kobj_type damon_sysfs_intervals_ktype = { struct damon_sysfs_prep { struct kobject kobj; + enum damon_prep_action action; }; static struct damon_sysfs_prep *damon_sysfs_prep_alloc(void) @@ -764,9 +765,61 @@ static struct damon_sysfs_prep *damon_sysfs_prep_alloc(void) prep = kzalloc_obj(struct damon_sysfs_prep); if (!prep) return prep; + prep->action = DAMON_PREP_SET_PGIDLE; return prep; } +struct damon_sysfs_prep_action_name { + const enum damon_prep_action action; + const char *name; +}; + +static const struct damon_sysfs_prep_action_name +damon_sysfs_prep_action_names[] = { + { + .action = DAMON_PREP_SET_PGIDLE, + .name = "set_pgidle", + }, +}; + +static ssize_t prep_action_show(struct kobject *kobj, + struct kobj_attribute *attr, char *buf) +{ + struct damon_sysfs_prep *prep = container_of(kobj, + struct damon_sysfs_prep, kobj); + int i; + + for (i = 0; i < ARRAY_SIZE(damon_sysfs_prep_action_names); i++) { + const struct damon_sysfs_prep_action_name *action_name; + + action_name = &damon_sysfs_prep_action_names[i]; + if (action_name->action == prep->action) + return sysfs_emit(buf, "%s\n", action_name->name); + } + return -EINVAL; +} + +static ssize_t prep_action_store(struct kobject *kobj, + struct kobj_attribute *attr, const char *buf, size_t count) +{ + struct damon_sysfs_prep *prep = container_of(kobj, + struct damon_sysfs_prep, kobj); + ssize_t ret = -EINVAL; + int i; + + for (i = 0; i < ARRAY_SIZE(damon_sysfs_prep_action_names); i++) { + const struct damon_sysfs_prep_action_name *action_name; + + action_name = &damon_sysfs_prep_action_names[i]; + if (sysfs_streq(buf, action_name->name)) { + prep->action = action_name->action; + ret = count; + break; + } + } + return ret; +} + static void damon_sysfs_prep_release(struct kobject *kobj) { struct damon_sysfs_prep *prep = container_of(kobj, @@ -775,7 +828,11 @@ static void damon_sysfs_prep_release(struct kobject *kobj) kfree(prep); } +static struct kobj_attribute damon_sysfs_prep_prep_action_attr = + __ATTR_RW_MODE(prep_action, 0600); + static struct attribute *damon_sysfs_prep_attrs[] = { + &damon_sysfs_prep_prep_action_attr.attr, NULL, }; ATTRIBUTE_GROUPS(damon_sysfs_prep); From 047b968220e630f1ba37f34ed933012cbedbf9db Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:25:00 -0700 Subject: [PATCH 0705/1352] mm/damon/sysfs: pass preps to DAMON core DAMON sysfs interface provides the files for setting DAMON probe preps. But the underlying code is not really passing the user-set values to DAMON core. Pass those. Link: https://lore.kernel.org/20260901132506.99243-14-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs.c | 28 ++++++++++++++++++++++++++++ 1 file changed, 28 insertions(+) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index 2118089dd9d8dd..7f340b6f1921b9 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -812,7 +812,10 @@ static ssize_t prep_action_store(struct kobject *kobj, action_name = &damon_sysfs_prep_action_names[i]; if (sysfs_streq(buf, action_name->name)) { + if (!mutex_trylock(&damon_sysfs_lock)) + return -EBUSY; prep->action = action_name->action; + mutex_unlock(&damon_sysfs_lock); ret = count; break; } @@ -2175,6 +2178,23 @@ static int damon_sysfs_set_attrs(struct damon_ctx *ctx, return damon_set_attrs(ctx, &attrs); } +static int damon_sysfs_set_preps(struct damon_probe *probe, + struct damon_sysfs_preps *sys_preps) +{ + int i; + + for (i = 0; i < sys_preps->nr; i++) { + struct damon_sysfs_prep *sys_prep = sys_preps->preps_arr[i]; + struct damon_prep *prep; + + prep = damon_new_prep(sys_prep->action); + if (!prep) + return -ENOMEM; + damon_add_prep(probe, prep); + } + return 0; +} + static int damon_sysfs_set_filters(struct damon_probe *probe, struct damon_sysfs_filters *sys_filters) { @@ -2210,7 +2230,15 @@ static int damon_sysfs_set_probe(struct damon_probe *probe, struct damon_sysfs_probe *sys_probe) { struct damon_sysfs_filters *sys_filters; + struct damon_sysfs_preps *sys_preps; + int err; + sys_preps = sys_probe->preps; + if (sys_preps) { + err = damon_sysfs_set_preps(probe, sys_preps); + if (err) + return err; + } sys_filters = sys_probe->filters; if (!sys_filters) return 0; From 526127746b401774b1d321ecc336bb53c31df376 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:25:01 -0700 Subject: [PATCH 0706/1352] selftests/damon/sysfs.sh: test probe prep sysfs files Add basic file operations test for newly introduced DAMON probe prep sysfs directories and files. Link: https://lore.kernel.org/20260901132506.99243-15-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/damon/sysfs.sh | 27 ++++++++++++++++++++++++++ 1 file changed, 27 insertions(+) diff --git a/tools/testing/selftests/damon/sysfs.sh b/tools/testing/selftests/damon/sysfs.sh index f7fb94b84e716d..ddebde6edabe4e 100755 --- a/tools/testing/selftests/damon/sysfs.sh +++ b/tools/testing/selftests/damon/sysfs.sh @@ -361,6 +361,32 @@ test_intervals() test_intervals_goal "$intervals_dir/intervals_goal" } +test_damon_prep() +{ + damon_prep_dir=$1 + ensure_file "$damon_prep_dir/prep_action" "exist" "600" + ensure_write_succ "$damon_prep_dir/prep_action" "set_pgidle" \ + "valid input" + ensure_write_fail "$damon_prep_dir/prep_action" "foo" "invalid input" +} + +test_damon_preps() +{ + preps_dir=$1 + ensure_dir "$preps_dir" "exist" + ensure_file "$preps_dir/nr_preps" "exist" "600" + ensure_write_succ "$preps_dir/nr_preps" "1" "valid input" + test_damon_prep "$preps_dir/0" + + ensure_write_succ "$preps_dir/nr_preps" "2" "valid input" + test_damon_prep "$preps_dir/0" + test_damon_prep "$preps_dir/1" + + ensure_write_succ "$preps_dir/nr_preps" "0" "valid input" + ensure_dir "$preps_dir/0" "not_exist" + ensure_dir "$preps_dir/1" "not_exist" +} + test_damon_filter() { damon_filter_dir=$1 @@ -392,6 +418,7 @@ test_probe() { probe_dir=$1 ensure_dir "$probe_dir" "exist" + test_damon_preps "$probe_dir/preps" test_damon_filters "$probe_dir/filters" } From d838a53d73d63a17d490983ad751609ca04afdbd Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:25:02 -0700 Subject: [PATCH 0707/1352] Docs/mm/damon/design: document probe preps Update DAMON design document for the newly added DAMON probe preps feature. Link: https://lore.kernel.org/20260901132506.99243-16-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/mm/damon/design.rst | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index 947d91ae24a392..d036340dae8afb 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -309,6 +309,13 @@ Users can therefore know how much of a given DAMON region has a specific data attribute by reading the per-region per-probe probe hits counter after each aggregation interval. +Users can optionally register probing preparation actions per probe. If such +actions are registered, DAMON applies the actions to each region's sampling +memory before starting the next sampling interval. Currently only one action, +``set_pgidle`` is supported. The action marks the page for the probing target +memory as access-idle. This can be useful to be used together with +``pgidle_unset`` probe filter. + This is a sampling based mechanism. Hence, it is lightweight but the output may include some measurement errors. The output should be used with good understanding of statistics. From b2ba9c0f4bb5cfd2897670eeb0074aa3ad9b47ea Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:25:03 -0700 Subject: [PATCH 0708/1352] Docs/admin-guide/mm/damon/usage: document probe preps sysfs files Update DAMON usage document for the newly added DAMON probe preps sysfs files. Link: https://lore.kernel.org/20260901132506.99243-17-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/admin-guide/mm/damon/usage.rst | 18 +++++++++++++++--- 1 file changed, 15 insertions(+), 3 deletions(-) diff --git a/Documentation/admin-guide/mm/damon/usage.rst b/Documentation/admin-guide/mm/damon/usage.rst index da5f9afd08aef8..023c6334024f8e 100644 --- a/Documentation/admin-guide/mm/damon/usage.rst +++ b/Documentation/admin-guide/mm/damon/usage.rst @@ -74,6 +74,9 @@ comma (","). │ │ │ │ │ │ nr_regions/min,max │ │ │ │ │ │ :ref:`probes `/nr_probes │ │ │ │ │ │ │ 0/weight + │ │ │ │ │ │ │ │ preps/nr_preps + │ │ │ │ │ │ │ │ │ 0/prep_action + │ │ │ │ │ │ │ │ │ ... │ │ │ │ │ │ │ │ filters/nr_filters │ │ │ │ │ │ │ │ │ 0/type,matching,allow,path │ │ │ │ │ │ │ │ │ ... @@ -283,9 +286,18 @@ In the beginning, this directory has only one file, ``nr_probes``. Writing a number (``N``) to the file creates the number of child directories named ``0`` to ``N-1``. Each directory represents each monitoring probe. -In each probe directory, one directory, ``filters`` exists. The directory -contains files for installing filters for the probe, that is used to determine -the data attribute for the probe. +In each probe directory, two directories, ``preps`` and ``filters`` exist. The +directories contain files for installing probing preparation actions and +filters for the probe, that are used to determine the data attribute for the +probe. + +In the beginning, ``preps`` directory has only one file, ``nr_preps``. +Writing a number (``N``) to the file creates the number of child directories +named ``0`` to ``N-1``. Each directory represents each preparation action. +Each directory has one file, ``prep_action``. The preparation action can be +selected by writing the name of the action to the ``prep_action`` file. Refer +to the :ref:`design doc ` for the list of +supported actions. Each probe directory also contains ``weight`` file. Reading from and writing to the file gets and sets the :ref:`attributes-only monitoring From a980ef17c23c89401eece292196717f46fd5d102 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:25:04 -0700 Subject: [PATCH 0709/1352] Docs/ABI/damon: document probe prep sysfs files Update DAMON ABI document for the newly added DAMON probe prep sysfs files. Link: https://lore.kernel.org/20260901132506.99243-18-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/ABI/testing/sysfs-kernel-mm-damon | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/Documentation/ABI/testing/sysfs-kernel-mm-damon b/Documentation/ABI/testing/sysfs-kernel-mm-damon index e675a57145e36d..f8d2601e829047 100644 --- a/Documentation/ABI/testing/sysfs-kernel-mm-damon +++ b/Documentation/ABI/testing/sysfs-kernel-mm-damon @@ -173,6 +173,19 @@ Contact: SJ Park Description: Writing to and reading from this file sets and gets the per-probe attribute weight. +What: /sys/kernel/mm/damon/admin/kdamonds//contexts//monitoring_attrs/probes/

/preps/nr_preps +Date: Jun 2026 +Contact: SJ Park +Description: Writing a number 'N' to this file creates the number of + directories for each DAMON probing preparation action named '0' + to 'N-1' under the preps/ directory. + +What: /sys/kernel/mm/damon/admin/kdamonds//contexts//monitoring_attrs/probes/

/preps//prep_action +Date: Jun 2026 +Contact: SJ Park +Description: Writing to and reading from this file sets and gets the probing + preparation action. + What: /sys/kernel/mm/damon/admin/kdamonds//contexts//monitoring_attrs/probes/

/filters/nr_filters Date: May 2026 Contact: SJ Park From 2fe74b7a5ffaa09b68dc2ee4cdb4cf36f0446b08 Mon Sep 17 00:00:00 2001 From: Qinyun Tan Date: Tue, 1 Sep 2026 19:51:04 +0800 Subject: [PATCH 0710/1352] mm/list_lru: don't copy stale shrinker id from non-memcg-aware shrinkers MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit With cgroup.memory=nokmem, shrinker_memcg_alloc() fails with -ENOSYS for shrinkers without SHRINKER_NONSLAB, and shrinker_alloc() falls back to a non-memcg-aware shrinker. On this fallback path, shrinker->id is never assigned and keeps 0 from kzalloc(), which is a valid id belonging to whichever memcg-aware shrinker registers first. __list_lru_init() copies shrinker->id unconditionally, so every list_lru backed by such a fallback shrinker (thp-deferred_split, zswap-shrinker, workingset shadow nodes, superblock lrus, ...) ends up with lru->shrinker_id == 0 instead of -1. Under nokmem the list_lru collapses to the shared per-node lists, but __list_lru_add() still calls set_shrinker_bit() against the memcg of the added object. Most list_lru users are unaffected because their objects resolve to a NULL memcg without kmem accounting, but the THP deferred split queue holds user folios, which are charged regardless of nokmem. On a system where no SHRINKER_NONSLAB shrinker registers, shrinker_nr_max stays 0 and every memcg's shrinker_info has map_nr_max == 0, so the first folio added by khugepaged triggers on every boot: WARNING: mm/shrinker.c:212 at set_shrinker_bit+0x99/0xa0 On systems where a SHRINKER_NONSLAB shrinker (btrfs, xfs) did register and expand the maps, there is no warning; instead bit 0 is set spuriously for an unrelated shrinker. shrinker->id is only meaningful while SHRINKER_MEMCG_AWARE is set, and all readers inside mm/shrinker.c already check the flag before using the id. Make __list_lru_init() do the same and fall back to -1, so set_shrinker_bit() is never reached with a bogus id. The stale shrinker->id itself is left as is; cleaning that up is a separate topic. Verified on a machine booting with cgroup.memory=nokmem and CONFIG_TRANSPARENT_HUGEPAGE=y: the warning fires once per boot from khugepaged, disappears when nokmem is removed from the command line, and no longer triggers with this fix applied and nokmem set. Link: https://lore.kernel.org/20260901115104.2944996-1-qinyuntan@linux.alibaba.com Fixes: fafaeceb89a5e ("mm: switch deferred split shrinker to list_lru") Signed-off-by: Qinyun Tan Signed-off-by: Andrew Morton Acked-by: Muchun Song Reviewed-by: Baolin Wang Reported-by: Wentao Guan Closes: https://lore.kernel.org/linux-mm/20260904190028.21542-1-guanwentao@uniontech.com/ Tested-by: Wentao Guan Cc: Michal Hocko Cc: Roman Gushchin Cc: Johannes Weiner Cc: Shakeel Butt Cc: Dave Chinner Cc: David Hildenbrand Cc: Lance Yang Cc: Michal Koutný Cc: Xunlei Pang Cc: --- mm/list_lru.c | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/mm/list_lru.c b/mm/list_lru.c index a4522ca93ebcb9..8a6dd0a489e12c 100644 --- a/mm/list_lru.c +++ b/mm/list_lru.c @@ -666,7 +666,12 @@ int __list_lru_init(struct list_lru *lru, bool memcg_aware, struct shrinker *shr int i; #ifdef CONFIG_MEMCG - if (shrinker) + /* + * If the shrinker fell back to being non-memcg-aware (e.g. with + * cgroup.memory=nokmem), its id was never assigned and holds a + * stale 0. Don't let set_shrinker_bit() act on it. + */ + if (shrinker && (shrinker->flags & SHRINKER_MEMCG_AWARE)) lru->shrinker_id = shrinker->id; else lru->shrinker_id = -1; From 67c7d5368770207d958d86877c975718a6da0e5a Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 1 Sep 2026 15:38:04 -0400 Subject: [PATCH 0711/1352] mm: remove unused mark_page_reserved() Patch series "mm: remove three unused helpers from mm.h", v2. I happened to notice these were unused. Two of them are relatively recently unused, and one has been unused for a few years. Remove them. This patch (of 2): mark_page_reserved() lost its last caller in commit 6215d9f4470f ("arch, mm: consolidate empty_zero_page"). Remove it. Link: https://lore.kernel.org/20260901-mm-remove-unused-helpers-v2-0-f6474e169c23@columbia.edu Link: https://lore.kernel.org/20260901-mm-remove-unused-helpers-v2-1-f6474e169c23@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Cc: Liam R. Howlett Cc: Michal Hocko Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/mm.h | 6 ------ 1 file changed, 6 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index 1b28e6fc8d5dd1..e3736c42c4dbc0 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -4080,12 +4080,6 @@ static inline void free_reserved_page(struct page *page) free_reserved_pages(page, 0); } -static inline void mark_page_reserved(struct page *page) -{ - SetPageReserved(page); - adjust_managed_page_count(page, -1); -} - static inline void free_reserved_ptdesc(struct ptdesc *pt) { free_reserved_page(ptdesc_page(pt)); From c3c3fead07f164c34f705e3e9642b86fa3ee14e7 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 1 Sep 2026 15:38:05 -0400 Subject: [PATCH 0712/1352] mm: remove unused totalram_pages_inc() and totalram_pages_dec() totalram_pages_inc() and totalram_pages_dec() have had no callers since commit 7fbc5e26123e ("memblock: extract page freeing from free_reserved_area() into a helper") and commit 287b89773d81 ("powerpc/pseries/cmm: Use adjust_managed_page_count() insted of totalram_pages_*"), respectively. Remove them. Drop the totalram_pages_inc() stub from tools mm.h too. Link: https://lore.kernel.org/20260901-mm-remove-unused-helpers-v2-2-f6474e169c23@columbia.edu Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Liam R. Howlett Cc: Michal Hocko Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/mm.h | 10 ---------- tools/include/linux/mm.h | 4 ---- 2 files changed, 14 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index e3736c42c4dbc0..c105a3758915b2 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -57,16 +57,6 @@ static inline unsigned long totalram_pages(void) return (unsigned long)atomic_long_read(&_totalram_pages); } -static inline void totalram_pages_inc(void) -{ - atomic_long_inc(&_totalram_pages); -} - -static inline void totalram_pages_dec(void) -{ - atomic_long_dec(&_totalram_pages); -} - static inline void totalram_pages_add(long count) { atomic_long_add(count, &_totalram_pages); diff --git a/tools/include/linux/mm.h b/tools/include/linux/mm.h index 84b5954f66c3d5..d586a510e6e18b 100644 --- a/tools/include/linux/mm.h +++ b/tools/include/linux/mm.h @@ -33,10 +33,6 @@ static inline phys_addr_t virt_to_phys(volatile void *address) return (phys_addr_t)address; } -static inline void totalram_pages_inc(void) -{ -} - static inline void totalram_pages_add(long count) { } From aa46f6fde8aa448e5435ce8cd93f964f6b44a059 Mon Sep 17 00:00:00 2001 From: Sang-Heon Jeon Date: Wed, 2 Sep 2026 01:53:01 +0900 Subject: [PATCH 0713/1352] percpu: remove redundant assignments to bits Patch series "percpu: remove code with no effect". While reading the percpu initialization path, I found some minor cleanups for unnecessary initializations and an obsolete return statement. No functional change. This patch (of 4): pcpu_chunk_refresh_hint() and pcpu_find_block_fit() set bits to 0 and later call pcpu_next_md_free_region() or pcpu_next_fit_region(), which unconditionally set *bits to 0. Nothing uses it in between, so the assignments have no effect. Remove redundant assignments. No functional change. Link: https://lore.kernel.org/20260901165307.1026248-1-ekffu200098@gmail.com Link: https://lore.kernel.org/20260901165307.1026248-2-ekffu200098@gmail.com Signed-off-by: Sang-Heon Jeon Signed-off-by: Andrew Morton Cc: Dennis Zhou Cc: Tejun Heo Cc: Christoph Lameter --- mm/percpu.c | 2 -- 1 file changed, 2 deletions(-) diff --git a/mm/percpu.c b/mm/percpu.c index 47a903fe3b5124..b094c617147cb5 100644 --- a/mm/percpu.c +++ b/mm/percpu.c @@ -758,7 +758,6 @@ static void pcpu_chunk_refresh_hint(struct pcpu_chunk *chunk, bool full_scan) chunk_md->contig_hint = 0; } - bits = 0; pcpu_for_each_md_free_region(chunk, bit_off, bits) pcpu_block_update(chunk_md, bit_off, bit_off + bits); } @@ -1122,7 +1121,6 @@ static int pcpu_find_block_fit(struct pcpu_chunk *chunk, int alloc_bits, return -1; bit_off = pcpu_next_hint(chunk_md, alloc_bits); - bits = 0; pcpu_for_each_fit_region(chunk, alloc_bits, align, bit_off, bits) { if (!pop_only || pcpu_is_populated(chunk, bit_off, bits, &next_off)) From 3380b1dcdc71d51a4f22f83809b3fbf7b6e2ae78 Mon Sep 17 00:00:00 2001 From: Sang-Heon Jeon Date: Wed, 2 Sep 2026 01:53:02 +0900 Subject: [PATCH 0714/1352] percpu: remove unnecessary initialization in pcpu_build_alloc_info() pcpu_build_alloc_info() initializes nr_groups to 1 and unconditionally sets it to the number of groups. Nothing uses it in between, so the initialization has no effect. Remove unnecessary initialization. No functional change. Link: https://lore.kernel.org/20260901165307.1026248-3-ekffu200098@gmail.com Signed-off-by: Sang-Heon Jeon Signed-off-by: Andrew Morton Cc: Christoph Lameter Cc: Dennis Zhou Cc: Tejun Heo --- mm/percpu.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/percpu.c b/mm/percpu.c index b094c617147cb5..0ec9b2adfc208e 100644 --- a/mm/percpu.c +++ b/mm/percpu.c @@ -2817,7 +2817,7 @@ static struct pcpu_alloc_info * __init __flatten pcpu_build_alloc_info( static int group_cnt[NR_CPUS] __initdata; static struct cpumask mask __initdata; const size_t static_size = __per_cpu_end - __per_cpu_start; - int nr_groups = 1, nr_units = 0; + int nr_groups, nr_units = 0; size_t size_sum, min_unit_size, alloc_size; int upa, max_upa, best_upa; /* units_per_alloc */ int last_allocs, group, unit; From fe1fe1301643e6ecdc7a29e738f3e7eee5e4a34c Mon Sep 17 00:00:00 2001 From: Sang-Heon Jeon Date: Wed, 2 Sep 2026 01:53:03 +0900 Subject: [PATCH 0715/1352] percpu: remove unnecessary cpumask_clear() in pcpu_build_alloc_info() pcpu_build_alloc_info() clears mask and unconditionally sets it to cpu_possible_mask. Nothing uses it in between, so cpumask_clear() has no effect. Remove unnecessary cpumask_clear(). No functional change. Link: https://lore.kernel.org/20260901165307.1026248-4-ekffu200098@gmail.com Signed-off-by: Sang-Heon Jeon Signed-off-by: Andrew Morton Cc: Christoph Lameter Cc: Dennis Zhou Cc: Tejun Heo --- mm/percpu.c | 1 - 1 file changed, 1 deletion(-) diff --git a/mm/percpu.c b/mm/percpu.c index 0ec9b2adfc208e..fe9fe5a0c3d597 100644 --- a/mm/percpu.c +++ b/mm/percpu.c @@ -2828,7 +2828,6 @@ static struct pcpu_alloc_info * __init __flatten pcpu_build_alloc_info( /* this function may be called multiple times */ memset(group_map, 0, sizeof(group_map)); memset(group_cnt, 0, sizeof(group_cnt)); - cpumask_clear(&mask); /* calculate size_sum and ensure dyn_size is enough for early alloc */ size_sum = PFN_ALIGN(static_size + reserved_size + From 28cd718bf9efe851d01ec85ed7355a1f54418001 Mon Sep 17 00:00:00 2001 From: Sang-Heon Jeon Date: Wed, 2 Sep 2026 01:53:04 +0900 Subject: [PATCH 0716/1352] percpu: remove unnecessary return in pcpu_populate_pte() Since commit c6f239796b55 ("mm/memblock: add memblock_alloc_or_panic interface"), pcpu_populate_pte() no longer has an error label after the return statement, so the return statement has no effect. Remove unnecessary return. No functional change. Link: https://lore.kernel.org/20260901165307.1026248-5-ekffu200098@gmail.com Signed-off-by: Sang-Heon Jeon Signed-off-by: Andrew Morton Cc: Christoph Lameter Cc: Dennis Zhou Cc: Tejun Heo --- mm/percpu.c | 2 -- 1 file changed, 2 deletions(-) diff --git a/mm/percpu.c b/mm/percpu.c index fe9fe5a0c3d597..3eff382e565ad5 100644 --- a/mm/percpu.c +++ b/mm/percpu.c @@ -3178,8 +3178,6 @@ void __init __weak pcpu_populate_pte(unsigned long addr) new = memblock_alloc_or_panic(PTE_TABLE_SIZE, PTE_TABLE_SIZE); pmd_populate_kernel(&init_mm, pmd, new); } - - return; } /** From 9bd2c1980a06bfdfd5da7c074c6567abeaef9962 Mon Sep 17 00:00:00 2001 From: Sergey Senozhatsky Date: Tue, 1 Sep 2026 14:13:05 +0900 Subject: [PATCH 0717/1352] zram: remove unreachable kernel_read_file_from_path() return check Sashiko reported that: kernel_read_file_from_path() returns negative error for zero-sized files, so we cannot have "sz == 0" return, remove it and use a generic error message instead. Link: https://lore.kernel.org/20260901051335.2202390-1-senozhatsky@chromium.org Signed-off-by: Sergey Senozhatsky Signed-off-by: Andrew Morton Cc: Haoqin Huang --- drivers/block/zram/zram_drv.c | 5 ----- 1 file changed, 5 deletions(-) diff --git a/drivers/block/zram/zram_drv.c b/drivers/block/zram/zram_drv.c index 4ba0f77b2abd80..2359eaa6f53184 100644 --- a/drivers/block/zram/zram_drv.c +++ b/drivers/block/zram/zram_drv.c @@ -1722,11 +1722,6 @@ static int comp_params_store(struct zram *zram, u32 prio, s32 level, dict_path, sz); return sz; } - if (sz == 0) { - pr_err("failed to load dictionary %s (empty file)\n", - dict_path); - return -EINVAL; - } } zram->params[prio].dict_sz = sz; From 9045e08b67f4ebd97bce5c1409e14eab71df71f9 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 17:27:20 -0700 Subject: [PATCH 0718/1352] mm/damon/tests/core-kunit: test committing psi goal to psi goal Patch series "mm/damon: fix misc bugs in kunit, quota goals and sysfs refresh_ms". DAMOS quota goals commit unit test is mistakenly not testing a test case that was designed to test. DAMOS quota goals and DAMON sysfs refresh_ms file have bugs that can produce non critical but still unexpected behaviors. Fix the bugs. Patch 1 fixes the DAMOS quota goals commit unit test to cover a mistakenly uncovered case. Patches 2 and 3 fix the bugs in DAMOS PSI goal initialization and eligible_mem_bp online commit, respectively. Patch 4 fixes the bug in DAMON sysfs refresh_ms file handling. This patch (of 4): damon_test_commit_quota_goal_for() is set to test committing a new psi goal on an existing psi goal. However, damon_test_commit_quota_goal() is mistakenly not covering the test case. Add the test case. Link: https://lore.kernel.org/20260902002725.108635-1-sj@kernel.org Link: https://lore.kernel.org/20260902002725.108635-2-sj@kernel.org Fixes: 99f89debafc5 ("mm/damon/tests/core-kunit: add damos_commit_quota_goal() test") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Lian Wang Tested-by: Lian Wang Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: Quanmin Yan Cc: # 6.19.x --- mm/damon/tests/core-kunit.h | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index 7071ec277b0072..f1e11548c771b7 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -793,6 +793,13 @@ static void damos_test_commit_quota_goal(struct kunit *test) .last_psi_total = 456, }; + damos_test_commit_quota_goal_for(test, &dst, + &(struct damos_quota_goal) { + .metric = DAMOS_QUOTA_SOME_MEM_PSI_US, + .target_value = 234, + .current_value = 345, + .last_psi_total = 567, + }); damos_test_commit_quota_goal_for(test, &dst, &(struct damos_quota_goal){ .metric = DAMOS_QUOTA_USER_INPUT, From c2cc20a3d97a980eb62b224658c70c002fd3d206 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 17:27:21 -0700 Subject: [PATCH 0719/1352] mm/damon/core: handle uninitialized damos_quota_goal->last_psi_total When DAMOS_QUOTA_SOME_MEM_PSI_US metric damos quota goal is set, the PSI delta for the feedback loop is calculated using damos_quota_goal->last_psi_total. However, it is initialized only after the first feedback loop. The first iteration of the loop uses the uninitialized value. As a result, the feedback loop can change the effective quota in an unexpected way at the first iteration. The user impact of the issue is not big, because the issue impacts only the first iteration of the feedback loop. The feedback loop also has an internal cap of the quota adjustment. The wrong adjustment will soon be corrected over a few iterations. For this reason, doing no initialization at commit time was intentional. It is also explicitly commented. That said, nobody likes behaviors that are unexpected or difficult to be expected. Check last_psi_total initialization and skip the tuning round when it is not initialized. For this, initialize last_psi_total with U64_MAX in the goal creation and the goal commit time. U64_MAX means the field is not initialized. The tuning round shows the value and adjusts it to guarantee the current quota is maintained for the round, and last_psi_total is correctly initialized on the next round. Before this change, committing a new PSI goal on an existing PSI goal with goal-only DAMON sysfs command (commit_schemes_quota_goals) just worked. After this change, the tuning round right after the commit will be unnecessarily skipped, because last_psi_total is unconditionally marked as not initialized in the damos_commit_quota_goal_union(). This is an intended tradeoff for simplicity. Skipping just one round of tuning is no problem. Meanwhile it makes both the code and the behavior simple to understand. Also update the quota goal commit unit test for changed last_psi_total setup behavior. The issue was discovered [1] by Sashiko. Link: https://lore.kernel.org/20260902002725.108635-3-sj@kernel.org Link: https://lore.kernel.org/20260718005316.89585-1-sj@kernel.org [1] Fixes: 2dbb60f789cb ("mm/damon/core: implement PSI metric DAMOS quota goal") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Lian Wang Tested-by: Lian Wang Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: Quanmin Yan Cc: # 6.9.x --- mm/damon/core.c | 13 +++++++++++-- mm/damon/tests/core-kunit.h | 9 +++------ 2 files changed, 14 insertions(+), 8 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 5ee2d0448a8091..f950a3b9fcb99f 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -697,6 +697,8 @@ struct damos_quota_goal *damos_new_quota_goal( return NULL; goal->metric = metric; goal->target_value = target_value; + if (metric == DAMOS_QUOTA_SOME_MEM_PSI_US) + goal->last_psi_total = U64_MAX; INIT_LIST_HEAD(&goal->list); return goal; } @@ -1190,6 +1192,9 @@ static void damos_commit_quota_goal_union( struct damos_quota_goal *dst, struct damos_quota_goal *src) { switch (dst->metric) { + case DAMOS_QUOTA_SOME_MEM_PSI_US: + dst->last_psi_total = U64_MAX; + break; case DAMOS_QUOTA_NODE_MEM_USED_BP: case DAMOS_QUOTA_NODE_MEM_FREE_BP: dst->nid = src->nid; @@ -1211,7 +1216,6 @@ static void damos_commit_quota_goal( dst->target_value = src->target_value; if (dst->metric == DAMOS_QUOTA_USER_INPUT) dst->current_value = src->current_value; - /* keep last_psi_total as is, since it will be updated in next cycle */ damos_commit_quota_goal_union(dst, src); } @@ -3139,7 +3143,12 @@ static void damos_set_quota_goal_current_value(struct damon_ctx *c, break; case DAMOS_QUOTA_SOME_MEM_PSI_US: now_psi_total = damos_get_some_mem_psi_total(); - goal->current_value = now_psi_total - goal->last_psi_total; + /* uninitialized last_psi_total; make no effect this round */ + if (goal->last_psi_total == U64_MAX) + goal->current_value = goal->target_value; + else + goal->current_value = now_psi_total - + goal->last_psi_total; goal->last_psi_total = now_psi_total; break; case DAMOS_QUOTA_NODE_MEM_USED_BP: diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index f1e11548c771b7..af26b3d60957b5 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -757,19 +757,16 @@ static void damos_test_commit_quota_goal_for(struct kunit *test, struct damos_quota_goal *dst, struct damos_quota_goal *src) { - u64 dst_last_psi_total = 0; - - if (dst->metric == DAMOS_QUOTA_SOME_MEM_PSI_US) - dst_last_psi_total = dst->last_psi_total; damos_commit_quota_goal(dst, src); KUNIT_EXPECT_EQ(test, dst->metric, src->metric); KUNIT_EXPECT_EQ(test, dst->target_value, src->target_value); if (src->metric == DAMOS_QUOTA_USER_INPUT) KUNIT_EXPECT_EQ(test, dst->current_value, src->current_value); - if (dst_last_psi_total && src->metric == DAMOS_QUOTA_SOME_MEM_PSI_US) - KUNIT_EXPECT_EQ(test, dst->last_psi_total, dst_last_psi_total); switch (dst->metric) { + case DAMOS_QUOTA_SOME_MEM_PSI_US: + KUNIT_EXPECT_EQ(test, dst->last_psi_total, U64_MAX); + break; case DAMOS_QUOTA_NODE_MEM_USED_BP: case DAMOS_QUOTA_NODE_MEM_FREE_BP: KUNIT_EXPECT_EQ(test, dst->nid, src->nid); From 7fa0ef7b1783b2543fff5c47c58e0a65e19102cb Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Tue, 22 Sep 2026 22:51:37 -0700 Subject: [PATCH 0720/1352] mm/damon/core: keep the temporal tuner quota over an unmeasured PSI round The patch in mm-unstable of subject "mm/damon/core: handle uninitialized damos_quota_goal->last_psi_total" scores a PSI quota goal without a previous sample as achieved. This leaves the consist tuner's input unchanged, but the temporal tuner sets its quota to zero for an achieved goal. A running scheme with a nonzero temporal quota therefore loses a charge window after a quota-goal commit. Use the effective quota to preserve the temporal tuner's previous goal-achievement state during an unmeasured round, as SJ suggested [1]. Score the goal as achieved when the effective quota is zero and as not achieved otherwise. A new scheme's initially zero quota stays zero, and the consist tuner's behavior is unchanged. Move the PSI current-value calculation and last_psi_total update into a helper that takes the current PSI total. This lets a unit test cover the unmeasured and measured rounds without depending on system memory pressure. Link: https://lore.kernel.org/20260923055139.2982-1-sj@kernel.org Link: https://lore.kernel.org/damon/20260916001311.101024-1-sj@kernel.org/ [1] Fixes: 2dbb60f789cb ("mm/damon/core: implement PSI metric DAMOS quota goal") Signed-off-by: Karl Mehltretter Signed-off-by: SJ Park Signed-off-by: Andrew Morton Suggested-by: SJ Park Reviewed-by: SJ Park Reviewed-by: Kunwu Chan Assisted-by: LLM Cc: Lian Wang Cc: # 6.9.x --- mm/damon/core.c | 30 +++++++++++++++++++++++------- 1 file changed, 23 insertions(+), 7 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index f950a3b9fcb99f..71e8cedbc64399 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2891,6 +2891,28 @@ static inline u64 damos_get_some_mem_psi_total(void) #endif /* CONFIG_PSI */ +static void damos_set_psi_current_val(u64 now_psi_total, + struct damos_quota_goal *goal, struct damos *s) +{ + u64 last_psi_total = goal->last_psi_total; + + goal->last_psi_total = now_psi_total; + if (last_psi_total != U64_MAX) { + goal->current_value = now_psi_total - last_psi_total; + return; + } + /* uninitialized last_psi_total; make no effect this round */ + if (s->quota.goal_tuner == DAMOS_QUOTA_GOAL_TUNER_CONSIST) { + goal->current_value = goal->target_value; + return; + } + /* let temporal tuner show the same achievement as in the last round */ + if (!s->quota.esz) + goal->current_value = goal->target_value; + else + goal->current_value = 0; +} + #ifdef CONFIG_NUMA static bool invalid_mem_node(int nid) { @@ -3143,13 +3165,7 @@ static void damos_set_quota_goal_current_value(struct damon_ctx *c, break; case DAMOS_QUOTA_SOME_MEM_PSI_US: now_psi_total = damos_get_some_mem_psi_total(); - /* uninitialized last_psi_total; make no effect this round */ - if (goal->last_psi_total == U64_MAX) - goal->current_value = goal->target_value; - else - goal->current_value = now_psi_total - - goal->last_psi_total; - goal->last_psi_total = now_psi_total; + damos_set_psi_current_val(now_psi_total, goal, s); break; case DAMOS_QUOTA_NODE_MEM_USED_BP: case DAMOS_QUOTA_NODE_MEM_FREE_BP: From eaa3d9b52f2b4a8b9351d6036eee2924dba73ed8 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 17:27:22 -0700 Subject: [PATCH 0721/1352] mm/damon/core: copy nid for eligible_mem_bp damos quota goal commit damos_commit_quota_goal_union() is not updating the ->nid union field when the goal metric is DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP. Hence, if a DAMOS quota goal of the type is online committed in a way that it will reuse other quota goal's memory space, the new goal will work with a garbage nid value. As a result, the DAMOS scheme can show unexpected aggressiveness. Do the update. The user impact is not catastrophic. No leak or crash happens. Doing the quota goal online commit that can reproduce the issue is expected to be not common. This issue was not found by real users but the AI review. That said, the issue can reliably be reproduced. This issue was discovered [1] by Sashiko. Link: https://lore.kernel.org/20260902002725.108635-4-sj@kernel.org Link: https://lore.kernel.org/20260827045035.94611-1-sj@kernel.org [1] Fixes: 9138e27a3bc3 ("mm/damon: add node_eligible_mem_bp goal metric") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Lian Wang Tested-by: Lian Wang Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: Quanmin Yan Cc: # 7.2.x --- mm/damon/core.c | 3 +++ 1 file changed, 3 insertions(+) diff --git a/mm/damon/core.c b/mm/damon/core.c index 71e8cedbc64399..cf4ec122a7f99d 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -1204,6 +1204,9 @@ static void damos_commit_quota_goal_union( dst->nid = src->nid; dst->memcg_id = src->memcg_id; break; + case DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP: + dst->nid = src->nid; + break; default: break; } From 3924082d2765b1c8084a827360d44f6030b17bc3 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 17:27:23 -0700 Subject: [PATCH 0722/1352] mm/damon/sysfs: set next refresh jiffies per sysfs context When 'refresh_ms' is set, DAMON sysfs interface periodically updates auto-tuned parameters and DAMOS stats. The timestamp for the next refresh is initialized when a DAMON context starts, and updated in its damon_call() callback function. That is, each DAMON context updates it. However, the timestamp is a global variable that is shared with all the contexts. When there are multiple DAMON contexts having different refresh_ms, the update frequency will be changed, depending on the order of the contexts. When there are multiple kdamonds, it will be even more chaotic. Fix the problem by having the timestamp per each context. The user impact is not very critical. It does not leak, corrupt or crash. The update will not be faster or slower than the lowest and largest refresh_ms values of the contexts, respectively. The user can also manually ask the updates on demand using kdamond state commands. That said, clearly this is a bug and can easily be reproduced. Link: https://lore.kernel.org/20260902002725.108635-5-sj@kernel.org Fixes: 9fd7bb5083d1 ("mm/damon/sysfs: change next_update_jiffies to a global variable") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Lian Wang Tested-by: Lian Wang Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: Quanmin Yan Cc: # 6.18.x --- mm/damon/sysfs.c | 11 +++++------ 1 file changed, 5 insertions(+), 6 deletions(-) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index 7f340b6f1921b9..7ec14f48d157a4 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -2034,6 +2034,7 @@ struct damon_sysfs_kdamond { struct damon_sysfs_contexts *contexts; struct damon_ctx *damon_ctx; unsigned int refresh_ms; + unsigned long next_refresh_jiffies; }; static struct damon_sysfs_kdamond *damon_sysfs_kdamond_alloc(void) @@ -2484,17 +2485,15 @@ static struct damon_ctx *damon_sysfs_build_ctx( return ctx; } -static unsigned long damon_sysfs_next_update_jiffies; - static int damon_sysfs_repeat_call_fn(void *data) { struct damon_sysfs_kdamond *sysfs_kdamond = data; if (!sysfs_kdamond->refresh_ms) return 0; - if (time_before(jiffies, damon_sysfs_next_update_jiffies)) + if (time_before(jiffies, sysfs_kdamond->next_refresh_jiffies)) return 0; - damon_sysfs_next_update_jiffies = jiffies + + sysfs_kdamond->next_refresh_jiffies = jiffies + msecs_to_jiffies(sysfs_kdamond->refresh_ms); if (!mutex_trylock(&damon_sysfs_lock)) @@ -2542,8 +2541,8 @@ static int damon_sysfs_turn_damon_on(struct damon_sysfs_kdamond *kdamond) } kdamond->damon_ctx = ctx; - damon_sysfs_next_update_jiffies = - jiffies + msecs_to_jiffies(kdamond->refresh_ms); + kdamond->next_refresh_jiffies = jiffies + + msecs_to_jiffies(kdamond->refresh_ms); repeat_call_control->fn = damon_sysfs_repeat_call_fn; repeat_call_control->data = kdamond; From f8986a0734c856897454648312550709b22605c1 Mon Sep 17 00:00:00 2001 From: "Barry Song (Xiaomi)" Date: Wed, 2 Sep 2026 07:24:15 +0800 Subject: [PATCH 0723/1352] mm/mglru: separate folio generation update from LRU accounting MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Patch series "mm/mglru: speed up inc_min_seq() and fix cold/hot inversions", v3. This is an aging speedup series split out from the MGLRU swappiness series [1], with the inc_min_seq changes separated to make them easier to review. Currently, inc_min_seq() has two primary issues that affect both performance and folio hotness assessment: 1. It processes each folio one by one, while many operations can be batched or skipped. For example, a batch of folios can be moved together from the oldest generation to the second-oldest generation, and the associated counting can also be done in batches. 2. It may cause potential cold/hot inversion by placing promoted folios (which have been scanned and found to have young PTEs) behind non-promoted folios. A similar inversion can also occur among non-promoted folios, as tail folios from the oldest generation are placed before head folios when moving them to the second-oldest generation. This series tries to batch operations as much as possible and fix the potential cold/hot inversion by keeping promoted folios ahead of non-promoted folios, while also preserving the order of non-promoted folios when moving them from the oldest generation to the second-oldest generation. Minor issue: inc_min_seq() also counts protected folios improperly, as promoted folios should be skipped, as in sort_folio(). We need a stable workload with a stable number of folios to measure aging and evaluate the speedup in inc_min_seq(). So I asked ChatGPT to generate the microbenchmark below. It ages an LRU vec containing 512 MB of memory 100 times: #define _GNU_SOURCE #include #include #include #include #include #include #include #include #include #define SIZE (512UL * 1024 * 1024) #define LRU_GEN "/sys/kernel/debug/lru_gen" #define TARGET_CGROUP "/system.slice/agetest.scope" #define START_GEN 3 #define END_GEN 103 static long long nsec_diff(const struct timespec *start, const struct timespec *end) { return (end->tv_sec - start->tv_sec) * 1000000000LL + (end->tv_nsec - start->tv_nsec); } static int find_memcg_id(void) { FILE *fp; char line[4096]; int memcg_id; fp = fopen(LRU_GEN, "r"); if (!fp) { perror("fopen lru_gen"); return -1; } while (fgets(line, sizeof(line), fp)) { char *p; if (strncmp(line, "memcg ", 6)) continue; p = line + 6; if (sscanf(p, "%d", &memcg_id) != 1) continue; /* * The memcg path follows the numeric ID. */ p = strchr(p, ' '); if (!p) continue; if (strstr(p, TARGET_CGROUP)) { fclose(fp); return memcg_id; } } fclose(fp); fprintf(stderr, "Cannot find %s\n", TARGET_CGROUP); return -1; } int main(void) { void *addr; int memcg_id; int fd; long long total_ns = 0; /* * mmap 512 MB and touch every page. */ addr = mmap(NULL, SIZE, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); if (addr == MAP_FAILED) { perror("mmap"); return 1; } memset(addr, 0x55, SIZE); printf("mmap: %p, size: %lu MB\n", addr, SIZE / 1024 / 1024); /* * Find the memcg ID automatically. */ memcg_id = find_memcg_id(); if (memcg_id < 0) return 1; printf("memcg: %d (%s)\n", memcg_id, TARGET_CGROUP); printf("aging generation %d -> %d\n", START_GEN, END_GEN); fd = open(LRU_GEN, O_WRONLY); if (fd < 0) { perror("open lru_gen"); return 1; } for (int gen = START_GEN; gen <= END_GEN; gen++) { char buf[128]; int len; struct timespec start, end; long long ns; len = snprintf(buf, sizeof(buf), "+ %d 0 %d\n", memcg_id, gen); clock_gettime(CLOCK_MONOTONIC, &start); if (write(fd, buf, len) != len) { perror("write lru_gen"); close(fd); return 1; } clock_gettime(CLOCK_MONOTONIC, &end); ns = nsec_diff(&start, &end); total_ns += ns; printf("gen %3d: %8.3f ms\n", gen, ns / 1000000.0); fflush(stdout); } close(fd); printf("\nTotal: %.3f ms\n", total_ns / 1000000.0); printf("Average: %.3f ms\n", total_ns / (double)(END_GEN - START_GEN + 1) / 1000000.0); while (1) sleep(1); return 0; } Run the above microbenchmark with: systemd-run --scope --unit=agetest -p MemoryMax=1024M ./agetest I’m seeing a significant speedup in inc_min_seq() on my x86 PC: W/o patch: Running scope as unit: agetest.scope mmap: 0x72c1b5a00000, size: 512 MB memcg: 12673 (/system.slice/agetest.scope) aging generation 3 -> 103 gen 3: 7.433 ms gen 4: 0.949 ms gen 5: 2.535 ms gen 6: 5.043 ms gen 7: 5.041 ms gen 8: 5.027 ms ... gen 100: 5.035 ms gen 101: 5.011 ms gen 102: 5.029 ms gen 103: 5.056 ms Total: 503.946 ms Average: 4.990 ms W/ patch: Running scope as unit: agetest.scope mmap: 0x775ec5200000, size: 512 MB memcg: 12717 (/system.slice/agetest.scope) aging generation 3 -> 103 gen 3: 7.558 ms gen 4: 0.916 ms gen 5: 2.277 ms gen 6: 2.328 ms … gen 100: 2.303 ms gen 101: 2.297 ms gen 102: 2.305 ms gen 103: 2.312 ms Total: 236.635 ms Average: 2.343 ms The average aging time drops from 4.990 ms to 2.343 ms! Thanks, Xueyuan, for testing this on ARM[2]. It actually shows an even larger improvement. Xueyuan tested this series on his arm64 machine (24 cores, 4K base pages) and reproduced the improvement: THP=never (PTE): baseline: 7.644 ms patched: 2.964 ms (-61.2%) THP=always (PMD): baseline: 0.0373 ms patched: 0.0292 ms (-21.8%) This patch (of 7): folio_inc_gen() currently updates both the folio's generation and the LRU size accounting. This makes it difficult to batch the LRU size updates when moving multiple folios. Extract the generation update into __folio_inc_gen(), which only updates the folio's generation and reports whether the generation was actually increased. Keep folio_inc_gen() as the wrapper that performs the LRU size accounting when needed. This separates the per-folio generation update from LRU accounting and allows the latter to be batched by subsequent changes. Link: https://lore.kernel.org/20260901232421.40157-1-baohua@kernel.org Link: https://lore.kernel.org/20260901232421.40157-2-baohua@kernel.org Link: https://lore.kernel.org/linux-mm/20260812121658.69965-1-baohua@kernel.org/ [1] Link: https://lore.kernel.org/linux-mm/20260827035416.3012015-1-xueyuan.chen21@gmail.com/ [2] Signed-off-by: Barry Song (Xiaomi) Signed-off-by: Andrew Morton Tested-by: Xueyuan Chen Reviewed-by: Lian Wang Reviewed-by: Baolin Wang Cc: Axel Rasmussen Cc: Baoquan He Cc: David Hildenbrand Cc: David Stevens Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie Cc: Kunwu Chan Cc: Ridong Chen --- mm/vmscan.c | 28 +++++++++++++++++++++------- 1 file changed, 21 insertions(+), 7 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index ce3bab78af3cd1..d1a051e7db1eb6 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -3300,21 +3300,21 @@ static int folio_update_gen(struct folio *folio, int gen, const vma_flags_t *vma return ((old_flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1; } -/* protect pages accessed multiple times through file descriptors */ -static int folio_inc_gen(struct lruvec *lruvec, struct folio *folio) +static int __folio_inc_gen(struct folio *folio, int old_gen, bool *increased) { - int type = folio_is_file_lru(folio); - struct lru_gen_folio *lrugen = &lruvec->lrugen; - int new_gen, old_gen = lru_gen_from_seq(lrugen->min_seq[type]); unsigned long new_flags, old_flags = READ_ONCE(folio->flags.f); + int new_gen; VM_WARN_ON_ONCE_FOLIO(!(old_flags & LRU_GEN_MASK), folio); do { new_gen = ((old_flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1; /* folio_update_gen() has promoted this page? */ - if (new_gen >= 0 && new_gen != old_gen) + if (new_gen >= 0 && new_gen != old_gen) { + if (increased) + *increased = false; return new_gen; + } new_gen = (old_gen + 1) % MAX_NR_GENS; @@ -3322,8 +3322,22 @@ static int folio_inc_gen(struct lruvec *lruvec, struct folio *folio) new_flags |= (new_gen + 1UL) << LRU_GEN_PGOFF; } while (!try_cmpxchg(&folio->flags.f, &old_flags, new_flags)); - lru_gen_update_size(lruvec, folio, old_gen, new_gen); + if (increased) + *increased = true; + return new_gen; +} + +/* protect pages accessed multiple times through file descriptors */ +static int folio_inc_gen(struct lruvec *lruvec, struct folio *folio) +{ + int type = folio_is_file_lru(folio); + struct lru_gen_folio *lrugen = &lruvec->lrugen; + int new_gen, old_gen = lru_gen_from_seq(lrugen->min_seq[type]); + bool gen_increased; + new_gen = __folio_inc_gen(folio, old_gen, &gen_increased); + if (gen_increased) + lru_gen_update_size(lruvec, folio, old_gen, new_gen); return new_gen; } From d063e9490a3025e3f060f1e70bb685251036a300 Mon Sep 17 00:00:00 2001 From: "Barry Song (Xiaomi)" Date: Wed, 2 Sep 2026 07:24:16 +0800 Subject: [PATCH 0724/1352] mm/mglru: batch update lrugen->nr_pages in inc_min_seq() Currently, folio_inc_gen() updates lrugen->nr_pages for every folio as it advances generations. Instead, accumulate the size changes and update lrugen->nr_pages in a batch after scanning the entire oldest generation, or when the scan stops because remaining reaches zero. Since we only move folios from the oldest generation to the second oldest generation, the active/inactive state cannot change. We can therefore skip __lru_update_size(). Link: https://lore.kernel.org/20260901232421.40157-3-baohua@kernel.org Signed-off-by: Barry Song (Xiaomi) Signed-off-by: Andrew Morton Tested-by: Xueyuan Chen Reviewed-by: Lian Wang Reviewed-by: Kunwu Chan Reviewed-by: Baolin Wang Cc: Axel Rasmussen Cc: Baoquan He Cc: David Hildenbrand Cc: David Stevens Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Ridong Chen Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie --- mm/vmscan.c | 23 ++++++++++++++++++----- 1 file changed, 18 insertions(+), 5 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index d1a051e7db1eb6..801345ca4da3c6 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -3923,6 +3923,7 @@ static bool inc_min_seq(struct lruvec *lruvec, int type, int swappiness) struct lru_gen_folio *lrugen = &lruvec->lrugen; int hist = lru_hist_from_seq(lrugen->min_seq[type]); int new_gen, old_gen = lru_gen_from_seq(lrugen->min_seq[type]); + int target_gen = (old_gen + 1) % MAX_NR_GENS; /* For file type, skip the check if swappiness is anon only */ if (type && (swappiness == SWAPPINESS_ANON_ONLY)) @@ -3932,35 +3933,47 @@ static bool inc_min_seq(struct lruvec *lruvec, int type, int swappiness) if (!type && !swappiness) goto done; + VM_WARN_ON_ONCE(get_nr_gens(lruvec, type) != MAX_NR_GENS); + VM_WARN_ON_ONCE(lru_gen_is_active(lruvec, old_gen) != + lru_gen_is_active(lruvec, target_gen)); /* prevent cold/hot inversion if the type is evictable */ for (zone = 0; zone < MAX_NR_ZONES; zone++) { struct list_head *head = &lrugen->folios[old_gen][type][zone]; + long delta = 0; while (!list_empty(head)) { struct folio *folio = lru_to_folio(head); + long nr_pages = folio_nr_pages(folio); int refs = folio_lru_refs(folio); bool workingset = folio_test_workingset(folio); + bool gen_increased; VM_WARN_ON_ONCE_FOLIO(folio_test_unevictable(folio), folio); VM_WARN_ON_ONCE_FOLIO(folio_test_active(folio), folio); VM_WARN_ON_ONCE_FOLIO(folio_is_file_lru(folio) != type, folio); VM_WARN_ON_ONCE_FOLIO(folio_zonenum(folio) != zone, folio); - new_gen = folio_inc_gen(lruvec, folio); + new_gen = __folio_inc_gen(folio, old_gen, &gen_increased); list_move_tail(&folio->lru, &lrugen->folios[new_gen][type][zone]); - + if (gen_increased) + delta += nr_pages; /* don't count the workingset being lazily promoted */ if (refs + workingset != BIT(LRU_REFS_WIDTH) + 1) { int tier = lru_tier_from_refs(refs, workingset); - int delta = folio_nr_pages(folio); WRITE_ONCE(lrugen->protected[hist][type][tier], - lrugen->protected[hist][type][tier] + delta); + lrugen->protected[hist][type][tier] + nr_pages); } if (!--remaining) - return false; + break; } + WRITE_ONCE(lrugen->nr_pages[old_gen][type][zone], + lrugen->nr_pages[old_gen][type][zone] - delta); + WRITE_ONCE(lrugen->nr_pages[target_gen][type][zone], + lrugen->nr_pages[target_gen][type][zone] + delta); + if (!remaining) + return false; } done: reset_ctrl_pos(lruvec, type, true); From f8963d902b7f8b1a5cea06f308d98d7176f27f69 Mon Sep 17 00:00:00 2001 From: "Barry Song (Xiaomi)" Date: Wed, 2 Sep 2026 07:24:17 +0800 Subject: [PATCH 0725/1352] mm/mglru: enhance cold/hot inversion handling in inc_min_seq() During aging, a folio's generation may already have been updated by folio_update_gen(), even though it has not yet been moved to the corresponding generation list. Such folios are hotter than those already in that generation. It makes sense for inc_min_seq() to increment the generation of folios that were never promoted during aging and move them to the tail of the new oldest generation. However, folios that were already promoted should instead be moved to the head of their updated generation, just as sort_folio() does in scan_folios(). Otherwise, promoted folios could end up behind folios that were never promoted, effectively inverting their hot/cold ordering. Link: https://lore.kernel.org/20260901232421.40157-4-baohua@kernel.org Signed-off-by: Barry Song (Xiaomi) Signed-off-by: Andrew Morton Reviewed-by: Kairui Song Tested-by: Xueyuan Chen Reviewed-by: Ridong Chen Reviewed-by: Lian Wang Reviewed-by: Baolin Wang Cc: Axel Rasmussen Cc: Baoquan He Cc: David Hildenbrand Cc: David Stevens Cc: Johannes Weiner Cc: Kunwu Chan Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie --- mm/vmscan.c | 12 ++++++++++-- 1 file changed, 10 insertions(+), 2 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 801345ca4da3c6..a608483ff9c7f9 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -3954,9 +3954,17 @@ static bool inc_min_seq(struct lruvec *lruvec, int type, int swappiness) VM_WARN_ON_ONCE_FOLIO(folio_zonenum(folio) != zone, folio); new_gen = __folio_inc_gen(folio, old_gen, &gen_increased); - list_move_tail(&folio->lru, &lrugen->folios[new_gen][type][zone]); - if (gen_increased) + /* + * If gen_increased is false, this is a promotion. Put folios + * at the head of the promoted gen. Otherwise, put them at + * the tail of the second-oldest gen. + */ + if (gen_increased) { delta += nr_pages; + list_move_tail(&folio->lru, &lrugen->folios[new_gen][type][zone]); + } else { + list_move(&folio->lru, &lrugen->folios[new_gen][type][zone]); + } /* don't count the workingset being lazily promoted */ if (refs + workingset != BIT(LRU_REFS_WIDTH) + 1) { int tier = lru_tier_from_refs(refs, workingset); From 7c715adb81065bf4785f641486861591f9da9cd9 Mon Sep 17 00:00:00 2001 From: "Barry Song (Xiaomi)" Date: Wed, 2 Sep 2026 07:24:18 +0800 Subject: [PATCH 0726/1352] mm/mglru: exclude folios promoted by aging from protected in inc_min_seq() Some folios may have been promoted during aging, so don't count them as protected, similar to sort_folio(). Link: https://lore.kernel.org/20260901232421.40157-5-baohua@kernel.org Signed-off-by: Barry Song (Xiaomi) Signed-off-by: Andrew Morton Reviewed-by: Baoquan He Tested-by: Xueyuan Chen Reviewed-by: Ridong Chen Reviewed-by: Lian Wang Cc: Axel Rasmussen Cc: Baolin Wang Cc: David Hildenbrand Cc: David Stevens Cc: Johannes Weiner Cc: Kairui Song Cc: Kunwu Chan Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie --- mm/vmscan.c | 16 ++++++++-------- 1 file changed, 8 insertions(+), 8 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index a608483ff9c7f9..79311f62d3e615 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -3962,17 +3962,17 @@ static bool inc_min_seq(struct lruvec *lruvec, int type, int swappiness) if (gen_increased) { delta += nr_pages; list_move_tail(&folio->lru, &lrugen->folios[new_gen][type][zone]); + + /* don't count the workingset being lazily promoted */ + if (refs + workingset != BIT(LRU_REFS_WIDTH) + 1) { + int tier = lru_tier_from_refs(refs, workingset); + + WRITE_ONCE(lrugen->protected[hist][type][tier], + lrugen->protected[hist][type][tier] + nr_pages); + } } else { list_move(&folio->lru, &lrugen->folios[new_gen][type][zone]); } - /* don't count the workingset being lazily promoted */ - if (refs + workingset != BIT(LRU_REFS_WIDTH) + 1) { - int tier = lru_tier_from_refs(refs, workingset); - - WRITE_ONCE(lrugen->protected[hist][type][tier], - lrugen->protected[hist][type][tier] + nr_pages); - } - if (!--remaining) break; } From c46d2bd7d325f70dc0e468e66a2df5c4a1b16d83 Mon Sep 17 00:00:00 2001 From: "Barry Song (Xiaomi)" Date: Wed, 2 Sep 2026 07:24:19 +0800 Subject: [PATCH 0727/1352] mm/mglru: make LRU folio prefetch helper an inline function `prefetchw_prev_lru_folio()` is currently implemented as a macro with a potentially unused argument. This makes the helper harder to read and can also trigger checkpatch warnings about unused macro arguments. Make it a `static inline` function and remove the unnecessary `_field` argument. The helper always prefetches the previous folio's `flags`, so there is no need to make the field configurable. Link: https://lore.kernel.org/20260901232421.40157-6-baohua@kernel.org Signed-off-by: Barry Song (Xiaomi) Signed-off-by: Andrew Morton Tested-by: Xueyuan Chen Reviewed-by: Lian Wang Reviewed-by: Baolin Wang Cc: Axel Rasmussen Cc: Baoquan He Cc: David Hildenbrand Cc: David Stevens Cc: Johannes Weiner Cc: Kairui Song Cc: Kunwu Chan Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Ridong Chen Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie --- mm/vmscan.c | 26 +++++++++++++++----------- 1 file changed, 15 insertions(+), 11 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 79311f62d3e615..4b0ed3c9aab295 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -182,17 +182,21 @@ struct scan_control { }; #ifdef ARCH_HAS_PREFETCHW -#define prefetchw_prev_lru_folio(_folio, _base, _field) \ - do { \ - if ((_folio)->lru.prev != _base) { \ - struct folio *prev; \ - \ - prev = lru_to_folio(&(_folio->lru)); \ - prefetchw(&prev->_field); \ - } \ - } while (0) +static inline void prefetchw_prev_lru_folio(struct folio *folio, + struct list_head *base) +{ + if (folio->lru.prev != base) { + struct folio *prev; + + prev = lru_to_folio(&folio->lru); + prefetchw(&prev->flags); + } +} #else -#define prefetchw_prev_lru_folio(_folio, _base, _field) do { } while (0) +static inline void prefetchw_prev_lru_folio(struct folio *folio, + struct list_head *base) +{ +} #endif /* @@ -1700,7 +1704,7 @@ static unsigned long isolate_lru_folios(unsigned long nr_to_scan, struct folio *folio; folio = lru_to_folio(src); - prefetchw_prev_lru_folio(folio, src, flags); + prefetchw_prev_lru_folio(folio, src); nr_pages = folio_nr_pages(folio); total_scan += nr_pages; From d34031d1bdfbb5f2dca192de171c2b4047b21b97 Mon Sep 17 00:00:00 2001 From: "Barry Song (Xiaomi)" Date: Wed, 2 Sep 2026 07:24:20 +0800 Subject: [PATCH 0728/1352] mm/mglru: move folios from oldest gen to second-oldest gen from head to tail For reclamation, it makes sense to reclaim folios from tail to head, as folios near the head are relatively hot. However, when moving folios from the oldest generation to the second-oldest generation, using the tail-to-head order would effectively cause a cold/hot inversion. The impact of the added prefetching might be arch-dependent. Some architectures could benefit more from prefetching, while others might see little to no impact. In my x86 test, it shows a very slight improvement: ***** no-prefetch: agetest: ... gen 100: 2.421 ms gen 101: 2.424 ms gen 102: 2.420 ms gen 103: 2.413 ms Total: 248.418 ms Average: 2.460 ms agetest: ... gen 100: 2.393 ms gen 101: 2.396 ms gen 102: 2.395 ms gen 103: 2.392 ms Total: 245.627 ms Average: 2.432 ms agetest: ... gen 100: 2.433 ms gen 101: 2.427 ms gen 102: 2.432 ms gen 103: 2.450 ms Total: 249.186 ms Average: 2.467 ms **** has-prefetch: agetest: .... gen 100: 2.314 ms gen 101: 2.310 ms gen 102: 2.321 ms gen 103: 2.303 ms Total: 237.619 ms Average: 2.353 ms agetest: gen 100: 2.342 ms gen 101: 2.343 ms gen 102: 2.339 ms gen 103: 2.335 ms Total: 239.929 ms Average: 2.376 ms agetest: gen 100: 2.344 ms gen 101: 2.347 ms gen 102: 2.352 ms gen 103: 2.348 ms Total: 241.188 ms Average: 2.388 ms Basically, its 2.3xx vs. 2.4xx, lower is better. Link: https://lore.kernel.org/20260901232421.40157-7-baohua@kernel.org Signed-off-by: Barry Song (Xiaomi) Signed-off-by: Andrew Morton Reviewed-by: Baoquan He Tested-by: Xueyuan Chen Reviewed-by: Lian Wang Cc: Axel Rasmussen Cc: Baolin Wang Cc: David Hildenbrand Cc: David Stevens Cc: Johannes Weiner Cc: Kairui Song Cc: Kunwu Chan Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Ridong Chen Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie --- mm/vmscan.c | 23 +++++++++++++++++++++-- 1 file changed, 21 insertions(+), 2 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 4b0ed3c9aab295..49fab93470ad87 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -192,11 +192,27 @@ static inline void prefetchw_prev_lru_folio(struct folio *folio, prefetchw(&prev->flags); } } + +static inline void prefetchw_next_lru_folio(struct folio *folio, + struct list_head *base) +{ + if (folio->lru.next != base) { + struct folio *next; + + next = list_entry(folio->lru.next, struct folio, lru); + prefetchw(&next->flags); + } +} #else static inline void prefetchw_prev_lru_folio(struct folio *folio, struct list_head *base) { } + +static inline void prefetchw_next_lru_folio(struct folio *folio, + struct list_head *base) +{ +} #endif /* @@ -3943,10 +3959,11 @@ static bool inc_min_seq(struct lruvec *lruvec, int type, int swappiness) /* prevent cold/hot inversion if the type is evictable */ for (zone = 0; zone < MAX_NR_ZONES; zone++) { struct list_head *head = &lrugen->folios[old_gen][type][zone]; + struct list_head *pos = head->next; long delta = 0; - while (!list_empty(head)) { - struct folio *folio = lru_to_folio(head); + while (pos != head) { + struct folio *folio = list_entry(pos, struct folio, lru); long nr_pages = folio_nr_pages(folio); int refs = folio_lru_refs(folio); bool workingset = folio_test_workingset(folio); @@ -3957,6 +3974,8 @@ static bool inc_min_seq(struct lruvec *lruvec, int type, int swappiness) VM_WARN_ON_ONCE_FOLIO(folio_is_file_lru(folio) != type, folio); VM_WARN_ON_ONCE_FOLIO(folio_zonenum(folio) != zone, folio); + prefetchw_next_lru_folio(folio, head); + pos = pos->next; new_gen = __folio_inc_gen(folio, old_gen, &gen_increased); /* * If gen_increased is false, this is a promotion. Put folios From 3515f6637ef8497acdb9207ad7451127cc9544fc Mon Sep 17 00:00:00 2001 From: "Barry Song (Xiaomi)" Date: Wed, 2 Sep 2026 07:24:21 +0800 Subject: [PATCH 0729/1352] mm/mglru: batch move folios to the second-oldest gen's LRU Detect folios that need to move from the oldest generation to the second-oldest generation, and batch-move them together. This can significantly reduce the sys time of inc_min_seq(), especially when the other type is significantly behind the preferred type. Link: https://lore.kernel.org/20260901232421.40157-8-baohua@kernel.org Signed-off-by: Barry Song (Xiaomi) Signed-off-by: Andrew Morton Reviewed-by: Baoquan He Tested-by: Xueyuan Chen Reviewed-by: Lian Wang Reviewed-by: Baolin Wang Assisted-by: gemini:gemini-3.6-flash Cc: Axel Rasmussen Cc: David Hildenbrand Cc: David Stevens Cc: Johannes Weiner Cc: Kairui Song Cc: Kunwu Chan Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Ridong Chen Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie --- mm/vmscan.c | 20 +++++++++++++++++++- 1 file changed, 19 insertions(+), 1 deletion(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 49fab93470ad87..ba7adf36e69f7b 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -3936,6 +3936,19 @@ static void clear_mm_walk(void) kfree(walk); } +static inline void flush_lru_batch(struct list_head *head, struct list_head **batch_end, + struct list_head *dst) +{ + LIST_HEAD(movable); + + if (!*batch_end) + return; + + list_cut_position(&movable, head, *batch_end); + list_splice_tail_init(&movable, dst); + *batch_end = NULL; +} + static bool inc_min_seq(struct lruvec *lruvec, int type, int swappiness) { int zone; @@ -3958,8 +3971,10 @@ static bool inc_min_seq(struct lruvec *lruvec, int type, int swappiness) lru_gen_is_active(lruvec, target_gen)); /* prevent cold/hot inversion if the type is evictable */ for (zone = 0; zone < MAX_NR_ZONES; zone++) { + struct list_head *target_list = &lrugen->folios[target_gen][type][zone]; struct list_head *head = &lrugen->folios[old_gen][type][zone]; struct list_head *pos = head->next; + struct list_head *batch_end = NULL; long delta = 0; while (pos != head) { @@ -3984,7 +3999,7 @@ static bool inc_min_seq(struct lruvec *lruvec, int type, int swappiness) */ if (gen_increased) { delta += nr_pages; - list_move_tail(&folio->lru, &lrugen->folios[new_gen][type][zone]); + batch_end = &folio->lru; /* don't count the workingset being lazily promoted */ if (refs + workingset != BIT(LRU_REFS_WIDTH) + 1) { @@ -3994,11 +4009,14 @@ static bool inc_min_seq(struct lruvec *lruvec, int type, int swappiness) lrugen->protected[hist][type][tier] + nr_pages); } } else { + flush_lru_batch(head, &batch_end, target_list); list_move(&folio->lru, &lrugen->folios[new_gen][type][zone]); } if (!--remaining) break; } + flush_lru_batch(head, &batch_end, target_list); + WRITE_ONCE(lrugen->nr_pages[old_gen][type][zone], lrugen->nr_pages[old_gen][type][zone] - delta); WRITE_ONCE(lrugen->nr_pages[target_gen][type][zone], From 3b628f6ba843632d9ba0da00f6e1bc2bc6d1e432 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 07:03:07 -0700 Subject: [PATCH 0730/1352] mm/damon/tests/core-kunit: test damon_commit_filter() Patch series "mm/damon: add kunit and selftests for probes and probe weights". DAMON recently introduced probes and probe weights. Add kunit and selftests for ensuring the parameters for features can be set using the core API and the sysfs ABI, respectively. This patch (of 6): Add kunit test to ensure damon_commit_filter() updates destination filter as expected for valid inputs. Link: https://lore.kernel.org/20260902140313.85983-1-sj@kernel.org Link: https://lore.kernel.org/20260902140313.85983-2-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Shuah Khan --- mm/damon/tests/core-kunit.h | 40 +++++++++++++++++++++++++++++++++++++ 1 file changed, 40 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index af26b3d60957b5..3fc5b631b45c3a 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -1316,6 +1316,45 @@ static void damon_test_commit_target_regions(struct kunit *test) (unsigned long[][2]) {{3, 8}, {8, 10}}, 2); } +static void damon_test_commit_filter_for(struct kunit *test, + struct damon_filter *dst, struct damon_filter *src) +{ + damon_commit_filter(dst, src); + KUNIT_EXPECT_EQ(test, dst->type, src->type); + KUNIT_EXPECT_EQ(test, dst->matching, src->matching); + KUNIT_EXPECT_EQ(test, dst->allow, src->allow); + switch (src->type) { + case DAMON_FILTER_TYPE_MEMCG: + KUNIT_EXPECT_EQ(test, dst->memcg_id, src->memcg_id); + break; + default: + break; + } +} + +static void damon_test_commit_filter(struct kunit *test) +{ + struct damon_filter dst = { + .type = DAMON_FILTER_TYPE_ANON, + .matching = false, + .allow = false, + }; + + damon_test_commit_filter_for(test, &dst, + &(struct damon_filter){ + .type = DAMON_FILTER_TYPE_ANON, + .matching = true, + .allow = true, + }); + damon_test_commit_filter_for(test, &dst, + &(struct damon_filter){ + .type = DAMON_FILTER_TYPE_MEMCG, + .matching = false, + .allow = false, + .memcg_id = 123, + }); +} + static void damon_test_commit_ctx(struct kunit *test) { struct damon_ctx *src, *dst; @@ -1722,6 +1761,7 @@ static struct kunit_case damon_test_cases[] = { KUNIT_CASE(damos_test_commit_pageout), KUNIT_CASE(damos_test_commit_migrate_hot), KUNIT_CASE(damon_test_commit_target_regions), + KUNIT_CASE(damon_test_commit_filter), KUNIT_CASE(damon_test_commit_ctx), KUNIT_CASE(damon_test_valid_probe_params), KUNIT_CASE(damos_test_filter_out), From eacad6f5f918eaf9df43b3c322a6e17ecb4ea6af Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 07:03:08 -0700 Subject: [PATCH 0731/1352] mm/damon/tests/core-kunit: add damon_commit_probes() test Add kunit test to ensure damon_commit_probes() updates destination DAMON context with source probes as expected. Link: https://lore.kernel.org/20260902140313.85983-3-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Shuah Khan --- mm/damon/tests/core-kunit.h | 84 +++++++++++++++++++++++++++++++++++++ 1 file changed, 84 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index 3fc5b631b45c3a..f2568fba552ec4 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -1355,6 +1355,89 @@ static void damon_test_commit_filter(struct kunit *test) }); } +static struct damon_ctx *damon_test_help_setup_probes(unsigned int weights[], + int nr_weights) +{ + struct damon_ctx *ctx; + struct damon_probe *probe; + int i; + + ctx = damon_new_ctx(); + if (!ctx) + return NULL; + for (i = 0; i < nr_weights; i++) { + probe = damon_new_probe(); + if (!probe) { + damon_destroy_ctx(ctx); + return NULL; + } + probe->weight = weights[i]; + damon_add_probe(ctx, probe); + } + return ctx; +} + +static void damon_test_commit_probes_for(struct kunit *test, + unsigned int dst_weights[], int nr_dst_probes, + unsigned int src_weights[], int nr_src_probes) +{ + struct damon_ctx *dst, *src; + int err; + struct damon_probe *dst_probe, *src_probe; + + dst = damon_test_help_setup_probes(dst_weights, nr_dst_probes); + if (!dst) + kunit_skip(test, "dst alloc fail"); + src = damon_test_help_setup_probes(src_weights, nr_src_probes); + if (!src) { + damon_destroy_ctx(dst); + kunit_skip(test, "src alloc fail"); + } + + err = damon_commit_probes(dst, src); + KUNIT_EXPECT_EQ(test, err, 0); + if (err) + goto out; + nr_dst_probes = 0; + damon_for_each_probe(dst_probe, dst) + nr_dst_probes++; + nr_src_probes = 0; + damon_for_each_probe(src_probe, src) + nr_src_probes++; + KUNIT_EXPECT_EQ(test, nr_dst_probes, nr_src_probes); + if (nr_dst_probes != nr_src_probes) + goto out; + nr_dst_probes = 0; + damon_for_each_probe(dst_probe, dst) { + src_probe = damon_nth_probe(nr_dst_probes, src); + KUNIT_EXPECT_EQ(test, src_probe->weight, dst_probe->weight); + nr_dst_probes++; + } +out: + damon_destroy_ctx(dst); + damon_destroy_ctx(src); +} + +static void damon_test_commit_probes(struct kunit *test) +{ + damon_test_commit_probes_for(test, + (unsigned int[]){}, 0, (unsigned int[]){}, 0); + damon_test_commit_probes_for(test, + (unsigned int[]){}, 0, (unsigned int[]){1}, 1); + damon_test_commit_probes_for(test, + (unsigned int[]){}, 0, (unsigned int[]){1, 2}, 2); + damon_test_commit_probes_for(test, + (unsigned int[]){1}, 1, (unsigned int[]){2}, 1); + damon_test_commit_probes_for(test, + (unsigned int[]){1}, 1, (unsigned int[]){2, 3}, 2); + damon_test_commit_probes_for(test, + (unsigned int[]){2, 3}, 2, (unsigned int[]){1}, 1); + damon_test_commit_probes_for(test, + (unsigned int[]){2, 3}, 2, (unsigned int[]){}, 0); + damon_test_commit_probes_for(test, + (unsigned int[]){2}, 1, (unsigned int[]){}, 0); +} + static void damon_test_commit_ctx(struct kunit *test) { struct damon_ctx *src, *dst; @@ -1762,6 +1845,7 @@ static struct kunit_case damon_test_cases[] = { KUNIT_CASE(damos_test_commit_migrate_hot), KUNIT_CASE(damon_test_commit_target_regions), KUNIT_CASE(damon_test_commit_filter), + KUNIT_CASE(damon_test_commit_probes), KUNIT_CASE(damon_test_commit_ctx), KUNIT_CASE(damon_test_valid_probe_params), KUNIT_CASE(damos_test_filter_out), From 99a86f7c2e10295e9cb557d5eb807b304aaf150f Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 07:03:09 -0700 Subject: [PATCH 0732/1352] selftests/damon/_damon_sysfs: implement DamonProbes Extend _damon_sysfs.py to support staging and committing DAMON probes. It will be used for setting DAMON probes via sysfs changes for testing purposes. Link: https://lore.kernel.org/20260902140313.85983-4-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Shuah Khan --- tools/testing/selftests/damon/_damon_sysfs.py | 125 +++++++++++++++++- 1 file changed, 124 insertions(+), 1 deletion(-) diff --git a/tools/testing/selftests/damon/_damon_sysfs.py b/tools/testing/selftests/damon/_damon_sysfs.py index f604b7d6530b3c..e7095c3365245a 100644 --- a/tools/testing/selftests/damon/_damon_sysfs.py +++ b/tools/testing/selftests/damon/_damon_sysfs.py @@ -570,6 +570,117 @@ def stage(self): return err return None +class DamonFilter: + type_ = None + matching = None + allow = None + filters = None + path = None + idx = None + + def __init__(self, type_='anon', matching=False, allow=False, path=None): + self.type_ = type_ + self.matching = matching + self.allow = allow + self.path = path + + def sysfs_dir(self): + return os.path.join(self.filters.sysfs_dir(), '%d' % self.idx) + + def stage(self): + err = write_file(os.path.join(self.sysfs_dir(), 'type'), self.type_) + if err is not None: + return err + err = write_file(os.path.join(self.sysfs_dir(), 'matching'), + 'Y' if self.matching else 'N') + if err is not None: + return err + err = write_file(os.path.join(self.sysfs_dir(), 'allow'), + 'Y' if self.allow else 'N') + if err is not None: + return err + if self.type_ == 'memcg': + err = write_file(os.path.join(self.sysfs_dir(), 'path'), self.path) + if err is not None: + return err + return None + +class DamonFilters: + filters = None + probe = None + + def __init__(self, filters=None): + if filters is None: + filters = [] + self.filters = filters + for idx, filter in enumerate(self.filters): + filter.filters = self + filter.idx = idx + + def sysfs_dir(self): + return os.path.join(self.probe.sysfs_dir(), 'filters') + + def stage(self): + err = write_file( + os.path.join(self.sysfs_dir(), 'nr_filters'), + len(self.filters)) + if err is not None: + return err + for filter in self.filters: + err = filter.stage() + if err is not None: + return err + return None + +class DamonProbe: + weight = None + filters = None + probes = None + idx = None + + def __init__(self, weight=0, filters=None): + self.weight = weight + if filters is None: + filters = DamonFilters() + self.filters = filters + self.filters.probe = self + + def sysfs_dir(self): + return os.path.join(self.probes.sysfs_dir(), '%d' % self.idx) + + def stage(self): + err = write_file( + os.path.join(self.sysfs_dir(), 'weight'), '%d' % self.weight) + if err is not None: + return err + return self.filters.stage() + +class DamonProbes: + probes = None + attrs = None + + def __init__(self, probes=None): + if probes is None: + probes = [] + self.probes = probes + for idx, probe in enumerate(self.probes): + probe.probes = self + probe.idx = idx + + def sysfs_dir(self): + return os.path.join(self.attrs.sysfs_dir(), 'probes') + + def stage(self): + err = write_file(os.path.join(self.sysfs_dir(), 'nr_probes'), + len(self.probes)) + if err is not None: + return err + for probe in self.probes: + err = probe.stage() + if err is not None: + return err + return None + class DamonAttrs: sample_us = None aggr_us = None @@ -577,11 +688,12 @@ class DamonAttrs: update_us = None min_nr_regions = None max_nr_regions = None + probes = None context = None def __init__(self, sample_us=5000, aggr_us=100000, intervals_goal=None, update_us=1000000, min_nr_regions=10, - max_nr_regions=1000): + max_nr_regions=1000, probes=None): self.sample_us = sample_us self.aggr_us = aggr_us if intervals_goal is None: @@ -591,6 +703,10 @@ def __init__(self, sample_us=5000, aggr_us=100000, self.update_us = update_us self.min_nr_regions = min_nr_regions self.max_nr_regions = max_nr_regions + if probes is None: + probes = DamonProbes() + self.probes = probes + self.probes.attrs = self def interval_sysfs_dir(self): return os.path.join(self.context.sysfs_dir(), 'monitoring_attrs', @@ -600,6 +716,9 @@ def nr_regions_range_sysfs_dir(self): return os.path.join(self.context.sysfs_dir(), 'monitoring_attrs', 'nr_regions') + def sysfs_dir(self): + return os.path.join(self.context.sysfs_dir(), 'monitoring_attrs') + def stage(self): err = write_file(os.path.join(self.interval_sysfs_dir(), 'sample_us'), self.sample_us) @@ -629,6 +748,10 @@ def stage(self): if err is not None: return err + err = self.probes.stage() + if err is not None: + return err + class DamonCtx: ops = None monitoring_attrs = None From db2b8b8bc149a216b6a63697f9035a6aef9b0e2b Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 07:03:10 -0700 Subject: [PATCH 0733/1352] selftests/damon/drgn_dump_damon_status: dump probes Extend drgn_dump_damon_status.py to dump damon_ctx->probes. It will be used to see if in-kernel DAMON status are changed as the user sets the probes via sysfs. Link: https://lore.kernel.org/20260902140313.85983-5-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Shuah Khan --- .../selftests/damon/drgn_dump_damon_status.py | 32 +++++++++++++++++++ 1 file changed, 32 insertions(+) diff --git a/tools/testing/selftests/damon/drgn_dump_damon_status.py b/tools/testing/selftests/damon/drgn_dump_damon_status.py index 09552e91bc7820..4622046fd01180 100755 --- a/tools/testing/selftests/damon/drgn_dump_damon_status.py +++ b/tools/testing/selftests/damon/drgn_dump_damon_status.py @@ -48,6 +48,37 @@ def attrs_to_dict(attrs): ['max_nr_regions', int], ]) +def filter_to_dict(damon_filter): + filter_type_keyword = { + 0: 'anon', + 1: 'memcg', + } + dict_ = { + 'type': filter_type_keyword[int(damon_filter.type)], + 'matching': bool(damon_filter.matching), + 'allow': bool(damon_filter.allow), + } + type_ = dict_['type'] + if type_ == 'memcg': + dict_['memcg_id'] = int(damon_filter.memcg_id) + return dict_ + +def filters_to_list(filters): + return [filter_to_dict(f) + for f in list_for_each_entry( + 'struct damon_filter', filters.address_of_(), 'list')] + +def probe_to_dict(probe): + return to_dict(probe, [ + ['weight', int], + ['filters', filters_to_list], + ]) + +def probes_to_list(probes): + return [probe_to_dict(p) + for p in list_for_each_entry( + 'struct damon_probe', probes.address_of_(), 'list')] + def addr_range_to_dict(addr_range): return to_dict(addr_range, [ ['start', int], @@ -199,6 +230,7 @@ def damon_ctx_to_dict(ctx): return to_dict(ctx, [ ['ops', ops_to_dict], ['attrs', attrs_to_dict], + ['probes', probes_to_list], ['adaptive_targets', targets_to_list], ['schemes', schemes_to_list], ['pause', bool], From 98959fd09230c46b52110581a92c02682c6e4d67 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 07:03:11 -0700 Subject: [PATCH 0734/1352] selftests/damon/sysfs.py: extend commit assertion function for probes Extend DAMON sysfs testing commit assertion helper function to check probes too. Link: https://lore.kernel.org/20260902140313.85983-6-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Shuah Khan --- tools/testing/selftests/damon/sysfs.py | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) diff --git a/tools/testing/selftests/damon/sysfs.py b/tools/testing/selftests/damon/sysfs.py index 88a26422ff44cf..159cbeac067f89 100755 --- a/tools/testing/selftests/damon/sysfs.py +++ b/tools/testing/selftests/damon/sysfs.py @@ -178,6 +178,24 @@ def assert_monitoring_attrs_committed(attrs, dump): assert_true(dump['max_nr_regions'] == attrs.max_nr_regions, 'max_nr_regions', dump) +def assert_damon_filters_committed(filters, dump): + assert_true(len(dump) == len(filters.filters), 'probe filters', dump) + for idx, damon_filter in enumerate(filters.filters): + filter_dump = dump[idx] + assert_true(filter_dump['type'] == damon_filter.type_, 'type', + filter_dump) + assert_true(filter_dump['matching'] == damon_filter.matching, + 'matching', filter_dump) + assert_true(filter_dump['allow'] == damon_filter.allow, 'allow', + filter_dump) + +def assert_probes_committed(probes, dump): + assert_true(len(dump) == len(probes.probes), 'probes length', dump) + for idx, probe in enumerate(probes.probes): + probe_dump = dump[idx] + assert_true(probe.weight == probe_dump['weight'], 'weight', probe_dump) + assert_damon_filters_committed(probe.filters, probe_dump['filters']) + def assert_monitoring_target_committed(target, dump): # target.pid is the pid "number", while dump['pid'] is 'struct pid' # pointer, and hence cannot be compared. @@ -196,6 +214,7 @@ def assert_ctx_committed(ctx, dump): } assert_true(dump['ops']['id'] == ops_val[ctx.ops], 'ops_id', dump) assert_monitoring_attrs_committed(ctx.monitoring_attrs, dump['attrs']) + assert_probes_committed(ctx.monitoring_attrs.probes, dump['probes']) assert_monitoring_targets_committed(ctx.targets, dump['adaptive_targets']) assert_schemes_committed(ctx.schemes, dump['schemes']) assert_true(dump['pause'] == ctx.pause, 'pause', dump) From 33ad8642064c37625e85fbb86ce7fab0ac1e95f9 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 07:03:12 -0700 Subject: [PATCH 0735/1352] selftests/damon/sysfs.py: test damon probes Extend sysfs.py to commit DAMON probes via sysfs, and see if it changed in-kernel DAMON status as expected using drgn. Link: https://lore.kernel.org/20260902140313.85983-7-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Shuah Khan --- tools/testing/selftests/damon/sysfs.py | 16 +++++++++++++++- 1 file changed, 15 insertions(+), 1 deletion(-) diff --git a/tools/testing/selftests/damon/sysfs.py b/tools/testing/selftests/damon/sysfs.py index 159cbeac067f89..66c826189320d7 100755 --- a/tools/testing/selftests/damon/sysfs.py +++ b/tools/testing/selftests/damon/sysfs.py @@ -319,7 +319,21 @@ def main(): intervals_goal=_damon_sysfs.IntervalsGoal( access_bp=400, aggrs=3, min_sample_us=5000, max_sample_us=10000000), - update_us=2000000), + update_us=2000000, + probes=_damon_sysfs.DamonProbes( + probes=[_damon_sysfs.DamonProbe( + weight=42, + filters=_damon_sysfs.DamonFilters( + filters=[ + _damon_sysfs.DamonFilter( + type_='anon', + matching=True, + allow=True, + ), + ]), + ), + ]), + ), schemes=[_damon_sysfs.Damos( action='pageout', access_pattern=_damon_sysfs.DamosAccessPattern( From 349558de7fe6ba28736498487a8e42b1ced7e828 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Sat, 26 Sep 2026 11:41:06 +0100 Subject: [PATCH 0736/1352] mm: move drivers/char/mem.c to mm/char-mem.c The memory character driver implements several mm-specific features and is always compiled into the kernel, so move it to mm/ where it belongs. Among other things the driver implements /dev/mem which provides raw access to physical memory, and /dev/zero which either allows mapping of a shmem region (if mapped with MAP_SHARED) or, uniquely, anonymous memory (if mapped MAP_PRIVATE). This change lays the foundations to allow MAP_PRIVATE-/dev/zero to be mapped precisely the same as anonymous memory is mapped as currently it is an edge case within mm. Also update a couple of comments that reference 'drivers/char/mem.c' to reference 'mm/char-mem.c'. Link: https://lore.kernel.org/20260926-map-private-dev-zero-v3-1-d4781e84ccfc@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: Arnd Bergmann Cc: Greg Kroah-Hartman Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Jann Horn Cc: Pedro Falcato Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Hugh Dickins Cc: Baolin Wang Cc: Matthew Wilcox (Oracle) Cc: Jan Kara --- MAINTAINERS | 4 ++-- drivers/char/Makefile | 2 +- mm/Makefile | 3 ++- drivers/char/mem.c => mm/char-mem.c | 2 +- mm/shmem.c | 2 +- 5 files changed, 7 insertions(+), 6 deletions(-) rename drivers/char/mem.c => mm/char-mem.c (99%) diff --git a/MAINTAINERS b/MAINTAINERS index 360977678f707e..3ff1a8f07e526f 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -17268,7 +17268,7 @@ F: Documentation/ABI/testing/sysfs-kernel-mm-cma F: Documentation/ABI/testing/sysfs-kernel-mm-numa F: Documentation/admin-guide/mm/ F: Documentation/mm/ -F: drivers/char/mem.c +F: mm/char-mem.c F: include/linux/cma.h F: include/linux/dmapool.h F: include/linux/ioremap.h @@ -17492,7 +17492,7 @@ L: linux-mm@kvack.org S: Maintained W: http://www.linux-mm.org T: git git://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm -F: drivers/char/mem.c +F: mm/char-mem.c F: include/trace/events/mmap.h F: fs/proc/task_mmu.c F: fs/proc/task_nommu.c diff --git a/drivers/char/Makefile b/drivers/char/Makefile index a46d7bf7c4c828..bb3bf66937b37b 100644 --- a/drivers/char/Makefile +++ b/drivers/char/Makefile @@ -3,7 +3,7 @@ # Makefile for the kernel character device drivers. # -obj-y += mem.o random.o +obj-y += random.o obj-$(CONFIG_TTY_PRINTK) += ttyprintk.o obj-y += misc.o obj-$(CONFIG_TEST_MISC_MINOR) += misc_minor_kunit.o diff --git a/mm/Makefile b/mm/Makefile index e7245cb88c6651..2a3ec53d62eef6 100644 --- a/mm/Makefile +++ b/mm/Makefile @@ -55,7 +55,8 @@ obj-y := filemap.o mempool.o oom_kill.o fadvise.o \ mm_init.o percpu.o slab_common.o \ compaction.o show_mem.o \ interval_tree.o list_lru.o workingset.o \ - debug.o gup.o mmap_lock.o vma_init.o $(mmu-y) + debug.o gup.o mmap_lock.o vma_init.o char-mem.o \ + $(mmu-y) # Give 'page_alloc' its own module-parameter namespace page-alloc-y := page_alloc.o diff --git a/drivers/char/mem.c b/mm/char-mem.c similarity index 99% rename from drivers/char/mem.c rename to mm/char-mem.c index 5b93c92c2cf194..3cc48054a66e71 100644 --- a/drivers/char/mem.c +++ b/mm/char-mem.c @@ -1,6 +1,6 @@ // SPDX-License-Identifier: GPL-2.0 /* - * linux/drivers/char/mem.c + * mm/char-mem.c * * Copyright (C) 1991, 1992 Linus Torvalds * diff --git a/mm/shmem.c b/mm/shmem.c index 84f0a2eb85fecd..05bc7c3aa52548 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -2984,7 +2984,7 @@ unsigned long shmem_get_unmapped_area(struct file *file, sb = file_inode(file)->i_sb; } else { /* - * Called directly from mm/mmap.c, or drivers/char/mem.c + * Called directly from mm/mmap.c, or mm/char-mem.c * for "/dev/zero", to create a shared anonymous object. */ if (IS_ERR(shm_mnt)) From 540a6238c652cfa2b2125389ca6cd48cfab8e4e5 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Sat, 26 Sep 2026 11:41:07 +0100 Subject: [PATCH 0737/1352] mm: implement file_is_dev_zero() to uniquely identify /dev/zero To lay the foundation for a future change that converts MAP_PRIVATE-/dev/zero mappings to be truly anonymous, add the ability to uniquely identify these mappings. With the memory character device now part of mm/ this is trivially achievable through a file_is_dev_zero() predicate that simply tests that the file operation hooks are zero_fops. Also update userland VMA tests to expose file_is_dev_zero() and provide a stub zero_fops for testing. Link: https://lore.kernel.org/20260926-map-private-dev-zero-v3-2-d4781e84ccfc@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Arnd Bergmann Cc: Greg Kroah-Hartman Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Jann Horn Cc: Pedro Falcato Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Hugh Dickins Cc: Baolin Wang Cc: Matthew Wilcox (Oracle) Cc: Jan Kara --- mm/char-mem.c | 13 +++++++++++++ mm/internal.h | 3 +++ tools/testing/vma/include/dup.h | 7 +++++++ tools/testing/vma/shared.c | 9 +++++++++ 4 files changed, 32 insertions(+) diff --git a/mm/char-mem.c b/mm/char-mem.c index 3cc48054a66e71..87fb71a011919f 100644 --- a/mm/char-mem.c +++ b/mm/char-mem.c @@ -31,6 +31,8 @@ #include #include +#include "internal.h" + #define DEVMEM_MINOR 1 #define DEVPORT_MINOR 4 @@ -707,6 +709,17 @@ static const struct memdev { #endif }; +/** + * file_is_dev_zero() - is the specified @file associated with the /dev/zero + * driver? + * @file: File to test. + * Returns: true if it is, false otherwise. + */ +bool file_is_dev_zero(const struct file *file) +{ + return file && file->f_op == &zero_fops; +} + static int memory_open(struct inode *inode, struct file *filp) { int minor; diff --git a/mm/internal.h b/mm/internal.h index e16f1250b25c80..5d474e5f77092a 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -1638,4 +1638,7 @@ static inline bool can_spin_trylock(void) return true; } +/* char-mem.c */ +bool file_is_dev_zero(const struct file *file); + #endif /* __MM_INTERNAL_H */ diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index 57046d8ac81d80..0d1a2ac8892261 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -1641,3 +1641,10 @@ static inline pgoff_t linear_anon_page_index(const struct vm_area_struct *vma, return pgoff; } + +extern const struct file_operations zero_fops; + +static inline bool file_is_dev_zero(const struct file *file) +{ + return file && file->f_op == &zero_fops; +} diff --git a/tools/testing/vma/shared.c b/tools/testing/vma/shared.c index 4a39c9d5048964..8c4826499f4054 100644 --- a/tools/testing/vma/shared.c +++ b/tools/testing/vma/shared.c @@ -12,6 +12,15 @@ const struct vm_operations_struct vma_dummy_vm_ops; struct anon_vma dummy_anon_vma; struct task_struct __current; +static int mmap_zero_prepare(struct vm_area_desc *desc) +{ + return 0; +} + +const struct file_operations zero_fops = { + .mmap_prepare = mmap_zero_prepare, +}; + struct vm_area_struct *alloc_vma(struct mm_struct *mm, unsigned long start, unsigned long end, pgoff_t pgoff, vma_flags_t vma_flags) From 44ecc4cba8ff4177f0ccc4b1c4297697c76ee07f Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Sat, 26 Sep 2026 11:41:08 +0100 Subject: [PATCH 0738/1352] mm/vma: only permit MAP_PRIVATE /dev/zero to be mapped anonymous In order to use mmap_prepare() with MAP_PRIVATE mappings of /dev/zero without the success_hook hack we explicitly permitted mmap_prepare handlers to set NULL vm_ops. However this is dangerous and we really only want to allow this for MAP_PRIVATE-mapped /dev/zero. Therefore use the newly introduced file_is_dev_zero() to uniquely identify MAP_PRIVATE-/dev/zero mappings and only permit this behaviour for them. Then, remove all ability for mmap_prepare or mmap hooks to set a VMA anonymous and update mmap_zero_prepare() to leave it to the core mmap code to do so. Note that this disallows nested MAP_PRIVATE-mappings of /dev/zero regions. Doing this would be broken in any case. We therefore do not need to update the mmap_prepare() compatibility layer to reflect these changes, as the mmap hook check suffices to disallow this behaviour. Now we're setting vma->vm_ops to NULL for an mmap_prepare-initialised MAP_PRIVATE-/dev/zero mapping, we have to avoid a subtle issue when updating user-defined fields via set_vma_user_defined_fields(). The default for vma->vm_ops for all mmap_prepare-initialised mappings is vma_dummy_vm_ops, so map->vm_ops will be set to this and setting vma->vm_ops to this will render the VMA mistakenly non-anon. In general, we should never be setting user-defined fields for an anonymous VMA, so explicitly check for this to avoid doing so for the one case where a mapping can be both mmap_prepare and anonymous. In the case of legacy ->mmap hooks some drivers may set vma->vm_ops NULL believing this is the equivalent of setting no VMA operations. Therefore update mmap_file() to correct this by setting dummy VMA operations if this occurs. An example of this is drm_gem_shmem_mmap() which deliberately clears vma->vm_ops before handing the VMA to dma-buf. Cases such as this will be updated when they are converted to mmap_prepare. Also, in order to avoid a single commit bisection hazard, add a temporary workaround to set the VMA anonymous only after vma->vm_file is assigned in __mmap_new_file_vma(). This is because vma_set_range() calls vma_set_pgoff() and assert_sane_pgoff() in turn, prior to the vma->vm_file being assigned. If we set the VMA anonymous early then this assert will fail. This is removed in the subsequent commit. Link: https://lore.kernel.org/20260926-map-private-dev-zero-v3-3-d4781e84ccfc@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Arnd Bergmann Cc: Greg Kroah-Hartman Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Jann Horn Cc: Pedro Falcato Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Hugh Dickins Cc: Baolin Wang Cc: Matthew Wilcox (Oracle) Cc: Jan Kara --- mm/char-mem.c | 6 +----- mm/internal.h | 17 ++++++++++------- mm/vma.c | 33 +++++++++++++++++++++++++-------- 3 files changed, 36 insertions(+), 20 deletions(-) diff --git a/mm/char-mem.c b/mm/char-mem.c index 87fb71a011919f..e9f94d2b5133ef 100644 --- a/mm/char-mem.c +++ b/mm/char-mem.c @@ -508,11 +508,7 @@ static int mmap_zero_prepare(struct vm_area_desc *desc) if (vma_desc_test(desc, VMA_MAYSHARE_BIT)) return shmem_zero_setup_desc(desc); - /* - * This is a highly unique situation where we mark a MAP_PRIVATE mapping - * of /dev/zero anonymous, despite it not being. - */ - vma_desc_set_anonymous(desc); + /* MAP_PRIVATE semantics are taken care of for us by core mm. */ return 0; } diff --git a/mm/internal.h b/mm/internal.h index 5d474e5f77092a..da14c56fb24e11 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -226,15 +226,18 @@ static inline int mmap_file(struct file *file, struct vm_area_struct *vma) { int err = vfs_mmap(file, vma); - if (likely(!err)) - return 0; - /* - * OK, we tried to call the file hook for mmap(), but an error - * arose. The mapping is in an inconsistent state and we must not invoke - * any further hooks on it. + * Either we tried to call the file hook for mmap() and an error arose + * or a driver set vma->vm_ops = NULL intending there to be no VMA + * operations. + * + * In the former case the VMA is in an inconsistent state and we mustn't + * invoke any further hooks on it, in the latter case the hook actually + * wanted no further hooks to be invoked, so fix both by setting dummy + * VMA ops. */ - vma->vm_ops = &vma_dummy_vm_ops; + if (unlikely(err || !vma->vm_ops)) + vma->vm_ops = &vma_dummy_vm_ops; return err; } diff --git a/mm/vma.c b/mm/vma.c index 6c35f0d775ab6e..61c04c063c5f56 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -2644,6 +2644,19 @@ static int __mmap_new_file_vma(struct mmap_state *map, return 0; } +static bool map_is_private(const struct mmap_state *map) +{ + return !vma_flags_test(&map->vma_flags, VMA_SHARED_BIT); +} + +static bool map_is_anon(const struct mmap_state *map) +{ + if (!map_is_private(map)) + return false; + + return !map->file || file_is_dev_zero(map->file); +} + /* * __mmap_new_vma() - Allocate a new VMA for the region, as merging was not * possible. @@ -2657,8 +2670,7 @@ static int __mmap_new_file_vma(struct mmap_state *map, static int __mmap_new_vma(struct mmap_state *map, struct vm_area_struct **vmap, struct mmap_action *action) { - const bool is_anon = !map->file && - !vma_flags_test(&map->vma_flags, VMA_SHARED_BIT); + const bool is_anon = map_is_anon(map); struct vma_iterator *vmi = map->vmi; int error = 0; struct vm_area_struct *vma; @@ -2674,7 +2686,7 @@ static int __mmap_new_vma(struct mmap_state *map, struct vm_area_struct **vmap, vma_iter_config(vmi, map->addr, map->end); - if (is_anon) + if (is_anon && !map->file) vma_set_anonymous(vma); vma_set_range(vma, map->addr, map->end, map->pgoff, map->anon_pgoff); @@ -2692,6 +2704,10 @@ static int __mmap_new_vma(struct mmap_state *map, struct vm_area_struct **vmap, else if (!is_anon) error = shmem_zero_setup(vma); + /* Temporary MAP_PRIVATE-/dev/zero workaround. */ + if (is_anon && map->file) + vma_set_anonymous(vma); + if (error) goto free_iter_vma; @@ -2800,6 +2816,10 @@ static int call_mmap_prepare(struct mmap_state *map, if (err) return err; + /* It's invalid for mmap_prepare hooks to clear vm_ops. */ + if (!desc->vm_ops) + return -EINVAL; + err = call_action_prepare(map, desc); if (err) return err; @@ -2822,10 +2842,7 @@ static int call_mmap_prepare(struct mmap_state *map, static void set_vma_user_defined_fields(struct vm_area_struct *vma, struct mmap_state *map) { - if (map->vm_ops) - vma->vm_ops = map->vm_ops; - else /* Only /dev/zero should do this. */ - vma_set_anonymous(vma); + vma->vm_ops = map->vm_ops; vma->vm_private_data = map->vm_private_data; } @@ -2907,7 +2924,7 @@ static unsigned long __mmap_region(struct file *file, unsigned long addr, allocated_new = true; } - if (have_mmap_prepare && allocated_new) + if (have_mmap_prepare && !map_is_anon(&map) && allocated_new) set_vma_user_defined_fields(vma, &map); __mmap_complete(&map, vma); From f3ca3ca28930f6cd188833dddf2919593646316c Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Sat, 26 Sep 2026 11:41:09 +0100 Subject: [PATCH 0739/1352] mm/vma: make MAP_PRIVATE-mapped /dev/zero mappings truly anonymous When mapping /dev/zero with MAP_PRIVATE, one ends up with strange VMAs originating from Linux's distant past. These have vma->vm_file set but NULL vma->vm_ops, meaning they satisfy vma_is_anonymous() but otherwise resemble a file-backed VMA. The introduction of anonymous page offsets and their subsequent use as indexes for MAP_PRIVATE-file-backed mappings mean the rmap does the right thing with these but we are left with inconsistencies. The vma_start_pgoff(vma) == vma_start_anon_pgoff(vma) invariant is true for all other anonymous VMAs, but not these. These VMAs are also observable as files in /proc//[maps, smaps, map_files] but otherwise behave like anonymous mappings. Therefore let's make these VMAs actually anonymous at mapping time which will activate the anonymous code path for mappings. This means we no longer have to account for this discrepancy anywhere and no longer have to think about these at all. This is user-observable, as MAP_PRIVATE-/dev/zero will no longer appear in procfs as a file-backed mapping, but the impact of this change should be low as likely nobody is relying upon this. However in any case, in using MAP_PRIVATE-/dev/zero they are explicitly asking anonymous memory, so no longer seeing these as file mappings is in fact correct. A previous commit gave us file_is_dev_zero() to positively identify these mappings, so we expressly only do so for these alone. Update assert_sane_pgoff(), the comment for vma_start_pgoff() and linear_anon_page_index() to reflect the change. We make this change in call_mmap_prepare() alone as /dev/zero has been converted to an mmap_prepare hook and we do not permit nested MAP_PRIVATE mapping of /dev/zero. We also remove the now defunct vma_desc_set_anonymous() and eliminate the temporary bisection hazard fix from the previous commit. Also update the VMA userland tests to reflect the change. Finally, update the procfs self tests proc-self-map-files-001 and proc-self-map-files-002 which both intend to map an arbitrary file MAP_PRIVATE then assert procfs state, but happen to choose /dev/zero. Fix them by updating these to /proc/self/exe which is guaranteed to be present if procfs is mounted. Link: https://lore.kernel.org/20260926-map-private-dev-zero-v3-4-d4781e84ccfc@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Arnd Bergmann Cc: Greg Kroah-Hartman Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Jann Horn Cc: Pedro Falcato Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Hugh Dickins Cc: Baolin Wang Cc: Matthew Wilcox (Oracle) Cc: Jan Kara --- include/linux/mm.h | 10 ++----- include/linux/pagemap.h | 3 +-- mm/vma.c | 26 ++++++++++++------- mm/vma.h | 3 --- .../selftests/proc/proc-self-map-files-001.c | 2 +- .../selftests/proc/proc-self-map-files-002.c | 2 +- tools/testing/vma/include/dup.h | 3 +-- 7 files changed, 23 insertions(+), 26 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index c105a3758915b2..c49ef99b4413b4 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -1529,11 +1529,6 @@ static inline void vma_set_anonymous(struct vm_area_struct *vma) vma->vm_ops = NULL; } -static inline void vma_desc_set_anonymous(struct vm_area_desc *desc) -{ - desc->vm_ops = NULL; -} - static inline bool vma_is_anonymous(const struct vm_area_struct *vma) { return !vma->vm_ops; @@ -4389,9 +4384,8 @@ static inline unsigned long vma_pages(const struct vm_area_struct *vma) * If @vma is a MAP_PRIVATE file-backed mapping, then this returns the * page offset within the file. * - * Edge cases: nommu does not abide by these, MAP_PRIVATE-/dev/zero satisfies - * vma_is_anonymous() but has file-backed page offset, and MAP_PRIVATE-pfnmap - * regions have their page offset set to the first PFN in the range. + * Edge cases: nommu does not abide by these and CoW MAP_PRIVATE-pfnmap regions + * have their page offset set to the first PFN in the range. * * Returns: The page offset of the start of @vma. */ diff --git a/include/linux/pagemap.h b/include/linux/pagemap.h index 0adfa6605653db..939f3a5e973f6b 100644 --- a/include/linux/pagemap.h +++ b/include/linux/pagemap.h @@ -1128,8 +1128,7 @@ static inline pgoff_t linear_anon_page_index(const struct vm_area_struct *vma, const pgoff_t pgoff = __linear_anon_page_index(vma, address); VM_WARN_ON_ONCE(!vma_is_cow_mapping(vma)); - /* Account for MAP_PRIVATE-/dev/zero which is only semi-anonymous. */ - if (vma_is_anonymous(vma) && !vma->vm_file) + if (vma_is_anonymous(vma)) VM_WARN_ON_ONCE(pgoff != linear_page_index(vma, address)); return pgoff; diff --git a/mm/vma.c b/mm/vma.c index 61c04c063c5f56..f56317ee824717 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -2644,6 +2644,13 @@ static int __mmap_new_file_vma(struct mmap_state *map, return 0; } +static void map_set_anon(struct mmap_state *map) +{ + map->file = NULL; + map->vm_ops = NULL; + map->pgoff = map->addr >> PAGE_SHIFT; +} + static bool map_is_private(const struct mmap_state *map) { return !vma_flags_test(&map->vma_flags, VMA_SHARED_BIT); @@ -2651,10 +2658,7 @@ static bool map_is_private(const struct mmap_state *map) static bool map_is_anon(const struct mmap_state *map) { - if (!map_is_private(map)) - return false; - - return !map->file || file_is_dev_zero(map->file); + return map_is_private(map) && !map->file; } /* @@ -2686,7 +2690,7 @@ static int __mmap_new_vma(struct mmap_state *map, struct vm_area_struct **vmap, vma_iter_config(vmi, map->addr, map->end); - if (is_anon && !map->file) + if (is_anon) vma_set_anonymous(vma); vma_set_range(vma, map->addr, map->end, map->pgoff, map->anon_pgoff); @@ -2704,10 +2708,6 @@ static int __mmap_new_vma(struct mmap_state *map, struct vm_area_struct **vmap, else if (!is_anon) error = shmem_zero_setup(vma); - /* Temporary MAP_PRIVATE-/dev/zero workaround. */ - if (is_anon && map->file) - vma_set_anonymous(vma); - if (error) goto free_iter_vma; @@ -2836,6 +2836,14 @@ static int call_mmap_prepare(struct mmap_state *map, map->vm_ops = desc->vm_ops; map->vm_private_data = desc->private_data; + /* + * MAP_PRIVATE-/dev/zero mappings are an ancient way of getting + * anonymous mappings. Rather than allowing these mappings to be odd + * outliers, simply make them truly anonymous. + */ + if (map_is_private(map) && file_is_dev_zero(map->file)) + map_set_anon(map); + return 0; } diff --git a/mm/vma.h b/mm/vma.h index 36973abaa015a8..f856d9ace3a6a6 100644 --- a/mm/vma.h +++ b/mm/vma.h @@ -267,9 +267,6 @@ static inline void assert_sane_pgoff(struct vm_area_struct *vma, pgoff_t pgoff) */ if (!vma_is_anonymous(vma)) return; - /* MAP_PRIVATE-/dev/zero is anon, non-NULL vm_file, but has file pgoff. */ - if (vma->vm_file) - return; /* If faulted in, could have been remapped. */ if (vma->anon_vma) return; diff --git a/tools/testing/selftests/proc/proc-self-map-files-001.c b/tools/testing/selftests/proc/proc-self-map-files-001.c index 4209c64283d6f6..bbca9f9e27439e 100644 --- a/tools/testing/selftests/proc/proc-self-map-files-001.c +++ b/tools/testing/selftests/proc/proc-self-map-files-001.c @@ -51,7 +51,7 @@ int main(void) int fd; unsigned long a, b; - fd = open("/dev/zero", O_RDONLY); + fd = open("/proc/self/exe", O_RDONLY); if (fd == -1) return 1; diff --git a/tools/testing/selftests/proc/proc-self-map-files-002.c b/tools/testing/selftests/proc/proc-self-map-files-002.c index e6aa00a183bcd9..5786cdffbbf636 100644 --- a/tools/testing/selftests/proc/proc-self-map-files-002.c +++ b/tools/testing/selftests/proc/proc-self-map-files-002.c @@ -57,7 +57,7 @@ int main(void) int fd; unsigned long a, b; - fd = open("/dev/zero", O_RDONLY); + fd = open("/proc/self/exe", O_RDONLY); if (fd == -1) return 1; diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index 0d1a2ac8892261..16c09dac59d9b4 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -1635,8 +1635,7 @@ static inline pgoff_t linear_anon_page_index(const struct vm_area_struct *vma, const pgoff_t pgoff = __linear_anon_page_index(vma, address); VM_WARN_ON_ONCE(!vma_is_cow_mapping(vma)); - /* Account for MAP_PRIVATE-/dev/zero which is only semi-anonymous. */ - if (vma_is_anonymous(vma) && !vma->vm_file) + if (vma_is_anonymous(vma)) VM_WARN_ON_ONCE(pgoff != linear_page_index(vma, address)); return pgoff; From b5532063495b0b5391017824018f7132b7bd14c0 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Sat, 26 Sep 2026 11:41:10 +0100 Subject: [PATCH 0740/1352] tools/testing/vma: add test to assert MAP_PRIVATE-/dev/zero is anon Now we've made MAP_PRIVATE-mapped /dev/zero mappings truly anonymous, add a VMA userland test to assert that this is the case and everything is as we would expect for an anonymous mapping. Link: https://lore.kernel.org/20260926-map-private-dev-zero-v3-5-d4781e84ccfc@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Arnd Bergmann Cc: Greg Kroah-Hartman Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Jann Horn Cc: Pedro Falcato Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Hugh Dickins Cc: Baolin Wang Cc: Matthew Wilcox (Oracle) Cc: Jan Kara --- tools/testing/vma/tests/mmap.c | 37 ++++++++++++++++++++++++++++++++++ 1 file changed, 37 insertions(+) diff --git a/tools/testing/vma/tests/mmap.c b/tools/testing/vma/tests/mmap.c index c85bc000d1cb7a..fa73faff226263 100644 --- a/tools/testing/vma/tests/mmap.c +++ b/tools/testing/vma/tests/mmap.c @@ -45,7 +45,44 @@ static bool test_mmap_region_basic(void) return true; } +static bool test_pure_anon_dev_zero(void) +{ + const vma_flags_t vma_flags = mk_vma_flags(VMA_READ_BIT, VMA_WRITE_BIT, + VMA_MAYREAD_BIT, VMA_MAYWRITE_BIT); + struct file file = { + .f_op = &zero_fops, + }; + struct mm_struct mm = {}; + struct vm_area_struct *vma; + unsigned long addr; + VMA_ITERATOR(vmi, &mm, 0); + + current->mm = &mm; + + /* + * Map a MAP_PRIVATE-/dev/zero mapping at address 0x300000 with a page + * offset of 0x10, which we expect to be reset to the anonymous page + * offset. + */ + addr = __mmap_region(&file, 0x300000, 0x3000, vma_flags, 0x10, NULL); + ASSERT_EQ(addr, 0x300000); + + /* Assert that it truly is an anonymous mapping. */ + vma = vma_lookup(&mm, addr); + ASSERT_NE(vma, NULL); + ASSERT_TRUE(vma_is_anonymous(vma)); + ASSERT_EQ(vma->vm_file, NULL); + ASSERT_EQ(vma->vm_private_data, NULL); + /* Expect anonymous page offsets. */ + ASSERT_EQ(vma->vm_pgoff, 0x300); + ASSERT_EQ(vma_start_anon_pgoff(vma), 0x300); + + cleanup_mm(&mm, &vmi); + return true; +} + static void run_mmap_tests(int *num_tests, int *num_fail) { TEST(mmap_region_basic); + TEST(pure_anon_dev_zero); } From 7d7b461e0dc5b07bd295f04faf002a8ee65d486c Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Sat, 26 Sep 2026 11:41:11 +0100 Subject: [PATCH 0741/1352] tools/testing/selftests/mm: add MAP_PRIVATE-/dev/zero merge tests Assert that MAP_PRIVATE-mapped /dev/zero mappings behave like they are anonymous. Test both unfaulted and faulted/unfaulted merges with page offset 0 which would not merge if the mappings were treated as if they were file-backed. With the recent change that makes them behave as pure anonymous mappings, the merges should succeed as their page offsets are equal to their anonymous page offsets. Link: https://lore.kernel.org/20260926-map-private-dev-zero-v3-6-d4781e84ccfc@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Arnd Bergmann Cc: Greg Kroah-Hartman Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Jann Horn Cc: Pedro Falcato Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Hugh Dickins Cc: Baolin Wang Cc: Matthew Wilcox (Oracle) Cc: Jan Kara --- tools/testing/selftests/mm/merge.c | 95 ++++++++++++++++++++++++++++++ 1 file changed, 95 insertions(+) diff --git a/tools/testing/selftests/mm/merge.c b/tools/testing/selftests/mm/merge.c index 52b8727b6628e0..7dd2933417c1d6 100644 --- a/tools/testing/selftests/mm/merge.c +++ b/tools/testing/selftests/mm/merge.c @@ -1362,6 +1362,101 @@ TEST_F(merge, anon_and_page_offset_mismatch_memfd) ASSERT_EQ(procmap->query.vma_end, (unsigned long)ptr + 5 * page_size); } +TEST_F(merge, merge_map_private_dev_zero_unfaulted) +{ + struct procmap_fd *procmap = &self->procmap; + unsigned int page_size = self->page_size; + char *carveout = self->carveout; + char *ptr, *ptr2; + int fd_zero; + + if (access("/dev/zero", F_OK)) + SKIP(return, "No /dev/zero."); + fd_zero = open("/dev/zero", O_RDWR); + ASSERT_NE(fd_zero, -1); + + /* + * Map two MAP_PRIVATE-/dev/zero VMAs next to one another with offset 0 + * each. + * + * With these being made truly anonymous upon mapping, they will + * merge. If they were file-backed VMAs the page offsets would prevent + * the merge: + * + * |-----||------| |-------------| + * | ptr || ptr2 | -> | ptr | + * |-----||------| |-------------| + */ + ptr = mmap(carveout, 5 * page_size, PROT_READ | PROT_WRITE, + MAP_FIXED | MAP_PRIVATE, fd_zero, 0); + ptr2 = mmap(&carveout[5 * page_size], 5 * page_size, + PROT_READ | PROT_WRITE, MAP_FIXED | MAP_PRIVATE, fd_zero, 0); + close(fd_zero); + ASSERT_NE(ptr, MAP_FAILED); + ASSERT_NE(ptr2, MAP_FAILED); + + /* Assert that they merged. */ + ASSERT_TRUE(find_vma_procmap(procmap, ptr)); + ASSERT_EQ(procmap->query.vma_start, (unsigned long)ptr); + ASSERT_EQ(procmap->query.vma_end, (unsigned long)ptr + 10 * page_size); +} + +TEST_F(merge, merge_map_private_dev_zero_faulted_unfaulted) +{ + struct procmap_fd *procmap = &self->procmap; + unsigned int page_size = self->page_size; + char *carveout = self->carveout; + char *ptr, *ptr2; + int fd_zero; + + if (access("/dev/zero", F_OK)) + SKIP(return, "No /dev/zero."); + fd_zero = open("/dev/zero", O_RDWR); + ASSERT_NE(fd_zero, -1); + + /* + * Map a MAP_PRIVATE mapping of /dev/zero with page offset 0, then fault + * it in: + * + * |-------------------------------| + * | faulted | + * |-------------------------------| + */ + ptr = mmap(carveout, 15 * page_size, PROT_READ | PROT_WRITE, + MAP_FIXED | MAP_PRIVATE, fd_zero, 0); + ASSERT_NE(ptr, MAP_FAILED); + memset(ptr, 'x', 15 * page_size); + + /* + * Unmap the middle: + * + * |---------| |---------| + * | faulted | | faulted | + * |---------| |---------| + */ + ASSERT_EQ(munmap(&ptr[5 * page_size], 5 * page_size), 0); + + /* + * Map in a new unfaulted mapping in the middle with page offset 0 - + * this should merge and would not if it were treated as a file rather + * than pure anon: + * + * |---------|-----------|---------| + * | faulted | unfaulted | faulted | + * |---------|-----------|---------| + */ + ptr2 = mmap(&carveout[5 * page_size], 5 * page_size, + PROT_READ | PROT_WRITE, MAP_FIXED | MAP_PRIVATE, + fd_zero, 0); + close(fd_zero); + ASSERT_NE(ptr2, MAP_FAILED); + + /* Assert that they merged. */ + ASSERT_TRUE(find_vma_procmap(procmap, ptr)); + ASSERT_EQ(procmap->query.vma_start, (unsigned long)ptr); + ASSERT_EQ(procmap->query.vma_end, (unsigned long)ptr + 15 * page_size); +} + TEST_F(merge_with_fork, mremap_faulted_to_unfaulted_prev) { struct procmap_fd *procmap = &self->procmap; From 6a903030a384ef2537de8291f2ceeabeba719ffe Mon Sep 17 00:00:00 2001 From: Kefeng Wang Date: Wed, 2 Sep 2026 21:16:50 +0800 Subject: [PATCH 0742/1352] xfs: remove dead kswapd flag inheritance from btree split worker Patch series "mm: replace PF_KCOMPACTD/PF_KSWAPD with kthread_func()". The task_struct->flags field is a limited 32-bit resource. Two flag bits, PF_KCOMPACTD and PF_KSWAPD, are used to identify whether a kernel thread is kcompactd or kswapd. Converting these to use kthread_func() frees up two valuable flag bits for future use. This patch (of 4): Commit 1f6d64829db7 ("xfs: block allocation work needs to be kswapd aware") added PF_MEMALLOC | PF_KSWAPD inheritance to xfs_btree_split_worker() so that block allocation offloaded from kswapd to a workqueue thread could access emergency memory reserves and avoid reclaim throttling. pageout() no longer calls ->writepage() for filesystem folios -- it returns PAGE_ACTIVATE for non-shmem, non-anon pages. kswapd therefore never enters XFS writeback and cannot reach btree split. The only path that offloads btree splits is unwritten extent conversion at IO completion (xfs_end_io -> xfs_iomap_write_unwritten), which runs in a workqueue context where current_is_kswapd() is always false. Let's remove the dead kswapd flag and related codes. Link: https://lore.kernel.org/20260902131653.1338227-1-wangkefeng.wang@huawei.com Link: https://lore.kernel.org/20260902131653.1338227-2-wangkefeng.wang@huawei.com Signed-off-by: Kefeng Wang Signed-off-by: Andrew Morton Reviewed-by: Christoph Hellwig Reviewed-by: Shakeel Butt Cc: Brendan Jackman Cc: Carlos Maiolino Cc: Christian Brauner Cc: "Darrick J. Wong" Cc: David Hildenbrand Cc: Johannes Weiner Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Zi Yan --- fs/xfs/libxfs/xfs_btree.c | 18 +----------------- fs/xfs/xfs_platform.h | 4 ---- 2 files changed, 1 insertion(+), 21 deletions(-) diff --git a/fs/xfs/libxfs/xfs_btree.c b/fs/xfs/libxfs/xfs_btree.c index 60ef7f08b1d300..6738d9d1511bc7 100644 --- a/fs/xfs/libxfs/xfs_btree.c +++ b/fs/xfs/libxfs/xfs_btree.c @@ -2994,7 +2994,6 @@ struct xfs_btree_split_args { struct xfs_btree_cur **curp; int *stat; /* success/failure */ int result; - bool kswapd; /* allocation in kswapd context */ struct completion *done; struct work_struct work; }; @@ -3008,33 +3007,18 @@ xfs_btree_split_worker( { struct xfs_btree_split_args *args = container_of(work, struct xfs_btree_split_args, work); - unsigned long pflags; - unsigned long new_pflags = 0; - - /* - * we are in a transaction context here, but may also be doing work - * in kswapd context, and hence we may need to inherit that state - * temporarily to ensure that we don't block waiting for memory reclaim - * in any way. - */ - if (args->kswapd) - new_pflags |= PF_MEMALLOC | PF_KSWAPD; - - current_set_flags_nested(&pflags, new_pflags); xfs_trans_set_context(args->cur->bc_tp); args->result = __xfs_btree_split(args->cur, args->level, args->ptrp, args->key, args->curp, args->stat); xfs_trans_clear_context(args->cur->bc_tp); - current_restore_flags_nested(&pflags, new_pflags); /* * Do not access args after complete() has run here. We don't own args * and the owner may run and free args before we return here. */ complete(args->done); - } /* @@ -3078,7 +3062,7 @@ xfs_btree_split( args.curp = curp; args.stat = stat; args.done = &done; - args.kswapd = current_is_kswapd(); + INIT_WORK_ONSTACK(&args.work, xfs_btree_split_worker); queue_work(xfs_alloc_wq, &args.work); wait_for_completion(&done); diff --git a/fs/xfs/xfs_platform.h b/fs/xfs/xfs_platform.h index 745d715b4c646b..e961e36c53fce7 100644 --- a/fs/xfs/xfs_platform.h +++ b/fs/xfs/xfs_platform.h @@ -115,10 +115,6 @@ typedef __u32 xfs_nlink_t; #define xfs_blockgc_secs xfs_params.blockgc_timer.val #define current_cpu() (raw_smp_processor_id()) -#define current_set_flags_nested(sp, f) \ - (*(sp) = current->flags, current->flags |= (f)) -#define current_restore_flags_nested(sp, f) \ - (current->flags = ((current->flags & ~(f)) | (*(sp) & (f)))) #define NBBY 8 /* number of bits per byte */ From 86ba7ad9929b3b4006cc43643b9405488b764524 Mon Sep 17 00:00:00 2001 From: Kefeng Wang Date: Wed, 2 Sep 2026 21:16:51 +0800 Subject: [PATCH 0743/1352] iomap: simplify writepages reclaim guard Now that kswapd can no longer reach filesystem writeback (pageout() returns PAGE_ACTIVATE for non-shmem, non-anon folios), any PF_MEMALLOC context in iomap_writepages() indicates a VM regression, simplify the check by removing PF_KSWAPD. Link: https://lore.kernel.org/20260902131653.1338227-3-wangkefeng.wang@huawei.com Signed-off-by: Kefeng Wang Signed-off-by: Andrew Morton Reviewed-by: Shakeel Butt Cc: Brendan Jackman Cc: Carlos Maiolino Cc: Christian Brauner Cc: Christoph Hellwig Cc: "Darrick J. Wong" Cc: David Hildenbrand Cc: Johannes Weiner Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Suren Baghdasaryan Cc: Vlastimil Babka (SUSE) Cc: Zi Yan --- fs/iomap/buffered-io.c | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/fs/iomap/buffered-io.c b/fs/iomap/buffered-io.c index 0a5ebfda90f12e..6306ca747f3ba5 100644 --- a/fs/iomap/buffered-io.c +++ b/fs/iomap/buffered-io.c @@ -2077,8 +2077,7 @@ iomap_writepages(struct iomap_writepage_ctx *wpc) * Writeback from reclaim context should never happen except in the case * of a VM regression so warn about it and refuse to write the data. */ - if (WARN_ON_ONCE((current->flags & (PF_MEMALLOC | PF_KSWAPD)) == - PF_MEMALLOC)) + if (WARN_ON_ONCE((current->flags & PF_MEMALLOC))) return -EIO; while ((folio = writeback_iter(mapping, wpc->wbc, folio, &error))) { From 011ee9ff937500e89ac81aa478ef05653e4197c7 Mon Sep 17 00:00:00 2001 From: Kefeng Wang Date: Wed, 2 Sep 2026 21:16:52 +0800 Subject: [PATCH 0744/1352] mm: replace PF_KSWAPD flag with kthread_func() check The preceding commits removed the last consumer that propagated PF_KSWAPD beyond kswapd itself (XFS btree split worker inheritance). The only remaining setter of PF_KSWAPD is kswapd(), and every current_is_kswapd() caller only needs to check whether the current task *is* the kswapd thread, not whether it inherited the flag. Replace the flag-based test with kthread_func(current) == kswapd, freeing the 0x00020000 PF flag bit. Link: https://lore.kernel.org/20260902131653.1338227-4-wangkefeng.wang@huawei.com Signed-off-by: Kefeng Wang Signed-off-by: Andrew Morton Acked-by: Shakeel Butt Acked-by: Vlastimil Babka (SUSE) Acked-by: Zi Yan Cc: Brendan Jackman Cc: Carlos Maiolino Cc: Christian Brauner Cc: Christoph Hellwig Cc: "Darrick J. Wong" Cc: David Hildenbrand Cc: Johannes Weiner Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Suren Baghdasaryan --- include/linux/sched.h | 2 +- include/linux/swap.h | 7 +------ mm/vmscan.c | 10 ++++++++-- tools/sched_ext/include/scx/common.bpf.h | 1 - 4 files changed, 10 insertions(+), 10 deletions(-) diff --git a/include/linux/sched.h b/include/linux/sched.h index d35ae49a991f7d..90393bd53bbc67 100644 --- a/include/linux/sched.h +++ b/include/linux/sched.h @@ -1810,7 +1810,7 @@ extern struct pid __rcu *cad_pid; #define PF_USER_WORKER 0x00004000 /* Kernel thread cloned from userspace thread */ #define PF_NOFREEZE 0x00008000 /* This thread should not be frozen */ #define PF_KCOMPACTD 0x00010000 /* I am kcompactd */ -#define PF_KSWAPD 0x00020000 /* I am kswapd */ +#define PF__HOLE__00020000 0x00020000 #define PF_MEMALLOC_NOFS 0x00040000 /* All allocations inherit GFP_NOFS. See memalloc_nfs_save() */ #define PF_MEMALLOC_NOIO 0x00080000 /* All allocations inherit GFP_NOIO. See memalloc_noio_save() */ #define PF_LOCAL_THROTTLE 0x00100000 /* Throttle writes only against the bdi I write to, diff --git a/include/linux/swap.h b/include/linux/swap.h index 7a43409879caed..74ce794042474c 100644 --- a/include/linux/swap.h +++ b/include/linux/swap.h @@ -25,12 +25,6 @@ #define SWAP_FLAGS_VALID (SWAP_FLAG_PRIO_MASK | SWAP_FLAG_PREFER | \ SWAP_FLAG_DISCARD | SWAP_FLAG_DISCARD_ONCE | \ SWAP_FLAG_DISCARD_PAGES) - -static inline int current_is_kswapd(void) -{ - return current->flags & PF_KSWAPD; -} - /* * MAX_SWAPFILES defines the maximum number of swaptypes: things which can * be swapped to. The swap type and the offset into that swap type are @@ -339,6 +333,7 @@ void check_move_unevictable_folios(struct folio_batch *fbatch); extern void __meminit kswapd_run(int nid); extern void __meminit kswapd_stop(int nid); +bool current_is_kswapd(void); #ifdef CONFIG_SWAP int add_swap_extent(struct swap_info_struct *sis, unsigned long start_page, diff --git a/mm/vmscan.c b/mm/vmscan.c index ba7adf36e69f7b..245f68c75b2894 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -7539,7 +7539,7 @@ static int kswapd(void *p) * us from recursively trying to free more memory as we're * trying to free the first piece of memory in the first place). */ - tsk->flags |= PF_MEMALLOC | PF_KSWAPD; + tsk->flags |= PF_MEMALLOC; set_freezable(); WRITE_ONCE(pgdat->kswapd_order, 0); @@ -7589,11 +7589,17 @@ static int kswapd(void *p) goto kswapd_try_sleep; } - tsk->flags &= ~(PF_MEMALLOC | PF_KSWAPD); + tsk->flags &= ~PF_MEMALLOC; return 0; } +bool current_is_kswapd(void) +{ + return kthread_func(current) == kswapd; +} +EXPORT_SYMBOL_GPL(current_is_kswapd); + /* * A zone is low on free memory or too fragmented for high-order memory. If * kswapd should reclaim (direct reclaim is deferred), wake it up for the zone's diff --git a/tools/sched_ext/include/scx/common.bpf.h b/tools/sched_ext/include/scx/common.bpf.h index 22f24ebef8a9ae..5a28a36c5120d6 100644 --- a/tools/sched_ext/include/scx/common.bpf.h +++ b/tools/sched_ext/include/scx/common.bpf.h @@ -32,7 +32,6 @@ #define PF_IO_WORKER 0x00000010 /* Task is an IO worker */ #define PF_WQ_WORKER 0x00000020 /* I'm a workqueue worker */ #define PF_KCOMPACTD 0x00010000 /* I am kcompactd */ -#define PF_KSWAPD 0x00020000 /* I am kswapd */ #define PF_KTHREAD 0x00200000 /* I am a kernel thread */ #define PF_EXITING 0x00000004 #define CLOCK_MONOTONIC 1 From a981524f52889aabe32be84d48fa98937fabdc0c Mon Sep 17 00:00:00 2001 From: Kefeng Wang Date: Wed, 2 Sep 2026 21:16:53 +0800 Subject: [PATCH 0745/1352] mm: replace PF_KCOMPACTD flag with kthread_func() check PF_KCOMPACTD was introduced by commit ce6d9c1c2b5c ("NFS: fix nfs_release_folio() to not deadlock via kcompactd writeback") so nfs_release_folio() could detect kcompactd context and skip writeback. The flag is only consumed by current_is_kcompactd(), whose sole caller is nfs_release_folio(). Replace the flag-based check with kthread_func(current) == kcompactd, freeing the 0x00010000 PF flag bit. Link: https://lore.kernel.org/20260902131653.1338227-5-wangkefeng.wang@huawei.com Signed-off-by: Kefeng Wang Signed-off-by: Andrew Morton Acked-by: Shakeel Butt Acked-by: Zi Yan Acked-by: Vlastimil Babka (SUSE) Reviewed-by: David Hildenbrand (Arm) Cc: Brendan Jackman Cc: Carlos Maiolino Cc: Christian Brauner Cc: Christoph Hellwig Cc: "Darrick J. Wong" Cc: Johannes Weiner Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Suren Baghdasaryan --- include/linux/compaction.h | 11 ++++++----- include/linux/sched.h | 2 +- mm/compaction.c | 9 ++++++--- tools/sched_ext/include/scx/common.bpf.h | 1 - 4 files changed, 13 insertions(+), 10 deletions(-) diff --git a/include/linux/compaction.h b/include/linux/compaction.h index 66a2f70e9e019d..691c09f0a6971d 100644 --- a/include/linux/compaction.h +++ b/include/linux/compaction.h @@ -81,10 +81,6 @@ static inline unsigned long compact_gap(unsigned int order) return min(2UL << order, COMPACT_CLUSTER_MAX); } -static inline int current_is_kcompactd(void) -{ - return current->flags & PF_KCOMPACTD; -} #ifdef CONFIG_COMPACTION @@ -103,7 +99,7 @@ extern void compaction_defer_reset(struct zone *zone, int order, bool compaction_zonelist_suitable(struct alloc_context *ac, int order, int alloc_flags, gfp_t gfp_mask); - +bool current_is_kcompactd(void); extern void __meminit kcompactd_run(int nid); extern void __meminit kcompactd_stop(int nid); extern void wakeup_kcompactd(pg_data_t *pgdat, int order, int highest_zoneidx); @@ -120,6 +116,11 @@ static inline bool compaction_suitable(struct zone *zone, int order, return false; } +static inline bool current_is_kcompactd(void) +{ + return false; +} + static inline void kcompactd_run(int nid) { } diff --git a/include/linux/sched.h b/include/linux/sched.h index 90393bd53bbc67..f45b7d43113ca9 100644 --- a/include/linux/sched.h +++ b/include/linux/sched.h @@ -1809,7 +1809,7 @@ extern struct pid __rcu *cad_pid; #define PF_USED_MATH 0x00002000 /* If unset the fpu must be initialized before use */ #define PF_USER_WORKER 0x00004000 /* Kernel thread cloned from userspace thread */ #define PF_NOFREEZE 0x00008000 /* This thread should not be frozen */ -#define PF_KCOMPACTD 0x00010000 /* I am kcompactd */ +#define PF__HOLE__00010000 0x00010000 #define PF__HOLE__00020000 0x00020000 #define PF_MEMALLOC_NOFS 0x00040000 /* All allocations inherit GFP_NOFS. See memalloc_nfs_save() */ #define PF_MEMALLOC_NOIO 0x00080000 /* All allocations inherit GFP_NOIO. See memalloc_noio_save() */ diff --git a/mm/compaction.c b/mm/compaction.c index a049415512c672..4994e200bbecd7 100644 --- a/mm/compaction.c +++ b/mm/compaction.c @@ -3197,7 +3197,6 @@ static int kcompactd(void *p) long default_timeout = msecs_to_jiffies(HPAGE_FRAG_CHECK_INTERVAL_MSEC); long timeout = default_timeout; - current->flags |= PF_KCOMPACTD; set_freezable(); pgdat->kcompactd_max_order = 0; @@ -3254,11 +3253,15 @@ static int kcompactd(void *p) pgdat->proactive_compact_trigger = false; } - current->flags &= ~PF_KCOMPACTD; - return 0; } +bool current_is_kcompactd(void) +{ + return kthread_func(current) == kcompactd; +} +EXPORT_SYMBOL_GPL(current_is_kcompactd); + /* * This kcompactd start function will be called by init and node-hot-add. * On node-hot-add, kcompactd will moved to proper cpus if cpus are hot-added. diff --git a/tools/sched_ext/include/scx/common.bpf.h b/tools/sched_ext/include/scx/common.bpf.h index 5a28a36c5120d6..ab8d676d8cd63f 100644 --- a/tools/sched_ext/include/scx/common.bpf.h +++ b/tools/sched_ext/include/scx/common.bpf.h @@ -31,7 +31,6 @@ #define PF_IDLE 0x00000002 /* I am an IDLE thread */ #define PF_IO_WORKER 0x00000010 /* Task is an IO worker */ #define PF_WQ_WORKER 0x00000020 /* I'm a workqueue worker */ -#define PF_KCOMPACTD 0x00010000 /* I am kcompactd */ #define PF_KTHREAD 0x00200000 /* I am a kernel thread */ #define PF_EXITING 0x00000004 #define CLOCK_MONOTONIC 1 From 571671c5913460de0bd73af4110c3502375f048f Mon Sep 17 00:00:00 2001 From: Ackerley Tng Date: Wed, 9 Sep 2026 10:11:00 -0700 Subject: [PATCH 0746/1352] mm: hugetlb: return -ENOSPC on memcg charge failure Patch series "Fix bugs in HugeTLB allocation when mem_cgroup_charge_hugetlb() fails", v2. In hugetlb_alloc_folio(), when mem_cgroup_charge_hugetlb() fails, there are 2 issues: 1. free_huge_folio() expects a non-refcounted folio and will VM_BUG_ON_FOLIO(). 2. -ENOMEM is returned, causing an infinite loop retrying the fault. This patch (of 2): When mem_cgroup_charge_hugetlb() fails with -ENOMEM, alloc_hugetlb_folio() currently propagates this error. This results in the page fault handler returning VM_FAULT_OOM. Because HugeTLB allocations are high-order and use __GFP_RETRY_MAYFAIL, they bypass the OOM killer. Returning VM_FAULT_OOM to the #PF handler without triggering the OOM killer (or having it make progress) leads to an infinite loop of retrying the fault. Avoid this loop by returning -ENOSPC when charging fails, which maps to VM_FAULT_SIGBUS, terminating the process cleanly. Make mem_cgroup_charge_hugetlb() fault handling use a common error handling path, the same handling used for hugetlb_cgroup_uncharge_cgroup{,_rsvd}(), which also don't trigger the OOM killer and hence opt to terminate the process with a SIGBUS. Link: https://lore.kernel.org/20260909-hugetlb-alloc-folio-memcg-charge-error-handling-v2-1-4b4a8a19a7f7@google.com Fixes: 991135774c0e ("memcg/hugetlb: introduce mem_cgroup_charge_hugetlb") Signed-off-by: Ackerley Tng Signed-off-by: Andrew Morton Reviewed-by: Muchun Song Reviewed-by: Joshua Hahn Cc: Alex Shi Cc: David Hildenbrand Cc: David Rientjes Cc: Dongliang Mu Cc: Frank van der Linden Cc: Hongxiang Lou Cc: James Houghton Cc: Johannes Weiner Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Ma Wupeng Cc: Miaohe Lin Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Oscar Salvador Cc: Peter Xu Cc: Roman Gushchin Cc: Shakeel Butt Cc: Suren Baghdasaryan Cc: Vishal Annapurve Cc: Vlastimil Babka Cc: Yanteng Si Cc: --- mm/hugetlb.c | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 7edc2a860a4007..dfee81e955fcc7 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -2851,7 +2851,6 @@ void wait_for_freed_hugetlb_folios(void) * * Return: A pointer to the allocated folio, or an ERR_PTR on failure. * -ENOSPC if cgroup charging fails or no folio is available. - * -ENOMEM if mem cgroup charging fails. */ struct folio *hugetlb_alloc_folio(struct hstate *h, struct mempolicy_interpreted *mpoli, u8 alloc_flags) @@ -2924,7 +2923,11 @@ struct folio *hugetlb_alloc_folio(struct hstate *h, * were committed to the folio and freeing the folio * would have cleared those up. */ - return ERR_PTR(ret); + /* + * Return -ENOSPC, since retrying the fault is futile: + * the OOM killer is not triggered for HugeTLB. + */ + return ERR_PTR(-ENOSPC); } return folio; From bcb55b0704c6d20de8d515bf9e19ee72170609f1 Mon Sep 17 00:00:00 2001 From: Ackerley Tng Date: Wed, 9 Sep 2026 10:11:01 -0700 Subject: [PATCH 0747/1352] mm: hugetlb: drop refcount before freeing on memcg charge failure When mem_cgroup_charge_hugetlb(folio, gfp) returns -ENOMEM, the folio has its refcount set to 1 via folio_ref_unfreeze(folio, 1). The error path calls free_huge_folio(folio) directly, which expects a refcount of 0. Hence, VM_BUG_ON_FOLIO(folio_ref_count(folio), folio) is triggered. Even with CONFIG_DEBUG_VM disabled, returning a folio with refcount 1 to the freelist can corrupt allocator state later. Use folio_put(folio) instead of free_huge_folio(folio) to properly drop the reference before freeing it. Link: https://lore.kernel.org/20260909-hugetlb-alloc-folio-memcg-charge-error-handling-v2-2-4b4a8a19a7f7@google.com Fixes: 991135774c0e ("memcg/hugetlb: introduce mem_cgroup_charge_hugetlb") Signed-off-by: Ackerley Tng Signed-off-by: Andrew Morton Reviewed-by: Muchun Song Reviewed-by: Joshua Hahn Cc: Alex Shi Cc: David Hildenbrand Cc: David Rientjes Cc: Dongliang Mu Cc: Frank van der Linden Cc: Hongxiang Lou Cc: James Houghton Cc: Johannes Weiner Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Ma Wupeng Cc: Miaohe Lin Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Oscar Salvador Cc: Peter Xu Cc: Roman Gushchin Cc: Shakeel Butt Cc: Suren Baghdasaryan Cc: Vishal Annapurve Cc: Vlastimil Babka Cc: Yanteng Si Cc: --- mm/hugetlb.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index dfee81e955fcc7..4d7c2ee126efda 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -2917,7 +2917,7 @@ struct folio *hugetlb_alloc_folio(struct hstate *h, lruvec_stat_mod_folio(folio, NR_HUGETLB, nr_pages); if (ret == -ENOMEM) { - free_huge_folio(folio); + folio_put(folio); /* * Skip uncharging hugetlb_cgroup since the charges * were committed to the folio and freeing the folio From c8ce137417a242f17239a76c8fd978544269214c Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Wed, 2 Sep 2026 11:10:11 +0300 Subject: [PATCH 0748/1352] docs/core-api: memory-allocation: add k[mz]alloc_obj() and clarify kmalloc Patch series "docs/core-api: memory-allocation: add k[mz]alloc_obj() and clarify kmalloc", v2. Update the memory-allocation guide to describe k[mz]alloc_obj() and clarify description of kmalloc() size limitations. And when I realized that get_maintainers.pl does not list any of mm people for that patch I added the MAINTAINERS update as well :) This patch (of 2): Since v7.0 the most used memory allocation function is kzalloc_obj(). Update the memory-allocation guide to describe k[mz]alloc_obj() family and make kzalloc_obj() the first answer to "How should I allocate memory?" question. While on it, clarify description of kmalloc() size limitations. Link: https://lore.kernel.org/20260902-docs-memalloc-guide-v2-0-218c1a4dcb80@kernel.org Link: https://lore.kernel.org/20260902-docs-memalloc-guide-v2-1-218c1a4dcb80@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Reviewed-by: Suren Baghdasaryan Acked-by: SJ Park Acked-by: Vlastimil Babka (SUSE) Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Randy Dunlap --- Documentation/core-api/memory-allocation.rst | 30 +++++++++++++++----- 1 file changed, 23 insertions(+), 7 deletions(-) diff --git a/Documentation/core-api/memory-allocation.rst b/Documentation/core-api/memory-allocation.rst index 0f19dd52432394..823f7fa57429b7 100644 --- a/Documentation/core-api/memory-allocation.rst +++ b/Documentation/core-api/memory-allocation.rst @@ -19,6 +19,12 @@ Diversity of the allocation APIs combined with the numerous GFP flags makes the question "How should I allocate memory?" not that easy to answer, although very likely you should use +:: + + kzalloc_obj(); + +or + :: kzalloc(, GFP_KERNEL); @@ -139,10 +145,13 @@ allocate memory for an array, there are kmalloc_array() and kcalloc() helpers. The helpers struct_size(), array_size() and array3_size() can be used to safely calculate object sizes without overflowing. -The maximal size of a chunk that can be allocated with `kmalloc` is -limited. The actual limit depends on the hardware and the kernel -configuration, but it is a good practice to use `kmalloc` for objects -smaller than page size. +Since 7.0 there are type aware kmalloc-family helpers that let you safely and +conveniently allocate a single object or arrays of objects with kzalloc_obj() +and kmalloc_obj() and their array versions kzalloc_objs() and +kmalloc_objs(). These helpers only need the type of the object that should be +allocated and the count of elements in the array for the array versions. + +As of v7.2, vast majority of the memory allocations use kzalloc_obj(). The address of a chunk allocated with `kmalloc` is aligned to at least ARCH_KMALLOC_MINALIGN bytes. For sizes which are a power of two, the @@ -154,9 +163,16 @@ Chunks allocated with kmalloc() can be resized with krealloc(). Similarly to kmalloc_array(): a helper for resizing arrays is provided in the form of krealloc_array(). -For large allocations you can use vmalloc() and vzalloc(), or directly -request pages from the page allocator. The memory allocated by `vmalloc` -and related functions is not physically contiguous. +`kmalloc` always allocates physically contiguous memory and the maximal size of +a chunk that can be allocated with `kmalloc` is limited by `KMALLOC_MAX_SIZE`, +which matches the page allocator's MAX_PAGE_ORDER limit. + +Internally, the slab allocator differentiates allocations of different orders +and delegates larger allocations to the page allocator, but for the users of +`kmalloc` family it is entirely transparent. + +For large allocations that do not require physically contiguous memory you can +use vmalloc() and vzalloc() family. If you are not sure whether the allocation size is too large for `kmalloc`, it is possible to use kvmalloc() and its derivatives. It will From 6fa9cf435e6319ac057609070968fc7bd60ad815 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Wed, 2 Sep 2026 11:10:12 +0300 Subject: [PATCH 0749/1352] MAINTAINERS: add memory related docs in core-mm/ to MM - MISC section Previous efforts to make sure that mm files are properly listed in MAINTAINERS missed Documentation/core-api/mm-api.rst Documentation/core-api/memory-allocation.rst Add them to "MEMORY MANAGEMENT - MISC" Link: https://lore.kernel.org/20260902-docs-memalloc-guide-v2-2-218c1a4dcb80@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: SJ Park Acked-by: Vlastimil Babka (SUSE) Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Michal Hocko Cc: Randy Dunlap Cc: Suren Baghdasaryan --- MAINTAINERS | 2 ++ 1 file changed, 2 insertions(+) diff --git a/MAINTAINERS b/MAINTAINERS index 3ff1a8f07e526f..ca76ab10fe68ed 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -17267,6 +17267,8 @@ F: Documentation/ABI/testing/sysfs-kernel-mm F: Documentation/ABI/testing/sysfs-kernel-mm-cma F: Documentation/ABI/testing/sysfs-kernel-mm-numa F: Documentation/admin-guide/mm/ +F: Documentation/core-api/memory-allocation.rst +F: Documentation/core-api/mm-api.rst F: Documentation/mm/ F: mm/char-mem.c F: include/linux/cma.h From f7d1bfcfbcea4d5fe313f76c134bb2ef2918e0b4 Mon Sep 17 00:00:00 2001 From: Meijing Zhao Date: Wed, 2 Sep 2026 14:45:08 +0800 Subject: [PATCH 0750/1352] mm: trace: decode arm64 and sparc64 VM_ARCH_1 flags Patch series "mm: trace: decode architecture-specific VMA flags", v2. show_vma_flags(), which is used by VMA tracepoints and %pGv, names generic VM_* bits but does not fully decode architecture-specific flag positions. As a result, trace output and VMA dumps can show a generic "arch_1" name or a raw hexadecimal value instead of the meaning assigned by the target architecture. This series adds symbolic names for the arm64 and sparc64 meanings of VM_ARCH_1, arm64 MTE flags, user shadow stack flags, and protection-key encoding bits under their corresponding configuration guards. This patch (of 3): The VM_ARCH_1 bit has architecture-specific meanings. show_vma_flags(), which is used by VMA tracepoints and %pGv, already reports the powerpc, parisc and no-MMU meanings, but falls back to the generic "arch_1" name on arm64 and sparc64. Report VM_ARM64_BTI as "bti" and VM_SPARC_ADI as "adi" so trace output and %pGv dumps expose the actual architecture-specific state. Link: https://lore.kernel.org/cover.1788330432.git.zhaomeijing@lixiang.com Link: https://lore.kernel.org/bcf0fea58240c4b6daa253739683ed7709e42fbb.1788330432.git.zhaomeijing@lixiang.com Signed-off-by: Meijing Zhao Signed-off-by: Andrew Morton Cc: Lorenzo Stoakes Cc: "Masami Hiramatsu (Google)" Cc: Mathieu Desnoyers Cc: Steven Rostedt --- include/trace/events/mmflags.h | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/include/trace/events/mmflags.h b/include/trace/events/mmflags.h index 935893e5ea53bc..02caeaf359ab18 100644 --- a/include/trace/events/mmflags.h +++ b/include/trace/events/mmflags.h @@ -168,6 +168,10 @@ IF_HAVE_PG_ARCH_3(arch_3) #define __VM_ARCH_SPECIFIC_1 {VM_SAO, "sao" } #elif defined(CONFIG_PARISC) #define __VM_ARCH_SPECIFIC_1 {VM_GROWSUP, "growsup" } +#elif defined(CONFIG_SPARC64) +#define __VM_ARCH_SPECIFIC_1 {VM_SPARC_ADI, "adi" } +#elif defined(CONFIG_ARM64) +#define __VM_ARCH_SPECIFIC_1 {VM_ARM64_BTI, "bti" } #elif !defined(CONFIG_MMU) #define __VM_ARCH_SPECIFIC_1 {VM_MAPPED_COPY,"mappedcopy" } #else From 8b129fa68833f06db913c610e3181d1600a30cda Mon Sep 17 00:00:00 2001 From: Meijing Zhao Date: Wed, 2 Sep 2026 14:45:09 +0800 Subject: [PATCH 0751/1352] mm: trace: decode MTE and shadow stack VMA flags show_vma_flags(), which is used by VMA tracepoints and %pGv, leaves architecture-specific HIGH_ARCH_* bits unnamed. Arm64 MTE flags and user shadow stack flags are therefore printed as raw hexadecimal values. These bit positions are shared between architectures. For example, bit 37 represents VM_MTE_ALLOWED on arm64 but VM_SHADOW_STACK on x86, so a shared HIGH_ARCH_* bit cannot be given an unconditional name. Add conditionally compiled names for VM_MTE, VM_MTE_ALLOWED and VM_SHADOW_STACK. Keep the configuration guards aligned with the definitions of these aliases so each shared bit position is decoded according to the target architecture. For example, an arm64 VMA containing VM_MTE_ALLOWED is now printed as: ...|account|mte_allowed|... instead of: ...|account|0x2000000000 Link: https://lore.kernel.org/1c7f7003cff8d00209baa8498f1efee113f5b5a4.1788330432.git.zhaomeijing@lixiang.com Signed-off-by: Meijing Zhao Signed-off-by: Andrew Morton Cc: Lorenzo Stoakes Cc: "Masami Hiramatsu (Google)" Cc: Mathieu Desnoyers Cc: Steven Rostedt --- include/trace/events/mmflags.h | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/include/trace/events/mmflags.h b/include/trace/events/mmflags.h index 02caeaf359ab18..2f980d20a9f640 100644 --- a/include/trace/events/mmflags.h +++ b/include/trace/events/mmflags.h @@ -202,6 +202,19 @@ IF_HAVE_PG_ARCH_3(arch_3) # define IF_HAVE_VM_DROPPABLE(flag, name) #endif +#ifdef CONFIG_ARM64_MTE +# define IF_HAVE_VM_MTE(flag, name) {flag, name}, +#else +# define IF_HAVE_VM_MTE(flag, name) +#endif + +#if defined(CONFIG_X86_USER_SHADOW_STACK) || defined(CONFIG_RISCV_USER_CFI) || \ + defined(CONFIG_ARM64_GCS) +# define IF_HAVE_VM_SHADOW_STACK(flag, name) {flag, name}, +#else +# define IF_HAVE_VM_SHADOW_STACK(flag, name) +#endif + #define __def_vmaflag_names \ {VM_READ, "read" }, \ {VM_WRITE, "write" }, \ @@ -237,6 +250,9 @@ IF_HAVE_VM_SOFTDIRTY(VM_SOFTDIRTY, "softdirty" ) \ {VM_HUGEPAGE, "hugepage" }, \ {VM_NOHUGEPAGE, "nohugepage" }, \ IF_HAVE_VM_DROPPABLE(VM_DROPPABLE, "droppable" ) \ +IF_HAVE_VM_MTE(VM_MTE, "mte") \ +IF_HAVE_VM_MTE(VM_MTE_ALLOWED, "mte_allowed") \ +IF_HAVE_VM_SHADOW_STACK(VM_SHADOW_STACK, "shadow_stack") \ {VM_MERGEABLE, "mergeable" } \ #define show_vma_flags(flags) \ From ee112e8f82cada65be82be9f719c1535e6174fd1 Mon Sep 17 00:00:00 2001 From: Meijing Zhao Date: Wed, 2 Sep 2026 14:45:10 +0800 Subject: [PATCH 0752/1352] mm: trace: name protection key encoding bits Protection keys are encoded in architecture-specific HIGH_ARCH_* VMA flag bits. show_vma_flags(), which is used by VMA tracepoints and %pGv, does not name those bits, leaving them as raw hexadecimal values. Name the protection-key encoding bits pkey_bit0 through pkey_bit4 under CONFIG_ARCH_HAS_PKEYS. The names make clear that these are bits of one protection-key value rather than independent protection keys. For example, protection key 3 is represented as: pkey_bit0|pkey_bit1 Honor CONFIG_ARCH_PKEY_BITS when exposing bit 3 and bit 4 so only bits provided by the architecture are included. Link: https://lore.kernel.org/1871da32001243b37147bac5af02d2f2ec64304e.1788330432.git.zhaomeijing@lixiang.com Signed-off-by: Meijing Zhao Signed-off-by: Andrew Morton Cc: Lorenzo Stoakes Cc: "Masami Hiramatsu (Google)" Cc: Mathieu Desnoyers Cc: Steven Rostedt --- include/trace/events/mmflags.h | 23 +++++++++++++++++++++++ 1 file changed, 23 insertions(+) diff --git a/include/trace/events/mmflags.h b/include/trace/events/mmflags.h index 2f980d20a9f640..ef9aa388b84f7d 100644 --- a/include/trace/events/mmflags.h +++ b/include/trace/events/mmflags.h @@ -202,6 +202,24 @@ IF_HAVE_PG_ARCH_3(arch_3) # define IF_HAVE_VM_DROPPABLE(flag, name) #endif +#ifdef CONFIG_ARCH_HAS_PKEYS +# define IF_HAVE_VM_PKEY(flag, name) {flag, name}, +#if CONFIG_ARCH_PKEY_BITS > 3 +# define IF_HAVE_VM_PKEY3(flag, name) {flag, name}, +#else +# define IF_HAVE_VM_PKEY3(flag, name) +#endif +#if CONFIG_ARCH_PKEY_BITS > 4 +# define IF_HAVE_VM_PKEY4(flag, name) {flag, name}, +#else +# define IF_HAVE_VM_PKEY4(flag, name) +#endif +#else +# define IF_HAVE_VM_PKEY(flag, name) +# define IF_HAVE_VM_PKEY3(flag, name) +# define IF_HAVE_VM_PKEY4(flag, name) +#endif + #ifdef CONFIG_ARM64_MTE # define IF_HAVE_VM_MTE(flag, name) {flag, name}, #else @@ -250,6 +268,11 @@ IF_HAVE_VM_SOFTDIRTY(VM_SOFTDIRTY, "softdirty" ) \ {VM_HUGEPAGE, "hugepage" }, \ {VM_NOHUGEPAGE, "nohugepage" }, \ IF_HAVE_VM_DROPPABLE(VM_DROPPABLE, "droppable" ) \ +IF_HAVE_VM_PKEY(VM_PKEY_BIT0, "pkey_bit0") \ +IF_HAVE_VM_PKEY(VM_PKEY_BIT1, "pkey_bit1") \ +IF_HAVE_VM_PKEY(VM_PKEY_BIT2, "pkey_bit2") \ +IF_HAVE_VM_PKEY3(VM_PKEY_BIT3, "pkey_bit3") \ +IF_HAVE_VM_PKEY4(VM_PKEY_BIT4, "pkey_bit4") \ IF_HAVE_VM_MTE(VM_MTE, "mte") \ IF_HAVE_VM_MTE(VM_MTE_ALLOWED, "mte_allowed") \ IF_HAVE_VM_SHADOW_STACK(VM_SHADOW_STACK, "shadow_stack") \ From ff1fa83b32ce352623eaf00449ad55bed9946fee Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 22:47:34 -0700 Subject: [PATCH 0753/1352] mm/damon/core: use damon_nr_samples_per_aggr() for max merge threshold Patch series "mm/damon: cleanup code, add test cases, and update guidances in docs". Misc cleanup, improvements and updates of code, test, and documents. Patches 1-5 cleanup DAMON code. Patches 6-10 adds kunit and selftest test cases for recently fixed bugs and a new feature. Patches 11 and 12 update guidelines for AI review and what document to read, on DAMON documents. This patch (of 12): kdamond_merge_regions() open-codes max region merge threshold calculation. What it does is fundamentally the same as damon_nr_samples_per_aggr() but missing a few corner cases. The unhandled corner cases should be rare and make only a negligible level of monitoring results degradation. But having the inconsistency could increase future maintenance burden. Use the dedicated function. Link: https://lore.kernel.org/20260902054747.99370-1-sj@kernel.org Link: https://lore.kernel.org/20260902054747.99370-2-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/core.c | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index cf4ec122a7f99d..1cf4d41f24dcdc 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -3533,8 +3533,7 @@ static void kdamond_merge_regions(struct damon_ctx *c, unsigned int threshold, unsigned int max_thres; bool count_age = true; - max_thres = c->attrs.aggr_interval / - (c->attrs.sample_interval ? c->attrs.sample_interval : 1); + max_thres = damon_nr_samples_per_aggr(&c->attrs); while (true) { nr_regions = 0; damon_for_each_target(t, c) { From 82fd6ab915f2567daec1c21f288efa5086ad937a Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 22:47:35 -0700 Subject: [PATCH 0754/1352] mm/damon/core: remove debug messages There are a few debug messages in DAMON core. Those have not really been used in a meaningful way for the last few years, though. Remove those. Link: https://lore.kernel.org/20260902054747.99370-3-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Kunwu Chan --- mm/damon/core.c | 9 --------- 1 file changed, 9 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 1cf4d41f24dcdc..4d817063a7a2e5 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -3756,10 +3756,6 @@ static unsigned long damos_wmark_wait_us(struct damos *scheme) /* higher than high watermark or lower than low watermark */ if (metric > scheme->wmarks.high || scheme->wmarks.low > metric) { - if (scheme->wmarks.activated) - pr_debug("deactivate a scheme (%d) for %s wmark\n", - scheme->action, - str_high_low(metric > scheme->wmarks.high)); scheme->wmarks.activated = false; return scheme->wmarks.interval; } @@ -3769,8 +3765,6 @@ static unsigned long damos_wmark_wait_us(struct damos *scheme) !scheme->wmarks.activated) return scheme->wmarks.interval; - if (!scheme->wmarks.activated) - pr_debug("activate a scheme (%d)\n", scheme->action); scheme->wmarks.activated = true; return 0; } @@ -3913,8 +3907,6 @@ static int kdamond_fn(void *data) struct damon_ctx *ctx = data; unsigned long sz_limit = 0; - pr_debug("kdamond (%d) starts\n", current->pid); - mutex_lock(&ctx->call_controls_lock); ctx->call_controls_obsolete = false; mutex_unlock(&ctx->call_controls_lock); @@ -4068,7 +4060,6 @@ static int kdamond_fn(void *data) mutex_unlock(&ctx->walk_control_lock); damos_walk_cancel(ctx); - pr_debug("kdamond (%d) finishes\n", current->pid); mutex_lock(&ctx->kdamond_lock); ctx->kdamond = NULL; mutex_unlock(&ctx->kdamond_lock); From d54c6a49213b19c0f36b8d27fdc2ab51af49c1e4 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 23:08:16 -0700 Subject: [PATCH 0755/1352] mm/damon/core: remove string_choices.h include It is no more being used. Link: https://lore.kernel.org/20260902061401.104419-1-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Kunwu Chan --- mm/damon/core.c | 1 - 1 file changed, 1 deletion(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 4d817063a7a2e5..88bd2f505cea07 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -15,7 +15,6 @@ #include #include #include -#include /* for damon_get_folio() used by node eligible memory metrics */ #include "ops-common.h" From 6d1f464dd0c33ea9d102e3871b50cdfcfd561634 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 22:47:36 -0700 Subject: [PATCH 0756/1352] mm/damon/vaddr: remove a debug message There is a debug message in the DAMON virtual address space operation set. It has not really been used in a meaningful way for the last few years, though. Remove it. Link: https://lore.kernel.org/20260902054747.99370-4-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/vaddr.c | 16 +++------------- 1 file changed, 3 insertions(+), 13 deletions(-) diff --git a/mm/damon/vaddr.c b/mm/damon/vaddr.c index 5b4d16c8db6285..91a0d441c1f940 100644 --- a/mm/damon/vaddr.c +++ b/mm/damon/vaddr.c @@ -189,22 +189,12 @@ static int damon_va_three_regions(struct damon_target *t, * * */ -static void __damon_va_init_regions(struct damon_ctx *ctx, - struct damon_target *t) +static void __damon_va_init_regions(struct damon_target *t) { - struct damon_target *ti; struct damon_addr_range regions[3]; - int tidx = 0; - if (damon_va_three_regions(t, regions)) { - damon_for_each_target(ti, ctx) { - if (ti == t) - break; - tidx++; - } - pr_debug("Failed to get three regions of %dth target\n", tidx); + if (damon_va_three_regions(t, regions)) return; - } damon_set_regions(t, regions, 3, DAMON_MIN_REGION_SZ); } @@ -217,7 +207,7 @@ static void damon_va_init(struct damon_ctx *ctx) damon_for_each_target(t, ctx) { /* the user may set the target regions as they want */ if (!damon_nr_regions(t)) - __damon_va_init_regions(ctx, t); + __damon_va_init_regions(t); } } From dac3ac152a78c5c8572778bcafd976625d5bb66b Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 22:47:37 -0700 Subject: [PATCH 0757/1352] mm/damon/core: validate number of probes in valid_probe_params() Each DAMON context is allowed to have only up to DAMON_MAX_PROBES probes. The central place for validating DAMON probe parameters, damon_valid_probe_params(), is not validating the upper limit, though. Do the validation. Link: https://lore.kernel.org/20260902054747.99370-5-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/core.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/mm/damon/core.c b/mm/damon/core.c index 88bd2f505cea07..87cc5d7568932c 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -1419,6 +1419,13 @@ static bool damon_valid_probe_params(struct damon_ctx *ctx) unsigned char max_probe_hits; struct damon_probe *probe; unsigned int wsum, wsum_to_add; + int nr_probes; + + nr_probes = 0; + damon_for_each_probe(probe, ctx) + nr_probes++; + if (nr_probes > DAMON_MAX_PROBES) + return false; if (!damon_has_probe_weights(ctx)) return true; From 19a01f0ed5aba62fc725cd505aae394153059482 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 22:47:38 -0700 Subject: [PATCH 0758/1352] mm/damon/sysfs: remove probes number validation DAMON sysfs interface is disallowing >DAMON_MAX_PROBES nr_probes input, since DAMON_MAX_PROBES is the upper limit of probes per DAMON context. The core layer is validating the upper limit again, though. It is preferred to let DAMON API callers such as sysfs interface to set parameters in flexible ways, and do parameters validation in the core layer. Drop the duplicated validation in the sysfs interface. Link: https://lore.kernel.org/20260902054747.99370-6-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index 7ec14f48d157a4..b576e97cbfdb8a 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -1479,7 +1479,7 @@ static ssize_t nr_probes_store(struct kobject *kobj, if (err) return err; - if (nr < 0 || nr > DAMON_MAX_PROBES) + if (nr < 0) return -EINVAL; probes = container_of(kobj, struct damon_sysfs_probes, kobj); From e057a9fda748a35d8629dbf49553f0faffcea385 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 22:47:39 -0700 Subject: [PATCH 0759/1352] mm/damon/tests/core-kunit: extend set_regions() test for error case damon_test_set_regions_for() is designed to test only success-expected damon_set_regions() calls. Extend it to cover error-expected calls, too. Link: https://lore.kernel.org/20260902054747.99370-7-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Kunwu Chan --- mm/damon/tests/core-kunit.h | 18 ++++++++++-------- 1 file changed, 10 insertions(+), 8 deletions(-) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index f2568fba552ec4..d20ea12ca61633 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -469,11 +469,12 @@ static void damon_test_set_regions_for(struct kunit *test, struct damon_addr_range *old_ranges, int sz_old_ranges, struct damon_addr_range *new_ranges, int sz_new_ranges, unsigned long min_region_sz, - struct damon_addr_range *expect_ranges, int sz_expect_ranges) + struct damon_addr_range *expect_ranges, int sz_expect_ranges, + int expect_err) { struct damon_target *t; struct damon_region *r; - int i; + int i, err; t = damon_new_target(); if (!t) @@ -487,7 +488,8 @@ static void damon_test_set_regions_for(struct kunit *test, damon_add_region(r, t); } - damon_set_regions(t, new_ranges, sz_new_ranges, min_region_sz); + err = damon_set_regions(t, new_ranges, sz_new_ranges, min_region_sz); + KUNIT_EXPECT_EQ(test, err, expect_err); KUNIT_EXPECT_EQ(test, damon_nr_regions(t), sz_expect_ranges); if (damon_nr_regions(t) != sz_expect_ranges) { @@ -516,7 +518,7 @@ static void damon_test_set_regions(struct kunit *test) (struct damon_addr_range[]){ {.start = 5, .end = 15}, {.start = 15, .end = 25}, - }, 2); + }, 2, 0); /* Un-intersecting regions should be removed. */ damon_test_set_regions_for(test, (struct damon_addr_range[]){ @@ -529,7 +531,7 @@ static void damon_test_set_regions(struct kunit *test) 1, (struct damon_addr_range[]){ {.start = 18, .end = 23}, - }, 1); + }, 1, 0); /* * Holes should be filled up with new regions. * @@ -550,7 +552,7 @@ static void damon_test_set_regions(struct kunit *test) {.start = 8, .end = 16}, {.start = 16, .end = 24}, {.start = 24, .end = 28}, - }, 3); + }, 3, 0); /* * New regions should be able to be appended. * @@ -572,7 +574,7 @@ static void damon_test_set_regions(struct kunit *test) {.start = 0, .end = 4}, {.start = 4, .end = 15}, {.start = 25, .end = 40}, - }, 3); + }, 3, 0); /* * New regions should be able to be inserted. * @@ -595,7 +597,7 @@ static void damon_test_set_regions(struct kunit *test) {.start = 0, .end = 15}, {.start = 25, .end = 40}, {.start = 44, .end = 50}, - }, 3); + }, 3, 0); } static void damon_test_update_monitoring_result(struct kunit *test) From 955e2c4c5a6e79336bcfa4c902d190a1fd1363f2 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 22:47:40 -0700 Subject: [PATCH 0760/1352] mm/damon/tests/core-kunit: test <=0 size damon_set_regions() inputs Commit 1292c0ecb1ca ("mm/damon/core: validate ranges in damon_set_regions()") disallowed passing zero or negative size input ranges to damon_set_regions(). Add kunit test cases for those inputs. Link: https://lore.kernel.org/20260902054747.99370-8-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Kunwu Chan --- mm/damon/tests/core-kunit.h | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index d20ea12ca61633..89f364a8ffeb67 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -598,6 +598,20 @@ static void damon_test_set_regions(struct kunit *test) {.start = 25, .end = 40}, {.start = 44, .end = 50}, }, 3, 0); + /* Zero size regions should return -EINVAL. */ + damon_test_set_regions_for(test, + (struct damon_addr_range[]){}, 0, + (struct damon_addr_range[]){ + {.start = 42, .end = 42}, + }, 1, 1, + (struct damon_addr_range[]){}, 0, -EINVAL); + /* Negative size regions should return -EINVAL. */ + damon_test_set_regions_for(test, + (struct damon_addr_range[]){}, 0, + (struct damon_addr_range[]){ + {.start = 42, .end = 21}, + }, 1, 1, + (struct damon_addr_range[]){}, 0, -EINVAL); } static void damon_test_update_monitoring_result(struct kunit *test) From c27b8f83de9944bb88eaa16c388154936f3c6e93 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 22:47:41 -0700 Subject: [PATCH 0761/1352] mm/damon/tests/core-kunit: test overlapping ranges for set_regions() Commit 954157679ec3 ("mm/damon/core: disallow overlapping input ranges for damon_set_regions()") disallowed passing overlapping input ranges to damon_set_regions(). Add a kunit test case for the overlapping input. Link: https://lore.kernel.org/20260902054747.99370-9-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/tests/core-kunit.h | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index 89f364a8ffeb67..ebb695090bb426 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -612,6 +612,17 @@ static void damon_test_set_regions(struct kunit *test) {.start = 42, .end = 21}, }, 1, 1, (struct damon_addr_range[]){}, 0, -EINVAL); + /* + * Regions resulting in same region after alignment should return + * -EINVAL. + */ + damon_test_set_regions_for(test, + (struct damon_addr_range[]){}, 0, + (struct damon_addr_range[]){ + {.start = 10, .end = 20}, + {.start = 20, .end = 30}, + }, 2, 4096, + (struct damon_addr_range[]){}, 0, -EINVAL); } static void damon_test_update_monitoring_result(struct kunit *test) From 40a519ec55863c242efd5bd833b612df974537e9 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 22:47:42 -0700 Subject: [PATCH 0762/1352] mm/damon/tests/core-kunit: test damon_nr_samples_per_aggr() damon_max_nr_accesses(), which is a previous version of damon_nr_samples_per_aggr() before the renaming, was wrongly returning zero or random overflowed values for extreme intervals setup. Commit 35d4a3cf70a8 ("mm/damon/ops-common: handle extreme intervals in damon_hot_score()") updated the function to return correct or more valid values. Add a kunit test to ensure it is working as expected. Link: https://lore.kernel.org/20260902054747.99370-10-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/tests/core-kunit.h | 22 ++++++++++++++++++++++ 1 file changed, 22 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index ebb695090bb426..c01e6a75cadc1e 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -625,6 +625,27 @@ static void damon_test_set_regions(struct kunit *test) (struct damon_addr_range[]){}, 0, -EINVAL); } +static void damon_test_nr_samples_per_aggr(struct kunit *test) +{ + struct damon_attrs attrs = { + .sample_interval = 0, + .aggr_interval = 0, + }; + + /* Zero aggregation interval doesn't cause division by zero */ + KUNIT_EXPECT_EQ(test, damon_nr_samples_per_aggr(&attrs), 1); + + /* + * Too large aggregation interval on 64 bit system doesn't cause + * overflow + */ + if (ULONG_MAX > UINT_MAX) { + attrs.aggr_interval = (unsigned long)UINT_MAX + 1; + KUNIT_EXPECT_EQ(test, damon_nr_samples_per_aggr(&attrs), + UINT_MAX); + } +} + static void damon_test_update_monitoring_result(struct kunit *test) { struct damon_attrs old_attrs = { @@ -1858,6 +1879,7 @@ static struct kunit_case damon_test_cases[] = { KUNIT_CASE(damon_test_split_above_half_progresses), KUNIT_CASE(damon_test_ops_registration), KUNIT_CASE(damon_test_set_regions), + KUNIT_CASE(damon_test_nr_samples_per_aggr), KUNIT_CASE(damon_test_update_monitoring_result), KUNIT_CASE(damon_test_set_attrs), KUNIT_CASE(damon_test_mvsum), From e9ba1bf546d4359718d37054fc5f81bc25d9ea45 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 22:47:43 -0700 Subject: [PATCH 0763/1352] selftests/damon/sysfs.sh: test hugepage_mem_bp quota goal DAMON sysfs quota goal target_metric file now accepts 'hugepage_mem_bp' input. Test it is accepted in fundamental DAMON sysfs file operation selftest. Link: https://lore.kernel.org/20260902054747.99370-11-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/damon/sysfs.sh | 1 + 1 file changed, 1 insertion(+) diff --git a/tools/testing/selftests/damon/sysfs.sh b/tools/testing/selftests/damon/sysfs.sh index ddebde6edabe4e..b66593c9ac471f 100755 --- a/tools/testing/selftests/damon/sysfs.sh +++ b/tools/testing/selftests/damon/sysfs.sh @@ -210,6 +210,7 @@ test_goal() ensure_write_succ "$fpath" "active_mem_bp" "valid input" ensure_write_succ "$fpath" "inactive_mem_bp" "valid input" ensure_write_succ "$fpath" "node_eligible_mem_bp" "valid input" + ensure_write_succ "$fpath" "hugepage_mem_bp" "valid input" ensure_write_fail "$fpath" "foo" "invalid input" ensure_file "$goal_dir/nid" "exist" "600" ensure_file "$goal_dir/path" "exist" "600" From 99e4929087adabc16fe044c6d0450f2889ea3387 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 22:47:44 -0700 Subject: [PATCH 0764/1352] Docs/mm/damon/maintainer-profile: update AI review for Sashiko replies In the past, sharing Sashiko review results was tedious. Also it was suggested to minimize the recipients on the sharing mails. Hence the AI review section of DAMON maintainer profile document was updated to give guidance about available tools for making the sharing easier, and how the recipients list should be managed. Now DAMON is onboarded [1] to Sashiko's automatic review replies feature. Sashiko directly sends its reviews as replies to the patch mail thread. It also reduces the recipients list to deliver the reporting to only the author and the mailing list. Remove the old guidance that is no more necessary but only confusing. Link: https://lore.kernel.org/20260902054747.99370-12-sj@kernel.org Link: https://github.com/sashiko-dev/sashiko/commit/b554c7b6e733 [1] Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Kunwu Chan --- Documentation/mm/damon/maintainer-profile.rst | 19 +++++-------------- 1 file changed, 5 insertions(+), 14 deletions(-) diff --git a/Documentation/mm/damon/maintainer-profile.rst b/Documentation/mm/damon/maintainer-profile.rst index fb2fa00cc9aa1b..a7c1352339378c 100644 --- a/Documentation/mm/damon/maintainer-profile.rst +++ b/Documentation/mm/damon/maintainer-profile.rst @@ -106,18 +106,9 @@ AI Review For patches that are publicly posted to DAMON mailing list (damon@lists.linux.dev), AI reviews of the patches will be available at -sashiko.dev. The reviews could also be sent as mails to the author of the -patch. - -Patch authors are encouraged to check the AI reviews and share their opinions. -The sharing could be done as a reply to the mail thread. Consider reducing the -recipients list for such sharing, since some people are not really interested -in AI reviews. As a rule of thumb, drop stable@vger.kernel.org and individuals -except DAMON maintainer. - -`hkml` also provides a `feature -`_ -for such sharing. Please feel free to use the feature. +sashiko.dev. The reviews will also be sent as replies to the author of the +patch and the mailing list. -It is only an optional recommendation. DAMON maintainer could also ask any -question about the AI reviews, though. +Patch authors are encouraged to check the AI reviews and share their opinions +by replying on the mail thread. It is only an optional recommendation. DAMON +maintainer could also ask any question about the AI reviews, though. From 8d22d6be78ecdf5defdcced73975c638b9dd2e38 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 22:47:45 -0700 Subject: [PATCH 0765/1352] Docs/ABI/damon: recommend subsystem doc instead of admin-guide DAMON ABI doc is recommending DAMON admin-guide for people who are willing to know further about DAMON. Nowadays the subsystem doc (Documentation/mm/damon) is a more recommended place for even beginners. Recommend the subsystem doc over admin-guide. Link: https://lore.kernel.org/20260902054747.99370-13-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/ABI/testing/sysfs-kernel-mm-damon | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Documentation/ABI/testing/sysfs-kernel-mm-damon b/Documentation/ABI/testing/sysfs-kernel-mm-damon index f8d2601e829047..ad21f58f3c9126 100644 --- a/Documentation/ABI/testing/sysfs-kernel-mm-damon +++ b/Documentation/ABI/testing/sysfs-kernel-mm-damon @@ -3,7 +3,7 @@ Date: Mar 2022 Contact: SJ Park Description: Interface for Data Access MONitoring (DAMON). Contains files for controlling DAMON. For more details on DAMON itself, - please refer to Documentation/admin-guide/mm/damon/index.rst. + please refer to Documentation/mm/damon/index.rst. What: /sys/kernel/mm/damon/admin/ Date: Mar 2022 From fa7869f6f5d0950f849abe99f901d64bc9306ac4 Mon Sep 17 00:00:00 2001 From: zhaozhengzhuo Date: Wed, 2 Sep 2026 11:12:29 +0800 Subject: [PATCH 0766/1352] mm/migrate_device: fix function name in kernel-doc The kernel-doc for migrate_device_range() says that migrate_vma_setup() is similar to itself. Refer to migrate_device_range() as the subject of the comparison, making the distinction between virtual-address-based and device-PFN-based migration clear. Link: https://lore.kernel.org/13768B0F4A5FC1F5+20260902031229.1821112-1-zhaozhengzhuo@uniontech.com Fixes: e778406b40db ("mm/migrate_device.c: add migrate_device_range()") Signed-off-by: zhaozhengzhuo Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Cc: Alistair Popple Cc: David Hildenbrand Cc: Byungchul Park Cc: Gregory Price Cc: "Huang, Ying" Cc: Joshua Hahn Cc: Matthew Brost Cc: Rakie Kim --- mm/migrate_device.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/mm/migrate_device.c b/mm/migrate_device.c index 009bfa8b212d5b..0c437004329d9c 100644 --- a/mm/migrate_device.c +++ b/mm/migrate_device.c @@ -1398,9 +1398,9 @@ static unsigned long migrate_device_pfn_lock(unsigned long pfn) * @start: starting pfn in the range to migrate. * @npages: number of pages to migrate. * - * migrate_vma_setup() is similar in concept to migrate_vma_setup() except that - * instead of looking up pages based on virtual address mappings a range of - * device pfns that should be migrated to system memory is used instead. + * migrate_device_range() is similar in concept to migrate_vma_setup(), except + * that instead of looking up pages based on virtual address mappings, a range + * of device pfns that should be migrated to system memory is used. * * This is useful when a driver needs to free device memory but doesn't know the * virtual mappings of every page that may be in device memory. For example this From 80a9ddcfef4a924813a174479e4b0434d44eba3b Mon Sep 17 00:00:00 2001 From: Wei Yang Date: Wed, 24 Jun 2026 08:23:59 +0000 Subject: [PATCH 0767/1352] mm/page_vma_mapped: guard check_pmd() with CONFIG_TRANSPARENT_HUGEPAGE The kernel test robot reported a build failure on the parisc architecture when expanding HPAGE_PMD_NR in check_pmd(). mm/page_vma_mapped.c:142:13: note: in expansion of macro 'HPAGE_PMD_NR' if ((pfn + HPAGE_PMD_NR - 1) < pvmw->pfn) ^~~~~~~~~~~~ The config [1] in report link shows neither TRANSPARENT_HUGEPAGE nor HUGETLB_PAGE is defined. Then trigger the BUILD_BUG. Fix it by define check_pmd() under CONFIG_TRANSPARENT_HUGEPAGE. Link: https://lore.kernel.org/20260624082359.2869-1-richard.weiyang@gmail.com Link: https://download.01.org/0day-ci/archive/20260624/202606240042.ffPsEXVc-lkp@intel.com/config [1] Fixes: 2aff7a4755be ("mm: Convert page_vma_mapped_walk to work on PFNs") Signed-off-by: Wei Yang Signed-off-by: Andrew Morton Reported-by: kernel test robot Closes: https://lore.kernel.org/oe-kbuild-all/202606240042.ffPsEXVc-lkp@intel.com/ Cc: David Hildenbrand Cc: Harry Yoo Cc: Jann Horn Cc: Lance Yang Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Matthew Wilcox (Oracle) Cc: Rik van Riel Cc: Vlastimil Babka --- mm/page_vma_mapped.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/mm/page_vma_mapped.c b/mm/page_vma_mapped.c index 4e964545e5e85a..28e306fdb3a5b8 100644 --- a/mm/page_vma_mapped.c +++ b/mm/page_vma_mapped.c @@ -142,6 +142,7 @@ static bool check_pte(struct page_vma_mapped_walk *pvmw, unsigned long pte_nr) return true; } +#ifdef CONFIG_TRANSPARENT_HUGEPAGE /* Returns true if the two ranges overlap. Careful to not overflow. */ static bool check_pmd(unsigned long pfn, struct page_vma_mapped_walk *pvmw) { @@ -151,6 +152,12 @@ static bool check_pmd(unsigned long pfn, struct page_vma_mapped_walk *pvmw) return false; return true; } +#else +static bool check_pmd(unsigned long pfn, struct page_vma_mapped_walk *pvmw) +{ + return false; +} +#endif static void step_forward(struct page_vma_mapped_walk *pvmw, unsigned long size) { From 8cf0cc5fd15fbdf10b476f7a2e4d071dda1fa51f Mon Sep 17 00:00:00 2001 From: "Zenghui Yu (Huawei)" Date: Thu, 3 Sep 2026 21:52:51 +0800 Subject: [PATCH 0768/1352] selftests/mm: remove unreachable returns after ksft exit helpers The ksft_exit*() helpers such as ksft_exit_fail_msg() are declared __noreturn, and the ksft_exit() and ksft_finished() macros expand to calls of them, always terminating the process via exit(). Any return statements following such calls are unreachable, both at the end of main() and on error paths of helper functions. Remove all of them. No functional change. Link: https://lore.kernel.org/20260903135251.39593-1-zenghui.yu@linux.dev Signed-off-by: Zenghui Yu (Huawei) Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: SJ Park Reviewed-by: Zi Yan Assisted-by: GLM-5.3 OpenCode Cc: Kiryl Shutsemau Cc: Baolin Wang Cc: Barry Song Cc: David Hildenbrand Cc: Dev Jain Cc: Lance Yang Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Ryan Roberts Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/mm/folio_split_race_test.c | 2 -- tools/testing/selftests/mm/mlock-random-test.c | 1 - tools/testing/selftests/mm/pkey_sighandler_tests.c | 1 - tools/testing/selftests/mm/split_huge_page_test.c | 4 ---- 4 files changed, 8 deletions(-) diff --git a/tools/testing/selftests/mm/folio_split_race_test.c b/tools/testing/selftests/mm/folio_split_race_test.c index 45b84f7b364e0a..1960635a953eb5 100644 --- a/tools/testing/selftests/mm/folio_split_race_test.c +++ b/tools/testing/selftests/mm/folio_split_race_test.c @@ -269,6 +269,4 @@ int main(void) NUM_ITERATIONS); ksft_exit(iter == NUM_ITERATIONS); - - return 0; } diff --git a/tools/testing/selftests/mm/mlock-random-test.c b/tools/testing/selftests/mm/mlock-random-test.c index 16294bc7dae6b2..58772914fd79e1 100644 --- a/tools/testing/selftests/mm/mlock-random-test.c +++ b/tools/testing/selftests/mm/mlock-random-test.c @@ -71,7 +71,6 @@ int get_proc_locked_vm_size(void) fclose(f); ksft_exit_fail_msg("cannot parse VmLck in /proc/self/status: %s\n", strerror(errno)); - return -1; } /* diff --git a/tools/testing/selftests/mm/pkey_sighandler_tests.c b/tools/testing/selftests/mm/pkey_sighandler_tests.c index 74bf79a5399dac..f9c728ba96a559 100644 --- a/tools/testing/selftests/mm/pkey_sighandler_tests.c +++ b/tools/testing/selftests/mm/pkey_sighandler_tests.c @@ -556,5 +556,4 @@ int main(int argc, char *argv[]) } ksft_finished(); - return 0; } diff --git a/tools/testing/selftests/mm/split_huge_page_test.c b/tools/testing/selftests/mm/split_huge_page_test.c index 86a6036928261d..c01d227d7fd6dd 100644 --- a/tools/testing/selftests/mm/split_huge_page_test.c +++ b/tools/testing/selftests/mm/split_huge_page_test.c @@ -101,7 +101,6 @@ static bool is_backed_by_folio(char *vaddr, int order, int pagemap_fd, return (pfn_flags & folio_tail_flags) != folio_tail_flags; fail: ksft_exit_fail_msg("Failed to get folio info\n"); - return false; } static int check_after_split_folio_orders(char *vaddr_start, size_t len, @@ -548,7 +547,6 @@ static int create_pagecache_thp_and_fd(const char *testfile, size_t fd_size, err_out_unlink: unlink(testfile); ksft_exit_fail_msg("Failed to create large pagecache folios\n"); - return -1; } static void split_thp_in_pagecache_to_order_at(size_t fd_size, @@ -711,6 +709,4 @@ int main(int argc, char **argv) free(expected_orders); ksft_finished(); - - return 0; } From 59731e1e59e8bb79c0b3ca2549c366b1b8c10d00 Mon Sep 17 00:00:00 2001 From: Christos Skarlos Date: Thu, 3 Sep 2026 12:22:00 +0300 Subject: [PATCH 0769/1352] mm/huge_memory: fix various coding style warnings Resolve coding style issues flagged by checkpatch.pl Specifically: - Add missing blank lines after variable declarations. - Remove unnecessary braces {} for a single statement block. No functional changes are introduced. Link: https://lore.kernel.org/20260903092200.88910-1-christosskarlos.kernel@gmail.com Signed-off-by: Christos Skarlos Signed-off-by: Andrew Morton Reviewed-by: Barry Song Reviewed-by: Zi Yan Reviewed-by: Lorenzo Stoakes (ARM) Cc: Baolin Wang Cc: David Hildenbrand Cc: Dev Jain Cc: Lance Yang Cc: Liam R. Howlett Cc: Ryan Roberts --- mm/huge_memory.c | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index c5d11147b69aec..dd66c6ad5af13c 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -1146,6 +1146,7 @@ subsys_initcall(hugepage_init); static int __init setup_transparent_hugepage(char *str) { int ret = 0; + if (!str) goto out; if (!strcmp(str, "always")) { @@ -1552,6 +1553,7 @@ static void set_huge_zero_folio(pgtable_t pgtable, struct mm_struct *mm, struct folio *zero_folio) { pmd_t entry; + entry = folio_mk_pmd(zero_folio, vma->vm_page_prot); entry = pmd_mkspecial(entry); pgtable_trans_huge_deposit(mm, pmd, pgtable); @@ -2661,6 +2663,7 @@ bool move_huge_pmd(struct vm_area_struct *vma, unsigned long old_addr, if (pmd_move_must_withdraw(new_ptl, old_ptl, vma)) { pgtable_t pgtable; + pgtable = pgtable_trans_huge_withdraw(mm, old_pmd); pgtable_trans_huge_deposit(mm, new_pmd, pgtable); } @@ -3949,9 +3952,8 @@ int folio_check_splittable(struct folio *folio, unsigned int new_order, * swapcache folio split. Only uniform split to order-0 can be used * here. */ - if ((split_type == SPLIT_TYPE_NON_UNIFORM || new_order) && folio_test_swapcache(folio)) { + if ((split_type == SPLIT_TYPE_NON_UNIFORM || new_order) && folio_test_swapcache(folio)) return -EINVAL; - } if (is_huge_zero_folio(folio)) return -EINVAL; From b901d01a6a4202b9907e481ca82efd9e9e1bcf89 Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Thu, 3 Sep 2026 17:21:26 +0800 Subject: [PATCH 0770/1352] mm/page_owner: preserve original free_pid/free_tgid during folio migration __update_page_owner_free_handle() accepts pid and tgid, but ignores them, always storing current->pid and current->tgid. __folio_copy_owner() forwards the original free pid/tgid during folio migration, yet these values were overwritten by the migrating task. Store the passed-in pid/tgid so the migrated folio keeps the original free attribution. Link: https://lore.kernel.org/20260903092126.24685-1-hongfu.li@linux.dev Signed-off-by: Hongfu Li Signed-off-by: Andrew Morton Acked-by: Vlastimil Babka (SUSE) Cc: Brendan Jackman Cc: Johannes Weiner Cc: Michal Hocko Cc: Suren Baghdasaryan Cc: Zi Yan --- mm/page_owner.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/mm/page_owner.c b/mm/page_owner.c index 3fc37d9b908ef0..cfc31c92d76570 100644 --- a/mm/page_owner.c +++ b/mm/page_owner.c @@ -307,8 +307,8 @@ static inline void __update_page_owner_free_handle(struct page *page, page_owner->free_handle = handle; } page_owner->free_ts_nsec = free_ts_nsec; - page_owner->free_pid = current->pid; - page_owner->free_tgid = current->tgid; + page_owner->free_pid = pid; + page_owner->free_tgid = tgid; } rcu_read_unlock(); } From aa7d2741a4350285fc4ee5060d9773b5fb636fde Mon Sep 17 00:00:00 2001 From: Jinmeng Zhou Date: Thu, 3 Sep 2026 15:50:48 +0800 Subject: [PATCH 0771/1352] mm/hugetlb: charge folios to the target mm's memcg HugeTLB folios are currently charged to the memcg of the allocating task. This gives the wrong result when a userfaultfd handler populates a HugeTLB VMA that belongs to another process. The UFFDIO_COPY ioctl operates on the userfaultfd context's mm, but get_mem_cgroup_from_current() charges the folio to the handler's memcg instead. This can be reproduced by placing the faulting process and its userfaultfd handler in different memory cgroups. Have the target process register a HugeTLB mapping with userfaultfd, trigger a missing fault, and let the handler resolve it with UFFDIO_COPY. The hugepage usage is then reported in the handler's memory.current instead of the target's. The generic userfaultfd population path avoids this problem by charging folios to dst_vma->vm_mm. Pass the target mm through hugetlb_alloc_folio() and charge the folio by using get_mem_cgroup_from_mm(). This preserves the existing charge timing and error handling while making HugeTLB userfaultfd population consistent with the generic path. Link: https://lore.kernel.org/20260903075048.3316-1-zhoujinmeng@bytedance.com Fixes: 8cba9576df60 ("hugetlb: memcg: account hugetlb-backed memory in memory controller") Signed-off-by: Jinmeng Zhou Signed-off-by: Andrew Morton Reviewed-by: Muchun Song Reviewed-by: Hongfu Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Michal Hocko Cc: Nhat Pham Cc: Oscar Salvador Cc: Roman Gushchin Cc: Shakeel Butt Cc: --- include/linux/hugetlb.h | 3 ++- include/linux/memcontrol.h | 8 +++++--- mm/hugetlb.c | 9 ++++++--- mm/memcontrol.c | 6 ++++-- 4 files changed, 17 insertions(+), 9 deletions(-) diff --git a/include/linux/hugetlb.h b/include/linux/hugetlb.h index 900c95e346b2e3..80a5a03e9cee72 100644 --- a/include/linux/hugetlb.h +++ b/include/linux/hugetlb.h @@ -694,7 +694,8 @@ enum hugetlb_alloc_flag { #define HUGETLB_ALLOC_USE_GLOBAL_RESERVATIONS BIT(HUGETLB_ALLOC_USE_GLOBAL_RESERVATIONS_BIT) struct folio *hugetlb_alloc_folio(struct hstate *h, - struct mempolicy_interpreted *mpoli, u8 alloc_flags); + struct mempolicy_interpreted *mpoli, struct mm_struct *mm, + u8 alloc_flags); struct folio *alloc_hugetlb_folio(struct vm_area_struct *vma, unsigned long addr, bool cow_from_owner); struct folio *alloc_hugetlb_folio_nodemask(struct hstate *h, int preferred_nid, diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index c799926435560f..058ebd73ff1605 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -647,7 +647,8 @@ static inline int mem_cgroup_charge(struct folio *folio, struct mm_struct *mm, return __mem_cgroup_charge(folio, mm, gfp); } -int mem_cgroup_charge_hugetlb(struct folio* folio, gfp_t gfp); +int mem_cgroup_charge_hugetlb(struct folio *folio, struct mm_struct *mm, + gfp_t gfp); int mem_cgroup_swapin_charge_folio(struct folio *folio, unsigned short id, struct mm_struct *mm, gfp_t gfp); @@ -1146,9 +1147,10 @@ static inline int mem_cgroup_charge(struct folio *folio, return 0; } -static inline int mem_cgroup_charge_hugetlb(struct folio* folio, gfp_t gfp) +static inline int mem_cgroup_charge_hugetlb(struct folio *folio, + struct mm_struct *mm, gfp_t gfp) { - return 0; + return 0; } static inline int mem_cgroup_swapin_charge_folio(struct folio *folio, diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 4d7c2ee126efda..afffaa3d2d3741 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -2844,6 +2844,7 @@ void wait_for_freed_hugetlb_folios(void) * hugetlb_alloc_folio - Allocate a hugetlb folio. * @h: Hugetlb state control block. * @mpoli: Interpreted memory policy to use for allocation. + * @mm: Memory descriptor of the allocation target. * @alloc_flags: Flags controlling the allocation behavior. * * Allocates a hugetlb folio and handles cgroup charging and global hstate @@ -2853,7 +2854,8 @@ void wait_for_freed_hugetlb_folios(void) * -ENOSPC if cgroup charging fails or no folio is available. */ struct folio *hugetlb_alloc_folio(struct hstate *h, - struct mempolicy_interpreted *mpoli, u8 alloc_flags) + struct mempolicy_interpreted *mpoli, struct mm_struct *mm, + u8 alloc_flags) { bool charge_hugetlb_cgroup_rsvd = alloc_flags & HUGETLB_ALLOC_CHARG_CGROUP_RSVD; @@ -2908,7 +2910,8 @@ struct folio *hugetlb_alloc_folio(struct hstate *h, spin_unlock_irq(&hugetlb_lock); - ret = mem_cgroup_charge_hugetlb(folio, gfp | __GFP_RETRY_MAYFAIL); + ret = mem_cgroup_charge_hugetlb(folio, mm, + gfp | __GFP_RETRY_MAYFAIL); /* * Unconditionally increment NR_HUGETLB here because if * mem_cgroup_charge_hugetlb failed, freeing the page will @@ -3051,7 +3054,7 @@ struct folio *alloc_hugetlb_folio(struct vm_area_struct *vma, .nodemask = nodemask, }; - folio = hugetlb_alloc_folio(h, &mpoli, alloc_flags); + folio = hugetlb_alloc_folio(h, &mpoli, vma->vm_mm, alloc_flags); mpol_cond_put(mpol); diff --git a/mm/memcontrol.c b/mm/memcontrol.c index bd1e7e15442659..86ff580c70183a 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -5265,6 +5265,7 @@ int __mem_cgroup_charge(struct folio *folio, struct mm_struct *mm, gfp_t gfp) /** * mem_cgroup_charge_hugetlb - charge the memcg for a hugetlb folio * @folio: folio being charged + * @mm: mm context of the allocation target * @gfp: reclaim mode * * This function is called when allocating a huge page folio, after the page has @@ -5274,9 +5275,10 @@ int __mem_cgroup_charge(struct folio *folio, struct mm_struct *mm, gfp_t gfp) * Returns ENOMEM if the memcg is already full. * Returns 0 if either the charge was successful, or if we skip the charging. */ -int mem_cgroup_charge_hugetlb(struct folio *folio, gfp_t gfp) +int mem_cgroup_charge_hugetlb(struct folio *folio, struct mm_struct *mm, + gfp_t gfp) { - struct mem_cgroup *memcg = get_mem_cgroup_from_current(); + struct mem_cgroup *memcg = get_mem_cgroup_from_mm(mm); int ret = 0; /* From 8a996fd5713e0bb7769ad2fcb6ad94578f06e9e1 Mon Sep 17 00:00:00 2001 From: Kaitao Cheng Date: Thu, 3 Sep 2026 13:35:34 +0800 Subject: [PATCH 0772/1352] mm/page-flags: define HWPoison test-and-change helpers unconditionally Page flag helpers for configuration-dependent flags provide false or no-op variants so that their users can be independent of the configuration. The HWPoison helpers do not fully follow this pattern. When CONFIG_MEMORY_FAILURE is enabled, PAGEFLAG() and TESTSCFLAG() provide the regular, test-and-set, and test-and-clear operations. When it is disabled, only PAGEFLAG_FALSE() is instantiated, leaving TestSetPageHWPoison() and TestClearPageHWPoison() undefined. Use TESTSCFLAG_FALSE() to provide the missing accessors when memory failure handling is disabled. Both accessors return false, which is consistent with HWPoison state being unavailable, and makes the accessor interface consistent across configurations. Also remove now-unneeded test_and_clear_pmem_poison() from nvdimm/pmem. Link: https://lore.kernel.org/20260903053535.17611-1-kaitao.cheng@linux.dev Link: https://lore.kernel.org/20260903053535.17611-2-kaitao.cheng@linux.dev Link: https://lore.kernel.org/20260903053535.17611-3-kaitao.cheng@linux.dev Signed-off-by: Kaitao Cheng Signed-off-by: Andrew Morton Acked-by: Muchun Song Acked-by: David Hildenbrand (Arm) Reviewed-by: Oscar Salvador Cc: Alison Schofield Cc: Dave Jiang Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vishal Verma Cc: Vlastimil Babka Cc: Ira Weiny --- drivers/nvdimm/pmem.c | 2 +- drivers/nvdimm/pmem.h | 12 ------------ include/linux/page-flags.h | 1 + 3 files changed, 2 insertions(+), 13 deletions(-) diff --git a/drivers/nvdimm/pmem.c b/drivers/nvdimm/pmem.c index 30a51c365ce8ba..5fb86595e8bd7f 100644 --- a/drivers/nvdimm/pmem.c +++ b/drivers/nvdimm/pmem.c @@ -80,7 +80,7 @@ static void pmem_mkpage_present(struct pmem_device *pmem, phys_addr_t offset, * here since we're in the driver I/O path and * outstanding I/O requests pin the dev_pagemap. */ - if (test_and_clear_pmem_poison(page)) + if (TestClearPageHWPoison(page)) clear_mce_nospec(pfn); } } diff --git a/drivers/nvdimm/pmem.h b/drivers/nvdimm/pmem.h index a48509f901968e..76870505dd7971 100644 --- a/drivers/nvdimm/pmem.h +++ b/drivers/nvdimm/pmem.h @@ -1,7 +1,6 @@ /* SPDX-License-Identifier: GPL-2.0 */ #ifndef __NVDIMM_PMEM_H__ #define __NVDIMM_PMEM_H__ -#include #include #include #include @@ -31,15 +30,4 @@ long __pmem_direct_access(struct pmem_device *pmem, pgoff_t pgoff, long nr_pages, enum dax_access_mode mode, void **kaddr, unsigned long *pfn); -#ifdef CONFIG_MEMORY_FAILURE -static inline bool test_and_clear_pmem_poison(struct page *page) -{ - return TestClearPageHWPoison(page); -} -#else -static inline bool test_and_clear_pmem_poison(struct page *page) -{ - return false; -} -#endif #endif /* __NVDIMM_PMEM_H__ */ diff --git a/include/linux/page-flags.h b/include/linux/page-flags.h index ae2ebaed6d4d96..3dc79c0c5adf03 100644 --- a/include/linux/page-flags.h +++ b/include/linux/page-flags.h @@ -655,6 +655,7 @@ TESTSCFLAG(HWPoison, hwpoison, PF_ANY) #define __PG_HWPOISON (1UL << PG_hwpoison) #else PAGEFLAG_FALSE(HWPoison, hwpoison) +TESTSCFLAG_FALSE(HWPoison, hwpoison) #define __PG_HWPOISON 0 #endif From 43f3db7ef14a748da38f49cda7837c5995eb4a0b Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Thu, 3 Sep 2026 11:01:34 +0800 Subject: [PATCH 0773/1352] mm/memfd: fix hugetlb reservation accounting in error paths If hugetlb_add_to_page_cache() in memfd_alloc_folio() fails with -EEXIST, a concurrent fault has already instantiated the folio in the page cache, and the reservation now belongs to that folio. Calling hugetlb_unreserve_pages() in that case incorrectly removes the region backing the cached folio and releases one reservation more than it should, leaving that folio in the page cache with no region recording it. That leaves resv_huge_pages one page short until that folio is removed from the page cache, so available_huge_pages() reports a page that is not actually free and the pool can grant one reservation more than it can back. Applications using hugetlb memfds can then fail to allocate or fault in a page they already reserved. They see -ENOMEM or -ENOSPC from the allocation or fault path even though the reservation and HugePages_Free still look healthy. So hold the hugetlb fault mutex from hugetlb_reserve_pages() until the error-path unreserve completes to make the reserve, allocate and instantiate steps atomic against concurrent faults. With the mutex held from the start, a concurrent fault can no longer consume the reservation between reserve and allocate/instantiate. If a fault completed before the mutex was taken, it has already added the region for that index, so hugetlb_reserve_pages() returns 0 and the error path leaves the region in place. Link: https://lore.kernel.org/20260927094757.31665-1-hongfu.li@linux.dev Link: https://lore.kernel.org/20260903030134.7407-1-hongfu.li@linux.dev Fixes: 717cf9357325 ("mm/memfd: reserve hugetlb folios before allocation") Signed-off-by: Hongfu Li Signed-off-by: Andrew Morton Cc: Baolin Wang Cc: David Hildenbrand Cc: Hugh Dickins Cc: Muchun Song Cc: Oscar Salvador Cc: Vivek Kasireddy Cc: --- mm/memfd.c | 31 ++++++++++++++++--------------- 1 file changed, 16 insertions(+), 15 deletions(-) diff --git a/mm/memfd.c b/mm/memfd.c index c708d92533f4ab..0f6fff004f5e4c 100644 --- a/mm/memfd.c +++ b/mm/memfd.c @@ -82,22 +82,31 @@ struct folio *memfd_alloc_folio(struct file *memfd, pgoff_t idx) struct hstate *h = hstate_file(memfd); int err = -ENOMEM; long nr_resv; + u32 hash; gfp_mask = htlb_alloc_mask(h); gfp_mask &= ~(__GFP_HIGHMEM | __GFP_MOVABLE); idx >>= huge_page_order(h); + /* + * Serialize hugepage allocation and instantiation to prevent + * races with concurrent allocations, as required by all other + * callers of hugetlb_add_to_page_cache(). + */ + hash = hugetlb_fault_mutex_hash(memfd->f_mapping, idx); + mutex_lock(&hugetlb_fault_mutex_table[hash]); + nr_resv = hugetlb_reserve_pages(inode, idx, idx + 1, NULL, EMPTY_VMA_FLAGS); - if (nr_resv < 0) - return ERR_PTR(nr_resv); + if (nr_resv < 0) { + err = nr_resv; + goto out_unlock; + } folio = alloc_hugetlb_folio_reserve(h, numa_node_id(), NULL, gfp_mask); if (folio) { - u32 hash; - /* * Zero the folio to prevent information leaks to userspace. * Use folio_zero_user() which is optimized for huge/gigantic @@ -112,20 +121,9 @@ struct folio *memfd_alloc_folio(struct file *memfd, pgoff_t idx) */ __folio_mark_uptodate(folio); - /* - * Serialize hugepage allocation and instantiation to prevent - * races with concurrent allocations, as required by all other - * callers of hugetlb_add_to_page_cache(). - */ - hash = hugetlb_fault_mutex_hash(memfd->f_mapping, idx); - mutex_lock(&hugetlb_fault_mutex_table[hash]); - err = hugetlb_add_to_page_cache(folio, memfd->f_mapping, idx); - - mutex_unlock(&hugetlb_fault_mutex_table[hash]); - if (err) { folio_put(folio); goto err_unresv; @@ -133,11 +131,14 @@ struct folio *memfd_alloc_folio(struct file *memfd, pgoff_t idx) hugetlb_set_folio_subpool(folio, subpool_inode(inode)); folio_unlock(folio); + mutex_unlock(&hugetlb_fault_mutex_table[hash]); return folio; } err_unresv: if (nr_resv > 0) hugetlb_unreserve_pages(inode, idx, idx + 1, 0); +out_unlock: + mutex_unlock(&hugetlb_fault_mutex_table[hash]); return ERR_PTR(err); } #endif From 19e8f9344eb444af1e1162587d4f59348a1e405f Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 18:07:19 -0700 Subject: [PATCH 0774/1352] mm/damon/core: error damos_commit_quota_goal() for zero target_value Patch series "mm/damon: move zero damos quota target_value handling to the core layer". Having zero DAMOS quota target value can cause division by zero. DAMON API callers are checking the target value parameters to avoid that. It is easy to make mistakes in some of the multiple API callers. Move that to the core layer. Patch 1 adds the corner case handling into the core layer DAMON parameters validation logic. Patches 2 and 3 remove no more needed DAMON API callers side handling of the corner case in DAMON_LRU_SORT and DMON_SAMPLIE_MTIER, respectively. This patch (of 3): If a DAMOS scheme has a damos_quota_goal of zero target_value, damos_quota_goal() could trigger division-by-zero error. Hence each DAMON API callers should do the zero target_value validation. It is easy to make mistakes. Actually such bugs in DAMON_LRU_SORT and DAMON_SAMPLE_MTIER were found and fixed [1]. It is better to handle the corner case only once in the core layer, instead of multiple places in all DAMON API callers. One straightforward option is using an alternative denominator for the corner case in the damos_quota_goal(). However, the zero target_value is meaningless. In this case, the quota goal is always evaluated as achieved or over-achieved. The quota will only keep being reduced. Simply avoid using zero target_value by adding a check in the core layer DAMOS quota goal parameters validation/commit path, damos_commit_quota_goal(). Update it to return an error in the case. Also update its caller to propagate the error. Link: https://lore.kernel.org/20260903010722.94244-1-sj@kernel.org Link: https://lore.kernel.org/20260903010722.94244-2-sj@kernel.org Link: https://lore.kernel.org/20260803134034.15217-1-sj@kernel.org [1] Signed-off-by: SJ Park Signed-off-by: Andrew Morton --- mm/damon/core.c | 22 ++++++++++++++++------ 1 file changed, 16 insertions(+), 6 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 87cc5d7568932c..15a773614935b9 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -1211,14 +1211,17 @@ static void damos_commit_quota_goal_union( } } -static void damos_commit_quota_goal( +static int damos_commit_quota_goal( struct damos_quota_goal *dst, struct damos_quota_goal *src) { + if (!src->target_value) + return -EINVAL; dst->metric = src->metric; dst->target_value = src->target_value; if (dst->metric == DAMOS_QUOTA_USER_INPUT) dst->current_value = src->current_value; damos_commit_quota_goal_union(dst, src); + return 0; } /** @@ -1236,14 +1239,17 @@ static void damos_commit_quota_goal( int damos_commit_quota_goals(struct damos_quota *dst, struct damos_quota *src) { struct damos_quota_goal *dst_goal, *next, *src_goal, *new_goal; - int i = 0, j = 0; + int i = 0, j = 0, err; damos_for_each_quota_goal_safe(dst_goal, next, dst) { src_goal = damos_nth_quota_goal(i++, src); - if (src_goal) - damos_commit_quota_goal(dst_goal, src_goal); - else + if (src_goal) { + err = damos_commit_quota_goal(dst_goal, src_goal); + if (err) + return err; + } else { damos_destroy_quota_goal(dst_goal); + } } damos_for_each_quota_goal_safe(src_goal, next, src) { if (j++ < i) @@ -1252,7 +1258,11 @@ int damos_commit_quota_goals(struct damos_quota *dst, struct damos_quota *src) src_goal->metric, src_goal->target_value); if (!new_goal) return -ENOMEM; - damos_commit_quota_goal(new_goal, src_goal); + err = damos_commit_quota_goal(new_goal, src_goal); + if (err) { + damos_free_quota_goal(new_goal); + return err; + } damos_add_quota_goal(dst, new_goal); } return 0; From 58a47bc544edbfdca5743513b3a8d4aac7692934 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 18:07:20 -0700 Subject: [PATCH 0775/1352] Revert "mm/damon/lru_sort: error out for >10000 active_mem_bp" This reverts commit 06befa61c427e74319781e6f35a364cfc32dbae8. The commit was made to avoid zero damos quota goal target value, because it can trigger division-by-zero. Now the core layer handles the corner case. It returns an error for any attempt setting the aero target_value. The corner case handling in DAMON_LRU_SORT is hence no more needed. Remove it. Note that this slightly changes the user behavior. It still disallows active_mem_bp of 10,002. But now it allows other >10,000 active_mem_bp values. Setting >10,000 active_mem_bp makes not much sense. But it doesn't cause critical problems such as memory leak or crash, either. Arguably that doesn't deserve additional code complexity. Just allow it. Link: https://lore.kernel.org/20260903010722.94244-3-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton --- mm/damon/lru_sort.c | 2 -- 1 file changed, 2 deletions(-) diff --git a/mm/damon/lru_sort.c b/mm/damon/lru_sort.c index bd847829a99072..7df45f9a0b3aeb 100644 --- a/mm/damon/lru_sort.c +++ b/mm/damon/lru_sort.c @@ -233,8 +233,6 @@ static int damon_lru_sort_add_quota_goals(struct damos *hot_scheme, if (!active_mem_bp) return 0; - if (10000 < active_mem_bp) - return -EINVAL; goal = damos_new_quota_goal(DAMOS_QUOTA_ACTIVE_MEM_BP, active_mem_bp); if (!goal) return -ENOMEM; From 62a0e5227c7a310fd31a08abdf273f913620fa52 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 18:07:21 -0700 Subject: [PATCH 0776/1352] Revert "samples/damon/mtier: error out for zero quota goal target values" This reverts commit a16fd3ad9d89b05475864da97327870464611736. The commit was made to avoid zero damos quota goal target value, because it can trigger division-by-zero. Now the core layer handles the corner case. It returns an error for any attempt setting the aero target_value. The corner case handling in DAMON_SAMPLE_MTIER is hence no more needed. Remove it. Link: https://lore.kernel.org/20260903010722.94244-4-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton --- samples/damon/mtier.c | 3 --- 1 file changed, 3 deletions(-) diff --git a/samples/damon/mtier.c b/samples/damon/mtier.c index bea45c87cc9bed..27dc88bdf7a0ef 100644 --- a/samples/damon/mtier.c +++ b/samples/damon/mtier.c @@ -161,9 +161,6 @@ static struct damon_ctx *damon_sample_mtier_build_ctx(bool promote) if (!scheme) goto free_out; damon_set_schemes(ctx, &scheme, 1); - /* zero target value causes division by zero in damos_quota_store() */ - if (!node0_mem_used_bp || !node0_mem_free_bp) - goto free_out; quota_goal = damos_new_quota_goal( promote ? DAMOS_QUOTA_NODE_MEM_USED_BP : DAMOS_QUOTA_NODE_MEM_FREE_BP, From 0579df55c8654baae852e7a7d2b31ad4b222e3f8 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 18:03:30 -0700 Subject: [PATCH 0777/1352] mm/damon/core: handle NULL ctx parameter in damon_call() Patch series "mm/damon: allow NULL or unstarted damon_ctx parameter for damon_call()". Callers of damon_call() should validate the damon_ctx object parameter. If it is NULL or never damon_start()-ed object, damon_call() could dereference the NULL pointer or indefinitely hang. Ensuring all callers doing the validation correctly has turned out to be difficult. Handle the corner cases inside the core layer and remove callers' validations. Patches 1 and 2 respectively allow passing NULL and not yet damon_start()-ed ctx parameter to damon_call(). Patches 3 and 4 remove the callers side validations in DMON_RECLAIM and DASMON_LRU_SORT, respectively. This patch (of 4): When NULL damon_ctx pointer parameter is passed, damon_call() could do NULL dereference. The caller is responsible to avoid that. It is easy to forget, and there are many damon_call() callers. Meanwhile, damon_call() is never meant to be performance critical. It uses mutex and completion. Add the NULL pointer check inside damon_call() so that callers can pass the parameter without NULL checks. Link: https://lore.kernel.org/20260903010334.93622-1-sj@kernel.org Link: https://lore.kernel.org/20260903010334.93622-2-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton --- mm/damon/core.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/mm/damon/core.c b/mm/damon/core.c index 15a773614935b9..a4f78d10ef4e5f 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2216,6 +2216,8 @@ int damon_kdamond_pid(struct damon_ctx *ctx) */ int damon_call(struct damon_ctx *ctx, struct damon_call_control *control) { + if (!ctx) + return -EINVAL; if (!control->repeat) init_completion(&control->completion); control->canceled = false; From ca1039d323001619258dc1e45d8dd8356aaca48b Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 18:03:31 -0700 Subject: [PATCH 0778/1352] mm/damon/core: set ctx->call_controls_obsolete in damon_new_ctx() damon_ctx->call_controls_obsolete is used to disallow damon_call() requests when the request cannot be served. The field is unset and set when the context execution is started and terminated, respectively. The intention is to allow damon_call() requests only while the context is actively being executed. damon_ctx constructor, damon_new_ctx() unsets the field, though. As a result, passing the damon_ctx parameter that never successfully damon_start()-ed to damon_call() can indefinitely hang. The callers should ensure to avoid the case. Such parameter validation is not always simple. Actually such bugs in DAMON_RECLAIM and DAMON_LRU_SORT have been found and fixed [1]. Set the field in damon_new_ctx(), so that DAMON API callers can pass the context parameter to damon_call() without the additional check. Link: https://lore.kernel.org/20260903010334.93622-3-sj@kernel.org Link: https://lore.kernel.org/20260803134646.16640-1-sj@kernel.org [1] Signed-off-by: SJ Park Signed-off-by: Andrew Morton --- mm/damon/core.c | 5 +---- 1 file changed, 1 insertion(+), 4 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index a4f78d10ef4e5f..ea6df4311ceb7f 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -933,6 +933,7 @@ struct damon_ctx *damon_new_ctx(void) INIT_LIST_HEAD(&ctx->adaptive_targets); INIT_LIST_HEAD(&ctx->schemes); + ctx->call_controls_obsolete = true; prandom_seed_state(&ctx->rnd_state, get_random_u64()); return ctx; @@ -2206,10 +2207,6 @@ int damon_kdamond_pid(struct damon_ctx *ctx) * synchronization. The return value of the function will be saved in * &damon_call_control->return_code. * - * Note that this function should be called only after damon_start() with the - * @ctx has succeeded. Otherwise, this function could fall into an indefinite - * wait. - * * When this function is failed, the @ctx is guaranteed to be stopped. * * Return: 0 on success, negative error code otherwise. From 6ee37fd3ede65262fc68c6c17c17a634f664d2f0 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 18:03:32 -0700 Subject: [PATCH 0779/1352] mm/damon/reclaim: remove unnecessary damon_call() param validation DAMON_RECLAIM avoids passing NULL or unstarted damon_ctx to damon_call() with its own validation. The validation is no longer needed, because the DAMON core layer now handles the corner cases itself. Remove the unnecessary check. Link: https://lore.kernel.org/20260903010334.93622-4-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton --- mm/damon/reclaim.c | 8 -------- 1 file changed, 8 deletions(-) diff --git a/mm/damon/reclaim.c b/mm/damon/reclaim.c index 45d5557cc575a0..42a2c9cb134310 100644 --- a/mm/damon/reclaim.c +++ b/mm/damon/reclaim.c @@ -271,8 +271,6 @@ static int damon_reclaim_commit_inputs_fn(void *arg) return damon_reclaim_apply_parameters(); } -static bool damon_reclaim_damon_has_started; - static int damon_reclaim_commit_inputs_store(const char *val, const struct kernel_param *kp) { @@ -293,10 +291,6 @@ static int damon_reclaim_commit_inputs_store(const char *val, if (!commit_inputs_request) return 0; - /* Skip damon_call() if ctx has not successfully started. */ - if (!damon_reclaim_damon_has_started) - return -EINVAL; - err = damon_call(ctx, &control); return err ? err : control.return_code; @@ -343,8 +337,6 @@ static int damon_reclaim_turn(bool on) err = damon_start(&ctx, 1, true); if (err) return err; - if (!damon_reclaim_damon_has_started) - damon_reclaim_damon_has_started = true; return damon_call(ctx, &call_control); } From a14e61e2bd54bb17b317daa59e4d199400aa5032 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 18:03:33 -0700 Subject: [PATCH 0780/1352] mm/damon/lru_sort: remove unnecessary damon_call() param validation DAMON_LRU_SORT avoids passing NULL or unstarted damon_ctx to damon_call() with its own validation. The validation is no longer needed, because the DAMON core layer now handles the corner cases itself. Remove the unnecessary check. Link: https://lore.kernel.org/20260903010334.93622-5-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton --- mm/damon/lru_sort.c | 8 -------- 1 file changed, 8 deletions(-) diff --git a/mm/damon/lru_sort.c b/mm/damon/lru_sort.c index 7df45f9a0b3aeb..ad8e86dd3a93e1 100644 --- a/mm/damon/lru_sort.c +++ b/mm/damon/lru_sort.c @@ -344,8 +344,6 @@ static int damon_lru_sort_commit_inputs_fn(void *arg) return damon_lru_sort_apply_parameters(); } -static bool damon_lru_sort_damon_has_started; - static int damon_lru_sort_commit_inputs_store(const char *val, const struct kernel_param *kp) { @@ -366,10 +364,6 @@ static int damon_lru_sort_commit_inputs_store(const char *val, if (!commit_inputs_request) return 0; - /* Skip damon_call() if ctx has not successfully started. */ - if (!damon_lru_sort_damon_has_started) - return -EINVAL; - err = damon_call(ctx, &control); return err ? err : control.return_code; @@ -420,8 +414,6 @@ static int damon_lru_sort_turn(bool on) err = damon_start(&ctx, 1, true); if (err) return err; - if (!damon_lru_sort_damon_has_started) - damon_lru_sort_damon_has_started = true; return damon_call(ctx, &call_control); } From 0e6ab155e04036c40326a4b1eabb2335f319531f Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Wed, 2 Sep 2026 15:55:07 -0400 Subject: [PATCH 0781/1352] mm/memory_hotplug: factor out node_is_memoryless() A memoryless node neither spans present pages (populated or ZONE_DEVICE) nor has an offline-but-added memory block still linked to it in sysfs. try_offline_node() presently open-codes this memoryless check. Pull that into a node_is_memoryless() helper and pull the existing check_no_memblock_for_node_cb() helper ahead of the add/online path so it's clearer what is happening here. No functional change. Link: https://lore.kernel.org/20260902195507.88655-1-gourry@gourry.net Signed-off-by: Gregory Price Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Oscar Salvador --- mm/memory_hotplug.c | 60 +++++++++++++++++++++++---------------------- 1 file changed, 31 insertions(+), 29 deletions(-) diff --git a/mm/memory_hotplug.c b/mm/memory_hotplug.c index 226ab9cb078ad1..d0e94057682af6 100644 --- a/mm/memory_hotplug.c +++ b/mm/memory_hotplug.c @@ -1491,6 +1491,36 @@ static int create_altmaps_and_memory_blocks(int nid, struct memory_group *group, return ret; } +static int check_no_memblock_for_node_cb(struct memory_block *mem, void *arg) +{ + int nid = *(int *)arg; + + /* + * If a memory block belongs to multiple nodes, the stored nid is not + * reliable. However, such blocks are always online (e.g., cannot get + * offlined) and, therefore, are still spanned by the node. + */ + return mem->nid == nid ? -EEXIST : 0; +} + +/* Caller must hold the memory hotplug lock for this check. */ +static bool node_is_memoryless(int nid) +{ + /* + * A node still spanning pages (especially ZONE_DEVICE) is not + * memoryless. A node spans memory after move_pfn_range_to_zone(), + * e.g. once a memory block has been onlined. + */ + if (node_spanned_pages(nid)) + return false; + /* + * Offline memory blocks may not be spanned by the node yet, but they + * link to it in sysfs and can be onlined later, so the node is not + * memoryless while any remain. + */ + return !for_each_memory_block(&nid, check_no_memblock_for_node_cb); +} + /* * NOTE: The caller must call lock_device_hotplug() to serialize hotplug * and online/offline operations (triggered e.g. by sysfs). @@ -2214,18 +2244,6 @@ static int check_cpu_on_node(int nid) return 0; } -static int check_no_memblock_for_node_cb(struct memory_block *mem, void *arg) -{ - int nid = *(int *)arg; - - /* - * If a memory block belongs to multiple nodes, the stored nid is not - * reliable. However, such blocks are always online (e.g., cannot get - * offlined) and, therefore, are still spanned by the node. - */ - return mem->nid == nid ? -EEXIST : 0; -} - /** * try_offline_node * @nid: the node ID @@ -2237,23 +2255,7 @@ static int check_no_memblock_for_node_cb(struct memory_block *mem, void *arg) */ void try_offline_node(int nid) { - int rc; - - /* - * If the node still spans pages (especially ZONE_DEVICE), don't - * offline it. A node spans memory after move_pfn_range_to_zone(), - * e.g., after the memory block was onlined. - */ - if (node_spanned_pages(nid)) - return; - - /* - * Especially offline memory blocks might not be spanned by the - * node. They will get spanned by the node once they get onlined. - * However, they link to the node in sysfs and can get onlined later. - */ - rc = for_each_memory_block(&nid, check_no_memblock_for_node_cb); - if (rc) + if (!node_is_memoryless(nid)) return; if (check_cpu_on_node(nid)) From 702078b3986b141432a3454d9eedb7824f387c46 Mon Sep 17 00:00:00 2001 From: Andrew Morton Date: Fri, 4 Sep 2026 18:54:43 -0700 Subject: [PATCH 0782/1352] mm-memory_hotplug-factor-out-node_is_memoryless-fix move node_is_memoryless() inside CONFIG_MEMORY_HOTREMOVE Reported-by: kernel test robot Closes: https://lore.kernel.org/oe-kbuild-all/202609050628.ywCLhOj5-lkp@intel.com/ Cc: David Hildenbrand Cc: Gregory Price Cc: Oscar Salvador Signed-off-by: Andrew Morton --- mm/memory_hotplug.c | 61 +++++++++++++++++++++++---------------------- 1 file changed, 31 insertions(+), 30 deletions(-) diff --git a/mm/memory_hotplug.c b/mm/memory_hotplug.c index d0e94057682af6..b428da66d279c0 100644 --- a/mm/memory_hotplug.c +++ b/mm/memory_hotplug.c @@ -1491,36 +1491,6 @@ static int create_altmaps_and_memory_blocks(int nid, struct memory_group *group, return ret; } -static int check_no_memblock_for_node_cb(struct memory_block *mem, void *arg) -{ - int nid = *(int *)arg; - - /* - * If a memory block belongs to multiple nodes, the stored nid is not - * reliable. However, such blocks are always online (e.g., cannot get - * offlined) and, therefore, are still spanned by the node. - */ - return mem->nid == nid ? -EEXIST : 0; -} - -/* Caller must hold the memory hotplug lock for this check. */ -static bool node_is_memoryless(int nid) -{ - /* - * A node still spanning pages (especially ZONE_DEVICE) is not - * memoryless. A node spans memory after move_pfn_range_to_zone(), - * e.g. once a memory block has been onlined. - */ - if (node_spanned_pages(nid)) - return false; - /* - * Offline memory blocks may not be spanned by the node yet, but they - * link to it in sysfs and can be onlined later, so the node is not - * memoryless while any remain. - */ - return !for_each_memory_block(&nid, check_no_memblock_for_node_cb); -} - /* * NOTE: The caller must call lock_device_hotplug() to serialize hotplug * and online/offline operations (triggered e.g. by sysfs). @@ -1815,6 +1785,37 @@ bool mhp_range_allowed(u64 start, u64 size, bool need_mapping) } #ifdef CONFIG_MEMORY_HOTREMOVE + +static int check_no_memblock_for_node_cb(struct memory_block *mem, void *arg) +{ + int nid = *(int *)arg; + + /* + * If a memory block belongs to multiple nodes, the stored nid is not + * reliable. However, such blocks are always online (e.g., cannot get + * offlined) and, therefore, are still spanned by the node. + */ + return mem->nid == nid ? -EEXIST : 0; +} + +/* Caller must hold the memory hotplug lock for this check. */ +static bool node_is_memoryless(int nid) +{ + /* + * A node still spanning pages (especially ZONE_DEVICE) is not + * memoryless. A node spans memory after move_pfn_range_to_zone(), + * e.g. once a memory block has been onlined. + */ + if (node_spanned_pages(nid)) + return false; + /* + * Offline memory blocks may not be spanned by the node yet, but they + * link to it in sysfs and can be onlined later, so the node is not + * memoryless while any remain. + */ + return !for_each_memory_block(&nid, check_no_memblock_for_node_cb); +} + /* * Scan pfn range [start,end) to find movable/migratable pages (LRU and * hugetlb folio, movable_ops pages). Will skip over most unmovable From 0a308a99c143514ce0d27c539d34674bb40ce0d9 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Sat, 5 Sep 2026 15:08:30 -0400 Subject: [PATCH 0783/1352] mm: remove PageWriteback The last caller of PageWriteback() was removed in commit f75987e543c2 ("libceph: remove pinning assertion in ceph_msg_data_iter_next()"). This flag is now only used on folios, so we can remove all the page accessors. folio_test_clear_writeback() is not used, so don't add FOLIO_TEST_CLEAR_FLAG() for it. Link: https://lore.kernel.org/20260905-remove-pagewriteback-v1-1-06f41c00db03@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Reviewed-by: SJ Park Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/page-flags.h | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/include/linux/page-flags.h b/include/linux/page-flags.h index 3dc79c0c5adf03..86dd0470da1173 100644 --- a/include/linux/page-flags.h +++ b/include/linux/page-flags.h @@ -588,8 +588,8 @@ FOLIO_FLAG(owner_2, FOLIO_HEAD_PAGE) * Only test-and-set exist for PG_writeback. The unconditional operators are * risky: they bypass page accounting. */ -TESTPAGEFLAG(Writeback, writeback, PF_NO_TAIL) - TESTSCFLAG(Writeback, writeback, PF_NO_TAIL) +FOLIO_TEST_FLAG(writeback, FOLIO_HEAD_PAGE) + FOLIO_TEST_SET_FLAG(writeback, FOLIO_HEAD_PAGE) FOLIO_FLAG(mappedtodisk, FOLIO_HEAD_PAGE) /* PG_readahead is only used for reads; PG_reclaim is only for writes */ From e1e9e26b12ea19d235685b802b91eb2f36b23c99 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Sun, 6 Sep 2026 00:51:06 +0800 Subject: [PATCH 0784/1352] mm/memcontrol: move the lru_zone_size sanity check to the reader side Patch series "mm/mglru: clean up folio counters and flag usage", v6. This is a cleanup series separated out from the MGLRU-FG series [1]. As that series is getting too long in following updates, separate out the clean up part for easier review and merge. No feature change is intended, except one bugfix. It mostly replaces the open-coded bit operations scattered throughout the MGLRU code with new helpers, with proper kdocs, sanity debug checks, and hardens a few MGLRU functions. A subtle generation counter leak is also found during the refactoring and the fix is included. Also collected review feedbacks on the cleanup part from the posted series. This patch (of 6): Instead of using an unsigned long and checking the counter value at the updater side, turn the counter into a signed long and check at the reader side. This reduces overhead and simplifies the code. commit ca707239e8a7 ("mm: update_lru_size warn and reset bad lru_size") added a sanity check for memcg counter underflow: lru_zone_size is unsigned, so an underflow wraps it around and returns an enormously large number, then the memcg shrinker loops almost forever as the calculated number of folios to shrink is huge. It also checked if a zero value matches the empty LRU list, so the positive and negative deltas had to be handled separately. However that emptiness check was already removed by commit b4536f0c829c ("mm, memcg: fix the active list aging for lowmem requests when memcg is enabled"), so handling the deltas separately is no longer needed. The remaining update-side check is costly and cannot really catch the leak it is after anyway. It runs on every LRU folio, and if a folio was removed without updating the counter while other folios remain on the LRU, the WARN only triggers much later, from a likely innocent callsite. While readers are much rarer than writers, only the reclaim and reparenting paths read it, once per batch. Checking at the reader side instead leaves the update path a plain addition, and puts the warning where the value is actually consumed. Note this changes the behavior on underflow: the correction is removed and a negative value is kept. A massive leak of the LRU size counter would indicate that something else has gone very wrong, and one should fix that leaking site instead. Besides, the original behavior might cause false positives, or make things worse if the accounting happens after the actual insertion: the value is not leaked, just delayed, so force-fixing it would cause a bigger problem. The warning now only kicks in when a consumer actually uses it, in which case the reader gets zero. Link: https://lore.kernel.org/20260906-mglru-flags-cleanup-v6-0-9aacbd77d4ca@tencent.com Link: https://lore.kernel.org/20260906-mglru-flags-cleanup-v6-1-9aacbd77d4ca@tencent.com Link: https://lore.kernel.org/linux-mm/20260804-mglru-fg-v1-0-4d8dad39dad6@tencent.com/ [1] Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Ridong Chen Reviewed-by: Barry Song Acked-by: Shakeel Butt Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie Cc: Yu Zhao Cc: Zi Yan Cc: Lian Wang Cc: Qi Zheng --- include/linux/memcontrol.h | 9 +++++++-- mm/memcontrol.c | 18 +----------------- 2 files changed, 8 insertions(+), 19 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 058ebd73ff1605..a03b6e3e547078 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -100,7 +100,7 @@ struct mem_cgroup_per_node { /* Fields which get updated often at the end. */ struct lruvec lruvec; CACHELINE_PADDING(_pad2_); - unsigned long lru_zone_size[MAX_NR_ZONES][NR_LRU_LISTS]; + long lru_zone_size[MAX_NR_ZONES][NR_LRU_LISTS]; struct mem_cgroup_reclaim_iter iter; /* @@ -888,10 +888,15 @@ static inline unsigned long mem_cgroup_get_zone_lru_size(struct lruvec *lruvec, enum lru_list lru, int zone_idx) { + long val; struct mem_cgroup_per_node *mz; mz = container_of(lruvec, struct mem_cgroup_per_node, lruvec); - return READ_ONCE(mz->lru_zone_size[zone_idx][lru]); + val = READ_ONCE(mz->lru_zone_size[zone_idx][lru]); + if (WARN_ON_ONCE(val < 0)) + return 0; + + return val; } void __mem_cgroup_handle_over_high(gfp_t gfp_mask); diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 86ff580c70183a..4cb2db8c0923a4 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -1529,28 +1529,12 @@ void mem_cgroup_update_lru_size(struct lruvec *lruvec, enum lru_list lru, int zid, long nr_pages) { struct mem_cgroup_per_node *mz; - unsigned long *lru_size; - long size; if (mem_cgroup_disabled()) return; mz = container_of(lruvec, struct mem_cgroup_per_node, lruvec); - lru_size = &mz->lru_zone_size[zid][lru]; - - if (nr_pages < 0) - *lru_size += nr_pages; - - size = *lru_size; - if (WARN_ONCE(size < 0, - "%s(%p, %d, %ld): lru_size %ld\n", - __func__, lruvec, lru, nr_pages, size)) { - VM_BUG_ON(1); - *lru_size = 0; - } - - if (nr_pages > 0) - *lru_size += nr_pages; + mz->lru_zone_size[zid][lru] += nr_pages; } /** From fcd74e09846327fca5150261725b29071ca4282a Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Sun, 6 Sep 2026 00:51:07 +0800 Subject: [PATCH 0785/1352] mm/mglru: introduce helpers for manipulating gen and refs flags Instead of doing bit ops on folio->flags.f, introduce helpers for adjusting a folio's refs and generation info, making the code easier to debug and understand. No functional change is intended: some combined atomic operations are split into two, which only creates harmless transient states. There is no measurable performance impact, and some paths even look slightly better in the generated assembly. Link: https://lore.kernel.org/20260906-mglru-flags-cleanup-v6-2-9aacbd77d4ca@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Reviewed-by: Baolin Wang Reviewed-by: Barry Song Cc: Axel Rasmussen Cc: Baoquan He Cc: Chris Li Cc: David Hildenbrand (Arm) Cc: Johannes Weiner Cc: Kairui Song Cc: Liam R. Howlett Cc: Lian Wang Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Ridong Chen Cc: Roman Gushchin Cc: Shakeel Butt Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie Cc: Yu Zhao Cc: Zi Yan --- include/linux/mm_inline.h | 84 +++++++++++++++++++++++++++++++++++---- include/linux/mmzone.h | 1 + mm/folio.c | 19 +++++---- mm/vmscan.c | 60 ++++++++++++++++------------ 4 files changed, 122 insertions(+), 42 deletions(-) diff --git a/include/linux/mm_inline.h b/include/linux/mm_inline.h index 621c8653d8f7eb..3f4bd5b02b54da 100644 --- a/include/linux/mm_inline.h +++ b/include/linux/mm_inline.h @@ -142,10 +142,66 @@ static inline int lru_tier_from_refs(int refs, bool workingset) return workingset ? MAX_NR_TIERS - 1 : order_base_2(refs); } -static inline int folio_lru_refs(const struct folio *folio) +/** + * lru_set_gen_flags - Set the LRU generation number to specified folio flags. + * @flags: pointer to the folio flags + * @gen: generation number, between 0 and (MAX_NR_GENS - 1), inclusive. + */ +static inline void lru_set_gen_flags(unsigned long *flags, int gen) +{ + BUILD_BUG_ON(LRU_GEN_MASK & LRU_REFS_MASK); + VM_WARN_ON_ONCE(gen >= MAX_NR_GENS || gen < 0); + /* Store gen offset by 1, zero means the folio is off-list. */ + *flags &= ~LRU_GEN_MASK; + *flags |= (gen + 1UL) << LRU_GEN_PGOFF; +} + +/** + * lru_get_gen_flags - Return the LRU generation number from folio flags. + * @flags: folio flags + * + * Returns: A number between 0 and (MAX_NR_GENS - 1), inclusive. Returns + * -1 if the flags indicate the folio is off the list (e.g., isolated). + */ +static inline int lru_get_gen_flags(unsigned long flags) { - unsigned long flags = READ_ONCE(folio->flags.f); + int gen = ((flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1; + /* Exclude the legal -1 from the unsigned MAX_NR_GENS comparison */ + VM_WARN_ON_ONCE(gen != -1 && gen >= MAX_NR_GENS); + return gen; +} + +/** + * lru_set_refs_flags - Set the LRU referenced count to folio flags. + * @flags: pointer to the folio flags + * @refs: referenced / access count number, between 0 and LRU_REFS_MAX, inclusive. + * + * For MGLRU, PG_referenced holds the first ref, and the extra bits hold the + * remaining refs. For classical LRU the extra bits are not used, so it can + * also be seen as the refs count never exceeds 1. In both cases, refs == 1 + * means PG_referenced is set and the extra bits are zero, and refs == 0 means + * PG_referenced and the extra bits are all unset. + */ +static inline void lru_set_refs_flags(unsigned long *flags, unsigned int refs) +{ + VM_WARN_ON_ONCE(refs > LRU_REFS_MAX); + BUILD_BUG_ON(LRU_REFS_MAX != (LRU_REFS_MASK >> LRU_REFS_PGOFF) + 1); + + *flags &= ~LRU_REFS_FLAGS; + if (!refs) + return; + *flags |= (BIT(PG_referenced) | ((refs - 1UL) << LRU_REFS_PGOFF)); +} + +/** + * lru_get_refs_flags - Return LRU referenced / access count from folio flags. + * @flags: folio flags + * + * Reads the LRU referenced count set by lru_set_refs_flags(). + */ +static inline int lru_get_refs_flags(unsigned long flags) +{ if (!(flags & BIT(PG_referenced))) return 0; /* @@ -155,11 +211,24 @@ static inline int folio_lru_refs(const struct folio *folio) return ((flags & LRU_REFS_MASK) >> LRU_REFS_PGOFF) + 1; } -static inline int folio_lru_gen(const struct folio *folio) +static inline int folio_lru_refs(const struct folio *folio) { - unsigned long flags = READ_ONCE(folio->flags.f); + return lru_get_refs_flags(READ_ONCE(*const_folio_flags(folio, 0))); +} + +static inline void folio_set_lru_refs(struct folio *folio, unsigned int refs) +{ + unsigned long new_flags, old_flags = READ_ONCE(*folio_flags(folio, 0)); + + do { + new_flags = old_flags; + lru_set_refs_flags(&new_flags, refs); + } while (!try_cmpxchg(folio_flags(folio, 0), &old_flags, new_flags)); +} - return ((flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1; +static inline int folio_lru_gen(const struct folio *folio) +{ + return lru_get_gen_flags(READ_ONCE(*const_folio_flags(folio, 0))); } static inline bool lru_gen_is_active(const struct lruvec *lruvec, int gen) @@ -270,7 +339,7 @@ static inline bool lru_gen_add_folio(struct lruvec *lruvec, struct folio *folio, gen = lru_gen_from_seq(seq); flags = (gen + 1UL) << LRU_GEN_PGOFF; /* see the comment on MIN_NR_GENS about PG_active */ - set_mask_bits(&folio->flags.f, LRU_GEN_MASK | BIT(PG_active), flags); + set_mask_bits(folio_flags(folio, 0), LRU_GEN_MASK | BIT(PG_active), flags); lru_gen_update_size(lruvec, folio, -1, gen); /* for folio_rotate_reclaimable() */ @@ -295,7 +364,7 @@ static inline bool lru_gen_del_folio(struct lruvec *lruvec, struct folio *folio, /* for folio_migrate_flags() */ flags = !reclaiming && lru_gen_is_active(lruvec, gen) ? BIT(PG_active) : 0; - flags = set_mask_bits(&folio->flags.f, LRU_GEN_MASK, flags); + flags = set_mask_bits(folio_flags(folio, 0), LRU_GEN_MASK, flags); gen = ((flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1; lru_gen_update_size(lruvec, folio, gen, -1); @@ -339,7 +408,6 @@ static inline bool lru_gen_del_folio(struct lruvec *lruvec, struct folio *folio, static inline void folio_migrate_refs(struct folio *new, const struct folio *old) { - } #endif /* CONFIG_LRU_GEN */ diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index c070b867e2f349..593cc92fe47146 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -500,6 +500,7 @@ enum lruvec_flags { #define LRU_GEN_MASK ((BIT(LRU_GEN_WIDTH) - 1) << LRU_GEN_PGOFF) #define LRU_REFS_MASK ((BIT(LRU_REFS_WIDTH) - 1) << LRU_REFS_PGOFF) +#define LRU_REFS_MAX BIT(LRU_REFS_WIDTH) /* * For folios accessed multiple times through file descriptors, diff --git a/mm/folio.c b/mm/folio.c index 50a6dbe55998e7..47a437e0f7fde6 100644 --- a/mm/folio.c +++ b/mm/folio.c @@ -354,26 +354,28 @@ static void __lru_cache_activate_folio(struct folio *folio) static void lru_gen_inc_refs(struct folio *folio) { - unsigned long new_flags, old_flags = READ_ONCE(folio->flags.f); + unsigned long new_flags, old_flags = READ_ONCE(*folio_flags(folio, 0)); + int refs; if (folio_test_unevictable(folio)) return; /* see the comment on LRU_REFS_FLAGS */ - if (!folio_test_referenced(folio)) { - set_mask_bits(&folio->flags.f, LRU_REFS_MASK, BIT(PG_referenced)); + if (!folio_lru_refs(folio)) { + folio_set_lru_refs(folio, 1); return; } do { - if ((old_flags & LRU_REFS_MASK) == LRU_REFS_MASK) { + new_flags = old_flags; + refs = lru_get_refs_flags(old_flags); + if (refs == LRU_REFS_MAX) { if (!folio_test_workingset(folio)) folio_set_workingset(folio); return; } - - new_flags = old_flags + BIT(LRU_REFS_PGOFF); - } while (!try_cmpxchg(&folio->flags.f, &old_flags, new_flags)); + lru_set_refs_flags(&new_flags, refs + 1); + } while (!try_cmpxchg(folio_flags(folio, 0), &old_flags, new_flags)); } static bool lru_gen_clear_refs(struct folio *folio) @@ -385,7 +387,8 @@ static bool lru_gen_clear_refs(struct folio *folio) if (gen < 0) return true; - set_mask_bits(&folio->flags.f, LRU_REFS_FLAGS | BIT(PG_workingset), 0); + folio_set_lru_refs(folio, 0); + folio_clear_workingset(folio); rcu_read_lock(); seq = READ_ONCE(folio_lruvec(folio)->lrugen.min_seq[type]); diff --git a/mm/vmscan.c b/mm/vmscan.c index 245f68c75b2894..c6e03a5047aeb9 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -863,19 +863,22 @@ static bool lru_gen_set_refs(struct folio *folio, const vma_flags_t *vma_flags) if (!folio_test_referenced(folio) && !folio_test_workingset(folio)) { /* Activate file-backed executable folios after first usage. */ if (is_exec_file_folio(folio, vma_flags)) { - set_mask_bits(&folio->flags.f, LRU_REFS_FLAGS, BIT(PG_workingset)); + folio_set_workingset(folio); + folio_set_lru_refs(folio, 0); return true; } - set_mask_bits(&folio->flags.f, LRU_REFS_MASK, BIT(PG_referenced)); + folio_set_lru_refs(folio, 1); return false; } /* Promote on second access */ - if (folio_lru_refs(folio) > 1) - set_mask_bits(&folio->flags.f, LRU_REFS_FLAGS, BIT(PG_workingset)); - else + if (folio_lru_refs(folio) > 1) { + folio_set_workingset(folio); + folio_set_lru_refs(folio, 0); + } else { folio_mark_accessed(folio); + } return true; } #else @@ -3291,11 +3294,10 @@ static bool positive_ctrl_err(struct ctrl_pos *sp, struct ctrl_pos *pv) ******************************************************************************/ /* promote pages accessed through page tables */ -static int folio_update_gen(struct folio *folio, int gen, const vma_flags_t *vma_flags) +static int folio_update_gen(struct folio *folio, int new_gen, const vma_flags_t *vma_flags) { - unsigned long new_flags, old_flags = READ_ONCE(folio->flags.f); - - VM_WARN_ON_ONCE(gen >= MAX_NR_GENS); + unsigned long new_flags, old_flags = READ_ONCE(*folio_flags(folio, 0)); + int old_gen; /* * See the comment on LRU_REFS_FLAGS, and activate file-backed @@ -3304,31 +3306,34 @@ static int folio_update_gen(struct folio *folio, int gen, const vma_flags_t *vma */ if (!folio_test_referenced(folio) && !folio_test_workingset(folio) && !is_exec_file_folio(folio, vma_flags)) { - set_mask_bits(&folio->flags.f, LRU_REFS_MASK, BIT(PG_referenced)); + folio_set_lru_refs(folio, 1); return -1; } do { + old_gen = lru_get_gen_flags(old_flags); + new_flags = old_flags; + /* lru_gen_del_folio() has isolated this page? */ - if (!(old_flags & LRU_GEN_MASK)) - return -1; + if (old_gen < 0) + break; - new_flags = old_flags & ~(LRU_GEN_MASK | LRU_REFS_FLAGS); - new_flags |= ((gen + 1UL) << LRU_GEN_PGOFF) | BIT(PG_workingset); - } while (!try_cmpxchg(&folio->flags.f, &old_flags, new_flags)); + lru_set_gen_flags(&new_flags, new_gen); + lru_set_refs_flags(&new_flags, 0); + new_flags |= BIT(PG_workingset); + } while (!try_cmpxchg(folio_flags(folio, 0), &old_flags, new_flags)); - return ((old_flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1; + return old_gen; } static int __folio_inc_gen(struct folio *folio, int old_gen, bool *increased) { - unsigned long new_flags, old_flags = READ_ONCE(folio->flags.f); + unsigned long new_flags, old_flags = READ_ONCE(*folio_flags(folio, 0)); int new_gen; - VM_WARN_ON_ONCE_FOLIO(!(old_flags & LRU_GEN_MASK), folio); - do { - new_gen = ((old_flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1; + new_gen = lru_get_gen_flags(old_flags); + /* folio_update_gen() has promoted this page? */ if (new_gen >= 0 && new_gen != old_gen) { if (increased) @@ -3336,11 +3341,12 @@ static int __folio_inc_gen(struct folio *folio, int old_gen, bool *increased) return new_gen; } + new_flags = old_flags; new_gen = (old_gen + 1) % MAX_NR_GENS; - new_flags = old_flags & ~(LRU_GEN_MASK | LRU_REFS_FLAGS); - new_flags |= (new_gen + 1UL) << LRU_GEN_PGOFF; - } while (!try_cmpxchg(&folio->flags.f, &old_flags, new_flags)); + lru_set_gen_flags(&new_flags, new_gen); + lru_set_refs_flags(&new_flags, 0); + } while (!try_cmpxchg(folio_flags(folio, 0), &old_flags, new_flags)); if (increased) *increased = true; @@ -4785,7 +4791,7 @@ static bool isolate_folio(struct lruvec *lruvec, struct folio *folio, struct sca /* see the comment on LRU_REFS_FLAGS */ if (!folio_test_referenced(folio)) - set_mask_bits(&folio->flags.f, LRU_REFS_MASK, 0); + folio_set_lru_refs(folio, 0); success = lru_gen_del_folio(lruvec, folio, true); VM_WARN_ON_ONCE_FOLIO(!success, folio); @@ -5017,8 +5023,10 @@ static int evict_folios(unsigned long nr_to_scan, struct lruvec *lruvec, } /* don't add rejected folios to the oldest generation */ - if (lru_gen_folio_seq(lruvec, folio, false) == min_seq[type]) - set_mask_bits(&folio->flags.f, LRU_REFS_FLAGS, BIT(PG_active)); + if (lru_gen_folio_seq(lruvec, folio, false) == min_seq[type]) { + folio_set_lru_refs(folio, 0); + folio_set_active(folio); + } } move_folios_to_lru(&list); From fbe559a82b5909b5608a749215d8feb179b06182 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Sun, 6 Sep 2026 00:51:08 +0800 Subject: [PATCH 0786/1352] mm/migrate: copy all referenced state via folio_migrate_lru_refs folio_migrate_flags() copies PG_referenced separately from the MGLRU refs counter, which folio_migrate_refs() transfers. Yet under MGLRU, PG_referenced and the refs counter bits together describe the referenced status of a folio. Consolidate the two: rename folio_migrate_refs() to folio_migrate_lru_refs() and let it copy the complete referenced status, i.e., the MGLRU refs count including PG_referenced, or just PG_referenced for the active/inactive LRU. Drop the open-coded PG_referenced copy so the referenced status is transferred in one place. No behavior change is intended: under the active/inactive LRU the extra bits are unused, so operating on them is a noop. Transfer the reference state first, before the destination folio is marked up to date, as a best effort to retain referenced status. Link: https://lore.kernel.org/20260906-mglru-flags-cleanup-v6-3-9aacbd77d4ca@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Baoquan He Reviewed-by: Baolin Wang Acked-by: David Hildenbrand (Arm) Reviewed-by: Lian Wang Reviewed-by: Barry Song Reviewed-by: Ridong Chen Cc: Axel Rasmussen Cc: Chris Li Cc: Johannes Weiner Cc: Kairui Song Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Qi Zheng Cc: Roman Gushchin Cc: Shakeel Butt Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie Cc: Yu Zhao Cc: Zi Yan --- include/linux/mm_inline.h | 20 +++++++++++++++----- mm/migrate.c | 6 +++--- 2 files changed, 18 insertions(+), 8 deletions(-) diff --git a/include/linux/mm_inline.h b/include/linux/mm_inline.h index 3f4bd5b02b54da..047295ae6e8ac8 100644 --- a/include/linux/mm_inline.h +++ b/include/linux/mm_inline.h @@ -373,11 +373,19 @@ static inline bool lru_gen_del_folio(struct lruvec *lruvec, struct folio *folio, return true; } -static inline void folio_migrate_refs(struct folio *new, const struct folio *old) +/** + * folio_migrate_lru_refs - copy the reference state to a new folio + * @new: the destination folio + * @old: the source folio + * + * Transfer the reference state to @new during migration: the MGLRU + * refs count, including PG_referenced, or just PG_referenced for the + * active/inactive LRU. + */ +static inline void folio_migrate_lru_refs(struct folio *new, const struct folio *old) { - unsigned long refs = READ_ONCE(old->flags.f) & LRU_REFS_MASK; - - set_mask_bits(&new->flags.f, LRU_REFS_MASK, refs); + BUILD_BUG_ON(LRU_REFS_MASK & BIT(PG_referenced)); + folio_set_lru_refs(new, folio_lru_refs(old)); } #else /* !CONFIG_LRU_GEN */ @@ -406,8 +414,10 @@ static inline bool lru_gen_del_folio(struct lruvec *lruvec, struct folio *folio, return false; } -static inline void folio_migrate_refs(struct folio *new, const struct folio *old) +static inline void folio_migrate_lru_refs(struct folio *new, const struct folio *old) { + if (folio_test_referenced(old)) + folio_set_referenced(new); } #endif /* CONFIG_LRU_GEN */ diff --git a/mm/migrate.c b/mm/migrate.c index 15b45832bcfa7e..a369d0c95c3860 100644 --- a/mm/migrate.c +++ b/mm/migrate.c @@ -776,8 +776,9 @@ void folio_migrate_flags(struct folio *newfolio, struct folio *folio) { int cpupid; - if (folio_test_referenced(folio)) - folio_set_referenced(newfolio); + /* Copy the reference state, including PG_referenced */ + folio_migrate_lru_refs(newfolio, folio); + if (folio_test_uptodate(folio)) folio_mark_uptodate(newfolio); if (folio_test_clear_active(folio)) { @@ -807,7 +808,6 @@ void folio_migrate_flags(struct folio *newfolio, struct folio *folio) if (folio_test_idle(folio)) folio_set_idle(newfolio); - folio_migrate_refs(newfolio, folio); /* * Copy NUMA information to the new page, to prevent over-eager * future migrations of this same page. From 69509309070f50df242ce4f97a4ab90a3ad92b27 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Sun, 6 Sep 2026 00:51:09 +0800 Subject: [PATCH 0787/1352] mm/mglru: move max_seq read into walk_update_folio walk_pte_range(), walk_pmd_range_locked(), and lru_gen_look_around() each read lrugen->max_seq to compute the target generation used by walk_update_folio(), then pass it as a parameter. Move the read into walk_update_folio() itself so the callers no longer need to compute or pass the value. The max_seq read now happens once per folio update rather than once per walk range, so folios always get promoted to the current youngest generation. Link: https://lore.kernel.org/20260906-mglru-flags-cleanup-v6-4-9aacbd77d4ca@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Baoquan He Reviewed-by: Baolin Wang Reviewed-by: Ridong Chen Reviewed-by: Lian Wang Reviewed-by: Barry Song Cc: Axel Rasmussen Cc: Chris Li Cc: David Hildenbrand (Arm) Cc: Johannes Weiner Cc: Kairui Song Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Qi Zheng Cc: Roman Gushchin Cc: Shakeel Butt Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie Cc: Yu Zhao Cc: Zi Yan --- mm/vmscan.c | 29 ++++++++++++----------------- 1 file changed, 12 insertions(+), 17 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index c6e03a5047aeb9..4732b987aa67b8 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -3557,13 +3557,15 @@ static bool suitable_to_scan(int total, int young) } static void walk_update_folio(struct lru_gen_mm_walk *walk, struct vm_area_struct *vma, - struct folio *folio, int new_gen, bool dirty) + struct lruvec *lruvec, struct folio *folio, bool dirty) { - int old_gen; + int new_gen, old_gen; if (!folio) return; + new_gen = lru_gen_from_seq(READ_ONCE(lruvec->lrugen.max_seq)); + if (dirty && !folio_test_dirty(folio) && !(folio_test_anon(folio) && folio_test_swapbacked(folio) && !folio_test_swapcache(folio))) @@ -3594,8 +3596,6 @@ static bool walk_pte_range(pmd_t *pmd, unsigned long start, unsigned long end, struct lru_gen_mm_walk *walk = args->private; struct mem_cgroup *memcg = lruvec_memcg(walk->lruvec); struct pglist_data *pgdat = lruvec_pgdat(walk->lruvec); - DEFINE_MAX_SEQ(walk->lruvec); - int gen = lru_gen_from_seq(max_seq); unsigned int nr; pmd_t pmdval; @@ -3646,7 +3646,7 @@ static bool walk_pte_range(pmd_t *pmd, unsigned long start, unsigned long end, continue; if (last != folio) { - walk_update_folio(walk, args->vma, last, gen, dirty); + walk_update_folio(walk, args->vma, walk->lruvec, last, dirty); last = folio; dirty = false; @@ -3659,7 +3659,7 @@ static bool walk_pte_range(pmd_t *pmd, unsigned long start, unsigned long end, walk->mm_stats[MM_LEAF_YOUNG] += nr; } - walk_update_folio(walk, args->vma, last, gen, dirty); + walk_update_folio(walk, args->vma, walk->lruvec, last, dirty); last = NULL; if (i < PTRS_PER_PTE && get_next_vma(PMD_MASK, PAGE_SIZE, args, &start, &end)) @@ -3682,8 +3682,6 @@ static void walk_pmd_range_locked(pud_t *pud, unsigned long addr, struct vm_area struct lru_gen_mm_walk *walk = args->private; struct mem_cgroup *memcg = lruvec_memcg(walk->lruvec); struct pglist_data *pgdat = lruvec_pgdat(walk->lruvec); - DEFINE_MAX_SEQ(walk->lruvec); - int gen = lru_gen_from_seq(max_seq); VM_WARN_ON_ONCE(pud_leaf(*pud)); @@ -3737,7 +3735,7 @@ static void walk_pmd_range_locked(pud_t *pud, unsigned long addr, struct vm_area goto next; if (last != folio) { - walk_update_folio(walk, vma, last, gen, dirty); + walk_update_folio(walk, vma, walk->lruvec, last, dirty); last = folio; dirty = false; @@ -3751,7 +3749,7 @@ static void walk_pmd_range_locked(pud_t *pud, unsigned long addr, struct vm_area i = i > MIN_LRU_BATCH ? 0 : find_next_bit(bitmap, MIN_LRU_BATCH, i) + 1; } while (i <= MIN_LRU_BATCH); - walk_update_folio(walk, vma, last, gen, dirty); + walk_update_folio(walk, vma, walk->lruvec, last, dirty); lazy_mmu_mode_disable(); spin_unlock(ptl); @@ -4357,8 +4355,6 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr) struct pglist_data *pgdat = folio_pgdat(folio); struct lruvec *lruvec; struct lru_gen_mm_state *mm_state; - unsigned long max_seq; - int gen; lockdep_assert_held(pvmw->ptl); VM_WARN_ON_ONCE_FOLIO(folio_test_lru(folio), folio); @@ -4395,8 +4391,6 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr) memcg = get_mem_cgroup_from_folio(folio); lruvec = mem_cgroup_lruvec(memcg, pgdat); - max_seq = READ_ONCE((lruvec)->lrugen.max_seq); - gen = lru_gen_from_seq(max_seq); mm_state = get_mm_state(lruvec); lazy_mmu_mode_enable(); @@ -4428,7 +4422,7 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr) continue; if (last != folio) { - walk_update_folio(walk, vma, last, gen, dirty); + walk_update_folio(walk, vma, lruvec, last, dirty); last = folio; dirty = false; @@ -4440,13 +4434,14 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr) young += nr; } - walk_update_folio(walk, vma, last, gen, dirty); + walk_update_folio(walk, vma, lruvec, last, dirty); lazy_mmu_mode_disable(); /* feedback from rmap walkers to page table walkers */ if (mm_state && suitable_to_scan(i, young)) - update_bloom_filter(mm_state, max_seq, pvmw->pmd); + update_bloom_filter(mm_state, READ_ONCE(lruvec->lrugen.max_seq), + pvmw->pmd); mem_cgroup_put(memcg); From 52094cce1ea3ad63510a6e0b83560cd9dbe5b4b4 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Sun, 6 Sep 2026 00:51:10 +0800 Subject: [PATCH 0788/1352] mm/mglru: use explicit tier range in read_ctrl_pos() read_ctrl_pos() encodes the tier range in a single "tier" parameter via "tier % MAX_NR_TIERS" as the start and "min(tier, MAX_NR_TIERS-1)" as the end. This is hard to follow, maintain, or extend. Tier values 0..3 select a single tier, while tier == MAX_NR_TIERS selects the full range. Replace it with explicit (tier_min, tier_max) parameters using a closed [tier_min, tier_max] interval, and add LRU_TIER_MIN and LRU_TIER_MAX for the tier bounds. The call sites now become self-documenting: - get_tier_idx: (LRU_TIER_MIN, LRU_TIER_MIN) for the first tier, (tier, tier) for each subsequent tier - get_type_to_scan: (LRU_TIER_MIN, LRU_TIER_MAX) for the full range No functional change. Link: https://lore.kernel.org/20260906-mglru-flags-cleanup-v6-5-9aacbd77d4ca@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Baolin Wang Reviewed-by: Baoquan He Reviewed-by: Barry Song Reviewed-by: Ridong Chen Cc: Axel Rasmussen Cc: Chris Li Cc: David Hildenbrand (Arm) Cc: Johannes Weiner Cc: Kairui Song Cc: Liam R. Howlett Cc: Lian Wang Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Qi Zheng Cc: Roman Gushchin Cc: Shakeel Butt Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie Cc: Yu Zhao Cc: Zi Yan --- include/linux/mmzone.h | 2 ++ mm/vmscan.c | 18 ++++++++++-------- 2 files changed, 12 insertions(+), 8 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 593cc92fe47146..acd94cecc0d399 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -495,6 +495,8 @@ enum lruvec_flags { * folio->flags, masked by LRU_REFS_MASK. */ #define MAX_NR_TIERS 4U +#define LRU_TIER_MIN 0U +#define LRU_TIER_MAX (MAX_NR_TIERS - 1) #ifndef __GENERATING_BOUNDS_H diff --git a/mm/vmscan.c b/mm/vmscan.c index 4732b987aa67b8..ec5832b0d8673f 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -3223,8 +3223,8 @@ struct ctrl_pos { int gain; }; -static void read_ctrl_pos(struct lruvec *lruvec, int type, int tier, int gain, - struct ctrl_pos *pos) +static void read_ctrl_pos(struct lruvec *lruvec, int type, int tier_min, + int tier_max, int gain, struct ctrl_pos *pos) { int i; struct lru_gen_folio *lrugen = &lruvec->lrugen; @@ -3233,7 +3233,7 @@ static void read_ctrl_pos(struct lruvec *lruvec, int type, int tier, int gain, pos->gain = gain; pos->refaulted = pos->total = 0; - for (i = tier % MAX_NR_TIERS; i <= min(tier, MAX_NR_TIERS - 1); i++) { + for (i = tier_min; i <= tier_max; i++) { pos->refaulted += lrugen->avg_refaulted[type][i] + atomic_long_read(&lrugen->refaulted[hist][type][i]); pos->total += lrugen->avg_total[type][i] + @@ -4879,9 +4879,9 @@ static int get_tier_idx(struct lruvec *lruvec, int type) * This value is chosen because any other tier would have at least twice * as many refaults as the first tier. */ - read_ctrl_pos(lruvec, type, 0, 2, &sp); - for (tier = 1; tier < MAX_NR_TIERS; tier++) { - read_ctrl_pos(lruvec, type, tier, 3, &pv); + read_ctrl_pos(lruvec, type, LRU_TIER_MIN, LRU_TIER_MIN, 2, &sp); + for (tier = LRU_TIER_MIN + 1; tier <= LRU_TIER_MAX; tier++) { + read_ctrl_pos(lruvec, type, tier, tier, 3, &pv); if (!positive_ctrl_err(&sp, &pv)) break; } @@ -4902,8 +4902,10 @@ static int get_type_to_scan(struct lruvec *lruvec, int swappiness) * Compare the sum of all tiers of anon with that of file to determine * which type to scan. */ - read_ctrl_pos(lruvec, LRU_GEN_ANON, MAX_NR_TIERS, swappiness, &sp); - read_ctrl_pos(lruvec, LRU_GEN_FILE, MAX_NR_TIERS, MAX_SWAPPINESS - swappiness, &pv); + read_ctrl_pos(lruvec, LRU_GEN_ANON, LRU_TIER_MIN, LRU_TIER_MAX, + swappiness, &sp); + read_ctrl_pos(lruvec, LRU_GEN_FILE, LRU_TIER_MIN, LRU_TIER_MAX, + MAX_SWAPPINESS - swappiness, &pv); return positive_ctrl_err(&sp, &pv); } From 363d4feb1e7641156fd455e5a50d920a8bb1356f Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Sun, 6 Sep 2026 00:51:11 +0800 Subject: [PATCH 0789/1352] mm/mglru: fix potential generation folio number leak Each generation of MGLRU accounts anon and file folio numbers separately, and the page table walker updates each generation's counters in batch once the walk is done. The walker promotes a folio's generation with a cmpxchg on folio->flags, and update_batch_size() then reads the live flags again to pick the anon/file column to charge. The walk holds neither the lruvec lock nor the folio lock, so the type can flip between the cmpxchg and that read: the lazyfree path clears PG_swapbacked, and reclaim sets it back on a dirty lazyfree folio. The batched delta pair is then recorded in the wrong type column. Nothing reconciles it afterwards, permanently skewing lrugen->nr_pages and the reclaim budgets derived from it. Fix it by capturing the type from the flags snapshot the cmpxchg linearized against: folio_update_gen() returns the type of the state it transitioned from, and update_batch_size() accounts with that. A folio's type only changes while it is off the LRU list, inside a del/add pair under the lruvec lock, with the gen bits cleared in between. The generation and PG_swapbacked sit in the same folio->flags word, so the cmpxchg snapshot captures them together. Let G be the generation that snapshot captured (old_gen) and G' the one it wrote (new_gen); the CAS can land in only three places: - before the del: the folio is anon at G; the batch records anon G -> G', and the del later removes the folio from the anon counters; - between del and add: gen == -1, so folio_update_gen() returns -1 without touching the flags and no batch is recorded; the del/add pair accounts for the move alone; - after the add: the folio is file at the fresh generation the add charged; the batch records file, that gen -> G', matching that charge. Unlike the drift of lazy promotions, which sort_folio() repairs under the lruvec lock, the phantom deltas from before this fix land in a column the folio never occupies again, so nothing ever repairs them. Link: https://lore.kernel.org/20260906-mglru-flags-cleanup-v6-6-9aacbd77d4ca@tencent.com Fixes: bd74fdaea146 ("mm: multi-gen LRU: support page table walks") Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Barry Song Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Chris Li Cc: David Hildenbrand (Arm) Cc: Johannes Weiner Cc: Kairui Song Cc: Liam R. Howlett Cc: Lian Wang Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Qi Zheng Cc: Ridong Chen Cc: Roman Gushchin Cc: Shakeel Butt Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie Cc: Yu Zhao Cc: Zi Yan --- include/linux/mm_inline.h | 11 +++++++++++ mm/vmscan.c | 13 +++++++------ 2 files changed, 18 insertions(+), 6 deletions(-) diff --git a/include/linux/mm_inline.h b/include/linux/mm_inline.h index 047295ae6e8ac8..ab69b9930893ff 100644 --- a/include/linux/mm_inline.h +++ b/include/linux/mm_inline.h @@ -30,6 +30,17 @@ static inline int folio_is_file_lru(const struct folio *folio) return !folio_test_swapbacked(folio); } +/** + * folio_flags_is_file_lru - Should the folio be on a file LRU or anon LRU? + * @flags: The folio's flags. + * + * Just like folio_is_file_lru but take the folio flags directly instead. + */ +static inline int folio_flags_is_file_lru(const unsigned long *flags) +{ + return !test_bit(PG_swapbacked, flags); +} + static __always_inline void __update_lru_size(struct lruvec *lruvec, enum lru_list lru, enum zone_type zid, long nr_pages) diff --git a/mm/vmscan.c b/mm/vmscan.c index ec5832b0d8673f..40d3f1b48a74cf 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -3294,7 +3294,8 @@ static bool positive_ctrl_err(struct ctrl_pos *sp, struct ctrl_pos *pv) ******************************************************************************/ /* promote pages accessed through page tables */ -static int folio_update_gen(struct folio *folio, int new_gen, const vma_flags_t *vma_flags) +static int folio_update_gen(struct folio *folio, int new_gen, int *type, + const vma_flags_t *vma_flags) { unsigned long new_flags, old_flags = READ_ONCE(*folio_flags(folio, 0)); int old_gen; @@ -3323,6 +3324,7 @@ static int folio_update_gen(struct folio *folio, int new_gen, const vma_flags_t new_flags |= BIT(PG_workingset); } while (!try_cmpxchg(folio_flags(folio, 0), &old_flags, new_flags)); + *type = folio_flags_is_file_lru(&old_flags); return old_gen; } @@ -3368,9 +3370,8 @@ static int folio_inc_gen(struct lruvec *lruvec, struct folio *folio) } static void update_batch_size(struct lru_gen_mm_walk *walk, struct folio *folio, - int old_gen, int new_gen) + int old_gen, int new_gen, int type) { - int type = folio_is_file_lru(folio); int zone = folio_zonenum(folio); int delta = folio_nr_pages(folio); @@ -3559,7 +3560,7 @@ static bool suitable_to_scan(int total, int young) static void walk_update_folio(struct lru_gen_mm_walk *walk, struct vm_area_struct *vma, struct lruvec *lruvec, struct folio *folio, bool dirty) { - int new_gen, old_gen; + int new_gen, old_gen, type; if (!folio) return; @@ -3572,9 +3573,9 @@ static void walk_update_folio(struct lru_gen_mm_walk *walk, struct vm_area_struc folio_mark_dirty(folio); if (walk) { - old_gen = folio_update_gen(folio, new_gen, &vma->flags); + old_gen = folio_update_gen(folio, new_gen, &type, &vma->flags); if (old_gen >= 0 && old_gen != new_gen) - update_batch_size(walk, folio, old_gen, new_gen); + update_batch_size(walk, folio, old_gen, new_gen, type); } else if (lru_gen_set_refs(folio, &vma->flags)) { old_gen = folio_lru_gen(folio); if (old_gen >= 0 && old_gen != new_gen) From 0da523cab104e5d14fbab4faac18a1d94c22d808 Mon Sep 17 00:00:00 2001 From: Arnd Bergmann Date: Wed, 16 Sep 2026 10:34:46 +0200 Subject: [PATCH 0790/1352] mm/vmscan: avoid false-positive -Wuninitialized warning, again I previously worked around a false-postive gcc-16 warning in the get_tier_idx() function, by adding a fake initializer. This happens with the -fsanitize=bounds sanitizer when the compiler creates a specialized variant of isolate_folios(): In function 'get_tier_idx', inlined from 'isolate_folios.constprop' at mm/vmscan.c:4982:9: mm/vmscan.c:4934:9: error: 'sp.refaulted' is used uninitialized [-Werror=uninitialized] 4934 | read_ctrl_pos(lruvec, type, LRU_TIER_MIN, LRU_TIER_MIN, 2, &sp); | ^~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ mm/vmscan.c: In function 'isolate_folios.constprop': mm/vmscan.c:4946:25: note: 'sp.refaulted' was declared here 4946 | struct ctrl_pos sp, pv = {}; | ^~ Adding another "= {}" would solve the problem as well, but to prevent this from happening again after the next code refactoring, try instead to prevent this by forbidding interprocedural optimizations on this function. Link: https://lore.kernel.org/all/20260213123902.3466040-1-arnd@kernel.org/ Link: https://lore.kernel.org/20260916083456.4136132-1-arnd@kernel.org Fixes: 3de705a43a46 ("mm/vmscan: avoid false-positive -Wuninitialized warning") Signed-off-by: Arnd Bergmann Signed-off-by: Andrew Morton Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie Cc: --- mm/vmscan.c | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 40d3f1b48a74cf..4059d130078960 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -3223,8 +3223,11 @@ struct ctrl_pos { int gain; }; -static void read_ctrl_pos(struct lruvec *lruvec, int type, int tier_min, - int tier_max, int gain, struct ctrl_pos *pos) +/* + * __noipa works around gcc-16 warning for uninitialized use of pos->refaulted + */ +static void __noipa read_ctrl_pos(struct lruvec *lruvec, int type, int tier_min, + int tier_max, int gain, struct ctrl_pos *pos) { int i; struct lru_gen_folio *lrugen = &lruvec->lrugen; @@ -4873,7 +4876,7 @@ static int scan_folios(unsigned long nr_to_scan, struct lruvec *lruvec, static int get_tier_idx(struct lruvec *lruvec, int type) { int tier; - struct ctrl_pos sp, pv = {}; + struct ctrl_pos sp, pv; /* * To leave a margin for fluctuations, use a larger gain factor (2:3). @@ -4892,7 +4895,7 @@ static int get_tier_idx(struct lruvec *lruvec, int type) static int get_type_to_scan(struct lruvec *lruvec, int swappiness) { - struct ctrl_pos sp, pv = {}; + struct ctrl_pos sp, pv; if (swappiness <= MIN_SWAPPINESS + 1) return LRU_GEN_FILE; From c16dd2594b5209171946540b3d7dcdc51a992551 Mon Sep 17 00:00:00 2001 From: Luxiao Xu Date: Thu, 10 Sep 2026 11:42:50 +0800 Subject: [PATCH 0791/1352] mm/memory: constrain generic_access_phys() to page boundary This patch addresses an issue in generic_access_phys() where accessing memory across page boundaries in PFNMAP VMAs assumes physical pages are contiguous, which can lead to accessing unintended physical memory or exceeding VMA boundaries. generic_access_phys() improperly validates the memory access range: it only validates the start address using follow_pfnmap_start() and passes PAGE_ALIGN(len + offset) to ioremap_prot(). This poses two problems: 1. In PFNMAP VMAs, consecutive virtual pages are not guaranteed to be physically contiguous, and individual PTEs may have different access permissions or writability. 2. The mapping may cross VMA boundaries if len extends beyond vma->vm_end. Constrain the access in generic_access_phys() to at most the current page boundary (PAGE_SIZE - offset) and map only a single PAGE_SIZE via ioremap_prot(). Since the caller __access_remote_vm() already loops over the requested length and handles partial transfers, it will naturally iterate over the remaining pages. Also add a missing (resource_size_t) cast during PFN re-validation to avoid truncation on 32-bit PAE systems. Link: https://lore.kernel.org/e06e28a46c2a176238f03b5740df0913e57c2861.1788842306.git.rakukuip@gmail.com Fixes: 9cb12d7b4cca ("mm/memory.c: actually remap enough memory") Signed-off-by: Luxiao Xu Signed-off-by: Ren Wei Signed-off-by: Andrew Morton Reported-by: Vega Suggested-by: David Hildenbrand Acked-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Grazvydas Ignotas Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: --- mm/memory.c | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/mm/memory.c b/mm/memory.c index ec63dd6212ac5a..926276d4192026 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -7131,6 +7131,12 @@ int generic_access_phys(struct vm_area_struct *vma, unsigned long addr, bool writable; struct follow_pfnmap_args args = { .vma = vma, .address = addr }; + /* + * Limit access to one page at a time, as that's what follow_pfnmap_start() + * guarantees; expect the caller to retry to read larger ranges. + */ + len = min_t(int, len, PAGE_SIZE - offset); + retry: if (follow_pfnmap_start(&args)) return -EINVAL; @@ -7142,7 +7148,7 @@ int generic_access_phys(struct vm_area_struct *vma, unsigned long addr, if ((write & FOLL_WRITE) && !writable) return -EINVAL; - maddr = ioremap_prot(phys_addr, PAGE_ALIGN(len + offset), prot); + maddr = ioremap_prot(phys_addr, PAGE_SIZE, prot); if (!maddr) return -ENOMEM; @@ -7150,7 +7156,7 @@ int generic_access_phys(struct vm_area_struct *vma, unsigned long addr, goto out_unmap; if ((pgprot_val(prot) != pgprot_val(args.pgprot)) || - (phys_addr != (args.pfn << PAGE_SHIFT)) || + (phys_addr != ((resource_size_t)args.pfn << PAGE_SHIFT)) || (writable != args.writable)) { follow_pfnmap_end(&args); iounmap(maddr); From d8e9399f6f5f1299f47606992c38f3867a209900 Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Sat, 5 Sep 2026 10:40:34 +0200 Subject: [PATCH 0792/1352] docs/mm: ksm: use the renamed ksm structure names The Reference section of ksm.rst asks mm/ksm.c for mm_slot, stable_node and rmap_item. Commit 21fbd59136e0 ("ksm: add the ksm prefix to the names of the ksm private structures") renamed them to ksm_mm_slot, ksm_stable_node and ksm_rmap_item. Since then only ksm_scan is rendered and the three structures are missing from the generated documentation. Use the current names. Link: https://lore.kernel.org/20260905084034.39521-1-kmehltretter@gmail.com Fixes: 21fbd59136e0 ("ksm: add the ksm prefix to the names of the ksm private structures") Signed-off-by: Karl Mehltretter Signed-off-by: Andrew Morton Reviewed-by: Randy Dunlap Acked-by: David Hildenbrand (Arm) Assisted-by: LLM --- Documentation/mm/ksm.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Documentation/mm/ksm.rst b/Documentation/mm/ksm.rst index 2b4f72f1f95343..5a80573162def0 100644 --- a/Documentation/mm/ksm.rst +++ b/Documentation/mm/ksm.rst @@ -78,7 +78,7 @@ The frequency of such scans is defined by Reference --------- .. kernel-doc:: mm/ksm.c - :functions: mm_slot ksm_scan stable_node rmap_item + :functions: ksm_mm_slot ksm_scan ksm_stable_node ksm_rmap_item -- Izik Eidus, From caf32ab4885b7e1199e4da45d8132493309151e4 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Fri, 4 Sep 2026 20:05:17 -0700 Subject: [PATCH 0793/1352] memcg: move per-node objcg to the read-mostly fields Patch series "memcg: group struct fields by access pattern". Every so often we get a memcg performance regression caused by nothing more than a field moving. Someone adds a field, removes one, or puts a few behind a config option. The layout shifts, fields with different access patterns land on the same cache line, and a bot reports a regression. Two examples. commit 98c9daf5ae6b ("mm: memcg: guard memcg1-specific members of struct mem_cgroup_per_node") moved lruvec next to lru_zone_size[] and needed commit f59adcf59332 ("mm: memcg: add cacheline padding after lruvec in mem_cgroup_per_node") to fix it. commit c1afbd5de131 ("mm/memcontrol: avoid false sharing between vmstats and events") had to add ____cacheline_aligned_in_smp for the same reason. Each fix was correct but nothing stops the next field addition from undoing it. This series makes the layout a contract the compiler checks, the same way struct net_device does it. Fields are sorted into named cache line groups by access pattern, and memcg_struct_check() verifies at build time that every field sits in its group. A field added in the wrong place now breaks the build instead of quietly costing a few percent. struct mem_cgroup gets three groups: memcg_write_hot written on the charge, reclaim and socket paths memcg_cold only the cgroup control paths touch these memcg_read_mostly set when the memcg is created, then only read Testing ======= The cgroup selftests give identical results with and without the series. For performance, two identical 30 core Xeon machines each ran both kernels, with the boot order swapped between them so that machine and order effects cancel. The useful tests run two workloads at once in one cgroup, because false sharing only shows up when one side reads a field that the other side writes. slab allocs + page faults, slab side +1.2% page faults + memory.stat readers, fault side +1.3% page faults + memory.stat readers, reader side +0.8% everything else no change No test regressed. The gains are small but the point of the series is the build time contract. This patch (of 6): current_obj_cgroup() reads memcg->nodeinfo[nid]->objcg on every accounted allocation. The field sits at the end of struct mem_cgroup_per_node, on the same cache line as lru_zone_size[] and iter. lru_zone_size[] is written on every LRU add and remove, and iter is written on every reclaim iteration. Move objcg next to the other read-mostly pointers at the start of the struct. No functional change. Link: https://lore.kernel.org/20260905030522.1887837-1-shakeel.butt@linux.dev Link: https://lore.kernel.org/20260905030522.1887837-2-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin --- include/linux/memcontrol.h | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index a03b6e3e547078..cfa1877915d45b 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -94,6 +94,7 @@ struct mem_cgroup_per_node { struct lruvec_stats_percpu __percpu *lruvec_stats_percpu; struct lruvec_stats *lruvec_stats; struct shrinker_info __rcu *shrinker_info; + struct obj_cgroup __rcu *objcg; CACHELINE_PADDING(_pad1_); @@ -104,11 +105,9 @@ struct mem_cgroup_per_node { struct mem_cgroup_reclaim_iter iter; /* - * objcg is wiped out as a part of the objcg repaprenting process. * orig_objcg preserves a pointer (and a reference) to the original - * objcg until the end of live of memcg. + * objcg until the end of life of memcg. */ - struct obj_cgroup __rcu *objcg; struct obj_cgroup *orig_objcg; /* list of inherited objcgs, protected by objcg_lock */ struct list_head objcg_list; From 3e0d297c3dece9110c471ad2b11952a4697db801 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Fri, 4 Sep 2026 20:05:18 -0700 Subject: [PATCH 0794/1352] memcg: split mem_cgroup_private_id into two fields The two members of struct mem_cgroup_private_id have different access patterns. The id is read on every eviction and refault through mem_cgroup_private_id(), and is only written when the memcg is created and destroyed. The ref is written on every swap charge and uncharge. Split them into private_id and private_id_ref so a later patch can put them into different cache line groups. A struct member cannot be split across two groups. No functional change. Link: https://lore.kernel.org/20260905030522.1887837-3-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin --- include/linux/memcontrol.h | 10 +++------- mm/memcontrol.c | 18 +++++++++--------- 2 files changed, 12 insertions(+), 16 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index cfa1877915d45b..32a959f1890abd 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -66,11 +66,6 @@ struct mem_cgroup_reclaim_cookie { #define MEM_CGROUP_ID_SHIFT 16 -struct mem_cgroup_private_id { - int id; - refcount_t ref; -}; - struct memcg_vmstats_percpu; struct memcg1_events_percpu; struct memcg_vmstats; @@ -189,7 +184,8 @@ struct mem_cgroup { struct cgroup_subsys_state css; /* Private memcg ID. Used to ID objects that outlive the cgroup */ - struct mem_cgroup_private_id id; + int private_id; + refcount_t private_id_ref; /* Accounted resources */ struct page_counter memory; /* Both v1 & v2 */ @@ -810,7 +806,7 @@ static inline unsigned short mem_cgroup_private_id(struct mem_cgroup *memcg) if (mem_cgroup_disabled()) return 0; - return memcg->id.id; + return memcg->private_id; } struct mem_cgroup *mem_cgroup_from_private_id(unsigned short id); diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 4cb2db8c0923a4..2f11a3a79a6bc9 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -3773,7 +3773,7 @@ static void memcg_online_kmem(struct mem_cgroup *memcg) static_branch_enable(&memcg_kmem_online_key); - memcg->kmemcg_id = memcg->id.id; + memcg->kmemcg_id = memcg->private_id; } static void memcg_offline_kmem(struct mem_cgroup *memcg) @@ -4032,15 +4032,15 @@ static DEFINE_XARRAY_ALLOC1(mem_cgroup_private_ids); static void mem_cgroup_private_id_remove(struct mem_cgroup *memcg) { - if (memcg->id.id > 0) { - xa_erase(&mem_cgroup_private_ids, memcg->id.id); - memcg->id.id = 0; + if (memcg->private_id > 0) { + xa_erase(&mem_cgroup_private_ids, memcg->private_id); + memcg->private_id = 0; } } static inline void mem_cgroup_private_id_put(struct mem_cgroup *memcg, unsigned int n) { - if (refcount_sub_and_test(n, &memcg->id.ref)) { + if (refcount_sub_and_test(n, &memcg->private_id_ref)) { mem_cgroup_private_id_remove(memcg); /* Memcg ID pins CSS */ @@ -4050,7 +4050,7 @@ static inline void mem_cgroup_private_id_put(struct mem_cgroup *memcg, unsigned struct mem_cgroup *mem_cgroup_private_id_get_online(struct mem_cgroup *memcg, unsigned int n) { - while (!refcount_add_not_zero(n, &memcg->id.ref)) { + while (!refcount_add_not_zero(n, &memcg->private_id_ref)) { /* * The root cgroup cannot be destroyed, so it's refcount must * always be >= 1. @@ -4174,7 +4174,7 @@ static struct mem_cgroup *mem_cgroup_alloc(struct mem_cgroup *parent) if (!memcg) return ERR_PTR(-ENOMEM); - error = xa_alloc(&mem_cgroup_private_ids, &memcg->id.id, NULL, + error = xa_alloc(&mem_cgroup_private_ids, &memcg->private_id, NULL, XA_LIMIT(1, MEM_CGROUP_ID_MAX), GFP_KERNEL); if (error) goto fail; @@ -4320,7 +4320,7 @@ static int mem_cgroup_css_online(struct cgroup_subsys_state *css) lru_gen_online_memcg(memcg); /* Online state pins memcg ID, memcg ID pins CSS */ - refcount_set(&memcg->id.ref, 1); + refcount_set(&memcg->private_id_ref, 1); css_get(css); /* @@ -4333,7 +4333,7 @@ static int mem_cgroup_css_online(struct cgroup_subsys_state *css) * publish it here at the end of onlining. This matches the * regular ID destruction during offlining. */ - xa_store(&mem_cgroup_private_ids, memcg->id.id, memcg, GFP_KERNEL); + xa_store(&mem_cgroup_private_ids, memcg->private_id, memcg, GFP_KERNEL); return 0; free_objcg: From 403a8a19d6613c9e99edd04b07f70ab1c8972786 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Fri, 4 Sep 2026 20:05:19 -0700 Subject: [PATCH 0795/1352] memcg: group the write-hot fields of struct mem_cgroup These fields are written on the charge, reclaim and socket paths: socket_pressure written by reclaim, read on every socket charge memory_events bumped for this memcg and every ancestor, so a busy child dirties the whole chain memory_events_local vmpressure written on every reclaim iteration private_id_ref written on every swap charge and uncharge kmem_stat high_irq_work, high_work They are spread over the struct today and share cache lines with read-mostly fields. Put them in one cache line group. socket_pressure is kept next to memory_events because mem_cgroup_sk_under_memory_pressure() reads one and bumps the other. Add memcg_struct_check() so the build fails if a field lands outside its group. No functional change. Link: https://lore.kernel.org/20260905030522.1887837-4-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin --- include/linux/memcontrol.h | 58 ++++++++++++++++++++++---------------- mm/memcontrol.c | 32 +++++++++++++++++++++ 2 files changed, 66 insertions(+), 24 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 32a959f1890abd..66431981f4f764 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -185,7 +185,6 @@ struct mem_cgroup { /* Private memcg ID. Used to ID objects that outlive the cgroup */ int private_id; - refcount_t private_id_ref; /* Accounted resources */ struct page_counter memory; /* Both v1 & v2 */ @@ -195,14 +194,45 @@ struct mem_cgroup { struct page_counter memsw; /* v1 only */ }; + /* Written on the charge, reclaim and socket paths. */ + __cacheline_group_begin_aligned(memcg_write_hot); + /* + * Hint of reclaim pressure for socket memory management. Note + * that this indicator should NOT be used in legacy cgroup mode + * where socket memory is accounted/charged separately. + */ + u64 socket_pressure; +#if BITS_PER_LONG < 64 + seqlock_t socket_pressure_seqlock; +#endif + /* + * memory.events is bumped for this memcg and all its ancestors, so a + * busy child dirties every ancestor. + */ + atomic_long_t memory_events[MEMCG_NR_MEMORY_EVENTS]; + atomic_long_t memory_events_local[MEMCG_NR_MEMORY_EVENTS]; + + /* vmpressure notifications. Written on every reclaim iteration. */ + struct vmpressure vmpressure; + + /* Written on every swap charge and uncharge. */ + refcount_t private_id_ref; + +#ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC + /* MEMCG_KMEM for nmi context */ + atomic_t kmem_stat; +#endif + + /* Range enforcement for interrupt charges */ + struct work_struct high_work; + + __cacheline_group_end_aligned(memcg_write_hot); + /* registered local peak watchers */ struct list_head memory_peaks; struct list_head swap_peaks; spinlock_t peaks_lock; - /* Range enforcement for interrupt charges */ - struct work_struct high_work; - #ifdef CONFIG_ZSWAP unsigned long zswap_max; @@ -213,9 +243,6 @@ struct mem_cgroup { bool zswap_writeback; #endif - /* vmpressure notifications */ - struct vmpressure vmpressure; - /* * Should the OOM killer kill all belonging tasks, had it kill one? */ @@ -231,23 +258,6 @@ struct mem_cgroup { /* memory.stat */ struct memcg_vmstats *vmstats; - /* memory.events */ - atomic_long_t memory_events[MEMCG_NR_MEMORY_EVENTS]; - atomic_long_t memory_events_local[MEMCG_NR_MEMORY_EVENTS]; - -#ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC - /* MEMCG_KMEM for nmi context */ - atomic_t kmem_stat; -#endif - /* - * Hint of reclaim pressure for socket memroy management. Note - * that this indicator should NOT be used in legacy cgroup mode - * where socket memory is accounted/charged separately. - */ - u64 socket_pressure; -#if BITS_PER_LONG < 64 - seqlock_t socket_pressure_seqlock; -#endif int kmemcg_id; #ifdef CONFIG_CGROUP_WRITEBACK diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 2f11a3a79a6bc9..56fc6d8c02a86a 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -5700,6 +5700,36 @@ __setup("cgroup.memory=", cgroup_memory); * basically everything that doesn't depend on a specific mem_cgroup structure * should be initialized from here. */ +/* + * Fields are grouped by access pattern. Putting a field in the wrong group + * breaks the build here. + */ +static void __init memcg_struct_check(void) +{ + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, + socket_pressure); +#if BITS_PER_LONG < 64 + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, + socket_pressure_seqlock); +#endif + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, + memory_events); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, + memory_events_local); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, + vmpressure); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, + private_id_ref); +#ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, + kmem_stat); +#endif + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, + high_irq_work); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, + high_work); +} + int __init mem_cgroup_init(void) { unsigned int memcg_size; @@ -5713,6 +5743,8 @@ int __init mem_cgroup_init(void) */ BUILD_BUG_ON(MEMCG_CHARGE_BATCH > S32_MAX / PAGE_SIZE); + memcg_struct_check(); + cpuhp_setup_state_nocalls(CPUHP_MM_MEMCQ_DEAD, "mm/memctrl:dead", NULL, memcg_hotplug_cpu_dead); From ee91be3f08b48dfa1d99b62c4e62712c4a024d2c Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Fri, 4 Sep 2026 20:05:20 -0700 Subject: [PATCH 0796/1352] memcg: group the cold fields of struct mem_cgroup These fields are only touched by the cgroup control paths: memory_peaks, swap_peaks, peaks_lock memory.peak open/read/release events_file, events_local_file, swap_events_file cgroup_file_notify() cgwb_list, cgwb_domain, cgwb_frn writeback setup and the foreign dirty slow path mm_list MGLRU mm list They sit in the middle of the struct today. The three cgroup_file members alone are 192 bytes of notify state next to the vmstats pointer. Put them in one cache line group. No functional change. Link: https://lore.kernel.org/20260905030522.1887837-5-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin --- include/linux/memcontrol.h | 49 ++++++++++++++++++++++---------------- mm/memcontrol.c | 25 +++++++++++++++++++ 2 files changed, 53 insertions(+), 21 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 66431981f4f764..3a7f8097130258 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -228,11 +228,37 @@ struct mem_cgroup { __cacheline_group_end_aligned(memcg_write_hot); + /* + * Off the charge and fault paths. Not write free: cgwb_domain is + * written on every writeout completion and mm_list on fork, exit and + * MGLRU aging. They are grouped here so those writes cannot land on + * a line that the fast paths read. + */ + __cacheline_group_begin_aligned(memcg_cold); /* registered local peak watchers */ struct list_head memory_peaks; struct list_head swap_peaks; spinlock_t peaks_lock; + /* memory.events and memory.events.local */ + struct cgroup_file events_file; + struct cgroup_file events_local_file; + + /* handle for "memory.swap.events" */ + struct cgroup_file swap_events_file; + +#ifdef CONFIG_CGROUP_WRITEBACK + struct list_head cgwb_list; + struct wb_domain cgwb_domain; + struct memcg_cgwb_frn cgwb_frn[MEMCG_CGWB_FRN_CNT]; +#endif + +#ifdef CONFIG_LRU_GEN_WALKS_MMU + /* per-memcg mm_struct list */ + struct lru_gen_mm_list mm_list; +#endif + __cacheline_group_end_aligned(memcg_cold); + #ifdef CONFIG_ZSWAP unsigned long zswap_max; @@ -248,37 +274,18 @@ struct mem_cgroup { */ bool oom_group; - /* memory.events and memory.events.local */ - struct cgroup_file events_file; - struct cgroup_file events_local_file; - - /* handle for "memory.swap.events" */ - struct cgroup_file swap_events_file; - /* memory.stat */ struct memcg_vmstats *vmstats; int kmemcg_id; -#ifdef CONFIG_CGROUP_WRITEBACK - struct list_head cgwb_list; -#endif - /* Keep the hot per-CPU stats pointer away from memory event counters. */ struct memcg_vmstats_percpu __percpu *vmstats_percpu ____cacheline_aligned_in_smp; -#ifdef CONFIG_CGROUP_WRITEBACK - struct wb_domain cgwb_domain; - struct memcg_cgwb_frn cgwb_frn[MEMCG_CGWB_FRN_CNT]; -#endif - -#ifdef CONFIG_LRU_GEN_WALKS_MMU - /* per-memcg mm_struct list */ - struct lru_gen_mm_list mm_list; -#endif - #ifdef CONFIG_MEMCG_V1 + /* v1 only. Not grouped: v1 is legacy, sorting it is not worth it. */ + /* Legacy consumer-oriented counters */ struct page_counter kmem; /* v1 only */ struct page_counter tcpmem; /* v1 only */ diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 56fc6d8c02a86a..51539265bfa9d5 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -5728,6 +5728,31 @@ static void __init memcg_struct_check(void) high_irq_work); CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, high_work); + + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_cold, + memory_peaks); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_cold, + swap_peaks); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_cold, + peaks_lock); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_cold, + events_file); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_cold, + events_local_file); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_cold, + swap_events_file); +#ifdef CONFIG_CGROUP_WRITEBACK + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_cold, + cgwb_list); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_cold, + cgwb_domain); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_cold, + cgwb_frn); +#endif +#ifdef CONFIG_LRU_GEN_WALKS_MMU + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_cold, + mm_list); +#endif } int __init mem_cgroup_init(void) From ad34141cd21312293e6b1599e572b16a1a019ae0 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Fri, 4 Sep 2026 20:05:21 -0700 Subject: [PATCH 0797/1352] memcg: group the read-mostly fields of struct mem_cgroup These fields are set when the memcg is created and only read after that: vmstats_percpu read on every stat update vmstats zswap_max, zswap_writeback private_id read on every eviction and refault kmemcg_id read on every list_lru lookup oom_group Put them in one cache line group at the end of the struct, right before nodeinfo[]. nodeinfo[] is read-mostly too but it is a flexible array, so it cannot sit inside a group. The group ends without padding so the two share a line. This also drops the ____cacheline_aligned_in_smp on vmstats_percpu added by commit c1afbd5de131 ("mm/memcontrol: avoid false sharing between vmstats and events"). That only aligned the start of the field. cgwb_domain followed it on the same line and is written on every writeout completion. A group boundary covers both sides. No functional change. Link: https://lore.kernel.org/20260905030522.1887837-6-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin --- include/linux/memcontrol.h | 62 +++++++++++++++++++++----------------- mm/memcontrol.c | 17 +++++++++++ 2 files changed, 52 insertions(+), 27 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 3a7f8097130258..73594783a2ef0e 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -183,9 +183,6 @@ struct obj_cgroup { struct mem_cgroup { struct cgroup_subsys_state css; - /* Private memcg ID. Used to ID objects that outlive the cgroup */ - int private_id; - /* Accounted resources */ struct page_counter memory; /* Both v1 & v2 */ @@ -259,30 +256,6 @@ struct mem_cgroup { #endif __cacheline_group_end_aligned(memcg_cold); -#ifdef CONFIG_ZSWAP - unsigned long zswap_max; - - /* - * Prevent pages from this memcg from being written back from zswap to - * swap, and from being swapped out on zswap store failures. - */ - bool zswap_writeback; -#endif - - /* - * Should the OOM killer kill all belonging tasks, had it kill one? - */ - bool oom_group; - - /* memory.stat */ - struct memcg_vmstats *vmstats; - - int kmemcg_id; - - /* Keep the hot per-CPU stats pointer away from memory event counters. */ - struct memcg_vmstats_percpu __percpu *vmstats_percpu - ____cacheline_aligned_in_smp; - #ifdef CONFIG_MEMCG_V1 /* v1 only. Not grouped: v1 is legacy, sorting it is not worth it. */ @@ -322,6 +295,41 @@ struct mem_cgroup { int swappiness; #endif /* CONFIG_MEMCG_V1 */ + /* + * Set when the memcg is created and cleared when it is offlined. + * Never written on a hot path. + */ + __cacheline_group_begin_aligned(memcg_read_mostly); + /* Read on every stat update */ + struct memcg_vmstats_percpu __percpu *vmstats_percpu; + + /* memory.stat */ + struct memcg_vmstats *vmstats; + +#ifdef CONFIG_ZSWAP + unsigned long zswap_max; +#endif + + /* Private memcg ID. Used to ID objects that outlive the cgroup */ + int private_id; + + int kmemcg_id; + + /* + * Should the OOM killer kill all belonging tasks, had it kill one? + */ + bool oom_group; + +#ifdef CONFIG_ZSWAP + /* + * Prevent pages from this memcg from being written back from zswap to + * swap, and from being swapped out on zswap store failures. + */ + bool zswap_writeback; +#endif + /* Not padded: nodeinfo[] is read-mostly too, let it share the line. */ + __cacheline_group_end(memcg_read_mostly); + struct mem_cgroup_per_node *nodeinfo[]; }; diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 51539265bfa9d5..a392f06a521344 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -5753,6 +5753,23 @@ static void __init memcg_struct_check(void) CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_cold, mm_list); #endif + + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_read_mostly, + vmstats_percpu); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_read_mostly, + vmstats); +#ifdef CONFIG_ZSWAP + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_read_mostly, + zswap_max); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_read_mostly, + zswap_writeback); +#endif + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_read_mostly, + private_id); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_read_mostly, + kmemcg_id); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_read_mostly, + oom_group); } int __init mem_cgroup_init(void) From c596acd5bf0a07ef4f43d02c3e46a9702f996c79 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Fri, 4 Sep 2026 20:05:22 -0700 Subject: [PATCH 0798/1352] memcg: group the fields of struct mem_cgroup_per_node Replace the two ad-hoc CACHELINE_PADDING members with named cache line groups: memcg_pn_read_mostly memcg, lruvec_stats_percpu, lruvec_stats, shrinker_info, objcg memcg_pn_lruvec lruvec memcg_pn_write_hot lru_zone_size, iter, nmi slab stats memcg_pn_cold orig_objcg, objcg_list The group markers give the same isolation the padding did, but they are named and the build now checks them. lruvec still gets its own lines. Commit f59adcf59332 ("mm: memcg: add cacheline padding after lruvec in mem_cgroup_per_node") showed why that matters: lru_zone_size[] is written under lru_lock but read without it by lruvec_lru_size(), so it must not share a line with lruvec. Splitting the cold fields out costs one extra cache line per node per memcg. No functional change. Link: https://lore.kernel.org/20260905030522.1887837-7-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin --- include/linux/memcontrol.h | 32 ++++++++++++++++++++++---------- mm/memcontrol.c | 30 ++++++++++++++++++++++++++++++ 2 files changed, 52 insertions(+), 10 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 73594783a2ef0e..c0c9805b6f0320 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -82,7 +82,8 @@ struct mem_cgroup_reclaim_iter { * per-node information in memory controller. */ struct mem_cgroup_per_node { - /* Keep the read-only fields at the start */ + /* Set when the memcg is created, then only read. */ + __cacheline_group_begin_aligned(memcg_pn_read_mostly); struct mem_cgroup *memcg; /* Back pointer, we cannot */ /* use container_of */ @@ -91,14 +92,30 @@ struct mem_cgroup_per_node { struct shrinker_info __rcu *shrinker_info; struct obj_cgroup __rcu *objcg; - CACHELINE_PADDING(_pad1_); + __cacheline_group_end_aligned(memcg_pn_read_mostly); - /* Fields which get updated often at the end. */ + /* + * Keep lruvec on its own lines. Sharing them with lru_zone_size[] + * regressed, see commit f59adcf59332 ("mm: memcg: add cacheline + * padding after lruvec in mem_cgroup_per_node"). + */ + __cacheline_group_begin_aligned(memcg_pn_lruvec); struct lruvec lruvec; - CACHELINE_PADDING(_pad2_); + __cacheline_group_end_aligned(memcg_pn_lruvec); + + /* Written on every LRU update and on every reclaim iteration. */ + __cacheline_group_begin_aligned(memcg_pn_write_hot); long lru_zone_size[MAX_NR_ZONES][NR_LRU_LISTS]; struct mem_cgroup_reclaim_iter iter; +#ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC + /* slab stats for nmi context */ + atomic_t slab_reclaimable; + atomic_t slab_unreclaimable; +#endif + __cacheline_group_end_aligned(memcg_pn_write_hot); + /* Touched only when the memcg is reparented or freed. */ + __cacheline_group_begin_aligned(memcg_pn_cold); /* * orig_objcg preserves a pointer (and a reference) to the original * objcg until the end of life of memcg. @@ -106,12 +123,7 @@ struct mem_cgroup_per_node { struct obj_cgroup *orig_objcg; /* list of inherited objcgs, protected by objcg_lock */ struct list_head objcg_list; - -#ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC - /* slab stats for nmi context */ - atomic_t slab_reclaimable; - atomic_t slab_unreclaimable; -#endif + __cacheline_group_end_aligned(memcg_pn_cold); }; struct mem_cgroup_threshold { diff --git a/mm/memcontrol.c b/mm/memcontrol.c index a392f06a521344..e539881f58f147 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -5770,6 +5770,36 @@ static void __init memcg_struct_check(void) kmemcg_id); CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_read_mostly, oom_group); + + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node, + memcg_pn_read_mostly, memcg); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node, + memcg_pn_read_mostly, lruvec_stats_percpu); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node, + memcg_pn_read_mostly, lruvec_stats); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node, + memcg_pn_read_mostly, shrinker_info); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node, + memcg_pn_read_mostly, objcg); + + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node, + memcg_pn_lruvec, lruvec); + + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node, + memcg_pn_write_hot, lru_zone_size); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node, + memcg_pn_write_hot, iter); +#ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node, + memcg_pn_write_hot, slab_reclaimable); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node, + memcg_pn_write_hot, slab_unreclaimable); +#endif + + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node, + memcg_pn_cold, orig_objcg); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node, + memcg_pn_cold, objcg_list); } int __init mem_cgroup_init(void) From 31826fd2e5ccf8bde34a850d87d54fe75e2d8627 Mon Sep 17 00:00:00 2001 From: Wilson Felipe Pereira Date: Fri, 4 Sep 2026 23:20:47 +0000 Subject: [PATCH 0799/1352] mm/zswap: convert zswap_store_page() and zswap_compress() to take a folio Currently, zswap_store() iterates through a folio and extracts individual struct page pointers using folio_page() to pass into zswap_store_page() and zswap_compress(). Pass the folio and the page index directly into these functions instead. This eliminates the need to materialize struct page pointers inside the zswap_store() loop and replaces legacy page-based accessors with their folio equivalents: - sg_set_page() -> sg_set_folio() - kmap_local_page() -> kmap_local_folio() - page_to_nid() -> folio_nid() - page_swap_entry() -> calculated via folio->swap and index Link: https://lore.kernel.org/20260904232108.3034333-1-wfelipe@google.com Signed-off-by: Wilson Felipe Pereira Signed-off-by: Andrew Morton Reviewed-by: Tal Zussman Cc: Chengming Zhou Cc: Johannes Weiner Cc: Nhat Pham --- mm/zswap.c | 26 ++++++++++++-------------- 1 file changed, 12 insertions(+), 14 deletions(-) diff --git a/mm/zswap.c b/mm/zswap.c index f3ae3c81e48eac..b894de1786fd3f 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -824,8 +824,8 @@ static int zswap_cpu_comp_prepare(unsigned int cpu, struct hlist_node *node) return ret; } -static bool zswap_compress(struct page *page, struct zswap_entry *entry, - struct zswap_pool *pool) +static bool zswap_compress(struct folio *folio, long index, + struct zswap_entry *entry, struct zswap_pool *pool) { struct crypto_acomp_ctx *acomp_ctx; struct scatterlist input, output; @@ -841,7 +841,7 @@ static bool zswap_compress(struct page *page, struct zswap_entry *entry, dst = acomp_ctx->buffer; sg_init_table(&input, 1); - sg_set_page(&input, page, PAGE_SIZE, 0); + sg_set_folio(&input, folio, PAGE_SIZE, index * PAGE_SIZE); sg_init_one(&output, dst, PAGE_SIZE); acomp_request_set_params(acomp_ctx->req, &input, &output, PAGE_SIZE, dlen); @@ -870,8 +870,7 @@ static bool zswap_compress(struct page *page, struct zswap_entry *entry, */ if (comp_ret || !dlen || dlen >= PAGE_SIZE) { rcu_read_lock(); - if (!mem_cgroup_zswap_writeback_enabled( - folio_memcg(page_folio(page)))) { + if (!mem_cgroup_zswap_writeback_enabled(folio_memcg(folio))) { rcu_read_unlock(); comp_ret = comp_ret ? comp_ret : -EINVAL; goto unlock; @@ -879,12 +878,12 @@ static bool zswap_compress(struct page *page, struct zswap_entry *entry, rcu_read_unlock(); comp_ret = 0; dlen = PAGE_SIZE; - dst = kmap_local_page(page); + dst = kmap_local_folio(folio, index * PAGE_SIZE); mapped = true; } gfp = GFP_NOWAIT | __GFP_NORETRY | __GFP_HIGHMEM | __GFP_MOVABLE; - handle = zs_malloc(pool->zs_pool, dlen, gfp, page_to_nid(page)); + handle = zs_malloc(pool->zs_pool, dlen, gfp, folio_nid(folio)); if (IS_ERR_VALUE(handle)) { alloc_ret = PTR_ERR((void *)handle); goto unlock; @@ -1392,21 +1391,22 @@ static void shrink_worker(struct work_struct *w) * main API **********************************/ -static bool zswap_store_page(struct page *page, +static bool zswap_store_page(struct folio *folio, long index, struct obj_cgroup *objcg, struct zswap_pool *pool) { - swp_entry_t page_swpentry = page_swap_entry(page); + swp_entry_t page_swpentry = swp_entry(swp_type(folio->swap), + swp_offset(folio->swap) + index); struct zswap_entry *entry, *old; /* allocate entry */ - entry = zswap_entry_cache_alloc(GFP_KERNEL, page_to_nid(page)); + entry = zswap_entry_cache_alloc(GFP_KERNEL, folio_nid(folio)); if (!entry) { zswap_reject_kmemcache_fail++; return false; } - if (!zswap_compress(page, entry, pool)) + if (!zswap_compress(folio, index, entry, pool)) goto compress_failed; old = xa_store(swap_zswap_tree(page_swpentry), @@ -1515,9 +1515,7 @@ bool zswap_store(struct folio *folio) } for (index = 0; index < nr_pages; ++index) { - struct page *page = folio_page(folio, index); - - if (!zswap_store_page(page, objcg, pool)) + if (!zswap_store_page(folio, index, objcg, pool)) goto put_pool; } From e80dcac66069dba5caa5afe2d46876394f11001e Mon Sep 17 00:00:00 2001 From: David Stevens Date: Fri, 4 Sep 2026 10:31:45 -0700 Subject: [PATCH 0800/1352] memcg: don't call schedule_work when no spinning is allowed Memcg charging can be done from any context, but calling schedule_work() isn't safe from an NMI. If memory.high is breached from a context where spinning isn't allowed, use irq_work to schedule the reclaim work. Found this via code inspection. I spent a little bit trying to trigger it for real, but the only way I managed was by writing a hacky driver absuing alloc_pages_nolock(). Link: https://lore.kernel.org/20260904173145.2028377-1-stevensd@google.com Fixes: 3ac4638a734a ("memcg: make memcg_rstat_updated nmi safe") Signed-off-by: David Stevens Signed-off-by: Andrew Morton Acked-by: Michal Hocko Reviewed-by: Johannes Weiner Acked-by: Shakeel Butt Cc: Lorenzo Stoakes Cc: Muchun Song Cc: Roman Gushchin --- include/linux/memcontrol.h | 2 ++ mm/memcontrol.c | 13 ++++++++++++- 2 files changed, 14 insertions(+), 1 deletion(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index c0c9805b6f0320..4f720791a31c95 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -23,6 +23,7 @@ #include #include #include +#include struct mem_cgroup; struct obj_cgroup; @@ -233,6 +234,7 @@ struct mem_cgroup { #endif /* Range enforcement for interrupt charges */ + struct irq_work high_irq_work; struct work_struct high_work; __cacheline_group_end_aligned(memcg_write_hot); diff --git a/mm/memcontrol.c b/mm/memcontrol.c index e539881f58f147..3e14e0ce9e7ec4 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -62,6 +62,7 @@ #include #include #include +#include #include "internal.h" #include "swap.h" #include "swap_table.h" @@ -2424,6 +2425,11 @@ static void high_work_func(struct work_struct *work) reclaim_high(memcg, MEMCG_CHARGE_BATCH, GFP_KERNEL); } +static void high_irq_work_func(struct irq_work *work) +{ + schedule_work(&container_of(work, struct mem_cgroup, high_irq_work)->high_work); +} + /* * Clamp the maximum sleep time per allocation batch to 2 seconds. This is * enough to still cause a significant slowdown in most cases, while still @@ -2832,7 +2838,10 @@ static int try_charge_memcg(struct mem_cgroup *memcg, gfp_t gfp_mask, /* Don't bother a random interrupted task */ if (!in_task()) { if (mem_high) { - schedule_work(&memcg->high_work); + if (allow_spinning) + schedule_work(&memcg->high_work); + else + irq_work_queue(&memcg->high_irq_work); break; } continue; @@ -4207,6 +4216,7 @@ static struct mem_cgroup *mem_cgroup_alloc(struct mem_cgroup *parent) goto fail; INIT_WORK(&memcg->high_work, high_work_func); + init_irq_work(&memcg->high_irq_work, high_irq_work_func); vmpressure_init(&memcg->vmpressure); INIT_LIST_HEAD(&memcg->memory_peaks); INIT_LIST_HEAD(&memcg->swap_peaks); @@ -4415,6 +4425,7 @@ static void mem_cgroup_css_free(struct cgroup_subsys_state *css) static_branch_dec(&memcg_bpf_enabled_key); vmpressure_cleanup(&memcg->vmpressure); + irq_work_sync(&memcg->high_irq_work); cancel_work_sync(&memcg->high_work); free_shrinker_info(memcg); mem_cgroup_free(memcg); From 33985cbf4ed8715c815eeba67a86be0c9b360d49 Mon Sep 17 00:00:00 2001 From: Joanne Koong Date: Thu, 3 Sep 2026 14:56:16 -0700 Subject: [PATCH 0801/1352] mm/memcontrol: skip non-hierarchical memcg-wide stats when v1 is unavailable memcg_vmstats keeps a non-hierarchical copy of every memcg-wide stat item and event alongside the hierarchical one. The only readers however are the legacy memory.stat and memory.numa_stat, and reparenting on offline. All of them live under CONFIG_MEMCG_V1, and their accessors (memcg_page_state_local() and memcg_events_local()) are already compiled out with it. This means on a CONFIG_MEMCG_V1=n kernel, memcg_vmstats's non-hierarchical arrays are written to on every rstat flush, despite their values never being read / accessed. The same holds when the kernel does support v1 but the controller has been blocked from v1 hierarchies with the boot param cgroup_no_v1={memory,all}. A v1 mount is refused in that case, so the legacy memory.stat can never exist and the arrays are just as unread / unaccessed. Compile out the non-hierarchical memcg-wide arrays if CONFIG_MEMCG_V1 is not set. If it is set but cgroup_no_v1= has blocked the controller, skip updates on the arrays. This makes flushes cheaper. mem_cgroup_stat_aggregate() can now skip the read-modify-write of ac->local[i]. Nothing else accesses state_local or events_local, so those cachelines were getting pulled in solely for the writes and they are separate from the ones the loop is already walking. On an 80-cpu x86_64 machine with 500 cgroups each running a workload that dirties anon, file, dirty/writeback, slab, kmem, mlock, and reclaim counters, timing mem_cgroup_css_rstat_flush() in-kernel in TSC ticks per flush showed roughly before after delta memcg-wide aggregation 1231 1180 -4.1% overall flush function 2452 2397 -2.2% These numbers are from taking the median of 70 samples, one per 20s window on each kernel. The 95% intervals observed on the two deltas are [-5.41%, -1.76%] and [-4.53%, -0.14%]. The values above include the timing overhead itself, so only the delta is meaningful here. Counting the items that actually changed, a median of 1.5 of the 77 memcg-wide items (57 state + 20 events) had a non-zero per-cpu delta at each flush, which means the benchmarks above are with one or two fewer cachelines pulled in per flush. The count is low because the benchmark reads memory.stat in a loop to keep the flush rate up. For cases where flushes are triggered only by the 2s periodic worker, more changes will have accumulated between flushes, so more cachelines are skipped and the per-flush saving should be larger. Link: https://lore.kernel.org/20260903215616.1456239-1-joannelkoong@gmail.com Signed-off-by: Joanne Koong Signed-off-by: Andrew Morton Reviewed-by: Johannes Weiner Acked-by: Shakeel Butt Reviewed-by: Yosry Ahmed Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin --- include/linux/cgroup.h | 3 +++ kernel/cgroup/cgroup-internal.h | 1 - mm/memcontrol.c | 48 ++++++++++++++++++++++++++++----- 3 files changed, 44 insertions(+), 8 deletions(-) diff --git a/include/linux/cgroup.h b/include/linux/cgroup.h index 5dfa915a630e8d..2afb4cb2bb4fa4 100644 --- a/include/linux/cgroup.h +++ b/include/linux/cgroup.h @@ -154,6 +154,9 @@ struct cgroup *cgroup_get_from_path(const char *path); struct cgroup *cgroup_get_from_fd(int fd); struct cgroup *cgroup_v1v2_get_from_fd(int fd); +/* Was this controller blocked from v1 hierarchies by cgroup_no_v1= ? */ +bool cgroup1_ssid_disabled(int ssid); + int cgroup_attach_task_all(struct task_struct *from, struct task_struct *); int cgroup_transfer_tasks(struct cgroup *to, struct cgroup *from); diff --git a/kernel/cgroup/cgroup-internal.h b/kernel/cgroup/cgroup-internal.h index 58797123b752f3..7c367c8d0cbe1d 100644 --- a/kernel/cgroup/cgroup-internal.h +++ b/kernel/cgroup/cgroup-internal.h @@ -285,7 +285,6 @@ extern struct kernfs_syscall_ops cgroup1_kf_syscall_ops; extern const struct fs_parameter_spec cgroup1_fs_parameters[]; int proc_cgroupstats_show(struct seq_file *m, void *v); -bool cgroup1_ssid_disabled(int ssid); void cgroup1_pidlist_destroy_all(struct cgroup *cgrp); void cgroup1_release_agent(struct work_struct *work); void cgroup1_check_for_release(struct cgroup *cgrp); diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 3e14e0ce9e7ec4..3e0f6edfd350d7 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -676,9 +676,11 @@ struct memcg_vmstats { long state[MEMCG_VMSTAT_SIZE]; unsigned long events[NR_MEMCG_EVENTS]; +#ifdef CONFIG_MEMCG_V1 /* Non-hierarchical (CPU aggregated) page state & events */ long state_local[MEMCG_VMSTAT_SIZE]; unsigned long events_local[NR_MEMCG_EVENTS]; +#endif /* Pending child counts during tree propagation */ long state_pending[MEMCG_VMSTAT_SIZE]; @@ -688,6 +690,31 @@ struct memcg_vmstats { atomic_long_t stats_updates; }; +/* + * The non-hierarchical memcg-wide counters are read back only by the legacy + * memory.stat and memory.numa_stat, and by reparenting on offline, all of which + * are v1-only. If the kernel is built without CONFIG_MEMCG_V1, or if the boot + * param cgroup_no_v1= has blocked the memory controller from v1 hierarchies, + * then nothing reads them and writers can skip the updates. + */ +static long *memcg_state_local_array(struct mem_cgroup *memcg) +{ +#ifdef CONFIG_MEMCG_V1 + if (!cgroup1_ssid_disabled(memory_cgrp_id)) + return memcg->vmstats->state_local; +#endif + return NULL; +} + +static unsigned long *memcg_events_local_array(struct mem_cgroup *memcg) +{ +#ifdef CONFIG_MEMCG_V1 + if (!cgroup1_ssid_disabled(memory_cgrp_id)) + return memcg->vmstats->events_local; +#endif + return NULL; +} + /* * memcg and lruvec stats flushing * @@ -4469,7 +4496,10 @@ static void mem_cgroup_css_reset(struct cgroup_subsys_state *css) struct aggregate_control { /* pointer to the aggregated (CPU and subtree aggregated) counters */ long *aggregate; - /* pointer to the non-hierarchichal (CPU aggregated) counters */ + /* + * pointer to the non-hierarchical (CPU aggregated) counters or NULL to + * skip updating them (see memcg_state_local_array()) + */ long *local; /* pointer to the pending child counters during tree propagation */ long *pending; @@ -4508,7 +4538,7 @@ static void mem_cgroup_stat_aggregate(struct aggregate_control *ac) } /* Aggregate counts on this level and propagate upwards */ - if (delta_cpu) + if (delta_cpu && ac->local) ac->local[i] += delta_cpu; if (delta) { @@ -4522,6 +4552,7 @@ static void mem_cgroup_stat_aggregate(struct aggregate_control *ac) #ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC static void flush_nmi_stats(struct mem_cgroup *memcg, struct mem_cgroup *parent) { + long *state_local = memcg_state_local_array(memcg); int nid; if (atomic_read(&memcg->kmem_stat)) { @@ -4529,7 +4560,8 @@ static void flush_nmi_stats(struct mem_cgroup *memcg, struct mem_cgroup *parent) int index = memcg_stats_index(MEMCG_KMEM); memcg->vmstats->state[index] += kmem; - memcg->vmstats->state_local[index] += kmem; + if (state_local) + state_local[index] += kmem; if (parent) parent->vmstats->state_pending[index] += kmem; } @@ -4551,7 +4583,8 @@ static void flush_nmi_stats(struct mem_cgroup *memcg, struct mem_cgroup *parent) if (plstats) plstats->state_pending[index] += slab; memcg->vmstats->state[index] += slab; - memcg->vmstats->state_local[index] += slab; + if (state_local) + state_local[index] += slab; if (parent) parent->vmstats->state_pending[index] += slab; } @@ -4564,7 +4597,8 @@ static void flush_nmi_stats(struct mem_cgroup *memcg, struct mem_cgroup *parent) if (plstats) plstats->state_pending[index] += slab; memcg->vmstats->state[index] += slab; - memcg->vmstats->state_local[index] += slab; + if (state_local) + state_local[index] += slab; if (parent) parent->vmstats->state_pending[index] += slab; } @@ -4589,7 +4623,7 @@ static void mem_cgroup_css_rstat_flush(struct cgroup_subsys_state *css, int cpu) ac = (struct aggregate_control) { .aggregate = memcg->vmstats->state, - .local = memcg->vmstats->state_local, + .local = memcg_state_local_array(memcg), .pending = memcg->vmstats->state_pending, .ppending = parent ? parent->vmstats->state_pending : NULL, .cstat = statc->state, @@ -4600,7 +4634,7 @@ static void mem_cgroup_css_rstat_flush(struct cgroup_subsys_state *css, int cpu) ac = (struct aggregate_control) { .aggregate = memcg->vmstats->events, - .local = memcg->vmstats->events_local, + .local = memcg_events_local_array(memcg), .pending = memcg->vmstats->events_pending, .ppending = parent ? parent->vmstats->events_pending : NULL, .cstat = statc->events, From 8ab5d49c10ff1bae5e20302b10f7a19faad01ff2 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 3 Sep 2026 20:08:39 +0100 Subject: [PATCH 0802/1352] mm/madvise: swap in CoW'd MAP_PRIVATE-file mappings on MADV_WILLNEED Currently MADV_WILLNEED treats file-backed and pure anonymous mappings entirely separately - using POSIX_FADV_WILLNEED (equivalent of a readahead) for the former and a tree walk and swap in to swap cache for the latter. MAP_PRIVATE-file backed mappings straddle the two and currently get treated as if they were purely file-backed, meaning any swapped out private pages remain swapped out. Resolve the issue by explicitly checking for CoW'd MAP_PRIVATE-file backed mappings and performing both walks in this case. Since the logic checks for vma->anon_vma this means un-CoW'd MAP_PRIVATE-file backed mappings retain only the single file walk. Link: https://lore.kernel.org/aprjOxDy3JCPb2oa@gremlin Reported-by: Mike Kaplinskiy Closes: https://lore.kernel.org/all/CABeknB_S2XJSHFgnHdgnN0rjzHhH4oQJs_APq9fvxHztQ_pgiA@mail.gmail.com/ Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Vlastimil Babka (SUSE) Reviewed-by: Pedro Falcato Cc: David Hildenbrand Cc: Jann Horn Cc: Liam R. Howlett --- mm/madvise.c | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/mm/madvise.c b/mm/madvise.c index 73c2901b9adbf0..963337f93a7a11 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -297,11 +297,12 @@ static long madvise_willneed(struct madvise_behavior *madv_behavior) loff_t offset; #ifdef CONFIG_SWAP - if (!file) { + if (vma_is_cow_mapping(vma) && vma->anon_vma) { walk_page_range_vma(vma, start, end, &swapin_walk_ops, vma); lru_add_drain(); /* Push any new pages onto the LRU now */ - return 0; } + if (!file) + return 0; if (shmem_mapping(file->f_mapping)) { shmem_swapin_range(vma, start, end, file->f_mapping); From 43b5498da3c000a8870d9890727c3377a5418b73 Mon Sep 17 00:00:00 2001 From: Hui Zhu Date: Fri, 11 Sep 2026 16:00:48 +0800 Subject: [PATCH 0803/1352] mm: memcg: redirect stats updates of dying memcgs for all hierarchies Patch series "mm: workingset: fix the shadow node budget under MGLRU", v5. Commit 7404bd37cfbe ("mm: workingset: use lruvec_lru_size() to get the number of lru pages") broke the workingset shadow node budget under MGLRU: lruvec_lru_size() reads mz->lru_zone_size, which MGLRU never maintains, so count_shadow_nodes() sees the evictable LRU lists as empty and the shadow shrinker reclaims eviction tokens almost as fast as they are created, losing thrashing protection. Patch 1 extends the dying-mcg stat redirection (previously cgroup v1 only) to all hierarchies, addressing the reparenting race that motivated 7404bd37cfbe. Patch 2 then switches count_shadow_nodes() back to lruvec_page_state_local(), which both classic LRU and MGLRU maintain. Patch 3 recovers the performance. Patch 1 added an unconditional rcu_read_lock() to the stat update fast path; patch 3 checks memcg_is_dying() first and takes the RCU lock only on the rare dying path. Patch 4 closes an accounting gap that patch 2 makes visible: on cgroup v2, reparent_state_local() never moves the dying memcg's non-hierarchical lruvec stats to the parent, so the parent receives the uncharges without the matching charges and its state_local underflows. Patch 4 reparents those stats, mirroring cgroup v1. Performance testing =================== The test script and the raw results are available at [1]. Environment: 10-vCPU QEMU guest, 8 GiB RAM, cgroup v2; 7 runs per configuration, medians reported. Workloads: w1-anon-churn: single-threaded anon fault/charge loop in a memcg (MADV_DONTNEED + re-fault, no reclaim). Every touch is a real fault with charge and memcg stat updates, so it stresses exactly the fast path patch 1 changes. This is the meaningful signal: it is not reclaim-bound, so the small fast-path overhead is not drowned out. w2-file-churn: file read loop under memory.high pressure (reclaim-bound; the differences below are within run-to-run noise and are shown for completeness only). w3-reparent: reparent accounting sanity check. w1-anon-churn (pages/s): classic LRU MGLRU base 4393028 4377122 patches 1-2 4385996 (-0.2%) 4352887 (-0.6%) patches 1-3 4381832 (-0.3%) 4377053 (+0.0%) w2-file-churn (MB/s): classic LRU MGLRU base 8277 8226 patches 1-2 8226 (-0.6%) 8123 (-1.3%) patches 1-3 8157 (-1.4%) 8294 (+0.8%) w3-reparent passed on all kernels. The small overhead visible with patches 1-2 comes from the redirection added by patch 1; patch 3 brings w1 back to the base level in both LRU configurations. The w2-file-churn differences are within run-to-run noise: that workload is reclaim-bound and too noisy to expose the small fast-path overhead, so w1-anon-churn is the meaningful signal. Patch 4 only touches the memcg offline path and is not exercised by these workloads. This patch (of 4): get_non_dying_memcg_start() redirects the stat updates of a dying memcg to its closest non-dying ancestor, but only on cgroup v1; on cgroup v2 the stats keep being accounted to the dying memcg itself. A later patch in this series restores lruvec_page_state_local() in count_shadow_nodes() to fix the broken workingset shadow node budget under MGLRU. count_shadow_nodes() is the only reader of those non-hierarchical state_locals on cgroup v2: when a memcg is offlined, its pages are reparented to the ancestor but their stat updates keep being accounted to the dying memcg, so count_shadow_nodes() computes a wrong shadow node budget and workingset thrashing protection is lost. This is user visible as premature reclaim of hot page cache and degraded performance under memory pressure. Apply the redirection to all hierarchies to fix this. Offlining is rare, so the added cost on the stat update fast path is limited to an rcu_read_lock() and a css_is_dying() check; the upward walk happens only while a memcg is dying. Link: https://lore.kernel.org/cover.1789096175.git.zhuhui@kylinos.cn Link: https://lore.kernel.org/c1ef4ef6a84cac479e573f4423b734dc8176f7d5.1789096175.git.zhuhui@kylinos.cn Link: https://gist.github.com/teawater/32f373ec41d185d840455eb167321a5a [1] Fixes: 7404bd37cfbe ("mm: workingset: use lruvec_lru_size() to get the number of lru pages") Signed-off-by: Hui Zhu Signed-off-by: Andrew Morton Acked-by: Shakeel Butt Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Cc: Wei Xu Cc: Yuanchu Xie Cc: --- mm/memcontrol.c | 30 +++++------------------------- 1 file changed, 5 insertions(+), 25 deletions(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 3e0f6edfd350d7..8a571e80a1030b 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -873,20 +873,14 @@ static long memcg_state_val_in_pages(int idx, long val) return val < 0 ? -res : res; } -#ifdef CONFIG_MEMCG_V1 /* - * Used in mod_memcg_state() and mod_memcg_lruvec_state() to avoid race with - * reparenting of non-hierarchical state_locals. + * Used in mod_memcg_state() and mod_memcg_lruvec_state() to avoid race + * with reparenting of non-hierarchical state_locals. Offlining a + * memcg is rare, so do the redirection for all cgroup hierarchies. */ -static inline struct mem_cgroup *get_non_dying_memcg_start(struct mem_cgroup *memcg, - bool *rcu_locked) +static inline struct mem_cgroup * +get_non_dying_memcg_start(struct mem_cgroup *memcg, bool *rcu_locked) { - /* Rebinding can cause this value to be changed at runtime */ - if (cgroup_subsys_on_dfl(memory_cgrp_subsys)) { - *rcu_locked = false; - return memcg; - } - rcu_read_lock(); *rcu_locked = true; @@ -898,22 +892,8 @@ static inline struct mem_cgroup *get_non_dying_memcg_start(struct mem_cgroup *me static inline void get_non_dying_memcg_end(bool rcu_locked) { - if (!rcu_locked) - return; - rcu_read_unlock(); } -#else -static inline struct mem_cgroup *get_non_dying_memcg_start(struct mem_cgroup *memcg, - bool *rcu_locked) -{ - return memcg; -} - -static inline void get_non_dying_memcg_end(bool rcu_locked) -{ -} -#endif static void __mod_memcg_state(struct mem_cgroup *memcg, enum memcg_stat_item idx, long val) From 1db0a955f9c02e6eccc23672690b35d7178e8a5d Mon Sep 17 00:00:00 2001 From: Hui Zhu Date: Fri, 11 Sep 2026 16:00:49 +0800 Subject: [PATCH 0804/1352] mm: workingset: use lruvec_page_state_local() to count lru pages Commit 7404bd37cfbe ("mm: workingset: use lruvec_lru_size() to get the number of lru pages") switched count_shadow_nodes() to lruvec_lru_size(). With CONFIG_MEMCG enabled, lruvec_lru_size() reads mz->lru_zone_size, which only the classic LRU paths maintain. MGLRU accounts its pages through __update_lru_size(), which skips that array, so with MGLRU on the four evictable LRU lists are always seen as empty. The shadow node budget (pages >> 3) then collapses to slab plus unevictable pages, and the workingset shadow shrinker reclaims eviction tokens almost as fast as they are created, losing thrashing protection. lruvec_page_state_local() reads lruvec_stats->state_local instead, which both classic LRU and MGLRU maintain. Switch back to it. The reparenting race this re-exposes on cgroup v2 is closed by the preceding patch that redirects dying-memcg stat updates for all hierarchies. Link: https://lore.kernel.org/2ed42f96aca124856ea30f774afb55cbe6d8ba58.1789096175.git.zhuhui@kylinos.cn Fixes: 7404bd37cfbe ("mm: workingset: use lruvec_lru_size() to get the number of lru pages") Signed-off-by: Hui Zhu Signed-off-by: Andrew Morton Acked-by: Shakeel Butt Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Cc: Wei Xu Cc: Yuanchu Xie Cc: --- mm/workingset.c | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/mm/workingset.c b/mm/workingset.c index 7ac2b88c80ae56..8412f4840ae35c 100644 --- a/mm/workingset.c +++ b/mm/workingset.c @@ -688,10 +688,9 @@ static unsigned long count_shadow_nodes(struct shrinker *shrinker, mem_cgroup_flush_stats_ratelimited(sc->memcg); lruvec = mem_cgroup_lruvec(sc->memcg, NODE_DATA(sc->nid)); - for (pages = 0, i = 0; i < NR_LRU_LISTS; i++) - pages += lruvec_lru_size(lruvec, i, MAX_NR_ZONES - 1); - + pages += lruvec_page_state_local(lruvec, + NR_LRU_BASE + i); pages += lruvec_page_state_local( lruvec, NR_SLAB_RECLAIMABLE_B) >> PAGE_SHIFT; pages += lruvec_page_state_local( From 157b2d3cc7e54ecdf53d4cea3b4b1e0c66127724 Mon Sep 17 00:00:00 2001 From: Hui Zhu Date: Fri, 11 Sep 2026 16:00:50 +0800 Subject: [PATCH 0805/1352] mm: memcg: skip the RCU lock when the memcg is not dying get_non_dying_memcg_start() takes rcu_read_lock() on every stat update, but the lock only protects the upward walk to a non-dying ancestor, which happens solely while a memcg is being offlined. The dying check itself reads the CSS_DYING flag of a memcg the caller already holds a reference to, so it is safe without the lock. Check memcg_is_dying() first and return immediately when the memcg is alive, taking the RCU lock only on the rare dying path. On an anon fault/charge churn workload in a memcg this recovers the ~0.6% overhead added by the previous patch (4368077 vs 4343159 pages/s before, back to ~4377000 pages/s after). Link: https://lore.kernel.org/9ffdbdfc96312e3e13cb8f056bfe26649492d949.1789096175.git.zhuhui@kylinos.cn Fixes: 7404bd37cfbe ("mm: workingset: use lruvec_lru_size() to get the number of lru pages") Signed-off-by: Hui Zhu Signed-off-by: Andrew Morton Acked-by: Shakeel Butt Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Cc: Wei Xu Cc: Yuanchu Xie Cc: --- mm/memcontrol.c | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 8a571e80a1030b..307203301a2a56 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -881,6 +881,17 @@ static long memcg_state_val_in_pages(int idx, long val) static inline struct mem_cgroup * get_non_dying_memcg_start(struct mem_cgroup *memcg, bool *rcu_locked) { + /* + * Fast path: the caller holds a reference to @memcg, so reading + * its CSS_DYING flag without the RCU lock is safe. The RCU lock + * is only needed to walk up to a non-dying ancestor, which + * happens only while a memcg is actually being offlined. + */ + if (!memcg_is_dying(memcg)) { + *rcu_locked = false; + return memcg; + } + rcu_read_lock(); *rcu_locked = true; @@ -892,6 +903,9 @@ get_non_dying_memcg_start(struct mem_cgroup *memcg, bool *rcu_locked) static inline void get_non_dying_memcg_end(bool rcu_locked) { + if (!rcu_locked) + return; + rcu_read_unlock(); } From 0892aca831b9f96de7b9dcfdcf9284f1fb1da53d Mon Sep 17 00:00:00 2001 From: Hui Zhu Date: Fri, 11 Sep 2026 16:00:51 +0800 Subject: [PATCH 0806/1352] mm: memcg: reparent non-hierarchical lruvec stats on cgroup v2 On cgroup v2, reparent_state_local() returns early and never moves the dying memcg's non-hierarchical state_local base counts to its parent. Meanwhile memcg_reparent_objcgs() rewrites objcg->memcg to the parent, so when the reparented folios are freed later, the negative deltas land on the parent's lruvec. The parent therefore receives the uncharges without ever having received the matching charges, and its state_local (NR_LRU_BASE + lru, MEMCG_SOCK, NR_SLAB_RECLAIMABLE_B, NR_SLAB_UNRECLAIMABLE_B) permanently underflows. Since lruvec_page_state_local() clamps negative values to zero, the underflow masks the parent's own legitimate pages. count_shadow_nodes() is the only reader of these non-hierarchical state_locals on cgroup v2, so the underflow directly distorts the workingset shadow node budget. Fix this by reparenting the lruvec state_locals on cgroup v2 as well, mirroring what cgroup v1 already does. Only the lruvec stats consumed by count_shadow_nodes() are moved; the memcg-level stats are left alone because on v2 they are exposed through the rstat hierarchical tree and are not read from state_local. Link: https://lore.kernel.org/4a7a64eed2b145ad535fedaea3624f8310c29d5b.1789096175.git.zhuhui@kylinos.cn Fixes: 8285917d6f38 ("mm: memcontrol: prepare for reparenting non-hierarchical stats") Signed-off-by: Hui Zhu Signed-off-by: Andrew Morton Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie Cc: --- mm/memcontrol-v1.h | 5 +++-- mm/memcontrol.c | 42 ++++++++++++++++++++++++++++-------------- 2 files changed, 31 insertions(+), 16 deletions(-) diff --git a/mm/memcontrol-v1.h b/mm/memcontrol-v1.h index b9a21f0fd2c3ac..2cd37e1792d79e 100644 --- a/mm/memcontrol-v1.h +++ b/mm/memcontrol-v1.h @@ -25,6 +25,9 @@ int memory_stat_show(struct seq_file *m, void *v); struct mem_cgroup *mem_cgroup_private_id_get_online(struct mem_cgroup *memcg, unsigned int n); +void reparent_memcg_lruvec_state_local(struct mem_cgroup *memcg, + struct mem_cgroup *parent, int idx); + /* Cgroup v1-specific declarations */ #ifdef CONFIG_MEMCG_V1 @@ -67,8 +70,6 @@ void reparent_memcg1_lruvec_state_local(struct mem_cgroup *memcg, struct mem_cgr void reparent_memcg_state_local(struct mem_cgroup *memcg, struct mem_cgroup *parent, int idx); -void reparent_memcg_lruvec_state_local(struct mem_cgroup *memcg, - struct mem_cgroup *parent, int idx); void memcg1_account_kmem(struct mem_cgroup *memcg, int nr_pages); static inline bool memcg1_tcpmem_active(struct mem_cgroup *memcg) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 307203301a2a56..cf53d4ac7ecc18 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -233,14 +233,29 @@ static inline struct obj_cgroup *__memcg_reparent_objcgs(struct mem_cgroup *memc return objcg; } -#ifdef CONFIG_MEMCG_V1 static void __mem_cgroup_flush_stats(struct mem_cgroup *memcg, bool force); -static inline void reparent_state_local(struct mem_cgroup *memcg, struct mem_cgroup *parent) +/* + * Reparent the non-hierarchical lruvec stats that count_shadow_nodes() reads + * to approximate the shadow node budget. They are not exposed to userspace + * on cgroup v2, but they must follow the reparented folios; otherwise the + * ancestor would only receive the negative deltas when the folios are freed + * without ever having received the positive base, and its local stats would + * permanently underflow. + */ +static void reparent_v2_lruvec_state_local(struct mem_cgroup *memcg, struct mem_cgroup *parent) { - if (cgroup_subsys_on_dfl(memory_cgrp_subsys)) - return; + int i; + + for (i = 0; i < NR_LRU_LISTS; i++) + reparent_memcg_lruvec_state_local(memcg, parent, NR_LRU_BASE + i); + + reparent_memcg_lruvec_state_local(memcg, parent, NR_SLAB_RECLAIMABLE_B); + reparent_memcg_lruvec_state_local(memcg, parent, NR_SLAB_UNRECLAIMABLE_B); +} +static inline void reparent_state_local(struct mem_cgroup *memcg, struct mem_cgroup *parent) +{ /* * Reparent stats exposed non-hierarchically. Flush @memcg's stats first * to read its stats accurately , and conservatively flush @parent's @@ -249,17 +264,18 @@ static inline void reparent_state_local(struct mem_cgroup *memcg, struct mem_cgr */ __mem_cgroup_flush_stats(memcg, true); - /* The following counts are all non-hierarchical and need to be reparented. */ - reparent_memcg1_state_local(memcg, parent); - reparent_memcg1_lruvec_state_local(memcg, parent); + if (cgroup_subsys_on_dfl(memory_cgrp_subsys)) { + reparent_v2_lruvec_state_local(memcg, parent); + } else { +#ifdef CONFIG_MEMCG_V1 + /* The following counts are all non-hierarchical and need to be reparented. */ + reparent_memcg1_state_local(memcg, parent); + reparent_memcg1_lruvec_state_local(memcg, parent); +#endif + } __mem_cgroup_flush_stats(parent, true); } -#else -static inline void reparent_state_local(struct mem_cgroup *memcg, struct mem_cgroup *parent) -{ -} -#endif static inline void reparent_locks(struct mem_cgroup *memcg, struct mem_cgroup *parent, int nid) { @@ -571,7 +587,6 @@ unsigned long lruvec_page_state_local(struct lruvec *lruvec, return x; } -#ifdef CONFIG_MEMCG_V1 static void __mod_memcg_lruvec_state(struct mem_cgroup_per_node *pn, enum node_stat_item idx, long val); @@ -593,7 +608,6 @@ void reparent_memcg_lruvec_state_local(struct mem_cgroup *memcg, __mod_memcg_lruvec_state(parent_pn, idx, value); } } -#endif /* Subset of vm_event_item to report for memcg event stats */ static const unsigned int memcg_vm_event_stat[] = { From cbe3c0fa84b796676fcd8818888f3849b3e65af8 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Thu, 3 Sep 2026 18:49:58 +0300 Subject: [PATCH 0807/1352] mm/execmem: free ROX cache chunks only when they span an entire vm area Patch series "mm/execmem: fixes and cleanups for the ROX cache". Sashiko review of the ROX cache refill path asked what happens when the PMD_SIZE allocation fails there. Pulling that thread uncovered a bit more than the fallback. Patch 1 fixes a real bug, although rare bug: execmem_cache_clean() can free a chunk that is still partially in use, because a PMD sized and PMD aligned free range in the middle of a larger chunk looks exactly like a chunk that nobody uses. Patch 2 handles maple tree allocation failures in the cache. They are unlikely, but silently dropping an area from the tree is not a great way to deal with them. Patch 3 deals with the fallback that started all this. The cache exists to keep the direct map free of unnecessary splits, and the fallback happily filled it with base page mapped areas that could never be freed from the cache again. Ask vmalloc for a huge mapping or nothing, and serve whatever does not fit outside the cache. Patches 4 and 5 convert the ROX cache to scope based cleanup. The series was tested on x86 with module load/unload cycles, with PTDUMP confirming that the cache is using 2M mappings, and with nohugevmalloc to exercise the uncached fallback. This patch (of 5): When execmem refills the ROX cache, it vmalloc()s multiples of PMD_SIZE aligned to PMD_SIZE. For every such allocation vmalloc creates a vm area. The first part of the vmalloc()ed chunk is returned to the allocation that triggered the cache refill and the remaining part is added to the cache and handed out for subsequent allocations with execmem_alloc(). When only the first part is freed, the entire vm area remains in the ROX cache and can be handed out again. In the case when the first allocation is larger than PMD_SIZE and the second allocation from the freed first part of the chunk is exactly PMD_SIZE, execmem_cache_clean() will free the entire chunk while part of it is still in use. For example: /* * vmalloc(4M), return p0 to the caller * add [p0 + 3M, p0 + 4M) to the cache */ p0 = execmem_alloc(3M); /* return p0 + 3M from the cache to the caller */ p1 = execmem_alloc(1M); /* put [p0, p0 + 3M) back into the cache */ execmem_free(p0); /* return p0 from the cache to the caller */ p2 = execmem_alloc(2M); /* return p0 + 2M from the cache to the caller */ p3 = execmem_alloc(1M); /* bah! execmem_cache_clean() frees the entire 4M chunk */ execmem_free(p2); Make sure that the ranges that execmem_cache_clean() frees cover the entire vm area. Link: https://lore.kernel.org/20260903-execmem-rox-cache-pmd-v1-v1-0-11beb2a3d249@kernel.org Link: https://lore.kernel.org/20260903-execmem-rox-cache-pmd-v1-v1-1-11beb2a3d249@kernel.org Fixes: 2e45474ab14f ("execmem: add support for cache of large ROX pages") Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Assisted-by: copilot:claude-opus-5 Cc: Benjamin Tissoires Cc: Jiri Kosina Cc: Luis Chamberalin Cc: "Uladzislau Rezki (Sony)" Cc: --- mm/execmem.c | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/mm/execmem.c b/mm/execmem.c index ad07cae9ed5854..ba277790e31328 100644 --- a/mm/execmem.c +++ b/mm/execmem.c @@ -143,9 +143,11 @@ static void execmem_cache_clean(struct work_struct *work) mutex_lock(mutex); mas_for_each(&mas, area, ULONG_MAX) { + struct vm_struct *vm = find_vm_area(area); size_t size = mas_range_len(&mas); - if (IS_ALIGNED(size, PMD_SIZE) && + if (vm && get_vm_area_size(vm) == size && + IS_ALIGNED(size, PMD_SIZE) && IS_ALIGNED(mas.index, PMD_SIZE)) { mas_store_gfp(&mas, NULL, GFP_KERNEL); vfree(area); From 847335c644c93719206aae0e8e751cfd03979a7f Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Thu, 3 Sep 2026 18:49:59 +0300 Subject: [PATCH 0808/1352] mm/execmem: handle potential allocation errors in the maple tree execmem_cache_clean() and execmem_cache_alloc_locked() ignore potential allocation failures in mas_store_gfp(). While in practice they are unlikely to happen, it's better to handle those errors and ensure the integrity of the ROX cache. Preallocate the maple tree nodes for the stores that must not fail and order the maple tree updates so that there won't be any failures once a tree has been modified. Link: https://lore.kernel.org/20260903-execmem-rox-cache-pmd-v1-v1-2-11beb2a3d249@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Assisted-by: copilot:claude-opus-5 Cc: Benjamin Tissoires Cc: Jiri Kosina Cc: Luis Chamberalin Cc: "Uladzislau Rezki (Sony)" --- mm/execmem.c | 32 ++++++++++++++++++++++---------- 1 file changed, 22 insertions(+), 10 deletions(-) diff --git a/mm/execmem.c b/mm/execmem.c index ba277790e31328..00dd6324cae01a 100644 --- a/mm/execmem.c +++ b/mm/execmem.c @@ -149,7 +149,15 @@ static void execmem_cache_clean(struct work_struct *work) if (vm && get_vm_area_size(vm) == size && IS_ALIGNED(size, PMD_SIZE) && IS_ALIGNED(mas.index, PMD_SIZE)) { - mas_store_gfp(&mas, NULL, GFP_KERNEL); + /* + * Preallocate to ensure mas_store does not fail + * If there is no memory for the tree update, bail out, + * next execmem_free() might be more lucky + */ + if (mas_preallocate(&mas, NULL, GFP_KERNEL)) + break; + + mas_store_prealloc(&mas, NULL); vfree(area); } } @@ -219,30 +227,34 @@ static void *execmem_cache_alloc_locked(struct execmem_range *range, size_t size addr = mas_free.index; last = mas_free.last; + mas_set_range(&mas_free, addr, addr + size - 1); + if (mas_preallocate(&mas_free, NULL, GFP_KERNEL)) + return NULL; + /* insert allocated size to busy_areas at range [addr, addr + size) */ mas_set_range(&mas_busy, addr, addr + size - 1); err = mas_store_gfp(&mas_busy, (void *)addr, GFP_KERNEL); if (err) - return NULL; + goto err_destroy_mas_free; - mas_store_gfp(&mas_free, NULL, GFP_KERNEL); + mas_store_prealloc(&mas_free, NULL); if (area_size > size) { - void *ptr = (void *)(addr + size); - /* * re-insert remaining free size to free_areas at range * [addr + size, last] + * the range matches an existing entry, so this cannot allocate */ + ptr = (void *)(addr + size); mas_set_range(&mas_free, addr + size, last); - err = mas_store_gfp(&mas_free, ptr, GFP_KERNEL); - if (err) { - mas_store_gfp(&mas_busy, NULL, GFP_KERNEL); - return NULL; - } + mas_store_gfp(&mas_free, ptr, GFP_KERNEL); } ptr = (void *)addr; return ptr; + +err_destroy_mas_free: + mas_destroy(&mas_free); + return NULL; } static void *__execmem_cache_alloc(struct execmem_range *range, size_t size) From 7e963438ba3d247d040456edf7d06d68979bc509 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Thu, 3 Sep 2026 18:50:00 +0300 Subject: [PATCH 0809/1352] mm/execmem: make sure ROX cache always contains multiples of PMD_SIZE The ROX cache relies on its chunks being PMD mapped. For a PMD mapped chunk, set_memory_rox() updates the direct map alias one PMD at a time and the large mappings there survive. When execmem refills the cache, it rounds up the requested allocation size to PMD_SIZE and tries to allocate that with vmalloc(VM_ALLOW_HUGE_VMAP). If that allocation fails, execmem falls back to vmalloc() of the original size. There are two issues with this approach: * If huge pages are not available, __vmalloc_node_range() silently falls back to base pages. The area execmem gets is virtually contiguous, but it is backed by 512 scattered base pages. Permission updates on such areas split large mappings in the direct map that contain those base pages, up to 512 PMD splits in the worst case. * execmem's own fallback adds a small base page mapped area to the cache. This adds the overhead of cache management to these allocations with no benefit of reducing fragmentation either in the vmalloc/modules address space or in the direct map. Worse, these areas are never freed from the cache, because execmem_cache_clean() releases only chunks that are a multiple of PMD_SIZE and aligned to PMD_SIZE, exactly to minimize the number of base page mappings. Add VM_REQUIRE_HUGE_VMAP option to vmalloc that fails if the allocation of huge pages fails or if such an allocation is not possible because huge page allocations in vmalloc were disabled or the architecture does not support them. Use this option when populating the ROX cache. If vmalloc(VM_REQUIRE_HUGE_VMAP) fails or vmalloc of huge pages is unavailable, handle the memory allocation outside the ROX cache with plain vmalloc(). Since the fallback allocation has to return ROX memory, add an execmem_alloc_rox() helper and use it for both populating the ROX cache and dealing with a fallback allocation in a ROX execmem_range. With that, the cache only ever contains PMD aligned chunks sized as a multiple of PMD_SIZE, and the PMD checks in execmem_cache_clean() become a VM_WARN_ON_ONCE() to ensure that the PMD mapping invariant does not change. Link: https://lore.kernel.org/20260903-execmem-rox-cache-pmd-v1-v1-3-11beb2a3d249@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Assisted-by: copilot:claude-opus-5 Cc: Benjamin Tissoires Cc: Jiri Kosina Cc: Luis Chamberalin Cc: "Uladzislau Rezki (Sony)" --- include/linux/vmalloc.h | 1 + mm/execmem.c | 76 +++++++++++++++++++++++++---------------- mm/vmalloc.c | 15 +++++++- 3 files changed, 62 insertions(+), 30 deletions(-) diff --git a/include/linux/vmalloc.h b/include/linux/vmalloc.h index aed121d729b013..6e555e31e62251 100644 --- a/include/linux/vmalloc.h +++ b/include/linux/vmalloc.h @@ -38,6 +38,7 @@ struct iov_iter; /* in uio.h */ #define VM_DEFER_KMEMLEAK 0 #endif #define VM_SPARSE 0x00001000 /* sparse vm_area. not all pages are present. */ +#define VM_REQUIRE_HUGE_VMAP 0x00002000 /* huge page mapping or nothing */ /* bits [20..32] reserved for arch specific ioremap internals */ diff --git a/mm/execmem.c b/mm/execmem.c index 00dd6324cae01a..77653b7f163dcb 100644 --- a/mm/execmem.c +++ b/mm/execmem.c @@ -51,7 +51,8 @@ static void *execmem_vmalloc(struct execmem_range *range, size_t size, } if (!p) { - pr_warn_ratelimited("unable to allocate memory\n"); + if (!(vm_flags & VM_REQUIRE_HUGE_VMAP)) + pr_warn_ratelimited("unable to allocate memory\n"); return NULL; } @@ -146,9 +147,10 @@ static void execmem_cache_clean(struct work_struct *work) struct vm_struct *vm = find_vm_area(area); size_t size = mas_range_len(&mas); - if (vm && get_vm_area_size(vm) == size && - IS_ALIGNED(size, PMD_SIZE) && - IS_ALIGNED(mas.index, PMD_SIZE)) { + if (vm && get_vm_area_size(vm) == size) { + VM_WARN_ON_ONCE(!IS_ALIGNED(mas.index, PMD_SIZE) || + !IS_ALIGNED(size, PMD_SIZE)); + /* * Preallocate to ensure mas_store does not fail * If there is no memory for the tree update, bail out, @@ -264,38 +266,41 @@ static void *__execmem_cache_alloc(struct execmem_range *range, size_t size) return execmem_cache_alloc_locked(range, size); } -static void *execmem_cache_populate_alloc(struct execmem_range *range, size_t size) +static void *execmem_vmalloc_rox(struct execmem_range *range, size_t size, + unsigned long vm_flags) { - unsigned long vm_flags = VM_ALLOW_HUGE_VMAP; - struct mutex *mutex = &execmem_cache.mutex; - struct vm_struct *vm; - size_t alloc_size; - int err = -ENOMEM; - void *p; - - alloc_size = round_up(size, PMD_SIZE); - p = execmem_vmalloc(range, alloc_size, PAGE_KERNEL, vm_flags); - if (!p) { - alloc_size = size; - p = execmem_vmalloc(range, alloc_size, PAGE_KERNEL, vm_flags); - } + void *p = execmem_vmalloc(range, size, PAGE_KERNEL, vm_flags); + int err; if (!p) return NULL; - vm = find_vm_area(p); - if (!vm) - goto err_free_mem; - /* fill memory with instructions that will trap */ - execmem_fill_trapping_insns(p, alloc_size); - + execmem_fill_trapping_insns(p, size); set_vm_flush_reset_perms(p); - - err = set_memory_rox((unsigned long)p, vm->nr_pages); + err = set_memory_rox((unsigned long)p, size >> PAGE_SHIFT); if (err) goto err_free_mem; + return p; + +err_free_mem: + vfree(p); + return NULL; +} + +static void *execmem_cache_populate_alloc(struct execmem_range *range, size_t size) +{ + unsigned long vm_flags = VM_REQUIRE_HUGE_VMAP; + size_t alloc_size = round_up(size, PMD_SIZE); + struct mutex *mutex = &execmem_cache.mutex; + int err; + void *p; + + p = execmem_vmalloc_rox(range, alloc_size, vm_flags); + if (!p) + return NULL; + /* * New memory blocks must be allocated and added to the cache * as an atomic operation, otherwise they may be consumed @@ -317,6 +322,11 @@ static void *execmem_cache_populate_alloc(struct execmem_range *range, size_t si return NULL; } +static void *execmem_alloc_rox(struct execmem_range *range, size_t size) +{ + return execmem_vmalloc_rox(range, size, 0); +} + static void *execmem_cache_alloc(struct execmem_range *range, size_t size) { void *p; @@ -444,6 +454,11 @@ static void *execmem_cache_alloc(struct execmem_range *range, size_t size) return NULL; } +static void *execmem_alloc_rox(struct execmem_range *range, size_t size) +{ + return NULL; +} + static bool execmem_cache_free(void *ptr) { return false; @@ -453,17 +468,20 @@ static bool execmem_cache_free(void *ptr) void *execmem_alloc(enum execmem_type type, size_t size) { struct execmem_range *range = &execmem_info->ranges[type]; - bool use_cache = range->flags & EXECMEM_ROX_CACHE; + bool use_rox_cache = range->flags & EXECMEM_ROX_CACHE; unsigned long vm_flags = VM_FLUSH_RESET_PERMS; pgprot_t pgprot = range->pgprot; void *p = NULL; size = PAGE_ALIGN(size); - if (use_cache) + if (use_rox_cache) { p = execmem_cache_alloc(range, size); - else + if (!p) + p = execmem_alloc_rox(range, size); + } else { p = execmem_vmalloc(range, size, pgprot, vm_flags); + } return kasan_reset_tag(p); } diff --git a/mm/vmalloc.c b/mm/vmalloc.c index db357a9bdd1250..aed70e4f8e4e4e 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -4034,6 +4034,12 @@ static gfp_t vmalloc_fix_flags(gfp_t flags) * %__GFP_SKIP_KASAN can be used to skip unpoisoning of mapped pages * (when prot=%PAGE_KERNEL). * + * %VM_ALLOW_HUGE_VMAP allocates huge pages when possible and falls back to + * base pages if huge page allocation fails. + * + * %VM_REQUIRE_HUGE_VMAP implies %VM_ALLOW_HUGE_VMAP and fails instead of + * silently falling back to base pages. + * * Can not be called from interrupt nor NMI contexts. * Return: the address of the area or %NULL on failure */ @@ -4059,6 +4065,10 @@ void *__vmalloc_node_range_noprof(unsigned long size, unsigned long align, return NULL; } + /* VM_REQUIRE_HUGE_VMAP implies VM_ALLOW_HUGE_VMAP */ + if (vm_flags & VM_REQUIRE_HUGE_VMAP) + vm_flags |= VM_ALLOW_HUGE_VMAP; + if (vmap_allow_huge && (vm_flags & VM_ALLOW_HUGE_VMAP)) { /* * Try huge pages. Only try for PAGE_KERNEL allocations, @@ -4075,6 +4085,9 @@ void *__vmalloc_node_range_noprof(unsigned long size, unsigned long align, align = max(original_align, 1UL << shift); } + if ((vm_flags & VM_REQUIRE_HUGE_VMAP) && shift == PAGE_SHIFT) + return NULL; + again: area = __get_vm_area_node(size, align, shift, VM_ALLOC | VM_UNINITIALIZED | vm_flags, start, end, node, @@ -4149,7 +4162,7 @@ void *__vmalloc_node_range_noprof(unsigned long size, unsigned long align, return area->addr; fail: - if (shift > PAGE_SHIFT) { + if (shift > PAGE_SHIFT && !(vm_flags & VM_REQUIRE_HUGE_VMAP)) { shift = PAGE_SHIFT; align = original_align; goto again; From f354049d316854e14d907cc77728e5068d2f080f Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Thu, 3 Sep 2026 18:50:01 +0300 Subject: [PATCH 0810/1352] mm/vmalloc: add DEFINE_FREE() for vfree() ... and use it in hid-core, the only place with a cleanup for a vmalloc() allocation. Link: https://lore.kernel.org/20260903-execmem-rox-cache-pmd-v1-v1-4-11beb2a3d249@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Cc: Benjamin Tissoires Cc: Jiri Kosina Cc: Luis Chamberalin Cc: "Uladzislau Rezki (Sony)" --- drivers/hid/hid-core.c | 4 ++-- include/linux/vmalloc.h | 3 +++ 2 files changed, 5 insertions(+), 2 deletions(-) diff --git a/drivers/hid/hid-core.c b/drivers/hid/hid-core.c index a3ff0514f9cdf8..ec7c2860c93efb 100644 --- a/drivers/hid/hid-core.c +++ b/drivers/hid/hid-core.c @@ -944,7 +944,7 @@ static int hid_scan_report(struct hid_device *hid) hid_parser_reserved }; - struct hid_parser *parser __free(kvfree) = vzalloc(sizeof(*parser)); + struct hid_parser *parser __free(vfree) = vzalloc(sizeof(*parser)); if (!parser) return -ENOMEM; @@ -1265,7 +1265,7 @@ static int hid_parse_collections(struct hid_device *device) hid_parser_reserved }; - struct hid_parser *parser __free(kvfree) = vzalloc(sizeof(*parser)); + struct hid_parser *parser __free(vfree) = vzalloc(sizeof(*parser)); if (!parser) return -ENOMEM; diff --git a/include/linux/vmalloc.h b/include/linux/vmalloc.h index 6e555e31e62251..034a693777ca05 100644 --- a/include/linux/vmalloc.h +++ b/include/linux/vmalloc.h @@ -3,6 +3,7 @@ #define _LINUX_VMALLOC_H #include +#include #include #include #include @@ -215,6 +216,8 @@ void *__must_check vrealloc_node_align_noprof(const void *p, size_t size, extern void vfree(const void *addr); extern void vfree_atomic(const void *addr); +DEFINE_FREE(vfree, void *, if (!IS_ERR_OR_NULL(_T)) vfree(_T)) + extern void *vmap(struct page **pages, unsigned int count, unsigned long flags, pgprot_t prot); void *vmap_pfn(unsigned long *pfns, unsigned int count, pgprot_t prot); From ed0c772dbb2b81c7290da0bbad199e074a1af6b8 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Thu, 3 Sep 2026 18:50:02 +0300 Subject: [PATCH 0811/1352] mm/execmem: use cleanup infrastructure in ROX cache functions After splitting out execmem_alloc_rox() from execmem_cache_populate_alloc(), the error paths of both functions became less complex and can be easily switched to use the cleanup infrastructure. Use __free(vfree) to free allocated memory on the error paths and guard(mutex) for synchronization in ROX cache functions. Link: https://lore.kernel.org/20260903-execmem-rox-cache-pmd-v1-v1-5-11beb2a3d249@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Cc: Benjamin Tissoires Cc: Jiri Kosina Cc: Luis Chamberalin Cc: "Uladzislau Rezki (Sony)" --- mm/execmem.c | 32 ++++++++++---------------------- 1 file changed, 10 insertions(+), 22 deletions(-) diff --git a/mm/execmem.c b/mm/execmem.c index 77653b7f163dcb..349cadd874863a 100644 --- a/mm/execmem.c +++ b/mm/execmem.c @@ -138,11 +138,10 @@ int execmem_restore_rox(void *ptr, size_t size) static void execmem_cache_clean(struct work_struct *work) { struct maple_tree *free_areas = &execmem_cache.free_areas; - struct mutex *mutex = &execmem_cache.mutex; MA_STATE(mas, free_areas, 0, ULONG_MAX); void *area; - mutex_lock(mutex); + guard(mutex)(&execmem_cache.mutex); mas_for_each(&mas, area, ULONG_MAX) { struct vm_struct *vm = find_vm_area(area); size_t size = mas_range_len(&mas); @@ -163,7 +162,6 @@ static void execmem_cache_clean(struct work_struct *work) vfree(area); } } - mutex_unlock(mutex); } static DECLARE_WORK(execmem_cache_clean_work, execmem_cache_clean); @@ -269,7 +267,7 @@ static void *__execmem_cache_alloc(struct execmem_range *range, size_t size) static void *execmem_vmalloc_rox(struct execmem_range *range, size_t size, unsigned long vm_flags) { - void *p = execmem_vmalloc(range, size, PAGE_KERNEL, vm_flags); + void *p __free(vfree) = execmem_vmalloc(range, size, PAGE_KERNEL, vm_flags); int err; if (!p) @@ -280,22 +278,17 @@ static void *execmem_vmalloc_rox(struct execmem_range *range, size_t size, set_vm_flush_reset_perms(p); err = set_memory_rox((unsigned long)p, size >> PAGE_SHIFT); if (err) - goto err_free_mem; - - return p; + return NULL; -err_free_mem: - vfree(p); - return NULL; + return no_free_ptr(p); } static void *execmem_cache_populate_alloc(struct execmem_range *range, size_t size) { unsigned long vm_flags = VM_REQUIRE_HUGE_VMAP; size_t alloc_size = round_up(size, PMD_SIZE); - struct mutex *mutex = &execmem_cache.mutex; + void *p __free(vfree) = NULL; int err; - void *p; p = execmem_vmalloc_rox(range, alloc_size, vm_flags); if (!p) @@ -306,20 +299,15 @@ static void *execmem_cache_populate_alloc(struct execmem_range *range, size_t si * as an atomic operation, otherwise they may be consumed * by a parallel call to the execmem_cache_alloc function. */ - mutex_lock(mutex); + guard(mutex)(&execmem_cache.mutex); err = execmem_cache_add_locked(p, alloc_size, GFP_KERNEL); - if (!err) - p = execmem_cache_alloc_locked(range, size); - mutex_unlock(mutex); - if (err) - goto err_free_mem; + return NULL; - return p; + /* the chunk belongs to the cache now */ + retain_and_null_ptr(p); -err_free_mem: - vfree(p); - return NULL; + return execmem_cache_alloc_locked(range, size); } static void *execmem_alloc_rox(struct execmem_range *range, size_t size) From 881bda43132d6564a94d28a0c577620f5c9e0242 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 8 Sep 2026 11:16:31 -0400 Subject: [PATCH 0812/1352] mm/swap: add folio_swap_entry() and folio_page_swap_entry() Patch series "mm: remove page_swap_entry()", v2. A folio in the swap cache occupies folio_nr_pages() contiguous swap entries starting at folio->swap, so a page's swap entry is just folio->swap plus the page's index in the folio. The swap entry is folio state, but the only helper for it is page-based: page_swap_entry() takes a page and recomputes the folio its callers already hold. Several callers avoid it by open-coding the arithmetic on folio->swap instead. Add folio_swap_entry() and folio_page_swap_entry(), convert all users, and remove page_swap_entry(), with a few other cleanups along the way. This patch (of 8): A folio in the swap cache occupies folio_nr_pages() contiguous swap entries starting at folio->swap, so a page's swap entry is just folio->swap plus the page's index in the folio. page_swap_entry() hides this behind a compound_head() call, and callers that already have the folio sometimes open-code the arithmetic instead. Add folio_swap_entry(), which takes a folio and a page index, and folio_page_swap_entry() for callers that have the page. Link: https://lore.kernel.org/20260908-folio_swap_entry-v2-0-ee6d01dfa5e1@columbia.edu Link: https://lore.kernel.org/20260908-folio_swap_entry-v2-1-ee6d01dfa5e1@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Dev Jain Cc: Harry Yoo Cc: Jann Horn Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Lance Yang Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Marc Rutland Cc: Nhat Pham Cc: Rik van Riel Cc: Ryan Roberts Cc: Vlastimil Babka Cc: Will Deacon Cc: Zi Yan --- include/linux/swap.h | 33 +++++++++++++++++++++++++++++++++ 1 file changed, 33 insertions(+) diff --git a/include/linux/swap.h b/include/linux/swap.h index 74ce794042474c..3da4e5843c0105 100644 --- a/include/linux/swap.h +++ b/include/linux/swap.h @@ -272,6 +272,39 @@ struct swap_info_struct { const struct swap_ops *ops; }; +/** + * folio_swap_entry - Return the swap entry for a page within a folio. + * @folio: The folio. + * @idx: The index of the page within the folio. + * + * A folio in the swap cache occupies folio_nr_pages() contiguous swap + * entries starting at folio->swap. The caller must ensure the folio is + * in the swap cache and that @idx is within the folio. + */ +static inline +swp_entry_t folio_swap_entry(const struct folio *folio, unsigned long idx) +{ + swp_entry_t entry = folio->swap; + + VM_WARN_ON_ONCE_FOLIO(idx >= folio_nr_pages(folio), folio); + entry.val += idx; + return entry; +} + +/** + * folio_page_swap_entry - Return the swap entry of a page in a folio. + * @folio: The folio containing @page. + * @page: A page within @folio. + * + * The caller must ensure the folio is in the swap cache and that @page + * is part of @folio. + */ +static inline swp_entry_t folio_page_swap_entry(const struct folio *folio, + const struct page *page) +{ + return folio_swap_entry(folio, folio_page_idx(folio, page)); +} + static inline swp_entry_t page_swap_entry(struct page *page) { struct folio *folio = page_folio(page); From 84c83b6658c3d005e55dcb3d2b396ce8cf609b3e Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Sun, 13 Sep 2026 17:52:28 -0400 Subject: [PATCH 0813/1352] mm-swap-add-folio_swap_entry-and-folio_page_swap_entry-fix adjust the folio_swap_entry() kerneldoc summary, per David Link: https://lore.kernel.org/20260913-folio_swap_entry-doc-fix-1@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Cc: David Hildenbrand --- include/linux/swap.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/include/linux/swap.h b/include/linux/swap.h index 3da4e5843c0105..a59737b7268173 100644 --- a/include/linux/swap.h +++ b/include/linux/swap.h @@ -273,7 +273,7 @@ struct swap_info_struct { }; /** - * folio_swap_entry - Return the swap entry for a page within a folio. + * folio_swap_entry - Return the swap entry at a page index within a folio. * @folio: The folio. * @idx: The index of the page within the folio. * From 801179fe4bc6209721e9ea486f4450db69263000 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 8 Sep 2026 11:16:32 -0400 Subject: [PATCH 0814/1352] mm/huge_memory: add a comment to the open-coded swap entry The swap entry of each new folio in __split_folio_to_order() is computed by hand from folio->swap rather than with folio_swap_entry(), because the folio's page count is not valid while it is being split. Add a comment explaining this. Link: https://lore.kernel.org/20260908-folio_swap_entry-v2-2-ee6d01dfa5e1@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Dev Jain Cc: Harry Yoo Cc: Jann Horn Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Lance Yang Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Marc Rutland Cc: Nhat Pham Cc: Rik van Riel Cc: Ryan Roberts Cc: Vlastimil Babka Cc: Will Deacon --- mm/huge_memory.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index dd66c6ad5af13c..009eb3adc2b783 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -3762,6 +3762,10 @@ static void __split_folio_to_order(struct folio *folio, int old_order, */ VM_WARN_ON_ONCE_PAGE(new_folio->private, new_head); + /* + * Not all folio fields are valid during a split, so open-code + * the swap entry rather than using folio_swap_entry(). + */ if (folio_test_swapcache(folio)) new_folio->swap.val = folio->swap.val + i; From c9520fd19dc9748e36835bf485a7ed7d3174e555 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 8 Sep 2026 11:16:33 -0400 Subject: [PATCH 0815/1352] mm/rmap: use folio_page_swap_entry() in ttu_anon_swapbacked_folio() We already have the folio here, so use folio_page_swap_entry() instead of going through page_swap_entry(). This saves a call to compound_head(). No functional change. Link: https://lore.kernel.org/20260908-folio_swap_entry-v2-3-ee6d01dfa5e1@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Dev Jain Cc: Harry Yoo Cc: Jann Horn Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Lance Yang Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Marc Rutland Cc: Nhat Pham Cc: Rik van Riel Cc: Ryan Roberts Cc: Vlastimil Babka Cc: Will Deacon Cc: Zi Yan --- mm/rmap.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/rmap.c b/mm/rmap.c index 0a3952706faf5c..5332c52909be18 100644 --- a/mm/rmap.c +++ b/mm/rmap.c @@ -2147,7 +2147,7 @@ static bool ttu_anon_swapbacked_folio(struct vm_area_struct *vma, { const bool anon_exclusive = folio_test_anon(folio) && PageAnonExclusive(page); - swp_entry_t entry = page_swap_entry(page); + swp_entry_t entry = folio_page_swap_entry(folio, page); struct mm_struct *mm = vma->vm_mm; if (folio_dup_swap(folio, page) < 0) From e9681f9764d715d871a6b7bb65355cf2f77a187d Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 8 Sep 2026 11:16:34 -0400 Subject: [PATCH 0816/1352] mm/zswap: use folio_swap_entry() in zswap_store_page() zswap_store_page() open-codes the swap entry computation from folio->swap and the page index. Use folio_swap_entry() instead. No functional change. Link: https://lore.kernel.org/20260908-folio_swap_entry-v2-4-ee6d01dfa5e1@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Dev Jain Cc: Harry Yoo Cc: Jann Horn Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Lance Yang Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Marc Rutland Cc: Nhat Pham Cc: Rik van Riel Cc: Ryan Roberts Cc: Vlastimil Babka Cc: Will Deacon Cc: Zi Yan --- mm/zswap.c | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/mm/zswap.c b/mm/zswap.c index b894de1786fd3f..3a6f8901764641 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -1395,8 +1395,7 @@ static bool zswap_store_page(struct folio *folio, long index, struct obj_cgroup *objcg, struct zswap_pool *pool) { - swp_entry_t page_swpentry = swp_entry(swp_type(folio->swap), - swp_offset(folio->swap) + index); + swp_entry_t page_swpentry = folio_swap_entry(folio, index); struct zswap_entry *entry, *old; /* allocate entry */ From 5178193c3364d567f0851cb8fe4628d5b41811d6 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 8 Sep 2026 11:16:35 -0400 Subject: [PATCH 0817/1352] mm/swapfile: use folio_page_swap_entry() folio_dup_swap() and folio_put_swap() open-code the swap entry computation from folio->swap and folio_page_idx(). Use folio_page_swap_entry() instead. No functional change. Link: https://lore.kernel.org/20260908-folio_swap_entry-v2-5-ee6d01dfa5e1@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Dev Jain Cc: Harry Yoo Cc: Jann Horn Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Lance Yang Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Marc Rutland Cc: Nhat Pham Cc: Rik van Riel Cc: Ryan Roberts Cc: Vlastimil Babka Cc: Will Deacon Cc: Zi Yan --- mm/swapfile.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/mm/swapfile.c b/mm/swapfile.c index 01e7b6b046b67d..48d3cd40defdb4 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -1822,7 +1822,7 @@ int folio_dup_swap(struct folio *folio, struct page *page) VM_WARN_ON_FOLIO(!folio_test_swapcache(folio), folio); if (page) { - entry.val += folio_page_idx(folio, page); + entry = folio_page_swap_entry(folio, page); nr_pages = 1; } @@ -1849,7 +1849,7 @@ void folio_put_swap(struct folio *folio, struct page *page) VM_WARN_ON_FOLIO(!folio_test_swapcache(folio), folio); if (page) { - entry.val += folio_page_idx(folio, page); + entry = folio_page_swap_entry(folio, page); nr_pages = 1; } From 87daa49d1b0f2b01e4d36becc3dd97095b640f43 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 8 Sep 2026 11:16:36 -0400 Subject: [PATCH 0818/1352] arm64: mte: make mte_save_tags() and mte_restore_tags() static mte_save_tags() and mte_restore_tags() are only used by arch_prepare_to_swap() and arch_swap_restore() in mteswap.c. Make them static and remove their declarations from mte.h. No functional change. Link: https://lore.kernel.org/20260908-folio_swap_entry-v2-6-ee6d01dfa5e1@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Dev Jain Cc: Harry Yoo Cc: Jann Horn Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Lance Yang Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Marc Rutland Cc: Nhat Pham Cc: Rik van Riel Cc: Ryan Roberts Cc: Vlastimil Babka Cc: Will Deacon Cc: Zi Yan --- arch/arm64/include/asm/mte.h | 2 -- arch/arm64/mm/mteswap.c | 4 ++-- 2 files changed, 2 insertions(+), 4 deletions(-) diff --git a/arch/arm64/include/asm/mte.h b/arch/arm64/include/asm/mte.h index 7f7b97e099968e..83f3b05fc78490 100644 --- a/arch/arm64/include/asm/mte.h +++ b/arch/arm64/include/asm/mte.h @@ -23,9 +23,7 @@ unsigned long mte_copy_tags_from_user(void *to, const void __user *from, unsigned long n); unsigned long mte_copy_tags_to_user(void __user *to, void *from, unsigned long n); -int mte_save_tags(struct page *page); void mte_save_page_tags(const void *page_addr, void *tag_storage); -void mte_restore_tags(swp_entry_t entry, struct page *page); void mte_restore_page_tags(void *page_addr, const void *tag_storage); void mte_invalidate_tags(int type, pgoff_t offset); void mte_invalidate_tags_area(int type); diff --git a/arch/arm64/mm/mteswap.c b/arch/arm64/mm/mteswap.c index 63e8d72f202a3b..3c54afc8b7585b 100644 --- a/arch/arm64/mm/mteswap.c +++ b/arch/arm64/mm/mteswap.c @@ -20,7 +20,7 @@ void mte_free_tag_storage(char *storage) kfree(storage); } -int mte_save_tags(struct page *page) +static int mte_save_tags(struct page *page) { void *tag_storage, *ret; @@ -47,7 +47,7 @@ int mte_save_tags(struct page *page) return 0; } -void mte_restore_tags(swp_entry_t entry, struct page *page) +static void mte_restore_tags(swp_entry_t entry, struct page *page) { void *tags = xa_load(&mte_pages, entry.val); From bb3ea36d7637b7c95f4d77a4183db6de600b52d0 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 8 Sep 2026 11:16:37 -0400 Subject: [PATCH 0819/1352] arm64: mte: pass the swap entry to mte_save_tags() arch_prepare_to_swap() derives a page from the folio only for mte_save_tags() to recompute the folio's swap entry from that page. Pass the entry in directly with folio_swap_entry(), matching mte_restore_tags(), and move the page_mte_tagged() check into the caller so we only compute the entry for tagged pages. __mte_invalidate_tags() loses its only user, so remove it and call mte_invalidate_tags() directly in the error path. This removes two calls to compound_head(). No functional change. Link: https://lore.kernel.org/20260908-folio_swap_entry-v2-7-ee6d01dfa5e1@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Dev Jain Cc: Harry Yoo Cc: Jann Horn Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Lance Yang Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Marc Rutland Cc: Nhat Pham Cc: Rik van Riel Cc: Ryan Roberts Cc: Vlastimil Babka Cc: Will Deacon Cc: Zi Yan --- arch/arm64/mm/mteswap.c | 30 +++++++++++++----------------- 1 file changed, 13 insertions(+), 17 deletions(-) diff --git a/arch/arm64/mm/mteswap.c b/arch/arm64/mm/mteswap.c index 3c54afc8b7585b..f64202b69309a7 100644 --- a/arch/arm64/mm/mteswap.c +++ b/arch/arm64/mm/mteswap.c @@ -20,22 +20,17 @@ void mte_free_tag_storage(char *storage) kfree(storage); } -static int mte_save_tags(struct page *page) +static int mte_save_tags(swp_entry_t entry, struct page *page) { void *tag_storage, *ret; - if (!page_mte_tagged(page)) - return 0; - tag_storage = mte_allocate_tag_storage(); if (!tag_storage) return -ENOMEM; mte_save_page_tags(page_address(page), tag_storage); - /* lookup the swap entry.val from the page */ - ret = xa_store(&mte_pages, page_swap_entry(page).val, tag_storage, - GFP_KERNEL); + ret = xa_store(&mte_pages, entry.val, tag_storage, GFP_KERNEL); if (WARN(xa_is_err(ret), "Failed to store MTE tags")) { mte_free_tag_storage(tag_storage); return xa_err(ret); @@ -68,13 +63,6 @@ void mte_invalidate_tags(int type, pgoff_t offset) mte_free_tag_storage(tags); } -static inline void __mte_invalidate_tags(struct page *page) -{ - swp_entry_t entry = page_swap_entry(page); - - mte_invalidate_tags(swp_type(entry), swp_offset(entry)); -} - void mte_invalidate_tags_area(int type) { swp_entry_t entry = swp_entry(type, 0); @@ -102,15 +90,23 @@ int arch_prepare_to_swap(struct folio *folio) nr = folio_nr_pages(folio); for (i = 0; i < nr; i++) { - err = mte_save_tags(folio_page(folio, i)); + struct page *page = folio_page(folio, i); + + if (!page_mte_tagged(page)) + continue; + + err = mte_save_tags(folio_swap_entry(folio, i), page); if (err) goto out; } return 0; out: - while (i--) - __mte_invalidate_tags(folio_page(folio, i)); + while (i--) { + swp_entry_t swap = folio_swap_entry(folio, i); + + mte_invalidate_tags(swp_type(swap), swp_offset(swap)); + } return err; } From 1d69b46c12cfbfc81914f91065521179c1039d22 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 8 Sep 2026 11:16:38 -0400 Subject: [PATCH 0820/1352] mm/swap: remove page_swap_entry() All callers have been converted to folio_swap_entry() and folio_page_swap_entry(), so remove it. Link: https://lore.kernel.org/20260908-folio_swap_entry-v2-8-ee6d01dfa5e1@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Dev Jain Cc: Harry Yoo Cc: Jann Horn Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Lance Yang Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Marc Rutland Cc: Nhat Pham Cc: Rik van Riel Cc: Ryan Roberts Cc: Vlastimil Babka Cc: Will Deacon Cc: Zi Yan --- include/linux/swap.h | 9 --------- 1 file changed, 9 deletions(-) diff --git a/include/linux/swap.h b/include/linux/swap.h index a59737b7268173..a37ad8375e011e 100644 --- a/include/linux/swap.h +++ b/include/linux/swap.h @@ -305,15 +305,6 @@ static inline swp_entry_t folio_page_swap_entry(const struct folio *folio, return folio_swap_entry(folio, folio_page_idx(folio, page)); } -static inline swp_entry_t page_swap_entry(struct page *page) -{ - struct folio *folio = page_folio(page); - swp_entry_t entry = folio->swap; - - entry.val += folio_page_idx(folio, page); - return entry; -} - /* linux/mm/page_alloc.c */ extern unsigned long totalreserve_pages; From 54a3b9db48c261df7f76239c8d17c2a04df08eea Mon Sep 17 00:00:00 2001 From: "Zenghui Yu (Huawei)" Date: Tue, 8 Sep 2026 06:52:55 -0700 Subject: [PATCH 0821/1352] Docs/mm/damon/design: fix broken :ref: usage and a typo The Statistics section refers readers to the DAMON sysfs interface documentation for how to read the statistics, using a ':ref:' role. However, the role name has a superfluous 's' appended (':ref:s' instead of ':ref:'). Moreover, its target label 'sysfs_stats' does not exist; the correct label is 'sysfs_schemes_stats'. Fix both. Also fix a typo, 'frequenceis' -> 'frequencies', in the Adaptive Regions Adjustment section. Link: https://lore.kernel.org/20260908135257.97523-1-sj@kernel.org Signed-off-by: Zenghui Yu (Huawei) Reviewed-by: SJ Park Signed-off-by: SJ Park Signed-off-by: Andrew Morton --- Documentation/mm/damon/design.rst | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index d036340dae8afb..aac84de261aa8a 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -249,7 +249,7 @@ For each ``aggregation interval``, it compares the access frequencies sum of the two regions' sizes is smaller than the size of total regions divided by the ``minimum number of regions``, DAMON merges the two regions. If the resulting number of total regions is still higher than ``maximum number of -regions``, it repeats the merging with increasing access frequenceis difference +regions``, it repeats the merging with increasing access frequencies difference threshold until the upper-limit of the number of regions is met, or the threshold becomes higher than possible maximum value (``aggregation interval`` divided by ``sampling interval``). Then, after it reports and clears the @@ -889,7 +889,7 @@ the scheme is deactivated. Note that, unlike watermarks, even if a scheme's ``nr_snapshots`` reaches ``max_nr_snapshots``, monitoring will not stop. To know how user-space can read the stats via :ref:`DAMON sysfs interface -`, refer to :ref:s`stats ` part of the +`, refer to :ref:`stats ` part of the documentation. Regions Walking From c145c69ad8b0c0953b51c26ff95392147d603783 Mon Sep 17 00:00:00 2001 From: Krishna Iyer Date: Tue, 8 Sep 2026 06:51:53 -0700 Subject: [PATCH 0822/1352] mm/damon: move damon_hugetlb_mkold() from vaddr to ops-common Patch series "mm/damon: support access monitoring of hugetlb-backed memory", v3. On virtualization hosts, most system memory is often backed by hugetlbfs. On our production hosts, for example, ~95% of RAM is 1 GiB hugetlb pages backing guest memory. DAMON's physical address space monitoring is blind to such memory: every access check starts at damon_get_folio(), which rejects folios that are not on the LRU lists, and hugetlb folios are managed outside of the LRU by design. As a result, all hugetlb-backed memory is silently reported as never accessed. In testing on a 1 TiB host, an hour of 4-thread random access over 842 GiB inside a guest was statistically indistinguishable from an idle host. The first patch moves damon_hugetlb_mkold() from vaddr to ops-common as a preparation. The second patch teaches the folio mkold/young rmap walkers to handle hugetlb folios, aging the huge PTE and notifying secondary MMUs across the whole huge page size; the secondary MMU notification is what surfaces guest-side (e.g., KVM/EPT) accessed bits. The third patch adds damon_get_monitor_folio() and uses it from the paddr monitoring primitives only. DAMOS action appliers such as DAMON_RECLAIM and DAMON_LRU_SORT keep the LRU-only lookup and are behaviorally unchanged. This series is the first half of an earlier six-patch series [1], split out as SJ suggested [2]. The second half (the 'aging_flush' TLB-flush-assisted aging) is deferred: we will gather more quantitative data on the gap it addresses, including the workload-side impact of the flushes and the working set measurement details SJ asked about, and post it separately once the data is in hand, aligned with the ongoing monitoring preparation actions work. Per Documentation/process/generated-content.rst, this series was developed with the assistance of an AI coding assistant (Anthropic Claude, via Claude Code). The assistant helped draft the code and changelogs, and applied the v1 review feedback. All changes were reviewed by the human submitter, who takes full responsibility for the contribution. The series as posted here was regression-tested on its base commit with a full x86_64 kernel build (no W=1 warnings in mm/damon), the DAMON kunit suite (41/41 passing) and the DAMON selftests (15/15 passing) on a kernel booted with virtme-ng. This patch (of 3): damon_hugetlb_mkold() clears the accessed bit of a hugetlb-mapping huge PTE and propagates the aging to secondary MMUs via mmu_notifier_clear_young(), spanning the whole huge page size. It currently lives in vaddr.c, and is thus usable only by the virtual address space monitoring operations set. The physical address space monitoring operations set will need the same logic, to support access monitoring of hugetlb-backed memory. Move the function to ops-common as-is, with no behavioral change. A follow-up change will use it from the folio-granular rmap walkers. Link: https://lore.kernel.org/20260908135156.97481-1-sj@kernel.org Link: https://lore.kernel.org/20260902025700.17975-2-kiyer@crusoe.ai Link: https://lore.kernel.org/20260908135156.97481-2-sj@kernel.org Signed-off-by: Krishna Iyer Reviewed-by: SJ Park Signed-off-by: SJ Park Signed-off-by: Andrew Morton Assisted-by: Claude:claude-fable-5 --- mm/damon/ops-common.c | 37 +++++++++++++++++++++++++++++++++++++ mm/damon/ops-common.h | 9 +++++++++ mm/damon/vaddr.c | 34 ---------------------------------- 3 files changed, 46 insertions(+), 34 deletions(-) diff --git a/mm/damon/ops-common.c b/mm/damon/ops-common.c index 7219c608b1952b..995cc1f3b9f32e 100644 --- a/mm/damon/ops-common.c +++ b/mm/damon/ops-common.c @@ -3,6 +3,7 @@ * Common Code for Data Access Monitoring */ +#include #include #include #include @@ -103,6 +104,42 @@ void damon_pmdp_mkold(pmd_t *pmd, struct vm_area_struct *vma, unsigned long addr #endif /* CONFIG_TRANSPARENT_HUGEPAGE */ } +#ifdef CONFIG_HUGETLB_PAGE +static bool damon_hugetlb_ptep_mkold(pte_t *pte, struct mm_struct *mm, + struct vm_area_struct *vma, unsigned long addr, pte_t *entry) +{ + unsigned long psize = huge_page_size(hstate_vma(vma)); + + if (!pte_young(*entry)) + return false; + *entry = huge_ptep_get_and_clear(mm, addr, pte, psize); + *entry = pte_mkold(*entry); + set_huge_pte_at(mm, addr, pte, *entry, psize); + return true; +} + +void damon_hugetlb_mkold(pte_t *pte, struct mm_struct *mm, + struct vm_area_struct *vma, unsigned long addr) +{ + bool referenced = false; + pte_t entry = huge_ptep_get(mm, addr, pte); + struct folio *folio = pfn_folio(pte_pfn(entry)); + + folio_get(folio); + + referenced = damon_hugetlb_ptep_mkold(pte, mm, vma, addr, &entry); + if (mmu_notifier_clear_young(mm, addr, + addr + huge_page_size(hstate_vma(vma)))) + referenced = true; + + if (referenced) + folio_set_young(folio); + + folio_set_idle(folio); + folio_put(folio); +} +#endif /* CONFIG_HUGETLB_PAGE */ + #define DAMON_MAX_SUBSCORE (100) #define DAMON_MAX_AGE_IN_LOG (32) diff --git a/mm/damon/ops-common.h b/mm/damon/ops-common.h index 38d295488fa181..f7811c9c7a024b 100644 --- a/mm/damon/ops-common.h +++ b/mm/damon/ops-common.h @@ -9,6 +9,15 @@ struct folio *damon_get_folio(unsigned long pfn); void damon_ptep_mkold(pte_t *pte, struct vm_area_struct *vma, unsigned long addr); void damon_pmdp_mkold(pmd_t *pmd, struct vm_area_struct *vma, unsigned long addr); +#ifdef CONFIG_HUGETLB_PAGE +void damon_hugetlb_mkold(pte_t *pte, struct mm_struct *mm, + struct vm_area_struct *vma, unsigned long addr); +#else +static inline void damon_hugetlb_mkold(pte_t *pte, struct mm_struct *mm, + struct vm_area_struct *vma, unsigned long addr) +{ +} +#endif /* CONFIG_HUGETLB_PAGE */ void damon_folio_mkold(struct folio *folio); bool damon_folio_young(struct folio *folio); diff --git a/mm/damon/vaddr.c b/mm/damon/vaddr.c index 91a0d441c1f940..af9e1b82454cc2 100644 --- a/mm/damon/vaddr.c +++ b/mm/damon/vaddr.c @@ -283,40 +283,6 @@ static int damon_mkold_pmd_entry(pmd_t *pmd, unsigned long addr, } #ifdef CONFIG_HUGETLB_PAGE -static bool damon_hugetlb_ptep_mkold(pte_t *pte, struct mm_struct *mm, - struct vm_area_struct *vma, unsigned long addr, pte_t *entry) -{ - unsigned long psize = huge_page_size(hstate_vma(vma)); - - if (!pte_young(*entry)) - return false; - *entry = huge_ptep_get_and_clear(mm, addr, pte, psize); - *entry = pte_mkold(*entry); - set_huge_pte_at(mm, addr, pte, *entry, psize); - return true; -} - -static void damon_hugetlb_mkold(pte_t *pte, struct mm_struct *mm, - struct vm_area_struct *vma, unsigned long addr) -{ - bool referenced = false; - pte_t entry = huge_ptep_get(mm, addr, pte); - struct folio *folio = pfn_folio(pte_pfn(entry)); - - folio_get(folio); - - referenced = damon_hugetlb_ptep_mkold(pte, mm, vma, addr, &entry); - if (mmu_notifier_clear_young(mm, addr, - addr + huge_page_size(hstate_vma(vma)))) - referenced = true; - - if (referenced) - folio_set_young(folio); - - folio_set_idle(folio); - folio_put(folio); -} - static int damon_mkold_hugetlb_entry(pte_t *pte, unsigned long hmask, unsigned long addr, unsigned long end, struct mm_walk *walk) From 69e6edafc1aa296f1a25f7c8706bb550bd830771 Mon Sep 17 00:00:00 2001 From: Krishna Iyer Date: Tue, 8 Sep 2026 06:51:54 -0700 Subject: [PATCH 0823/1352] mm/damon/ops-common: handle hugetlb folios in folio mkold/young rmap walkers damon_folio_mkold_one() and damon_folio_young_one() assume the folios they walk are mapped by normal PTEs or THP PMDs. When the folio is a hugetlb folio, page_vma_mapped_walk() returns the huge PTE in pvmw.pte with its page table lock held, but the walkers treat it as a normal PTE: they read and age it with PAGE_SIZE-granularity helpers, which is wrong for huge PTEs (up to PUD level), and notify secondary MMUs for only PAGE_SIZE of the mapping. Add hugetlb branches to both walkers. The mkold walker reuses damon_hugetlb_mkold(), which the virtual address space operations set has been using for hugetlb aging: it clears the young bit of the huge PTE via set_huge_pte_at() and calls mmu_notifier_clear_young() spanning the whole huge page size. The young walker gets an equivalent new helper, damon_hugetlb_young(), which reads the huge PTE with huge_ptep_get() and consults the page idle flag and mmu_notifier_test_young() like the existing PTE branch. Locking mirrors what page_vma_mapped_walk() provides: the huge PTE's page table lock is held inside the walk, and for shared hugetlb mappings (the only ones subject to huge PMD sharing), rmap_walk_file() already holds i_mmap_rwsem, satisfying hugetlb_walk()'s locking requirements. This is currently dead code: both rmap walkers are only reachable through damon_get_folio(), which rejects hugetlb folios since they are not on the LRU lists. A following commit will let the physical address space monitoring primitives opt in to hugetlb folios. Link: https://lore.kernel.org/20260902025700.17975-3-kiyer@crusoe.ai Link: https://lore.kernel.org/20260908135156.97481-3-sj@kernel.org Signed-off-by: Krishna Iyer Reviewed-by: SJ Park Signed-off-by: SJ Park Signed-off-by: Andrew Morton Assisted-by: Claude:claude-fable-5 --- mm/damon/ops-common.c | 61 +++++++++++++++++++++++++++++++++---------- 1 file changed, 47 insertions(+), 14 deletions(-) diff --git a/mm/damon/ops-common.c b/mm/damon/ops-common.c index 995cc1f3b9f32e..349e1604cc1b1f 100644 --- a/mm/damon/ops-common.c +++ b/mm/damon/ops-common.c @@ -205,10 +205,15 @@ static bool damon_folio_mkold_one(struct folio *folio, while (page_vma_mapped_walk(&pvmw)) { addr = pvmw.address; - if (pvmw.pte) - damon_ptep_mkold(pvmw.pte, vma, addr); - else + if (pvmw.pte) { + if (folio_test_hugetlb(folio)) + damon_hugetlb_mkold(pvmw.pte, vma->vm_mm, vma, + addr); + else + damon_ptep_mkold(pvmw.pte, vma, addr); + } else { damon_pmdp_mkold(pvmw.pmd, vma, addr); + } } return true; } @@ -233,27 +238,55 @@ void damon_folio_mkold(struct folio *folio) } +#ifdef CONFIG_HUGETLB_PAGE +static bool damon_hugetlb_young(pte_t *pte, struct vm_area_struct *vma, + unsigned long addr, struct folio *folio) +{ + pte_t entry = huge_ptep_get(vma->vm_mm, addr, pte); + + return (pte_present(entry) && pte_young(entry)) || + !folio_test_idle(folio) || + mmu_notifier_test_young(vma->vm_mm, addr); +} +#else +static bool damon_hugetlb_young(pte_t *pte, struct vm_area_struct *vma, + unsigned long addr, struct folio *folio) +{ + return false; +} +#endif /* CONFIG_HUGETLB_PAGE */ + +static bool damon_pte_young(pte_t *pte, struct vm_area_struct *vma, + unsigned long addr, struct folio *folio) +{ + pte_t entry = ptep_get(pte); + + /* + * PFN swap PTEs, such as device-exclusive ones, that actually map + * pages are "old" from a CPU perspective. The MMU notifier takes care + * of any device aspects. + */ + return (pte_present(entry) && pte_young(entry)) || + !folio_test_idle(folio) || + mmu_notifier_test_young(vma->vm_mm, addr); +} + static bool damon_folio_young_one(struct folio *folio, struct vm_area_struct *vma, unsigned long addr, void *arg) { bool *accessed = arg; DEFINE_FOLIO_VMA_WALK(pvmw, folio, vma, addr, 0); - pte_t pte; *accessed = false; while (page_vma_mapped_walk(&pvmw)) { addr = pvmw.address; if (pvmw.pte) { - pte = ptep_get(pvmw.pte); - - /* - * PFN swap PTEs, such as device-exclusive ones, that - * actually map pages are "old" from a CPU perspective. - * The MMU notifier takes care of any device aspects. - */ - *accessed = (pte_present(pte) && pte_young(pte)) || - !folio_test_idle(folio) || - mmu_notifier_test_young(vma->vm_mm, addr); + if (folio_test_hugetlb(folio)) + *accessed = damon_hugetlb_young(pvmw.pte, vma, + addr, folio); + else + *accessed = damon_pte_young(pvmw.pte, vma, + addr, folio); } else { #ifdef CONFIG_TRANSPARENT_HUGEPAGE pmd_t pmd = pmdp_get(pvmw.pmd); From 6d56ce1b689f7c74e4833667c6d15e88bb93bdb4 Mon Sep 17 00:00:00 2001 From: Krishna Iyer Date: Tue, 8 Sep 2026 06:51:55 -0700 Subject: [PATCH 0824/1352] mm/damon/paddr: support hugetlb folios in access monitoring DAMON's physical address space monitoring is blind to hugetlb-backed memory. Every access check starts at damon_get_folio(), which rejects folios that are not on the LRU lists. Hugetlb folios are managed outside of the LRU by design, so every sampling attempt on hugetlb-backed memory silently fails and the pages are reported as never accessed. This is a significant blind spot on virtualization hosts. Cloud hypervisor hosts commonly back guest memory with 1 GiB hugetlbfs pages, covering the vast majority of the machine's memory. On such hosts, modules like DAMON_STAT observe only the host-side remainder (page cache, daemons) and report all guest working sets as permanently idle, defeating the purpose of host-level access monitoring. In testing on a 1 TiB host, an hour of 4-thread random access over 842 GiB inside a guest was statistically indistinguishable from an idle host, while a 40x smaller host-side workload produced a quantitatively correct response. Add damon_get_monitor_folio(), which additionally accepts hugetlb folios, and use it in the two paddr access monitoring primitives, damon_pa_mkold() and damon_pa_young(). With the previous commit teaching the folio-granular rmap walkers to age huge PTEs and to call the mmu notifiers spanning the whole huge page, this makes guest accesses visible through secondary MMU (e.g. KVM/EPT) young bits. Free hugetlb pool folios have a zero refcount, so folio_try_get() naturally keeps rejecting them. The DAMOS action appliers (damon_pa_pageout(), damon_pa_mark_accessed_or_deactivate(), damon_pa_migrate(), damon_pa_stat()) keep using damon_get_folio(): reclaim, LRU manipulation and migration cannot act on hugetlb folios, so their behavior is unchanged. Note that the access check granularity for hugetlb-backed memory is the huge page size: one touched byte reports the whole (up to 1 GiB) page as accessed. Also, DAMON now consumes secondary MMU young bits that KVM's own aging uses; at DAMON's sampling rate (one page per region per sampling interval) the interference is negligible. Link: https://lore.kernel.org/20260902025700.17975-4-kiyer@crusoe.ai Link: https://lore.kernel.org/20260908135156.97481-4-sj@kernel.org Signed-off-by: Krishna Iyer Reviewed-by: SJ Park Signed-off-by: SJ Park Signed-off-by: Andrew Morton Assisted-by: Claude:claude-fable-5 --- mm/damon/ops-common.c | 25 +++++++++++++++++++++---- mm/damon/ops-common.h | 1 + mm/damon/paddr.c | 4 ++-- 3 files changed, 24 insertions(+), 6 deletions(-) diff --git a/mm/damon/ops-common.c b/mm/damon/ops-common.c index 349e1604cc1b1f..acf8f216c51cc8 100644 --- a/mm/damon/ops-common.c +++ b/mm/damon/ops-common.c @@ -15,14 +15,20 @@ #include "../internal.h" #include "ops-common.h" +static bool damon_folio_acceptable(struct folio *folio, bool monitor) +{ + return folio_test_lru(folio) || + (monitor && folio_test_hugetlb(folio)); +} + /* - * Get an online page for a pfn if it's in the LRU list. Otherwise, returns - * NULL. + * Get an online page for a pfn if it's in the LRU list, or a hugetlb folio if + * @monitor is set. Otherwise, returns NULL. * * The body of this function is stolen from the 'page_idle_get_folio()'. We * steal rather than reuse it because the code is quite simple. */ -struct folio *damon_get_folio(unsigned long pfn) +static struct folio *__damon_get_folio(unsigned long pfn, bool monitor) { struct page *page = pfn_to_online_page(pfn); struct folio *folio; @@ -33,13 +39,24 @@ struct folio *damon_get_folio(unsigned long pfn) folio = page_folio(page); if (!folio_try_get(folio)) return NULL; - if (unlikely(page_folio(page) != folio) || !folio_test_lru(folio)) { + if (unlikely(page_folio(page) != folio) || + !damon_folio_acceptable(folio, monitor)) { folio_put(folio); folio = NULL; } return folio; } +struct folio *damon_get_folio(unsigned long pfn) +{ + return __damon_get_folio(pfn, false); +} + +struct folio *damon_get_monitor_folio(unsigned long pfn) +{ + return __damon_get_folio(pfn, true); +} + void damon_ptep_mkold(pte_t *pte, struct vm_area_struct *vma, unsigned long addr) { pte_t pteval = ptep_get(pte); diff --git a/mm/damon/ops-common.h b/mm/damon/ops-common.h index f7811c9c7a024b..172f0f17c4a84d 100644 --- a/mm/damon/ops-common.h +++ b/mm/damon/ops-common.h @@ -6,6 +6,7 @@ #include struct folio *damon_get_folio(unsigned long pfn); +struct folio *damon_get_monitor_folio(unsigned long pfn); void damon_ptep_mkold(pte_t *pte, struct vm_area_struct *vma, unsigned long addr); void damon_pmdp_mkold(pmd_t *pmd, struct vm_area_struct *vma, unsigned long addr); diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c index c1e7d7a4f40df3..d2173a448d0b0c 100644 --- a/mm/damon/paddr.c +++ b/mm/damon/paddr.c @@ -37,7 +37,7 @@ static unsigned long damon_pa_core_addr( static void damon_pa_mkold(phys_addr_t paddr) { - struct folio *folio = damon_get_folio(PHYS_PFN(paddr)); + struct folio *folio = damon_get_monitor_folio(PHYS_PFN(paddr)); if (!folio) return; @@ -67,7 +67,7 @@ static void damon_pa_prepare_access_checks(struct damon_ctx *ctx) static bool damon_pa_young(phys_addr_t paddr) { - struct folio *folio = damon_get_folio(PHYS_PFN(paddr)); + struct folio *folio = damon_get_monitor_folio(PHYS_PFN(paddr)); bool accessed; if (!folio) From 4f239bee50b02fdbf1763ce3a2631483a0bed878 Mon Sep 17 00:00:00 2001 From: "Zenghui Yu (Huawei)" Date: Tue, 8 Sep 2026 21:41:15 +0800 Subject: [PATCH 0825/1352] selftests/mm: fix size truncation in pagemap_ioctl test Patch series "selftests/mm: pagemap_ioctl test fixes and cleanups", v2. This series fixes a size truncation bug in the pagemap_ioctl test that breaks it on arm64 systems with 64K base pages, and applies two small cleanups suggested during the review of the fix. This patch (of 3): On arm64 with 64K base pages, the huge page size is 512 MiB, and hpage_unit_tests() builds a 5 GiB range (10 * 512 MiB) for its tests. This exceeds the range of the int size parameters of gethugepage(), wp_addr_range() and pagemap_ioctl(). The implicit truncation to 1 GiB makes gethugepage() allocate a too small buffer, while the callers keep operating on the original 5 GiB range, resulting in spurious failures or SIGSEGV. Fix the truncation by changing those size parameters to size_t, and for consistency, also convert the remaining size-related parameters and variables that use int, long or unsigned long long to size_t. Link: https://lore.kernel.org/20260908134117.84405-1-zenghui.yu@linux.dev Link: https://lore.kernel.org/20260908134117.84405-2-zenghui.yu@linux.dev Fixes: 46fd75d4a3c9 ("selftests: mm: add pagemap ioctl tests") Signed-off-by: Zenghui Yu (Huawei) Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand (Arm) Acked-by: David Hildenbrand (Arm) Assisted-by: GLM-5.3 OpenCode Cc: Gregory Price Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/mm/pagemap_ioctl.c | 55 ++++++++++++---------- 1 file changed, 29 insertions(+), 26 deletions(-) diff --git a/tools/testing/selftests/mm/pagemap_ioctl.c b/tools/testing/selftests/mm/pagemap_ioctl.c index eadc7159ca5b90..3665530eda7688 100644 --- a/tools/testing/selftests/mm/pagemap_ioctl.c +++ b/tools/testing/selftests/mm/pagemap_ioctl.c @@ -44,7 +44,7 @@ const char *progname; #define LEN(region) ((region.end - region.start)/page_size) -static long pagemap_ioctl(void *start, int len, void *vec, int vec_len, int flag, +static long pagemap_ioctl(void *start, size_t len, void *vec, size_t vec_len, int flag, int max_pages, long required_mask, long anyof_mask, long excluded_mask, long return_mask) { @@ -65,7 +65,7 @@ static long pagemap_ioctl(void *start, int len, void *vec, int vec_len, int flag return ioctl(pagemap_fd, PAGEMAP_SCAN, &arg); } -static long pagemap_ioc(void *start, int len, void *vec, int vec_len, int flag, +static long pagemap_ioc(void *start, size_t len, void *vec, size_t vec_len, int flag, int max_pages, long required_mask, long anyof_mask, long excluded_mask, long return_mask, long *walk_end) { @@ -116,7 +116,7 @@ int init_uffd(void) return 0; } -int wp_init(void *addr, long size) +int wp_init(void *addr, size_t size) { struct uffdio_register uffdio_register; struct uffdio_writeprotect wp; @@ -140,7 +140,7 @@ int wp_init(void *addr, long size) return 0; } -int wp_free(void *addr, long size) +int wp_free(void *addr, size_t size) { struct uffdio_register uffdio_register; @@ -152,7 +152,7 @@ int wp_free(void *addr, long size) return 0; } -int wp_addr_range(void *addr, int size) +int wp_addr_range(void *addr, size_t size) { if (pagemap_ioctl(addr, size, NULL, 0, PM_SCAN_WP_MATCHING | PM_SCAN_CHECK_WPASYNC, @@ -162,7 +162,7 @@ int wp_addr_range(void *addr, int size) return 0; } -void *gethugetlb_mem(int size, int *shmid) +void *gethugetlb_mem(size_t size, int *shmid) { char *mem; @@ -188,7 +188,8 @@ void *gethugetlb_mem(int size, int *shmid) int userfaultfd_tests(void) { - long mem_size, vec_size, written, num_pages = 16; + size_t mem_size, vec_size, num_pages = 16; + long written; char *mem, *vec; mem_size = num_pages * page_size; @@ -229,9 +230,10 @@ int userfaultfd_tests(void) return 0; } -int get_reads(struct page_region *vec, int vec_size) +int get_reads(struct page_region *vec, size_t vec_size) { - int i, sum = 0; + size_t i; + int sum = 0; for (i = 0; i < vec_size; i++) sum += LEN(vec[i]); @@ -241,7 +243,7 @@ int get_reads(struct page_region *vec, int vec_size) int sanity_tests_sd(void) { - unsigned long long mem_size, vec_size, i, total_pages = 0; + size_t mem_size, vec_size, i, total_pages = 0; long ret, ret2, ret3; int num_pages = 1000; int total_writes, total_reads, reads, count; @@ -331,7 +333,7 @@ int sanity_tests_sd(void) if (ret < 0) ksft_exit_fail_msg("error %ld %d %s\n", ret, errno, strerror(errno)); - ksft_test_result((unsigned long long)ret == mem_size/(page_size * 2), + ksft_test_result((size_t)ret == mem_size/(page_size * 2), "%s Repeated pattern of written and non-written pages\n", __func__); /* 4. Repeated pattern of written and non-written pages in parts */ @@ -682,9 +684,9 @@ int sanity_tests_sd(void) return 0; } -int base_tests(char *prefix, char *mem, unsigned long long mem_size, int skip) +int base_tests(char *prefix, char *mem, size_t mem_size, int skip) { - unsigned long long vec_size; + size_t vec_size; int written; struct page_region *vec, *vec2; @@ -787,7 +789,7 @@ int base_tests(char *prefix, char *mem, unsigned long long mem_size, int skip) return 0; } -void *gethugepage(int map_size) +void *gethugepage(size_t map_size) { int ret; char *map; @@ -810,8 +812,8 @@ int hpage_unit_tests(void) char *map; int ret, ret2; size_t num_pages = 10; - unsigned long long map_size = hpage_size * num_pages; - unsigned long long vec_size = map_size/page_size; + size_t map_size = hpage_size * num_pages; + size_t vec_size = map_size/page_size; struct page_region *vec, *vec2; vec = calloc(vec_size, sizeof(struct page_region)); @@ -1002,8 +1004,9 @@ int hpage_unit_tests(void) int unmapped_region_tests(void) { void *start = (void *)0x10000000; - int written, len = 0x00040000; - long vec_size = len / page_size; + int written; + size_t len = 0x00040000; + size_t vec_size = len / page_size; struct page_region *vec = calloc(vec_size, sizeof(struct page_region)); if (!vec) ksft_exit_fail_msg("error nomem\n"); @@ -1072,7 +1075,7 @@ static void test_simple(void) * with no page table, exercising pagemap_scan_pte_hole(); a base-page range * leaves pte_none entries. */ -static void unpopulated_written_test(const char *name, char *mem, long size, +static void unpopulated_written_test(const char *name, char *mem, size_t size, bool use_thp) { long npages = size / page_size, fast = 0, slow = 0, ret; @@ -1115,7 +1118,7 @@ static void unpopulated_written_test(const char *name, char *mem, long size, static void unpopulated_scan_test(void) { - long mem_size = 16 * page_size; + size_t mem_size = 16 * page_size; char *mem; mem = mmap(NULL, mem_size, PROT_READ | PROT_WRITE, @@ -1157,8 +1160,8 @@ static void unpopulated_thp_scan_test(void) int sanity_tests(void) { - unsigned long long mem_size, vec_size; - long ret, fd, i, buf_size, nr_pages; + size_t mem_size, vec_size, i, buf_size; + long ret, fd, nr_pages; struct page_region *vec; char *mem, *fmem; struct stat sbuf; @@ -1582,9 +1585,9 @@ static void transact_test(int page_size) void zeropfn_tests(void) { - unsigned long long mem_size; + size_t mem_size, i; struct page_region vec; - int i, ret; + int ret; char *mmap_mem, *mem; /* Test with normal memory */ @@ -1642,8 +1645,8 @@ void zeropfn_tests(void) int main(int __attribute__((unused)) argc, char *argv[]) { - int shmid, buf_size, fd, i, ret; - unsigned long long mem_size; + int shmid, fd, ret; + size_t mem_size, buf_size, i; char *mem, *map, *fmem; struct stat sbuf; From 31b27542be901e7cb92717dbc1aef3c01d32cf02 Mon Sep 17 00:00:00 2001 From: "Zenghui Yu (Huawei)" Date: Tue, 8 Sep 2026 21:43:14 +0800 Subject: [PATCH 0826/1352] selftests/mm: mark file-local symbols of pagemap_ioctl.c static The file-scope variables (pagemap_fd, uffd, page_size, hpage_size and progname) and most functions of the pagemap_ioctl test are only used locally, but lack the static storage class. Mark them static so that the compiler can catch accidental outer references. Link: https://lore.kernel.org/20260908134315.84431-1-zenghui.yu@linux.dev Signed-off-by: Zenghui Yu (Huawei) Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand (Arm) Acked-by: David Hildenbrand (Arm) Cc: Gregory Price Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/mm/pagemap_ioctl.c | 43 +++++++++++----------- 1 file changed, 21 insertions(+), 22 deletions(-) diff --git a/tools/testing/selftests/mm/pagemap_ioctl.c b/tools/testing/selftests/mm/pagemap_ioctl.c index 3665530eda7688..1a87b7483316d8 100644 --- a/tools/testing/selftests/mm/pagemap_ioctl.c +++ b/tools/testing/selftests/mm/pagemap_ioctl.c @@ -36,11 +36,11 @@ #define TEST_ITERATIONS 100 #define PAGEMAP "/proc/self/pagemap" -int pagemap_fd; -int uffd; -size_t page_size; -size_t hpage_size; -const char *progname; +static int pagemap_fd; +static int uffd; +static size_t page_size; +static size_t hpage_size; +static const char *progname; #define LEN(region) ((region.end - region.start)/page_size) @@ -92,8 +92,7 @@ static long pagemap_ioc(void *start, size_t len, void *vec, size_t vec_len, int return ret; } - -int init_uffd(void) +static int init_uffd(void) { struct uffdio_api uffdio_api; @@ -116,7 +115,7 @@ int init_uffd(void) return 0; } -int wp_init(void *addr, size_t size) +static int wp_init(void *addr, size_t size) { struct uffdio_register uffdio_register; struct uffdio_writeprotect wp; @@ -140,7 +139,7 @@ int wp_init(void *addr, size_t size) return 0; } -int wp_free(void *addr, size_t size) +static int wp_free(void *addr, size_t size) { struct uffdio_register uffdio_register; @@ -152,7 +151,7 @@ int wp_free(void *addr, size_t size) return 0; } -int wp_addr_range(void *addr, size_t size) +static int wp_addr_range(void *addr, size_t size) { if (pagemap_ioctl(addr, size, NULL, 0, PM_SCAN_WP_MATCHING | PM_SCAN_CHECK_WPASYNC, @@ -162,7 +161,7 @@ int wp_addr_range(void *addr, size_t size) return 0; } -void *gethugetlb_mem(size_t size, int *shmid) +static void *gethugetlb_mem(size_t size, int *shmid) { char *mem; @@ -186,7 +185,7 @@ void *gethugetlb_mem(size_t size, int *shmid) return mem; } -int userfaultfd_tests(void) +static int userfaultfd_tests(void) { size_t mem_size, vec_size, num_pages = 16; long written; @@ -230,7 +229,7 @@ int userfaultfd_tests(void) return 0; } -int get_reads(struct page_region *vec, size_t vec_size) +static int get_reads(struct page_region *vec, size_t vec_size) { size_t i; int sum = 0; @@ -241,7 +240,7 @@ int get_reads(struct page_region *vec, size_t vec_size) return sum; } -int sanity_tests_sd(void) +static int sanity_tests_sd(void) { size_t mem_size, vec_size, i, total_pages = 0; long ret, ret2, ret3; @@ -684,7 +683,7 @@ int sanity_tests_sd(void) return 0; } -int base_tests(char *prefix, char *mem, size_t mem_size, int skip) +static int base_tests(char *prefix, char *mem, size_t mem_size, int skip) { size_t vec_size; int written; @@ -789,7 +788,7 @@ int base_tests(char *prefix, char *mem, size_t mem_size, int skip) return 0; } -void *gethugepage(size_t map_size) +static void *gethugepage(size_t map_size) { int ret; char *map; @@ -807,7 +806,7 @@ void *gethugepage(size_t map_size) return map; } -int hpage_unit_tests(void) +static int hpage_unit_tests(void) { char *map; int ret, ret2; @@ -1001,7 +1000,7 @@ int hpage_unit_tests(void) return 0; } -int unmapped_region_tests(void) +static int unmapped_region_tests(void) { void *start = (void *)0x10000000; int written; @@ -1158,7 +1157,7 @@ static void unpopulated_thp_scan_test(void) munmap(area, 2 * hpage_size); } -int sanity_tests(void) +static int sanity_tests(void) { size_t mem_size, vec_size, i, buf_size; long ret, fd, nr_pages; @@ -1330,7 +1329,7 @@ int sanity_tests(void) return 0; } -int mprotect_tests(void) +static int mprotect_tests(void) { int ret; char *mem, *mem2; @@ -1450,7 +1449,7 @@ static ssize_t get_dirty_pages_reset(char *mem, unsigned int count, return cnt; } -void *thread_proc(void *mem) +static void *thread_proc(void *mem) { int *m = mem; long curr_faults, faults; @@ -1583,7 +1582,7 @@ static void transact_test(int page_size) extra_thread_faults); } -void zeropfn_tests(void) +static void zeropfn_tests(void) { size_t mem_size, i; struct page_region vec; From e517edfff97b14e198fd781c64a6f3c611727ccf Mon Sep 17 00:00:00 2001 From: "Zenghui Yu (Huawei)" Date: Tue, 8 Sep 2026 21:44:05 +0800 Subject: [PATCH 0827/1352] selftests/mm: init page sizes early in pagemap_ioctl test Initialize page_size and hpage_size before calling init_uffd(), hugetlb_setup_default(), etc. That won't fix anything, but it is safer and saner to get these globals set up before doing other things. While at it, drop the page_size parameter of transact_test(), which is actually unnecessary. Link: https://lore.kernel.org/20260908134405.84448-1-zenghui.yu@linux.dev Signed-off-by: Zenghui Yu (Huawei) Signed-off-by: Andrew Morton Suggested-by: Andrew Morton Link: https://lore.kernel.org/20260628111329.9cfcd9c67925869307020aba@linux-foundation.org/ Cc: David Hildenbrand (Arm) Cc: Gregory Price Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/mm/pagemap_ioctl.c | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/tools/testing/selftests/mm/pagemap_ioctl.c b/tools/testing/selftests/mm/pagemap_ioctl.c index 1a87b7483316d8..d9a4fb782ecfe7 100644 --- a/tools/testing/selftests/mm/pagemap_ioctl.c +++ b/tools/testing/selftests/mm/pagemap_ioctl.c @@ -1489,7 +1489,7 @@ static void *thread_proc(void *mem) return NULL; } -static void transact_test(int page_size) +static void transact_test(void) { unsigned int i, count, extra_pages; unsigned int c; @@ -1653,6 +1653,9 @@ int main(int __attribute__((unused)) argc, char *argv[]) ksft_print_header(); + page_size = getpagesize(); + hpage_size = read_pmd_pagesize(); + if (init_uffd()) ksft_exit_skip("Failed to initialize userfaultfd\n"); @@ -1661,9 +1664,6 @@ int main(int __attribute__((unused)) argc, char *argv[]) ksft_set_plan(119); - page_size = getpagesize(); - hpage_size = read_pmd_pagesize(); - pagemap_fd = open(PAGEMAP, O_RDONLY); if (pagemap_fd < 0) ksft_exit_fail_msg("Failed to open " PAGEMAP "\n"); @@ -1823,7 +1823,7 @@ int main(int __attribute__((unused)) argc, char *argv[]) mprotect_tests(); /* 13. Transact test */ - transact_test(page_size); + transact_test(); /* 14. Sanity testing */ sanity_tests(); From 768c50236fd015647e5a05e01d46c9ac197a8d39 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Tue, 8 Sep 2026 14:28:21 +0100 Subject: [PATCH 0828/1352] mm/huge_memory: add folio_reset_partially_mapped() __folio_unqueue_deferred_split() and __folio_freeze_and_split_unmapped() both clear PG_partially_mapped and take the folio out of MTHP_STAT_NR_ANON_PARTIALLY_MAPPED with the same five lines. Move the block into folio_reset_partially_mapped() and call it from both places. The helper asserts what both callers rely on: the folio is frozen, so deferred_split_folio() cannot set the flag again under it, and the folio is already off the deferred split queue. The list check sits behind the flag test because order-1 folios have no _deferred_list. folio_order() is safe to use at this point in the split process: it still shows the pre-split order. Link: https://lore.kernel.org/20260908132821.1517475-1-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand (Arm) Acked-by: David Hildenbrand (Arm) Acked-by: Balbir Singh Reviewed-by: Zi Yan Reviewed-by: Baolin Wang Reviewed-by: Lorenzo Stoakes (ARM) Assisted-by: LLM --- mm/huge_memory.c | 32 +++++++++++++++++++++----------- 1 file changed, 21 insertions(+), 11 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 009eb3adc2b783..194188c292af3d 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -3976,6 +3976,25 @@ static unsigned int folio_cache_ref_count(const struct folio *folio) return folio_nr_pages(folio); } +static void folio_reset_partially_mapped(struct folio *folio) +{ + /* Folio must be frozen. */ + VM_WARN_ON_FOLIO(folio_ref_count(folio), folio); + + if (!folio_test_partially_mapped(folio)) + return; + + /* + * Order-1 folios have no _deferred_list. The flag is only ever set + * on folios that do, so the list can be checked after the flag. + */ + VM_WARN_ON_FOLIO(!list_empty(&folio->_deferred_list), folio); + + folio_clear_partially_mapped(folio); + mod_mthp_stat(folio_order(folio), + MTHP_STAT_NR_ANON_PARTIALLY_MAPPED, -1); +} + static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int new_order, struct page *split_at, struct xa_state *xas, struct address_space *mapping, bool do_lru, @@ -3984,7 +4003,6 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n { struct folio *end_folio = folio_next(folio); struct folio *new_folio, *next; - int old_order = folio_order(folio); int ret = 0; VM_WARN_ON_ONCE(!mapping && end); @@ -4002,11 +4020,7 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n * leaves PG_partially_mapped set. * Clear it here: the flag does not survive the split. */ - if (folio_test_partially_mapped(folio)) { - folio_clear_partially_mapped(folio); - mod_mthp_stat(old_order, - MTHP_STAT_NR_ANON_PARTIALLY_MAPPED, -1); - } + folio_reset_partially_mapped(folio); if (mapping) { int nr = folio_nr_pages(folio); @@ -4520,11 +4534,7 @@ bool __folio_unqueue_deferred_split(struct folio *folio) memcg = folio_memcg(folio); lru = list_lru_lock_irqsave(&deferred_split_lru, nid, &memcg, &flags); if (__list_lru_del(&deferred_split_lru, lru, &folio->_deferred_list, nid)) { - if (folio_test_partially_mapped(folio)) { - folio_clear_partially_mapped(folio); - mod_mthp_stat(folio_order(folio), - MTHP_STAT_NR_ANON_PARTIALLY_MAPPED, -1); - } + folio_reset_partially_mapped(folio); unqueued = true; } list_lru_unlock_irqrestore(lru, &flags); From c2644c2c959425ac08a57b96fff2a10b36d8389d Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 25 Sep 2026 21:09:36 +0100 Subject: [PATCH 0829/1352] mm/khugepaged: deposit a newly allocated page table on collapse Patch series "mm: make userland page table freeing RCU-safe", v5. The majority of architectures in the kernel defer page table freeing until an RCU grace period has elapsed, this series converts all remaining architectures to do so too and eliminates CONFIG_MMU_GATHER_RCU_TABLE_FREE altogether. This is important because it enables safe lockless page table walking under RCU alone. Doing so allows for reduced lock contention, avoids lock ordering concerns and enables fast, efficient and correct page table walking as a result. Additionally it removes a bunch of code and architecture-specific behaviour which is always a beneficial thing to do. There has been much recent work on this: * In 2023 Hugh Dickins RCU-deferred khugepaged page table retraction in commit 13cf577e6b66 ("mm/pgtable: add pte_free_defer() for pgtable as page"). * Qi Zheng has done most of the work that made this possible starting with the critical commit 718b13861d22 ("x86: mm: free page table pages by RCU instead of semi RCU"). * Qi then went on to convert a large number of architectures in commit e3ecf7c7d082 ("mm: pgtable: convert some architectures to use tlb_remove_ptdesc()"), commit 44b079583f7d ("alpha: mm: enable MMU_GATHER_RCU_TABLE_FREE") and the series to which it belongs. * Qi then introduced the important CONFIG_HAVE_ARCH_TLB_REMOVE_TABLE option in commit 086498aed3f6 ("mm: convert __HAVE_ARCH_TLB_REMOVE_TABLE to CONFIG_HAVE_ARCH_TLB_REMOVE_TABLE config"). * Finally, and critically, Lance Yang then converted the batch allocation fallback case to be RCU-safe in commit 1fb3d8c20bfa ("mm/mmu_gather: replace IPI with synchronize_rcu() when batch allocation fails"). The work I do here is only possible due to the work Hugh, Qi, Lance and others have done previously. An initial task this series addresses is to deposit a freshly allocated PTE page table and RCU-free the existing PTE page table. Not doing so is currently safe, but for page table walkers relying on RCU alone, it would not be. Additionally, it makes it possible to implement lockless RCU-only page table walkers which otherwise would have required the PTE PTL. The changes are largely mechanical - the majority of arches already have the machinery required to support CONFIG_MMU_GATHER_RCU_TABLE_FREE and simply needed configuration changes or small implementation changes to switch over. However some arches required extra attention - sh-X2, m68k-motorola and sparc32. sh-X2 allocates PMDs from the slab allocator and PTEs as normal. Therefore CONFIG_HAVE_ARCH_TLB_REMOVE_TABLE is set to customise page table freeing and the LSB is used to encode which page table level is used, with __tlb_remove_table() doing the right thing depending on this. This pattern is repeated for m68k-motorola and sparc32 to account for different page table levels. In each case, the page tables are aligned such that sufficient bits are available in each case for encoding this information. m68k-motorola required the biggest change - since RCU page table freeing uses call_rcu(), this means page table freeing can arise from softirq context. This was fixed with a BH-disabling spin lock used in both get_pointer_table() and free_pointer_table() (softirq being the only asynchronous context in which the lock is taken). As part of this change, the logic for allocation of a new pointer table was separated out into add_pointer_table() to make the locking more obviously correct. Finally, sparc32 was similar to m68k-motorola in that locking was required, however this was already implemented via a spinlock, and only had to be updated to be IRQ-safe. Additionally, the nocache pool's bit_map lock was updated to be IRQ-safe for softirq frees. Separately, the PTE path can't take mm->page_table_lock from softirq (no mm there), which is fine because the page reference count transitions are atomic and fully ordered. The series finally removes CONFIG_MMU_GATHER_RCU_TABLE_FREE and all related configurations and code that supported !CONFIG_MMU_GATHER_RCU_TABLE_FREE. As a result, page table walks can now be performed safely under RCU without any risk of page tables being freed underneath a walker. However, this is the only guarantee that this work provides - page table walkers must still ensure that page table entries are as expected throughout. All changes have been build tested. As most of the conversions are simply utilising existing mechanics that are known to work, this suffices for most cases. However those arches where significant changes have been made - m68k-motorola, sparc32 and sh-X2 - have been tested further. For each of these a boot test and stress test has been performed - fork 400 children, each mmap()'ing 2 MiB and touching every page then partially munmap()'ing then exiting to trigger as much page table freeing as possible. All were found to be working correctly. (Note that sparc32 LEON SMP is not possible to emulate.) This patch (of 12): collapse_huge_page() deposits a PTE page table on PMD collapse in order that it can be utilised for subsequent split operations, meaning that those operations do not need to perform an allocation (as they are in a context where it might be unwise). However the PTE page table which is deposited is the one which is currently mapped by the PMD entry that is in the process of being collapsed. Once deposited, the PTE page table may be used in a split of any other unrelated PMD entry. This is currently not an issue as this operation is performed with VMA/mmap write lock + anon rmap locks held, so ordinary page table walkers will never accidentally end up walking the wrong thing, and GUP-fast is protected by an IPI via tlb_remove_table_sync_one(). However, the series to which this commit belongs implements RCU-safe page table traversal, at which point this becomes problematic. This can be resolved by using pte_offset_map_lock() which gates on a PTE PTL and a pmd_same() check, but lockless walks are unsafe as things stand. Resolve this by simply allocating a new, zeroed, PTE page table to deposit at the point of collapse. This path is already costly and an allocation has already been performed for the huge folio, so this allocation is statistical noise in terms of performance and memory usage at this point. With this PTE page table deposited, RCU-free the existing PTE page table so it is safe for page table walkers to traverse within a grace period. This also brings this deposit case in line with all other page table deposit logic which deposit a fresh page table. Additionally, this was the only place in the kernel that displaced a page table like this, so eliminating it also helps consistency. Since khugepaged runs as a kernel thread, do a little dance in alloc_deposit_pte_table() to correctly charge the allocation. This is already done for the folio allocation via alloc_charge_folio() but no such wrapper exists for a page table allocation. Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-0-31e91065fea4@kernel.org Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-1-31e91065fea4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Baolin Wang Acked-by: David Hildenbrand (Arm) Reviewed-by: Lance Yang Cc: Zi Yan Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Usama Arif Cc: Kiryl Shutsemau Cc: Guo Ren Cc: Brian Cain Cc: Geert Uytterhoeven Cc: Dinh Nguyen Cc: Simon Schuster Cc: Jonas Bonn Cc: Stefan Kristiansson Cc: Stafford Horne Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti Cc: Russell King Cc: Vineet Gupta Cc: Michal Simek Cc: Chris Zankel Cc: Max Filippov Cc: Will Deacon Cc: Aneesh Kumar K.V Cc: Nicholas Piggin Cc: Peter Zijlstra Cc: David S. Miller Cc: Andreas Larsson Cc: Richard Henderson Cc: Matt Turner Cc: Magnus Lindholm Cc: Catalin Marinas Cc: Mark Rutland Cc: Huacai Chen Cc: WANG Xuerui Cc: Thomas Bogendoerfer Cc: James Bottomley Cc: Helge Deller Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Christophe Leroy Cc: Heiko Carstens Cc: Vasily Gorbik Cc: Alexander Gordeev Cc: Christian Borntraeger Cc: Sven Schnelle Cc: Richard Weinberger Cc: Anton Ivanov Cc: Johannes Berg Cc: Thomas Gleixner Cc: Ingo Molnar Cc: Borislav Petkov Cc: Dave Hansen Cc: H. Peter Anvin Cc: Arnd Bergmann Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Cc: Yoshinori Sato Cc: Shakeel Butt Cc: Jonathan Corbet Cc: Randy Dunlap Cc: Hugh Dickins Cc: Qi Zheng --- mm/khugepaged.c | 34 ++++++++++++++++++++++++++++++++-- 1 file changed, 32 insertions(+), 2 deletions(-) diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 792166950bba81..85ea095906fb70 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -1278,6 +1278,23 @@ static enum scan_result alloc_charge_folio(struct folio **foliop, struct mm_stru return SCAN_SUCCEED; } +static pgtable_t alloc_deposit_pte_table(struct mm_struct *mm) +{ + /* + * khugepaged is run from a kernel thread, so need to manually set the + * correct memcg so the allocation gets charged correctly. + */ + struct mem_cgroup *memcg = get_mem_cgroup_from_mm(mm); + struct mem_cgroup *old_memcg = set_active_memcg(memcg); + pgtable_t pgtable; + + pgtable = pte_alloc_one(mm); + + set_active_memcg(old_memcg); + mem_cgroup_put(memcg); + return pgtable; +} + /* * collapse_huge_page() expects the mmap_lock to be unlocked before entering and * will always return with the lock unlocked, to avoid holding the mmap_lock @@ -1293,7 +1310,7 @@ static enum scan_result collapse_huge_page(struct mm_struct *mm, unsigned long s LIST_HEAD(compound_pagelist); pmd_t *pmd, _pmd; pte_t *pte = NULL; - pgtable_t pgtable; + pgtable_t pgtable = NULL; struct folio *folio; spinlock_t *pmd_ptl, *pte_ptl; enum scan_result result = SCAN_FAIL; @@ -1310,6 +1327,14 @@ static enum scan_result collapse_huge_page(struct mm_struct *mm, unsigned long s goto out_nolock; } + if (is_pmd_order(order)) { + pgtable = alloc_deposit_pte_table(mm); + if (!pgtable) { + result = SCAN_ALLOC_HUGE_PAGE_FAIL; + goto out_nolock; + } + } + mmap_read_lock(mm); result = hugepage_vma_revalidate(mm, pmd_addr, /*expect_anon=*/ true, &vma, cc, order); @@ -1433,8 +1458,8 @@ static enum scan_result collapse_huge_page(struct mm_struct *mm, unsigned long s spin_lock(pmd_ptl); VM_WARN_ON_ONCE(!pmd_none(*pmd)); if (is_pmd_order(order)) { - pgtable = pmd_pgtable(_pmd); pgtable_trans_huge_deposit(mm, pmd, pgtable); + pgtable = NULL; map_anon_folio_pmd_nopf(folio, pmd, vma, pmd_addr); } else { /* @@ -1453,6 +1478,9 @@ static enum scan_result collapse_huge_page(struct mm_struct *mm, unsigned long s } spin_unlock(pmd_ptl); + if (is_pmd_order(order)) + pte_free_defer(mm, pmd_pgtable(_pmd)); + folio = NULL; result = SCAN_SUCCEED; @@ -1463,6 +1491,8 @@ static enum scan_result collapse_huge_page(struct mm_struct *mm, unsigned long s anon_vma_unlock_write(vma->anon_vma); mmap_write_unlock(mm); out_nolock: + if (pgtable) + pte_free(mm, pgtable); if (folio) folio_put(folio); trace_mm_collapse_huge_page(mm, result == SCAN_SUCCEED, result, order); From 88d63b260b6ffb6c8d73cf403aa526be7823cece Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 25 Sep 2026 21:09:37 +0100 Subject: [PATCH 0830/1352] mm: enable MMU_GATHER_RCU_TABLE_FREE for most 2-level architectures Commit e3ecf7c7d082 ("mm: pgtable: convert some architectures to use tlb_remove_ptdesc()") updated a number of architectures from using pagetable_dtor() + tlb_remove_page_ptdesc() to using tlb_remove_ptdesc() in __pte_free_tlb(). This is meaningful as tlb_remove_ptdesc() allows for RCU page table freeing if CONFIG_MMU_GATHER_RCU_TABLE_FREE is specified. The csky, hexagon, nios2, openrisc, sh (except X2) and m68k-sun3 architectures all have 2 levels of page tables, so the only page tables ever freed by mmu_gather are PTEs, so this update suffices to ensure that every page table freed by the mmu_gather mechanism is freed under RCU. Therefore, update all of these architectures to select CONFIG_MMU_GATHER_RCU_TABLE_FREE. This forms part of an overall effort to switch every architecture to this mode. Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-2-31e91065fea4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Kiryl Shutsemau (Meta) Reviewed-by: Lance Yang Cc: David Hildenbrand Cc: Zi Yan Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Usama Arif Cc: Guo Ren Cc: Brian Cain Cc: Geert Uytterhoeven Cc: Dinh Nguyen Cc: Simon Schuster Cc: Jonas Bonn Cc: Stefan Kristiansson Cc: Stafford Horne Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti Cc: Russell King Cc: Vineet Gupta Cc: Michal Simek Cc: Chris Zankel Cc: Max Filippov Cc: Will Deacon Cc: Aneesh Kumar K.V Cc: Nicholas Piggin Cc: Peter Zijlstra Cc: David S. Miller Cc: Andreas Larsson Cc: Richard Henderson Cc: Matt Turner Cc: Magnus Lindholm Cc: Catalin Marinas Cc: Mark Rutland Cc: Huacai Chen Cc: WANG Xuerui Cc: Thomas Bogendoerfer Cc: James Bottomley Cc: Helge Deller Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Christophe Leroy Cc: Heiko Carstens Cc: Vasily Gorbik Cc: Alexander Gordeev Cc: Christian Borntraeger Cc: Sven Schnelle Cc: Richard Weinberger Cc: Anton Ivanov Cc: Johannes Berg Cc: Thomas Gleixner Cc: Ingo Molnar Cc: Borislav Petkov Cc: Dave Hansen Cc: H. Peter Anvin Cc: Arnd Bergmann Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Cc: Yoshinori Sato Cc: Shakeel Butt Cc: Jonathan Corbet Cc: Randy Dunlap Cc: Hugh Dickins Cc: Qi Zheng --- arch/csky/Kconfig | 1 + arch/hexagon/Kconfig | 1 + arch/m68k/Kconfig | 1 + arch/nios2/Kconfig | 1 + arch/openrisc/Kconfig | 1 + arch/sh/Kconfig | 1 + 6 files changed, 6 insertions(+) diff --git a/arch/csky/Kconfig b/arch/csky/Kconfig index 4331313a42ff30..80f89ef1d9622c 100644 --- a/arch/csky/Kconfig +++ b/arch/csky/Kconfig @@ -96,6 +96,7 @@ config CSKY select HAVE_SYSCALL_TRACEPOINTS select HOTPLUG_CORE_SYNC_DEAD if HOTPLUG_CPU select LOCK_MM_AND_FIND_VMA + select MMU_GATHER_RCU_TABLE_FREE select MAY_HAVE_SPARSE_IRQ select MODULES_USE_ELF_RELA if MODULES select OF diff --git a/arch/hexagon/Kconfig b/arch/hexagon/Kconfig index b4849114001335..d9b3fb86556be3 100644 --- a/arch/hexagon/Kconfig +++ b/arch/hexagon/Kconfig @@ -23,6 +23,7 @@ config HEXAGON # select HAVE_CLK select GENERIC_ATOMIC64 select HAVE_PERF_EVENTS + select MMU_GATHER_RCU_TABLE_FREE # GENERIC_ALLOCATOR is used by dma_alloc_coherent() select GENERIC_ALLOCATOR select GENERIC_IRQ_PROBE diff --git a/arch/m68k/Kconfig b/arch/m68k/Kconfig index 11835eb59d94db..e29610fd1240f6 100644 --- a/arch/m68k/Kconfig +++ b/arch/m68k/Kconfig @@ -36,6 +36,7 @@ config M68K select HAVE_MOD_ARCH_SPECIFIC select HAVE_UID16 select MMU_GATHER_NO_RANGE if MMU + select MMU_GATHER_RCU_TABLE_FREE if MMU && SUN3 select MODULES_USE_ELF_REL select MODULES_USE_ELF_RELA select NO_DMA if !MMU && !COLDFIRE diff --git a/arch/nios2/Kconfig b/arch/nios2/Kconfig index 9c0e6eaeb005cd..b0ccfc3b7a7e1b 100644 --- a/arch/nios2/Kconfig +++ b/arch/nios2/Kconfig @@ -19,6 +19,7 @@ config NIOS2 select HAVE_PAGE_SIZE_4KB select IRQ_DOMAIN select LOCK_MM_AND_FIND_VMA + select MMU_GATHER_RCU_TABLE_FREE select MODULES_USE_ELF_RELA select OF select OF_EARLY_FLATTREE diff --git a/arch/openrisc/Kconfig b/arch/openrisc/Kconfig index 5eb995c13074c0..d90b24dd3bce3c 100644 --- a/arch/openrisc/Kconfig +++ b/arch/openrisc/Kconfig @@ -35,6 +35,7 @@ config OPENRISC select GENERIC_ATOMIC64 select GENERIC_CLOCKEVENTS_BROADCAST select GENERIC_SMP_IDLE_THREAD + select MMU_GATHER_RCU_TABLE_FREE select MODULES_USE_ELF_RELA select HAVE_DEBUG_STACKOVERFLOW select OR1K_PIC diff --git a/arch/sh/Kconfig b/arch/sh/Kconfig index d60f1d5a94c0f4..204f64912f0e42 100644 --- a/arch/sh/Kconfig +++ b/arch/sh/Kconfig @@ -61,6 +61,7 @@ config SUPERH select HAVE_SYSCALL_TRACEPOINTS select IRQ_FORCED_THREADING select LOCK_MM_AND_FIND_VMA + select MMU_GATHER_RCU_TABLE_FREE if MMU && !X2TLB select MODULES_USE_ELF_RELA select NEED_SG_DMA_LENGTH select NO_DMA if !MMU && !DMA_COHERENT From 338dd7e871ca16dce8aac98564a1097fb25d1291 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 25 Sep 2026 21:09:38 +0100 Subject: [PATCH 0831/1352] mm: enable MMU_GATHER_RCU_TABLE_FREE for MMU riscv Currently riscv gates MMU_GATHER_RCU_TABLE_FREE on CONFIG_SMP and CONFIG_MMU. Commit 69be3fb111e7 ("riscv: enable MMU_GATHER_RCU_TABLE_FREE for SMP && MMU") enabled CONFIG_MMU_GATHER_RCU_TABLE_FREE for CONFIG_SMP, CONFIG_MMU riscv builds. This is expressly for the safety of GUP-fast walkers (CONFIG_HAVE_GUP_FAST is enabled if CONFIG_MMU is enabled). Naturally a single core system does not encounter issues with software page table walkers being correctly synchronised across cores, as there is only a single core. However, CONFIG_PREEMPT_RCU is still available on a riscv UP system, so for a future RCU-only page table walker, this guarantee is required to prevent concurrent page table teardown. All page table freeing is already done via tlb_remove_ptdesc() so the conditions of CONFIG_MMU_GATHER_RCU_TABLE_FREE are already met. This forms part of an overall effort to switch every architecture to this mode. Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-3-31e91065fea4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Kiryl Shutsemau (Meta) Acked-by: Lance Yang Tested-by: Lance Yang Cc: David Hildenbrand Cc: Zi Yan Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Usama Arif Cc: Guo Ren Cc: Brian Cain Cc: Geert Uytterhoeven Cc: Dinh Nguyen Cc: Simon Schuster Cc: Jonas Bonn Cc: Stefan Kristiansson Cc: Stafford Horne Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti Cc: Russell King Cc: Vineet Gupta Cc: Michal Simek Cc: Chris Zankel Cc: Max Filippov Cc: Will Deacon Cc: Aneesh Kumar K.V Cc: Nicholas Piggin Cc: Peter Zijlstra Cc: David S. Miller Cc: Andreas Larsson Cc: Richard Henderson Cc: Matt Turner Cc: Magnus Lindholm Cc: Catalin Marinas Cc: Mark Rutland Cc: Huacai Chen Cc: WANG Xuerui Cc: Thomas Bogendoerfer Cc: James Bottomley Cc: Helge Deller Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Christophe Leroy Cc: Heiko Carstens Cc: Vasily Gorbik Cc: Alexander Gordeev Cc: Christian Borntraeger Cc: Sven Schnelle Cc: Richard Weinberger Cc: Anton Ivanov Cc: Johannes Berg Cc: Thomas Gleixner Cc: Ingo Molnar Cc: Borislav Petkov Cc: Dave Hansen Cc: H. Peter Anvin Cc: Arnd Bergmann Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Cc: Yoshinori Sato Cc: Shakeel Butt Cc: Jonathan Corbet Cc: Randy Dunlap Cc: Hugh Dickins Cc: Qi Zheng --- arch/riscv/Kconfig | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/arch/riscv/Kconfig b/arch/riscv/Kconfig index 5965666194b0f0..c13ef9cd688d33 100644 --- a/arch/riscv/Kconfig +++ b/arch/riscv/Kconfig @@ -208,7 +208,7 @@ config RISCV select IRQ_FORCED_THREADING select KASAN_VMALLOC if KASAN select LOCK_MM_AND_FIND_VMA - select MMU_GATHER_RCU_TABLE_FREE if SMP && MMU + select MMU_GATHER_RCU_TABLE_FREE if MMU select MODULES_USE_ELF_RELA if MODULES select OF select OF_EARLY_FLATTREE From 9017dc40369e582d5ea26787f151bbb48faa37a2 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 25 Sep 2026 21:09:39 +0100 Subject: [PATCH 0832/1352] mm: enable MMU_GATHER_RCU_TABLE_FREE for MMU arm Commit a0ad5496b2b3 ("arm: mm: enable HAVE_RCU_TABLE_FREE logic") enabled CONFIG_MMU_GATHER_RCU_TABLE_FREE (then named HAVE_RCU_TABLE_FREE) for SMP arm architectures with LPAE enabled. Regardless of whether CONFIG_ARM_LPAE is enabled or not, the same page table freeing functions __pte_free_tlb() and __pmd_free_tlb() are used. Non-LPAE PMD page tables are folded into the PGD and freed by pgd_free() (PGD freeing is not part of mmu_gather page table freeing in any case), so this is a noop in this case. Since commit 358d1c39c82a ("arm: convert various functions to use ptdescs") both LPAE and non-LPAE PTE page table freeing uses tlb_remove_ptdesc(). Thus all page table freeing is performed under RCU with CONFIG_MMU_GATHER_RCU_TABLE_FREE enabled for LPAE and non-LPAE and thus it need not be gated on LPAE. A UP arm system can set CONFIG_PREEMPT_RCU, so a future pure RCU page table walker requires MMU_GATHER_RCU_TABLE_FREE to be enabled on UP as well, even if concurrent GUP fast is not possible there. Therefore, it is both safe and desirable to set CONFIG_MMU_GATHER_RCU_TABLE_FREE for all MMU arm architectures (nommu does not perform mmu_gather operations). This forms part of an overall effort to switch every architecture to this mode. Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-4-31e91065fea4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Kiryl Shutsemau (Meta) Tested-by: Lance Yang Acked-by: Lance Yang Cc: David Hildenbrand Cc: Zi Yan Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Usama Arif Cc: Guo Ren Cc: Brian Cain Cc: Geert Uytterhoeven Cc: Dinh Nguyen Cc: Simon Schuster Cc: Jonas Bonn Cc: Stefan Kristiansson Cc: Stafford Horne Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti Cc: Russell King Cc: Vineet Gupta Cc: Michal Simek Cc: Chris Zankel Cc: Max Filippov Cc: Will Deacon Cc: Aneesh Kumar K.V Cc: Nicholas Piggin Cc: Peter Zijlstra Cc: David S. Miller Cc: Andreas Larsson Cc: Richard Henderson Cc: Matt Turner Cc: Magnus Lindholm Cc: Catalin Marinas Cc: Mark Rutland Cc: Huacai Chen Cc: WANG Xuerui Cc: Thomas Bogendoerfer Cc: James Bottomley Cc: Helge Deller Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Christophe Leroy Cc: Heiko Carstens Cc: Vasily Gorbik Cc: Alexander Gordeev Cc: Christian Borntraeger Cc: Sven Schnelle Cc: Richard Weinberger Cc: Anton Ivanov Cc: Johannes Berg Cc: Thomas Gleixner Cc: Ingo Molnar Cc: Borislav Petkov Cc: Dave Hansen Cc: H. Peter Anvin Cc: Arnd Bergmann Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Cc: Yoshinori Sato Cc: Shakeel Butt Cc: Jonathan Corbet Cc: Randy Dunlap Cc: Hugh Dickins Cc: Qi Zheng --- arch/arm/Kconfig | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/arch/arm/Kconfig b/arch/arm/Kconfig index 408aa58a2a5bbc..72b9afc6ae1058 100644 --- a/arch/arm/Kconfig +++ b/arch/arm/Kconfig @@ -134,7 +134,7 @@ config ARM select HAVE_PERF_REGS select HAVE_PERF_USER_STACK_DUMP select HAVE_POSIX_CPU_TIMERS_TASK_WORK - select MMU_GATHER_RCU_TABLE_FREE if SMP && ARM_LPAE + select MMU_GATHER_RCU_TABLE_FREE if MMU select HAVE_REGS_AND_STACK_ACCESS_API select HAVE_RSEQ select HAVE_RUST if CPU_LITTLE_ENDIAN && CPU_32v7 && !KASAN From cc267a3a838fd522ec86acbbef27872df73e9ff2 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 25 Sep 2026 21:09:40 +0100 Subject: [PATCH 0833/1352] mm: enable MMU_GATHER_RCU_TABLE_FREE for arc, microblaze, xtensa Each of these architectures directly free page tables without routing these changes through tlb_remove_ptdesc(). The use of tlb_remove_ptdesc() is required for CONFIG_MMU_GATHER_RCU_TABLE_FREE to correctly free page tables under RCU, so simply update these architectures to use these functions. Since none of the architectures share page tables or do anything unusual, nothing complicated is required here. Therefore this is simply a mechanical change - convert __pud_free_tlb(), __pmd_free_tlb() and __pte_free_tlb() to use tlb_remove_ptdesc() as required. At the point this is in place, all mmu_gather page table freeing is performed under RCU, and thus MMU_GATHER_RCU_TABLE_FREE is selected for each architecture. Note that CONFIG_MMU_GATHER_RCU_TABLE_FREE is dependent on CONFIG_MMU for xtensa to reflect the fact that nommu does not implement page table gathering. This forms part of an overall effort to switch every architecture to this mode. Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-5-31e91065fea4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Kiryl Shutsemau (Meta) Acked-by: Lance Yang Cc: David Hildenbrand Cc: Zi Yan Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Usama Arif Cc: Guo Ren Cc: Brian Cain Cc: Geert Uytterhoeven Cc: Dinh Nguyen Cc: Simon Schuster Cc: Jonas Bonn Cc: Stefan Kristiansson Cc: Stafford Horne Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti Cc: Russell King Cc: Vineet Gupta Cc: Michal Simek Cc: Chris Zankel Cc: Max Filippov Cc: Will Deacon Cc: Aneesh Kumar K.V Cc: Nicholas Piggin Cc: Peter Zijlstra Cc: David S. Miller Cc: Andreas Larsson Cc: Richard Henderson Cc: Matt Turner Cc: Magnus Lindholm Cc: Catalin Marinas Cc: Mark Rutland Cc: Huacai Chen Cc: WANG Xuerui Cc: Thomas Bogendoerfer Cc: James Bottomley Cc: Helge Deller Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Christophe Leroy Cc: Heiko Carstens Cc: Vasily Gorbik Cc: Alexander Gordeev Cc: Christian Borntraeger Cc: Sven Schnelle Cc: Richard Weinberger Cc: Anton Ivanov Cc: Johannes Berg Cc: Thomas Gleixner Cc: Ingo Molnar Cc: Borislav Petkov Cc: Dave Hansen Cc: H. Peter Anvin Cc: Arnd Bergmann Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Cc: Yoshinori Sato Cc: Shakeel Butt Cc: Jonathan Corbet Cc: Randy Dunlap Cc: Hugh Dickins Cc: Qi Zheng --- arch/arc/Kconfig | 1 + arch/arc/include/asm/pgalloc.h | 6 +++--- arch/microblaze/Kconfig | 1 + arch/microblaze/include/asm/pgalloc.h | 2 +- arch/xtensa/Kconfig | 1 + arch/xtensa/include/asm/tlb.h | 2 +- 6 files changed, 8 insertions(+), 5 deletions(-) diff --git a/arch/arc/Kconfig b/arch/arc/Kconfig index 80e61175bf526b..0c1679f4d386f8 100644 --- a/arch/arc/Kconfig +++ b/arch/arc/Kconfig @@ -47,6 +47,7 @@ config ARC select HAVE_SYSCALL_TRACEPOINTS select IRQ_DOMAIN select LOCK_MM_AND_FIND_VMA + select MMU_GATHER_RCU_TABLE_FREE select MODULES_USE_ELF_RELA select OF select OF_EARLY_FLATTREE diff --git a/arch/arc/include/asm/pgalloc.h b/arch/arc/include/asm/pgalloc.h index dfae070fe8d556..9b6c37f92e97f3 100644 --- a/arch/arc/include/asm/pgalloc.h +++ b/arch/arc/include/asm/pgalloc.h @@ -72,7 +72,7 @@ static inline void p4d_populate(struct mm_struct *mm, p4d_t *p4dp, pud_t *pudp) set_p4d(p4dp, __p4d((unsigned long)pudp)); } -#define __pud_free_tlb(tlb, pmd, addr) pud_free((tlb)->mm, pmd) +#define __pud_free_tlb(tlb, pmd, addr) tlb_remove_ptdesc((tlb), virt_to_ptdesc(pmd)) #endif @@ -83,10 +83,10 @@ static inline void pud_populate(struct mm_struct *mm, pud_t *pudp, pmd_t *pmdp) set_pud(pudp, __pud((unsigned long)pmdp)); } -#define __pmd_free_tlb(tlb, pmd, addr) pmd_free((tlb)->mm, pmd) +#define __pmd_free_tlb(tlb, pmd, addr) tlb_remove_ptdesc((tlb), virt_to_ptdesc(pmd)) #endif -#define __pte_free_tlb(tlb, pte, addr) pte_free((tlb)->mm, pte) +#define __pte_free_tlb(tlb, pte, addr) tlb_remove_ptdesc((tlb), page_ptdesc(pte)) #endif /* _ASM_ARC_PGALLOC_H */ diff --git a/arch/microblaze/Kconfig b/arch/microblaze/Kconfig index 484ebb3baedf15..af7e821e96c1da 100644 --- a/arch/microblaze/Kconfig +++ b/arch/microblaze/Kconfig @@ -41,6 +41,7 @@ config MICROBLAZE select PCI_SYSCALL if PCI select CPU_NO_EFFICIENT_FFS select MMU_GATHER_NO_RANGE + select MMU_GATHER_RCU_TABLE_FREE select SPARSE_IRQ select ZONE_DMA select TRACE_IRQFLAGS_SUPPORT diff --git a/arch/microblaze/include/asm/pgalloc.h b/arch/microblaze/include/asm/pgalloc.h index 084a8a0dc23952..ffee6a009219ac 100644 --- a/arch/microblaze/include/asm/pgalloc.h +++ b/arch/microblaze/include/asm/pgalloc.h @@ -25,7 +25,7 @@ extern void __bad_pte(pmd_t *pmd); extern pte_t *pte_alloc_one_kernel(struct mm_struct *mm); -#define __pte_free_tlb(tlb, pte, addr) pte_free((tlb)->mm, (pte)) +#define __pte_free_tlb(tlb, pte, addr) tlb_remove_ptdesc((tlb), page_ptdesc(pte)) #define pmd_populate(mm, pmd, pte) \ (pmd_val(*(pmd)) = (unsigned long)page_address(pte)) diff --git a/arch/xtensa/Kconfig b/arch/xtensa/Kconfig index f2f9cd9cde505d..33c4caee30e27b 100644 --- a/arch/xtensa/Kconfig +++ b/arch/xtensa/Kconfig @@ -55,6 +55,7 @@ config XTENSA select HAVE_VIRT_CPU_ACCOUNTING_GEN select IRQ_DOMAIN select LOCK_MM_AND_FIND_VMA + select MMU_GATHER_RCU_TABLE_FREE if MMU select MODULES_USE_ELF_RELA select PERF_USE_VMALLOC select TRACE_IRQFLAGS_SUPPORT diff --git a/arch/xtensa/include/asm/tlb.h b/arch/xtensa/include/asm/tlb.h index 8c3ceb4270180b..6fb7b78154f62a 100644 --- a/arch/xtensa/include/asm/tlb.h +++ b/arch/xtensa/include/asm/tlb.h @@ -16,7 +16,7 @@ #include -#define __pte_free_tlb(tlb, pte, address) pte_free((tlb)->mm, pte) +#define __pte_free_tlb(tlb, pte, address) tlb_remove_ptdesc((tlb), page_ptdesc(pte)) void check_tlb_sanity(void); From 0c06e34a29da5d7eee84178e07c9d3a009e0ed68 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 25 Sep 2026 21:09:41 +0100 Subject: [PATCH 0834/1352] mm: enable MMU_GATHER_RCU_TABLE_FREE for sparc64 Commit 4a0100f7546f ("sparc64: use RCU page table freeing") enabled CONFIG_MMU_GATHER_RCU_TABLE_FREE for SMP sparc64 architectures, expressly for GUP-fast page table walkers. Naturally, UP systems do not have to worry about concurrent GUP fast operations. However, CONFIG_PREEMPT_RCU is also available even on a UP system, so a future pure-RCU page table walker requires MMU_GATHER_RCU_TABLE_FREE to be enabled on UP, even if concurrent GUP fast is not possible there. To enable future pure-RCU page table walkers, enable MMU_GATHER_RCU_TABLE_FREE unconditionally. With this change, it is no longer necessary to have !CONFIG_SMP pgtable_free_tlb(), so also remove this now dead code. This forms part of an overall effort to switch every architecture to this mode. Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-6-31e91065fea4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Kiryl Shutsemau (Meta) Tested-by: Lance Yang Acked-by: Lance Yang Cc: David Hildenbrand Cc: Zi Yan Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Usama Arif Cc: Guo Ren Cc: Brian Cain Cc: Geert Uytterhoeven Cc: Dinh Nguyen Cc: Simon Schuster Cc: Jonas Bonn Cc: Stefan Kristiansson Cc: Stafford Horne Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti Cc: Russell King Cc: Vineet Gupta Cc: Michal Simek Cc: Chris Zankel Cc: Max Filippov Cc: Will Deacon Cc: Aneesh Kumar K.V Cc: Nicholas Piggin Cc: Peter Zijlstra Cc: David S. Miller Cc: Andreas Larsson Cc: Richard Henderson Cc: Matt Turner Cc: Magnus Lindholm Cc: Catalin Marinas Cc: Mark Rutland Cc: Huacai Chen Cc: WANG Xuerui Cc: Thomas Bogendoerfer Cc: James Bottomley Cc: Helge Deller Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Christophe Leroy Cc: Heiko Carstens Cc: Vasily Gorbik Cc: Alexander Gordeev Cc: Christian Borntraeger Cc: Sven Schnelle Cc: Richard Weinberger Cc: Anton Ivanov Cc: Johannes Berg Cc: Thomas Gleixner Cc: Ingo Molnar Cc: Borislav Petkov Cc: Dave Hansen Cc: H. Peter Anvin Cc: Arnd Bergmann Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Cc: Yoshinori Sato Cc: Shakeel Butt Cc: Jonathan Corbet Cc: Randy Dunlap Cc: Hugh Dickins Cc: Qi Zheng --- arch/sparc/Kconfig | 4 ++-- arch/sparc/include/asm/pgalloc_64.h | 8 -------- 2 files changed, 2 insertions(+), 10 deletions(-) diff --git a/arch/sparc/Kconfig b/arch/sparc/Kconfig index ab77d3f2536e1a..8d42ebc6d3029c 100644 --- a/arch/sparc/Kconfig +++ b/arch/sparc/Kconfig @@ -75,8 +75,8 @@ config SPARC64 select HAVE_FUNCTION_GRAPH_TRACER select HAVE_KRETPROBES select HAVE_KPROBES - select MMU_GATHER_RCU_TABLE_FREE if SMP - select HAVE_ARCH_TLB_REMOVE_TABLE if SMP + select MMU_GATHER_RCU_TABLE_FREE + select HAVE_ARCH_TLB_REMOVE_TABLE select MMU_GATHER_MERGE_VMAS select MMU_GATHER_NO_FLUSH_CACHE select HAVE_ARCH_TRANSPARENT_HUGEPAGE diff --git a/arch/sparc/include/asm/pgalloc_64.h b/arch/sparc/include/asm/pgalloc_64.h index caa7632be4c2ae..b5055d259b74d6 100644 --- a/arch/sparc/include/asm/pgalloc_64.h +++ b/arch/sparc/include/asm/pgalloc_64.h @@ -74,8 +74,6 @@ void pte_free_defer(struct mm_struct *mm, pgtable_t pgtable); void pgtable_free(void *table, bool is_page); -#ifdef CONFIG_SMP - struct mmu_gather; void tlb_remove_table(struct mmu_gather *, void *); @@ -96,12 +94,6 @@ static inline void __tlb_remove_table(void *_table) is_page = true; pgtable_free(table, is_page); } -#else /* CONFIG_SMP */ -static inline void pgtable_free_tlb(struct mmu_gather *tlb, void *table, bool is_page) -{ - pgtable_free(table, is_page); -} -#endif /* !CONFIG_SMP */ static inline void __pte_free_tlb(struct mmu_gather *tlb, pte_t *pte, unsigned long address) From 235e6fa1f084b9fd60234be75dd70b94d06c6c0c Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 25 Sep 2026 21:09:42 +0100 Subject: [PATCH 0835/1352] mm: enable MMU_GATHER_RCU_TABLE_FREE for m68k-coldfire Similar to sun3, the coldfire variant of m68k uses 2-level page tables. Update its __pte_free_tlb() function to use tlb_remove_ptdesc() in order that, with CONFIG_MMU_GATHER_RCU_TABLE_FREE, page tables are freed under RCU. The page tables occupy a page each and have no odd semantics, so this change suffices to allow enabling of CONFIG_MMU_GATHER_RCU_TABLE_FREE for m68k-coldfire, so do so. This forms part of an overall effort to switch every architecture to this mode. Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-7-31e91065fea4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Kiryl Shutsemau (Meta) Acked-by: Greg Ungerer Tested-by: Greg Ungerer Acked-by: Lance Yang Cc: David Hildenbrand Cc: Zi Yan Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Usama Arif Cc: Guo Ren Cc: Brian Cain Cc: Geert Uytterhoeven Cc: Dinh Nguyen Cc: Simon Schuster Cc: Jonas Bonn Cc: Stefan Kristiansson Cc: Stafford Horne Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti Cc: Russell King Cc: Vineet Gupta Cc: Michal Simek Cc: Chris Zankel Cc: Max Filippov Cc: Will Deacon Cc: Aneesh Kumar K.V Cc: Nicholas Piggin Cc: Peter Zijlstra Cc: David S. Miller Cc: Andreas Larsson Cc: Richard Henderson Cc: Matt Turner Cc: Magnus Lindholm Cc: Catalin Marinas Cc: Mark Rutland Cc: Huacai Chen Cc: WANG Xuerui Cc: Thomas Bogendoerfer Cc: James Bottomley Cc: Helge Deller Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Christophe Leroy Cc: Heiko Carstens Cc: Vasily Gorbik Cc: Alexander Gordeev Cc: Christian Borntraeger Cc: Sven Schnelle Cc: Richard Weinberger Cc: Anton Ivanov Cc: Johannes Berg Cc: Thomas Gleixner Cc: Ingo Molnar Cc: Borislav Petkov Cc: Dave Hansen Cc: H. Peter Anvin Cc: Arnd Bergmann Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Cc: Yoshinori Sato Cc: Shakeel Butt Cc: Jonathan Corbet Cc: Randy Dunlap Cc: Hugh Dickins Cc: Qi Zheng --- arch/m68k/Kconfig | 2 +- arch/m68k/include/asm/mcf_pgalloc.h | 5 +---- 2 files changed, 2 insertions(+), 5 deletions(-) diff --git a/arch/m68k/Kconfig b/arch/m68k/Kconfig index e29610fd1240f6..6b8ec67c86fde5 100644 --- a/arch/m68k/Kconfig +++ b/arch/m68k/Kconfig @@ -36,7 +36,7 @@ config M68K select HAVE_MOD_ARCH_SPECIFIC select HAVE_UID16 select MMU_GATHER_NO_RANGE if MMU - select MMU_GATHER_RCU_TABLE_FREE if MMU && SUN3 + select MMU_GATHER_RCU_TABLE_FREE if MMU && (SUN3 || COLDFIRE) select MODULES_USE_ELF_REL select MODULES_USE_ELF_RELA select NO_DMA if !MMU && !COLDFIRE diff --git a/arch/m68k/include/asm/mcf_pgalloc.h b/arch/m68k/include/asm/mcf_pgalloc.h index fc5454d37da318..b53ff0950db2e3 100644 --- a/arch/m68k/include/asm/mcf_pgalloc.h +++ b/arch/m68k/include/asm/mcf_pgalloc.h @@ -39,10 +39,7 @@ extern inline pmd_t *pmd_alloc_kernel(pgd_t *pgd, unsigned long address) static inline void __pte_free_tlb(struct mmu_gather *tlb, pgtable_t pgtable, unsigned long address) { - struct ptdesc *ptdesc = virt_to_ptdesc(pgtable); - - pagetable_dtor(ptdesc); - pagetable_free(ptdesc); + tlb_remove_ptdesc(tlb, virt_to_ptdesc(pgtable)); } static inline pgtable_t pte_alloc_one(struct mm_struct *mm) From 1f787740c34474414e7838008b2b434b930e69c4 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 25 Sep 2026 21:09:43 +0100 Subject: [PATCH 0836/1352] mm: enable MMU_GATHER_RCU_TABLE_FREE for sh-X2 Currently, non-x2 sh specifies CONFIG_MMU_GATHER_RCU_TABLE_FREE allowing RCU page table freeing. sh-X2 is problematic because it utilises slab-allocated PMD page tables, and thus tlb_remove_ptdesc() cannot be used in these cases. All other sh variants are fine as commit e3ecf7c7d082 ("mm: pgtable: convert some architectures to use tlb_remove_ptdesc()") already converted page table freeing to use tlb_remove_ptdesc(), which does so after an RCU grace period when CONFIG_MMU_GATHER_RCU_TABLE_FREE is specified. Resolve this issue by firstly specifying CONFIG_HAVE_ARCH_TLB_REMOVE_TABLE for sh-X2, so the arch can provide its own __tlb_remove_table() implementation (called after the RCU grace period). Then, convert __pmd_free_tlb() to tag the pointer to the PMD, and have __tlb_remove_table() check this tag to determine whether to free via the slab or to use pagetable_dtor_free(). This follows the pattern used by sparc64 as implemented in commit 4a0100f7546f ("sparc64: use RCU page table freeing"). Previously __pmd_free_tlb() freed PMD page tables immediately, before any TLB flush IPI. This seems to be a pre-existing bug, which this change also resolves. CONFIG_HAVE_ARCH_TLB_REMOVE_TABLE is only specified for sh-X2, as setting it disables CONFIG_PT_RECLAIM and causes __tlb_remove_table_one() to call tlb_remove_table_sync_rcu() and synchronize_rcu() in turn, and this is not necessary for other sh variants. This forms part of an overall effort to switch every architecture to this mode. Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-8-31e91065fea4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Kiryl Shutsemau (Meta) Acked-by: Lance Yang Cc: David Hildenbrand Cc: Zi Yan Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Usama Arif Cc: Guo Ren Cc: Brian Cain Cc: Geert Uytterhoeven Cc: Dinh Nguyen Cc: Simon Schuster Cc: Jonas Bonn Cc: Stefan Kristiansson Cc: Stafford Horne Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti Cc: Russell King Cc: Vineet Gupta Cc: Michal Simek Cc: Chris Zankel Cc: Max Filippov Cc: Will Deacon Cc: Aneesh Kumar K.V Cc: Nicholas Piggin Cc: Peter Zijlstra Cc: David S. Miller Cc: Andreas Larsson Cc: Richard Henderson Cc: Matt Turner Cc: Magnus Lindholm Cc: Catalin Marinas Cc: Mark Rutland Cc: Huacai Chen Cc: WANG Xuerui Cc: Thomas Bogendoerfer Cc: James Bottomley Cc: Helge Deller Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Christophe Leroy Cc: Heiko Carstens Cc: Vasily Gorbik Cc: Alexander Gordeev Cc: Christian Borntraeger Cc: Sven Schnelle Cc: Richard Weinberger Cc: Anton Ivanov Cc: Johannes Berg Cc: Thomas Gleixner Cc: Ingo Molnar Cc: Borislav Petkov Cc: Dave Hansen Cc: H. Peter Anvin Cc: Arnd Bergmann Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Cc: Yoshinori Sato Cc: Shakeel Butt Cc: Jonathan Corbet Cc: Randy Dunlap Cc: Hugh Dickins Cc: Qi Zheng --- arch/sh/Kconfig | 3 ++- arch/sh/include/asm/pgalloc.h | 6 +++++- arch/sh/mm/pgtable.c | 20 ++++++++++++++++++++ 3 files changed, 27 insertions(+), 2 deletions(-) diff --git a/arch/sh/Kconfig b/arch/sh/Kconfig index 204f64912f0e42..75236bef6f16ee 100644 --- a/arch/sh/Kconfig +++ b/arch/sh/Kconfig @@ -33,6 +33,7 @@ config SUPERH select HAVE_ARCH_AUDITSYSCALL select HAVE_ARCH_KGDB select HAVE_ARCH_SECCOMP_FILTER + select HAVE_ARCH_TLB_REMOVE_TABLE if X2TLB select HAVE_ARCH_TRACEHOOK select HAVE_DEBUG_BUGVERBOSE select HAVE_DEBUG_KMEMLEAK @@ -61,7 +62,7 @@ config SUPERH select HAVE_SYSCALL_TRACEPOINTS select IRQ_FORCED_THREADING select LOCK_MM_AND_FIND_VMA - select MMU_GATHER_RCU_TABLE_FREE if MMU && !X2TLB + select MMU_GATHER_RCU_TABLE_FREE if MMU select MODULES_USE_ELF_RELA select NEED_SG_DMA_LENGTH select NO_DMA if !MMU && !DMA_COHERENT diff --git a/arch/sh/include/asm/pgalloc.h b/arch/sh/include/asm/pgalloc.h index 6fe7123d38fa9e..67ce7fa23fa128 100644 --- a/arch/sh/include/asm/pgalloc.h +++ b/arch/sh/include/asm/pgalloc.h @@ -17,7 +17,11 @@ extern void pgd_free(struct mm_struct *mm, pgd_t *pgd); extern void pud_populate(struct mm_struct *mm, pud_t *pudp, pmd_t *pmd); extern pmd_t *pmd_alloc_one(struct mm_struct *mm, unsigned long address); extern void pmd_free(struct mm_struct *mm, pmd_t *pmd); -#define __pmd_free_tlb(tlb, pmdp, addr) pmd_free((tlb)->mm, (pmdp)) +extern void __tlb_remove_table(void *table); + +/* PMDs are slab-allocated, tag so they are freed correctly. */ +#define __pmd_free_tlb(tlb, pmdp, addr) \ + tlb_remove_table((tlb), (void *)((unsigned long)(pmdp) | 1)) #endif static inline void pmd_populate_kernel(struct mm_struct *mm, pmd_t *pmd, diff --git a/arch/sh/mm/pgtable.c b/arch/sh/mm/pgtable.c index 3a4085ea0161fe..f6184b86b89c6c 100644 --- a/arch/sh/mm/pgtable.c +++ b/arch/sh/mm/pgtable.c @@ -56,4 +56,24 @@ void pmd_free(struct mm_struct *mm, pmd_t *pmd) { kmem_cache_free(pmd_cachep, pmd); } + +static void __tlb_remove_table_slab(void *table) +{ + kmem_cache_free(pmd_cachep, table); +} + +static void __tlb_remove_table_pgtable(void *table) +{ + pagetable_dtor_free(table); +} + +void __tlb_remove_table(void *table) +{ + const unsigned long addr = (unsigned long)table; + + if (addr & 1) + __tlb_remove_table_slab((void *)(addr & ~1UL)); + else + __tlb_remove_table_pgtable(table); +} #endif /* PAGETABLE_LEVELS > 2 */ From e19dab39d4b86b1782ada1db3a3b8a3d00520ea1 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 25 Sep 2026 21:09:44 +0100 Subject: [PATCH 0837/1352] mm: enable MMU_GATHER_RCU_TABLE_FREE for m68k-motorola sun3 and coldfire are already supported, however motorola requires a little more care. Here, custom table removal logic is required, so CONFIG_HAVE_ARCH_TLB_REMOVE_TABLE is enabled for m68k-motorola. Firstly as part of this change, the page table level must be communicated to the underlying __tlb_remove_table() implementation. Take advantage of the fact that page tables are aligned by more than enough to permit setting TABLE_PTE or TABLE_PMD in the low bits of the pointer, and store this there. Then update __pte_free_tlb() and __pmd_free_tlb() to pass this through, then have __tlb_remove_table() decode this and pass it to free_pointer_table(). The page table freeing is performed via call_rcu(), so free_pointer_table() now will be invoked from softirq context, and as such may be re-entrant. Introduce a spinlock to handle this and hold it over the time a given ptable entry is being referenced in both get_pointer_table() and free_pointer_table(). As softirq is the only asynchronous context in which the lock is taken, it suffices to disable bottom halves while holding it. In order to make things a little easier in this respect, separate out the logic for adding a new ptable entry into add_pointer_table() and only hold the lock during ptable entry insertion in this case. Note that original list_add_tail(new, dp) added new prior to dp, which is ptable_list[type].next, i.e. after ptable_list[type]. The equivalent therefore is list_add(new, &ptable_list[type]), which adds new after ptable_list[type], only without needing to make reference to dp. Note that, as m68k-motorola specifies CONFIG_HAVE_ARCH_TLB_REMOVE_TABLE, it does not enable CONFIG_PT_RECLAIM. This isn't meaningfully impactful. With this applied, all of m68k implements CONFIG_MMU_GATHER_RCU_TABLE_FREE. This forms part of an overall effort to switch every architecture to this mode. Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-9-31e91065fea4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Kiryl Shutsemau (Meta) Acked-by: Lance Yang Tested-by: Lance Yang Cc: David Hildenbrand Cc: Zi Yan Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Usama Arif Cc: Guo Ren Cc: Brian Cain Cc: Geert Uytterhoeven Cc: Dinh Nguyen Cc: Simon Schuster Cc: Jonas Bonn Cc: Stefan Kristiansson Cc: Stafford Horne Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti Cc: Russell King Cc: Vineet Gupta Cc: Michal Simek Cc: Chris Zankel Cc: Max Filippov Cc: Will Deacon Cc: Aneesh Kumar K.V Cc: Nicholas Piggin Cc: Peter Zijlstra Cc: David S. Miller Cc: Andreas Larsson Cc: Richard Henderson Cc: Matt Turner Cc: Magnus Lindholm Cc: Catalin Marinas Cc: Mark Rutland Cc: Huacai Chen Cc: WANG Xuerui Cc: Thomas Bogendoerfer Cc: James Bottomley Cc: Helge Deller Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Christophe Leroy Cc: Heiko Carstens Cc: Vasily Gorbik Cc: Alexander Gordeev Cc: Christian Borntraeger Cc: Sven Schnelle Cc: Richard Weinberger Cc: Anton Ivanov Cc: Johannes Berg Cc: Thomas Gleixner Cc: Ingo Molnar Cc: Borislav Petkov Cc: Dave Hansen Cc: H. Peter Anvin Cc: Arnd Bergmann Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Cc: Yoshinori Sato Cc: Shakeel Butt Cc: Jonathan Corbet Cc: Randy Dunlap Cc: Hugh Dickins Cc: Qi Zheng --- arch/m68k/Kconfig | 3 +- arch/m68k/include/asm/motorola_pgalloc.h | 9 +- arch/m68k/mm/motorola.c | 119 +++++++++++++++-------- 3 files changed, 84 insertions(+), 47 deletions(-) diff --git a/arch/m68k/Kconfig b/arch/m68k/Kconfig index 6b8ec67c86fde5..fa5d39549da96a 100644 --- a/arch/m68k/Kconfig +++ b/arch/m68k/Kconfig @@ -29,6 +29,7 @@ config M68K select HAVE_ARCH_LIBGCC_H select HAVE_ARCH_SECCOMP select HAVE_ARCH_SECCOMP_FILTER + select HAVE_ARCH_TLB_REMOVE_TABLE if MMU_MOTOROLA select HAVE_ASM_MODVERSIONS select HAVE_DEBUG_BUGVERBOSE select HAVE_EFFICIENT_UNALIGNED_ACCESS if !CPU_HAS_NO_UNALIGNED @@ -36,7 +37,7 @@ config M68K select HAVE_MOD_ARCH_SPECIFIC select HAVE_UID16 select MMU_GATHER_NO_RANGE if MMU - select MMU_GATHER_RCU_TABLE_FREE if MMU && (SUN3 || COLDFIRE) + select MMU_GATHER_RCU_TABLE_FREE if MMU select MODULES_USE_ELF_REL select MODULES_USE_ELF_RELA select NO_DMA if !MMU && !COLDFIRE diff --git a/arch/m68k/include/asm/motorola_pgalloc.h b/arch/m68k/include/asm/motorola_pgalloc.h index 1091fb0affbee4..dcde40e8b5c6a1 100644 --- a/arch/m68k/include/asm/motorola_pgalloc.h +++ b/arch/m68k/include/asm/motorola_pgalloc.h @@ -17,6 +17,7 @@ enum m68k_table_types { extern void init_pointer_table(void *table, int type); extern void *get_pointer_table(struct mm_struct *mm, int type); extern int free_pointer_table(void *table, int type); +extern void __tlb_remove_table(void *table); /* * Allocate and free page tables. The xxx_kernel() versions are @@ -47,7 +48,7 @@ static inline void pte_free(struct mm_struct *mm, pgtable_t pgtable) static inline void __pte_free_tlb(struct mmu_gather *tlb, pgtable_t pgtable, unsigned long address) { - free_pointer_table(pgtable, TABLE_PTE); + tlb_remove_table(tlb, (void *)((unsigned long)pgtable | TABLE_PTE)); } @@ -61,10 +62,10 @@ static inline int pmd_free(struct mm_struct *mm, pmd_t *pmd) return free_pointer_table(pmd, TABLE_PMD); } -static inline int __pmd_free_tlb(struct mmu_gather *tlb, pmd_t *pmd, - unsigned long address) +static inline void __pmd_free_tlb(struct mmu_gather *tlb, pmd_t *pmd, + unsigned long address) { - return free_pointer_table(pmd, TABLE_PMD); + tlb_remove_table(tlb, (void *)((unsigned long)pmd | TABLE_PMD)); } diff --git a/arch/m68k/mm/motorola.c b/arch/m68k/mm/motorola.c index b30aa69a73a6ad..f3efa0d139634c 100644 --- a/arch/m68k/mm/motorola.c +++ b/arch/m68k/mm/motorola.c @@ -20,6 +20,7 @@ #include #include #include +#include #include #include @@ -103,6 +104,8 @@ static struct list_head ptable_list[3] = { LIST_HEAD_INIT(ptable_list[2]), }; +static DEFINE_SPINLOCK(ptable_lock); + #define PD_PTABLE(ptdesc) ((ptable_desc *)&(virt_to_ptdesc((void *)(ptdesc))->pt_list)) #define PD_PTDESC(ptable) (list_entry(ptable, struct ptdesc, pt_list)) #define PD_MARKBITS(dp) (*(unsigned int *)&PD_PTDESC(dp)->pt_index) @@ -139,52 +142,65 @@ void __init init_pointer_table(void *table, int type) return; } -void *get_pointer_table(struct mm_struct *mm, int type) +/* + * For a pointer table for a user process address space, a + * table is taken from a ptdesc allocated for the purpose. Each + * ptdesc can hold 8 pointer tables. The ptdesc is remapped in + * virtual address space to be noncacheable. + */ +static void *add_pointer_table(struct mm_struct *mm, int type) { - ptable_desc *dp = ptable_list[type].next; - unsigned int mask = list_empty(&ptable_list[type]) ? 0 : PD_MARKBITS(dp); - unsigned int tmp, off; + struct ptdesc *ptdesc; + ptable_desc *new; + void *pt_addr; - /* - * For a pointer table for a user process address space, a - * table is taken from a ptdesc allocated for the purpose. Each - * ptdesc can hold 8 pointer tables. The ptdesc is remapped in - * virtual address space to be noncacheable. - */ - if (mask == 0) { - struct ptdesc *ptdesc; - ptable_desc *new; - void *pt_addr; - - ptdesc = pagetable_alloc(GFP_KERNEL | __GFP_ZERO, 0); - if (!ptdesc) - return NULL; - - pt_addr = ptdesc_address(ptdesc); - - switch (type) { - case TABLE_PTE: - /* - * m68k doesn't have SPLIT_PTE_PTLOCKS for not having - * SMP. - */ - pagetable_pte_ctor(mm, ptdesc); - break; - case TABLE_PMD: - pagetable_pmd_ctor(mm, ptdesc); - break; - case TABLE_PGD: - pagetable_pgd_ctor(ptdesc); - break; - } + ptdesc = pagetable_alloc(GFP_KERNEL | __GFP_ZERO, 0); + if (!ptdesc) + return NULL; + + pt_addr = ptdesc_address(ptdesc); + + switch (type) { + case TABLE_PTE: + /* + * m68k doesn't have SPLIT_PTE_PTLOCKS for not having + * SMP. + */ + pagetable_pte_ctor(mm, ptdesc); + break; + case TABLE_PMD: + pagetable_pmd_ctor(mm, ptdesc); + break; + case TABLE_PGD: + pagetable_pgd_ctor(ptdesc); + break; + } + + mmu_page_ctor(pt_addr); + + new = PD_PTABLE(pt_addr); - mmu_page_ctor(pt_addr); + PD_MARKBITS(new) = ptable_mask(type) - 1; + scoped_guard(spinlock_bh, &ptable_lock) + list_add(new, &ptable_list[type]); - new = PD_PTABLE(pt_addr); - PD_MARKBITS(new) = ptable_mask(type) - 1; - list_add_tail(new, dp); + return (pmd_t *)pt_addr; +} + +void *get_pointer_table(struct mm_struct *mm, int type) +{ + unsigned int tmp, off; + unsigned long mask; + ptable_desc *dp; + void *ret; - return (pmd_t *)pt_addr; + spin_lock_bh(&ptable_lock); + dp = ptable_list[type].next; + mask = list_empty(&ptable_list[type]) ? 0 : PD_MARKBITS(dp); + + if (mask == 0) { + spin_unlock_bh(&ptable_lock); + return add_pointer_table(mm, type); } for (tmp = 1, off = 0; (mask & tmp) == 0; tmp <<= 1, off += ptable_size(type)) @@ -194,7 +210,10 @@ void *get_pointer_table(struct mm_struct *mm, int type) /* move to end of list */ list_move_tail(dp, &ptable_list[type]); } - return ptdesc_address(PD_PTDESC(dp)) + off; + + ret = ptdesc_address(PD_PTDESC(dp)) + off; + spin_unlock_bh(&ptable_lock); + return ret; } int free_pointer_table(void *table, int type) @@ -204,6 +223,8 @@ int free_pointer_table(void *table, int type) unsigned long pt_addr = ptable & PAGE_MASK; unsigned int mask = 1U << ((ptable - pt_addr)/ptable_size(type)); + spin_lock_bh(&ptable_lock); + dp = PD_PTABLE(pt_addr); if (PD_MARKBITS (dp) & mask) panic ("table already free!"); @@ -213,6 +234,8 @@ int free_pointer_table(void *table, int type) if (PD_MARKBITS(dp) == ptable_mask(type)) { /* all tables in ptdesc are free, free ptdesc */ list_del(dp); + spin_unlock_bh(&ptable_lock); + mmu_page_dtor((void *)pt_addr); pagetable_dtor_free(virt_to_ptdesc((void *)pt_addr)); return 1; @@ -223,9 +246,21 @@ int free_pointer_table(void *table, int type) */ list_move(dp, &ptable_list[type]); } + + spin_unlock_bh(&ptable_lock); return 0; } +void __tlb_remove_table(void *table) +{ + /* The bottom 2 bits are used to encode page table type. */ + const unsigned long encoded = (unsigned long)table; + void *addr = (void *)(encoded & ~3UL); + const int type = encoded & 3; + + free_pointer_table(addr, type); +} + /* size of memory already mapped in head.S */ extern __initdata unsigned long m68k_init_mapped_size; From 3359a257ca6065489ccca4524725d0eb90a2a733 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 25 Sep 2026 21:09:45 +0100 Subject: [PATCH 0838/1352] mm: enable MMU_GATHER_RCU_TABLE_FREE for sparc32 Careful handling is required for sparc32 which implements page tables as part of a shared backing page. To support this, a custom __tlb_remove_table() function is required, as specified by CONFIG_HAVE_ARCH_TLB_REMOVE_TABLE. This allows __pte_free_tlb() and __pmd_free_tlb() to specify which page table level is being freed, which is transmitted to __tlb_remove_table() through setting the lowest bit of the page table to 1 for a PMD and 0 for a PTE (the page tables are 256-byte aligned so this is safe to do). Next, since the page table freeing is done via RCU callback, and thus might be executed in softirq context, update the spin locks to IRQ save/restore. Then, in __tlb_remove_table(), figure out whether to free a PMD page table via free_pmd_fast() or a PTE via the newly introduced __pte_free() function, using the lower bit encoded in __pte_free_tlb() or __pmd_free_tlb() to determine which to call. As part of this change use this spin lock rather than mm->page_table_lock for all shared page table exclusion, as RCU freeing means that page tables can be freed from soft IRQ context so both don't have an mm and also mm->page_table_lock is not IRQ-safe. Note that the specification of CONFIG_HAVE_ARCH_TLB_REMOVE_TABLE disables CONFIG_PT_RECLAIM for sparc32, which mirrors sparc64. This forms part of an overall effort to switch every architecture to this mode, and with it complete, means every architecture now supports CONFIG_MMU_GATHER_RCU_TABLE_FREE. Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-10-31e91065fea4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Kiryl Shutsemau (Meta) Tested-by: Lance Yang Acked-by: Lance Yang Cc: David Hildenbrand Cc: Zi Yan Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Usama Arif Cc: Guo Ren Cc: Brian Cain Cc: Geert Uytterhoeven Cc: Dinh Nguyen Cc: Simon Schuster Cc: Jonas Bonn Cc: Stefan Kristiansson Cc: Stafford Horne Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti Cc: Russell King Cc: Vineet Gupta Cc: Michal Simek Cc: Chris Zankel Cc: Max Filippov Cc: Will Deacon Cc: Aneesh Kumar K.V Cc: Nicholas Piggin Cc: Peter Zijlstra Cc: David S. Miller Cc: Andreas Larsson Cc: Richard Henderson Cc: Matt Turner Cc: Magnus Lindholm Cc: Catalin Marinas Cc: Mark Rutland Cc: Huacai Chen Cc: WANG Xuerui Cc: Thomas Bogendoerfer Cc: James Bottomley Cc: Helge Deller Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Christophe Leroy Cc: Heiko Carstens Cc: Vasily Gorbik Cc: Alexander Gordeev Cc: Christian Borntraeger Cc: Sven Schnelle Cc: Richard Weinberger Cc: Anton Ivanov Cc: Johannes Berg Cc: Thomas Gleixner Cc: Ingo Molnar Cc: Borislav Petkov Cc: Dave Hansen Cc: H. Peter Anvin Cc: Arnd Bergmann Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Cc: Yoshinori Sato Cc: Shakeel Butt Cc: Jonathan Corbet Cc: Randy Dunlap Cc: Hugh Dickins Cc: Qi Zheng --- arch/sparc/Kconfig | 2 ++ arch/sparc/include/asm/pgalloc_32.h | 7 +++++-- arch/sparc/lib/bitext.c | 14 ++++++------- arch/sparc/mm/srmmu.c | 32 ++++++++++++++++++++++++----- 4 files changed, 41 insertions(+), 14 deletions(-) diff --git a/arch/sparc/Kconfig b/arch/sparc/Kconfig index 8d42ebc6d3029c..79c09d6ee466c4 100644 --- a/arch/sparc/Kconfig +++ b/arch/sparc/Kconfig @@ -64,6 +64,8 @@ config SPARC32 select HAVE_UID16 select HAVE_PAGE_SIZE_4KB select LOCK_MM_AND_FIND_VMA + select MMU_GATHER_RCU_TABLE_FREE + select HAVE_ARCH_TLB_REMOVE_TABLE select OLD_SIGACTION select ZONE_DMA diff --git a/arch/sparc/include/asm/pgalloc_32.h b/arch/sparc/include/asm/pgalloc_32.h index 4f73e87b22a32b..36010852ba0c04 100644 --- a/arch/sparc/include/asm/pgalloc_32.h +++ b/arch/sparc/include/asm/pgalloc_32.h @@ -48,7 +48,9 @@ static inline void free_pmd_fast(pmd_t * pmd) } #define pmd_free(mm, pmd) free_pmd_fast(pmd) -#define __pmd_free_tlb(tlb, pmd, addr) pmd_free((tlb)->mm, pmd) + +#define __pmd_free_tlb(tlb, pmd, addr) \ + tlb_remove_table((tlb), (void *)((unsigned long)(pmd) | 1UL)) #define pmd_populate(mm, pmd, pte) pmd_set(pmd, pte) @@ -72,6 +74,7 @@ static inline void free_pte_fast(pte_t *pte) #define pte_free_kernel(mm, pte) free_pte_fast(pte) void pte_free(struct mm_struct * mm, pgtable_t pte); -#define __pte_free_tlb(tlb, pte, addr) pte_free((tlb)->mm, pte) +void __tlb_remove_table(void *table); +#define __pte_free_tlb(tlb, pte, addr) tlb_remove_table((tlb), (void *)(pte)) #endif /* _SPARC_PGALLOC_H */ diff --git a/arch/sparc/lib/bitext.c b/arch/sparc/lib/bitext.c index 32a5c1d9459cde..c309e27973ce6f 100644 --- a/arch/sparc/lib/bitext.c +++ b/arch/sparc/lib/bitext.c @@ -22,8 +22,6 @@ * @align: requested alignment * * Returns offset in the map or -1 if out of space. - * - * Not safe to call from an interrupt (uses spin_lock). */ int bit_map_string_get(struct bit_map *t, int len, int align) { @@ -31,6 +29,7 @@ int bit_map_string_get(struct bit_map *t, int len, int align) int off_new; int align1; int i, color; + unsigned long flags; if (t->num_colors) { /* align is overloaded to be the page color */ @@ -50,7 +49,7 @@ int bit_map_string_get(struct bit_map *t, int len, int align) BUG(); color &= align1; - spin_lock(&t->lock); + spin_lock_irqsave(&t->lock, flags); if (len < t->last_size) offset = t->first_free; else @@ -64,7 +63,7 @@ int bit_map_string_get(struct bit_map *t, int len, int align) if (offset >= t->size) offset = 0; if (count + len > t->size) { - spin_unlock(&t->lock); + spin_unlock_irqrestore(&t->lock, flags); /* P3 */ printk(KERN_ERR "bitmap out: size %d used %d off %d len %d align %d count %d\n", t->size, t->used, offset, len, align, count); @@ -90,7 +89,7 @@ int bit_map_string_get(struct bit_map *t, int len, int align) t->last_off = 0; t->used += len; t->last_size = len; - spin_unlock(&t->lock); + spin_unlock_irqrestore(&t->lock, flags); return offset; } } @@ -103,10 +102,11 @@ int bit_map_string_get(struct bit_map *t, int len, int align) void bit_map_clear(struct bit_map *t, int offset, int len) { int i; + unsigned long flags; if (t->used < len) BUG(); /* Much too late to do any good, but alas... */ - spin_lock(&t->lock); + spin_lock_irqsave(&t->lock, flags); for (i = 0; i < len; i++) { if (test_bit(offset + i, t->map) == 0) BUG(); @@ -115,7 +115,7 @@ void bit_map_clear(struct bit_map *t, int offset, int len) if (offset < t->first_free) t->first_free = offset; t->used -= len; - spin_unlock(&t->lock); + spin_unlock_irqrestore(&t->lock, flags); } void bit_map_init(struct bit_map *t, unsigned long *map, int size) diff --git a/arch/sparc/mm/srmmu.c b/arch/sparc/mm/srmmu.c index 9a74902ad18147..1c277ab3cdb848 100644 --- a/arch/sparc/mm/srmmu.c +++ b/arch/sparc/mm/srmmu.c @@ -340,38 +340,60 @@ pgd_t *get_pgd_fast(void) * Alignments up to the page size are the same for physical and virtual * addresses of the nocache area. */ + +static DEFINE_SPINLOCK(pte_page_lock); + pgtable_t pte_alloc_one(struct mm_struct *mm) { + unsigned long flags; pte_t *ptep; struct page *page; if (!(ptep = pte_alloc_one_kernel(mm))) return NULL; page = pfn_to_page(__nocache_pa((unsigned long)ptep) >> PAGE_SHIFT); - spin_lock(&mm->page_table_lock); + spin_lock_irqsave(&pte_page_lock, flags); if (page_ref_inc_return(page) == 2 && !pagetable_pte_ctor(mm, page_ptdesc(page))) { page_ref_dec(page); ptep = NULL; } - spin_unlock(&mm->page_table_lock); + spin_unlock_irqrestore(&pte_page_lock, flags); return ptep; } -void pte_free(struct mm_struct *mm, pgtable_t ptep) +static void __pte_free(pgtable_t ptep) { struct page *page; + unsigned long flags; page = pfn_to_page(__nocache_pa((unsigned long)ptep) >> PAGE_SHIFT); - spin_lock(&mm->page_table_lock); + spin_lock_irqsave(&pte_page_lock, flags); if (page_ref_dec_return(page) == 1) pagetable_dtor(page_ptdesc(page)); - spin_unlock(&mm->page_table_lock); + spin_unlock_irqrestore(&pte_page_lock, flags); srmmu_free_nocache(ptep, SRMMU_PTE_TABLE_SIZE); } +void pte_free(struct mm_struct *mm, pgtable_t ptep) +{ + __pte_free(ptep); +} + +void __tlb_remove_table(void *table) +{ + const unsigned long encoded = (unsigned long)table; + const unsigned long addr = encoded & ~1UL; + const bool is_pmd = encoded & 1; + + if (is_pmd) + free_pmd_fast((pmd_t *)addr); + else + __pte_free((pgtable_t)addr); +} + /* context handling - a dynamically sized pool is used */ #define NO_CONTEXT -1 From 5365e506822b9b8b3b7b57e0a49278d16ada2a15 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 25 Sep 2026 21:09:46 +0100 Subject: [PATCH 0839/1352] mm: userland pgtable freeing is RCU-safe now, remove leftover bits Now every architecture has been converted to support CONFIG_MMU_GATHER_RCU_TABLE_FREE, this configuration option no longer makes any sense to keep around. Therefore remove it, and remove all the dead code that existed for !CONFIG_MMU_GATHER_RCU_TABLE_FREE architectures previously. Additionally, CONFIG_MMU_GATHER_TABLE_FREE is no longer necessary, as all architectures instead use CONFIG_HAVE_ARCH_TLB_REMOVE_TABLE when a custom __tlb_remove_table() is required, so remove this too. A number of architectures only enabled CONFIG_MMU_GATHER_RCU_TABLE_FREE if CONFIG_MMU was set, however the mmu_gather logic only actually does something meaningful if CONFIG_MMU is set (mmu_gather.c is only compiled in this case, for instance). As a result, there's no need to gate any of this logic on CONFIG_MMU explicitly. CONFIG_PT_RECLAIM however does have a strict dependency on CONFIG_MMU, so make this dependency explicit. Additionally, correct comments to remove references to non-RCU page table gathering and make it clear that this is not 'semi-RCU', nor has it been since commit 1fb3d8c20bfa ("mm/mmu_gather: replace IPI with synchronize_rcu() when batch allocation fails"). With this change in place the kernel policy is now that userspace page tables are freed after an RCU grace period, and thus it is now safe to unconditionally perform page table walks under RCU, safe in the knowledge that page tables will not be freed underneath the walker. This is all that is guaranteed, however, so naturally it is still incumbent upon page table walkers to ensure that the page table entries are as expected. Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-11-31e91065fea4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Kiryl Shutsemau (Meta) Reviewed-by: Lance Yang Acked-by: David Hildenbrand (Arm) Cc: Zi Yan Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Usama Arif Cc: Guo Ren Cc: Brian Cain Cc: Geert Uytterhoeven Cc: Dinh Nguyen Cc: Simon Schuster Cc: Jonas Bonn Cc: Stefan Kristiansson Cc: Stafford Horne Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti Cc: Russell King Cc: Vineet Gupta Cc: Michal Simek Cc: Chris Zankel Cc: Max Filippov Cc: Will Deacon Cc: Aneesh Kumar K.V Cc: Nicholas Piggin Cc: Peter Zijlstra Cc: David S. Miller Cc: Andreas Larsson Cc: Richard Henderson Cc: Matt Turner Cc: Magnus Lindholm Cc: Catalin Marinas Cc: Mark Rutland Cc: Huacai Chen Cc: WANG Xuerui Cc: Thomas Bogendoerfer Cc: James Bottomley Cc: Helge Deller Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Christophe Leroy Cc: Heiko Carstens Cc: Vasily Gorbik Cc: Alexander Gordeev Cc: Christian Borntraeger Cc: Sven Schnelle Cc: Richard Weinberger Cc: Anton Ivanov Cc: Johannes Berg Cc: Thomas Gleixner Cc: Ingo Molnar Cc: Borislav Petkov Cc: Dave Hansen Cc: H. Peter Anvin Cc: Arnd Bergmann Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Cc: Yoshinori Sato Cc: Shakeel Butt Cc: Jonathan Corbet Cc: Randy Dunlap Cc: Hugh Dickins Cc: Qi Zheng --- arch/Kconfig | 8 ---- arch/alpha/Kconfig | 1 - arch/arc/Kconfig | 1 - arch/arm/Kconfig | 1 - arch/arm64/Kconfig | 1 - arch/csky/Kconfig | 1 - arch/hexagon/Kconfig | 1 - arch/loongarch/Kconfig | 1 - arch/m68k/Kconfig | 1 - arch/microblaze/Kconfig | 1 - arch/mips/Kconfig | 1 - arch/nios2/Kconfig | 1 - arch/openrisc/Kconfig | 1 - arch/parisc/Kconfig | 1 - arch/powerpc/Kconfig | 1 - arch/riscv/Kconfig | 1 - arch/s390/Kconfig | 1 - arch/sh/Kconfig | 1 - arch/sparc/Kconfig | 2 - arch/sparc/include/asm/tlb_64.h | 2 - arch/um/Kconfig | 1 - arch/x86/Kconfig | 1 - arch/xtensa/Kconfig | 1 - include/asm-generic/tlb.h | 66 +++++---------------------------- mm/Kconfig | 2 +- mm/gup.c | 5 ++- mm/mmu_gather.c | 30 +++------------ 27 files changed, 18 insertions(+), 117 deletions(-) diff --git a/arch/Kconfig b/arch/Kconfig index 45c65777236231..6f7516916797eb 100644 --- a/arch/Kconfig +++ b/arch/Kconfig @@ -526,13 +526,6 @@ config HAVE_ARCH_JUMP_LABEL config HAVE_ARCH_JUMP_LABEL_RELATIVE bool -config MMU_GATHER_TABLE_FREE - bool - -config MMU_GATHER_RCU_TABLE_FREE - bool - select MMU_GATHER_TABLE_FREE - config MMU_GATHER_PAGE_SIZE bool @@ -548,7 +541,6 @@ config MMU_GATHER_MERGE_VMAS config MMU_GATHER_NO_GATHER bool - depends on MMU_GATHER_TABLE_FREE config ARCH_WANT_IRQS_OFF_ACTIVATE_MM bool diff --git a/arch/alpha/Kconfig b/arch/alpha/Kconfig index e53ef2d8846360..9063c7bda4e41e 100644 --- a/arch/alpha/Kconfig +++ b/arch/alpha/Kconfig @@ -42,7 +42,6 @@ config ALPHA select ARCH_STACKWALK select CPU_NO_EFFICIENT_FFS if !ALPHA_EV67 select MMU_GATHER_NO_RANGE - select MMU_GATHER_RCU_TABLE_FREE select SPARSEMEM_EXTREME if SPARSEMEM select ZONE_DMA select TRACE_IRQFLAGS_SUPPORT diff --git a/arch/arc/Kconfig b/arch/arc/Kconfig index 0c1679f4d386f8..80e61175bf526b 100644 --- a/arch/arc/Kconfig +++ b/arch/arc/Kconfig @@ -47,7 +47,6 @@ config ARC select HAVE_SYSCALL_TRACEPOINTS select IRQ_DOMAIN select LOCK_MM_AND_FIND_VMA - select MMU_GATHER_RCU_TABLE_FREE select MODULES_USE_ELF_RELA select OF select OF_EARLY_FLATTREE diff --git a/arch/arm/Kconfig b/arch/arm/Kconfig index 72b9afc6ae1058..0cc289a7184ab1 100644 --- a/arch/arm/Kconfig +++ b/arch/arm/Kconfig @@ -134,7 +134,6 @@ config ARM select HAVE_PERF_REGS select HAVE_PERF_USER_STACK_DUMP select HAVE_POSIX_CPU_TIMERS_TASK_WORK - select MMU_GATHER_RCU_TABLE_FREE if MMU select HAVE_REGS_AND_STACK_ACCESS_API select HAVE_RSEQ select HAVE_RUST if CPU_LITTLE_ENDIAN && CPU_32v7 && !KASAN diff --git a/arch/arm64/Kconfig b/arch/arm64/Kconfig index 2bbeded33da0da..b6c2dd8b26124d 100644 --- a/arch/arm64/Kconfig +++ b/arch/arm64/Kconfig @@ -221,7 +221,6 @@ config ARM64 select HAVE_RELIABLE_STACKTRACE select HAVE_POSIX_CPU_TIMERS_TASK_WORK select HAVE_FUNCTION_ARG_ACCESS_API - select MMU_GATHER_RCU_TABLE_FREE select HAVE_RSEQ select HAVE_RUST if RUSTC_SUPPORTS_ARM64 select HAVE_STACKPROTECTOR diff --git a/arch/csky/Kconfig b/arch/csky/Kconfig index 80f89ef1d9622c..4331313a42ff30 100644 --- a/arch/csky/Kconfig +++ b/arch/csky/Kconfig @@ -96,7 +96,6 @@ config CSKY select HAVE_SYSCALL_TRACEPOINTS select HOTPLUG_CORE_SYNC_DEAD if HOTPLUG_CPU select LOCK_MM_AND_FIND_VMA - select MMU_GATHER_RCU_TABLE_FREE select MAY_HAVE_SPARSE_IRQ select MODULES_USE_ELF_RELA if MODULES select OF diff --git a/arch/hexagon/Kconfig b/arch/hexagon/Kconfig index d9b3fb86556be3..b4849114001335 100644 --- a/arch/hexagon/Kconfig +++ b/arch/hexagon/Kconfig @@ -23,7 +23,6 @@ config HEXAGON # select HAVE_CLK select GENERIC_ATOMIC64 select HAVE_PERF_EVENTS - select MMU_GATHER_RCU_TABLE_FREE # GENERIC_ALLOCATOR is used by dma_alloc_coherent() select GENERIC_ALLOCATOR select GENERIC_IRQ_PROBE diff --git a/arch/loongarch/Kconfig b/arch/loongarch/Kconfig index 1d8fb1e456d6d8..0bd8503fd5c4c2 100644 --- a/arch/loongarch/Kconfig +++ b/arch/loongarch/Kconfig @@ -188,7 +188,6 @@ config LOONGARCH select IRQ_LOONGARCH_CPU select LOCK_MM_AND_FIND_VMA select MMU_GATHER_MERGE_VMAS if MMU - select MMU_GATHER_RCU_TABLE_FREE select MODULES_USE_ELF_RELA if MODULES select NEED_PER_CPU_EMBED_FIRST_CHUNK select NEED_PER_CPU_PAGE_FIRST_CHUNK diff --git a/arch/m68k/Kconfig b/arch/m68k/Kconfig index fa5d39549da96a..eb84c3af92c02c 100644 --- a/arch/m68k/Kconfig +++ b/arch/m68k/Kconfig @@ -37,7 +37,6 @@ config M68K select HAVE_MOD_ARCH_SPECIFIC select HAVE_UID16 select MMU_GATHER_NO_RANGE if MMU - select MMU_GATHER_RCU_TABLE_FREE if MMU select MODULES_USE_ELF_REL select MODULES_USE_ELF_RELA select NO_DMA if !MMU && !COLDFIRE diff --git a/arch/microblaze/Kconfig b/arch/microblaze/Kconfig index af7e821e96c1da..484ebb3baedf15 100644 --- a/arch/microblaze/Kconfig +++ b/arch/microblaze/Kconfig @@ -41,7 +41,6 @@ config MICROBLAZE select PCI_SYSCALL if PCI select CPU_NO_EFFICIENT_FFS select MMU_GATHER_NO_RANGE - select MMU_GATHER_RCU_TABLE_FREE select SPARSE_IRQ select ZONE_DMA select TRACE_IRQFLAGS_SUPPORT diff --git a/arch/mips/Kconfig b/arch/mips/Kconfig index d7c67cebe065f9..ed17c9b1624cac 100644 --- a/arch/mips/Kconfig +++ b/arch/mips/Kconfig @@ -97,7 +97,6 @@ config MIPS select IRQ_FORCED_THREADING select ISA if EISA select LOCK_MM_AND_FIND_VMA - select MMU_GATHER_RCU_TABLE_FREE select MODULES_USE_ELF_REL if MODULES select MODULES_USE_ELF_RELA if MODULES && 64BIT select PERF_USE_VMALLOC diff --git a/arch/nios2/Kconfig b/arch/nios2/Kconfig index b0ccfc3b7a7e1b..9c0e6eaeb005cd 100644 --- a/arch/nios2/Kconfig +++ b/arch/nios2/Kconfig @@ -19,7 +19,6 @@ config NIOS2 select HAVE_PAGE_SIZE_4KB select IRQ_DOMAIN select LOCK_MM_AND_FIND_VMA - select MMU_GATHER_RCU_TABLE_FREE select MODULES_USE_ELF_RELA select OF select OF_EARLY_FLATTREE diff --git a/arch/openrisc/Kconfig b/arch/openrisc/Kconfig index d90b24dd3bce3c..5eb995c13074c0 100644 --- a/arch/openrisc/Kconfig +++ b/arch/openrisc/Kconfig @@ -35,7 +35,6 @@ config OPENRISC select GENERIC_ATOMIC64 select GENERIC_CLOCKEVENTS_BROADCAST select GENERIC_SMP_IDLE_THREAD - select MMU_GATHER_RCU_TABLE_FREE select MODULES_USE_ELF_RELA select HAVE_DEBUG_STACKOVERFLOW select OR1K_PIC diff --git a/arch/parisc/Kconfig b/arch/parisc/Kconfig index d3afac2f0d9be9..77f67028ad89ce 100644 --- a/arch/parisc/Kconfig +++ b/arch/parisc/Kconfig @@ -80,7 +80,6 @@ config PARISC select GENERIC_CLOCKEVENTS select CPU_NO_EFFICIENT_FFS select THREAD_INFO_IN_TASK - select MMU_GATHER_RCU_TABLE_FREE select NEED_DMA_MAP_STATE select NEED_SG_DMA_LENGTH select HAVE_ARCH_KGDB diff --git a/arch/powerpc/Kconfig b/arch/powerpc/Kconfig index 2580e27e432874..0767cfcbaa422b 100644 --- a/arch/powerpc/Kconfig +++ b/arch/powerpc/Kconfig @@ -307,7 +307,6 @@ config PPC select KASAN_VMALLOC if KASAN && EXECMEM select LOCK_MM_AND_FIND_VMA select MMU_GATHER_PAGE_SIZE - select MMU_GATHER_RCU_TABLE_FREE select HAVE_ARCH_TLB_REMOVE_TABLE select MMU_GATHER_MERGE_VMAS select MMU_LAZY_TLB_SHOOTDOWN if PPC_BOOK3S_64 diff --git a/arch/riscv/Kconfig b/arch/riscv/Kconfig index c13ef9cd688d33..0db108ea146626 100644 --- a/arch/riscv/Kconfig +++ b/arch/riscv/Kconfig @@ -208,7 +208,6 @@ config RISCV select IRQ_FORCED_THREADING select KASAN_VMALLOC if KASAN select LOCK_MM_AND_FIND_VMA - select MMU_GATHER_RCU_TABLE_FREE if MMU select MODULES_USE_ELF_RELA if MODULES select OF select OF_EARLY_FLATTREE diff --git a/arch/s390/Kconfig b/arch/s390/Kconfig index b88b8504213692..a34376c05f6e3c 100644 --- a/arch/s390/Kconfig +++ b/arch/s390/Kconfig @@ -267,7 +267,6 @@ config S390 select LOCK_MM_AND_FIND_VMA select MMU_GATHER_MERGE_VMAS select MMU_GATHER_NO_GATHER - select MMU_GATHER_RCU_TABLE_FREE select MODULES_USE_ELF_RELA select NEED_DMA_MAP_STATE if PCI select NEED_PER_CPU_EMBED_FIRST_CHUNK diff --git a/arch/sh/Kconfig b/arch/sh/Kconfig index 75236bef6f16ee..fe859def918cc3 100644 --- a/arch/sh/Kconfig +++ b/arch/sh/Kconfig @@ -62,7 +62,6 @@ config SUPERH select HAVE_SYSCALL_TRACEPOINTS select IRQ_FORCED_THREADING select LOCK_MM_AND_FIND_VMA - select MMU_GATHER_RCU_TABLE_FREE if MMU select MODULES_USE_ELF_RELA select NEED_SG_DMA_LENGTH select NO_DMA if !MMU && !DMA_COHERENT diff --git a/arch/sparc/Kconfig b/arch/sparc/Kconfig index 79c09d6ee466c4..742ffff8c37f21 100644 --- a/arch/sparc/Kconfig +++ b/arch/sparc/Kconfig @@ -64,7 +64,6 @@ config SPARC32 select HAVE_UID16 select HAVE_PAGE_SIZE_4KB select LOCK_MM_AND_FIND_VMA - select MMU_GATHER_RCU_TABLE_FREE select HAVE_ARCH_TLB_REMOVE_TABLE select OLD_SIGACTION select ZONE_DMA @@ -77,7 +76,6 @@ config SPARC64 select HAVE_FUNCTION_GRAPH_TRACER select HAVE_KRETPROBES select HAVE_KPROBES - select MMU_GATHER_RCU_TABLE_FREE select HAVE_ARCH_TLB_REMOVE_TABLE select MMU_GATHER_MERGE_VMAS select MMU_GATHER_NO_FLUSH_CACHE diff --git a/arch/sparc/include/asm/tlb_64.h b/arch/sparc/include/asm/tlb_64.h index 3037187482db7e..f5f9631685d505 100644 --- a/arch/sparc/include/asm/tlb_64.h +++ b/arch/sparc/include/asm/tlb_64.h @@ -29,9 +29,7 @@ void flush_tlb_pending(void); * and therefore we don't need a TLBI when freeing page-table pages. */ -#ifdef CONFIG_MMU_GATHER_RCU_TABLE_FREE #define tlb_needs_table_invalidate() (false) -#endif #include diff --git a/arch/um/Kconfig b/arch/um/Kconfig index d9541d13d9eb06..94b8ff70f578b5 100644 --- a/arch/um/Kconfig +++ b/arch/um/Kconfig @@ -44,7 +44,6 @@ config UML select HAVE_SYSCALL_TRACEPOINTS select THREAD_INFO_IN_TASK select SPARSE_IRQ - select MMU_GATHER_RCU_TABLE_FREE config MMU bool diff --git a/arch/x86/Kconfig b/arch/x86/Kconfig index a8c3b3d31a2761..6e5e462ec059a1 100644 --- a/arch/x86/Kconfig +++ b/arch/x86/Kconfig @@ -283,7 +283,6 @@ config X86 select HAVE_PERF_REGS select HAVE_PERF_USER_STACK_DUMP select ASYNC_KERNEL_PGTABLE_FREE if IOMMU_SVA - select MMU_GATHER_RCU_TABLE_FREE select MMU_GATHER_MERGE_VMAS select HAVE_POSIX_CPU_TIMERS_TASK_WORK select HAVE_REGS_AND_STACK_ACCESS_API diff --git a/arch/xtensa/Kconfig b/arch/xtensa/Kconfig index 33c4caee30e27b..f2f9cd9cde505d 100644 --- a/arch/xtensa/Kconfig +++ b/arch/xtensa/Kconfig @@ -55,7 +55,6 @@ config XTENSA select HAVE_VIRT_CPU_ACCOUNTING_GEN select IRQ_DOMAIN select LOCK_MM_AND_FIND_VMA - select MMU_GATHER_RCU_TABLE_FREE if MMU select MODULES_USE_ELF_RELA select PERF_USE_VMALLOC select TRACE_IRQFLAGS_SUPPORT diff --git a/include/asm-generic/tlb.h b/include/asm-generic/tlb.h index bdcc2778ac64f4..9d827076db1969 100644 --- a/include/asm-generic/tlb.h +++ b/include/asm-generic/tlb.h @@ -67,11 +67,8 @@ * - tlb_remove_table() * * tlb_remove_table() is the basic primitive to free page-table directories - * (__p*_free_tlb()). In it's most primitive form it is an alias for - * tlb_remove_page() below, for when page directories are pages and have no - * additional constraints. - * - * See also MMU_GATHER_TABLE_FREE and MMU_GATHER_RCU_TABLE_FREE. + * (__p*_free_tlb()). Page directories are freed after an RCU grace + * period - see the comment in mm/mmu_gather.c. * * - tlb_remove_page() / tlb_remove_page_size() * - __tlb_remove_folio_pages() / __tlb_remove_page_size() @@ -151,24 +148,15 @@ * This might be useful if your architecture has size specific TLB * invalidation instructions. * - * MMU_GATHER_TABLE_FREE - * - * This provides tlb_remove_table(), to be used instead of tlb_remove_page() - * for page directores (__p*_free_tlb()). - * - * Useful if your architecture has non-page page directories. + * Page directories (__p*_free_tlb()) are always freed via tlb_remove_table(), + * after an RCU grace period (see mm/mmu_gather.c). * - * When used, an architecture is expected to provide __tlb_remove_table() or - * use the generic __tlb_remove_table(), which does the actual freeing of these - * pages. + * This serialises against software page-table walkers, including architectures + * which do not use IPIs for remote TLB invalidates. * - * MMU_GATHER_RCU_TABLE_FREE - * - * Like MMU_GATHER_TABLE_FREE, and adds semi-RCU semantics to the free (see - * comment below). - * - * Useful if your architecture doesn't use IPIs for remote TLB invalidates - * and therefore doesn't naturally serialize with software page-table walkers. + * An architecture is expected to provide __tlb_remove_table() (see + * HAVE_ARCH_TLB_REMOVE_TABLE) or use the generic __tlb_remove_table(), which + * does the actual freeing of these pages. * * MMU_GATHER_NO_FLUSH_CACHE * @@ -200,12 +188,8 @@ * various ptep_get_and_clear() functions. */ -#ifdef CONFIG_MMU_GATHER_TABLE_FREE - struct mmu_table_batch { -#ifdef CONFIG_MMU_GATHER_RCU_TABLE_FREE struct rcu_head rcu; -#endif unsigned int nr; void *tables[]; }; @@ -224,23 +208,6 @@ static inline void __tlb_remove_table(void *table) extern void tlb_remove_table(struct mmu_gather *tlb, void *table); -#else /* !CONFIG_MMU_GATHER_TABLE_FREE */ - -static inline void tlb_remove_page(struct mmu_gather *tlb, struct page *page); -/* - * Without MMU_GATHER_TABLE_FREE the architecture is assumed to have page based - * page directories and we can use the normal page batching to free them. - */ -static inline void tlb_remove_table(struct mmu_gather *tlb, void *table) -{ - struct ptdesc *ptdesc = (struct ptdesc *)table; - - pagetable_dtor(ptdesc); - tlb_remove_page(tlb, ptdesc_page(ptdesc)); -} -#endif /* CONFIG_MMU_GATHER_TABLE_FREE */ - -#ifdef CONFIG_MMU_GATHER_RCU_TABLE_FREE /* * This allows an architecture that does not use the linux page-tables for * hardware to skip the TLBI when freeing page tables. @@ -253,19 +220,6 @@ void tlb_remove_table_sync_one(void); void tlb_remove_table_sync_rcu(void); -#else - -#ifdef tlb_needs_table_invalidate -#error tlb_needs_table_invalidate() requires MMU_GATHER_RCU_TABLE_FREE -#endif - -static inline void tlb_remove_table_sync_one(void) { } - -static inline void tlb_remove_table_sync_rcu(void) { } - -#endif /* CONFIG_MMU_GATHER_RCU_TABLE_FREE */ - - #ifndef CONFIG_MMU_GATHER_NO_GATHER /* * If we can't allocate a page to make a big batch of page pointers @@ -325,9 +279,7 @@ static inline void tlb_flush_rmaps(struct mmu_gather *tlb, struct vm_area_struct struct mmu_gather { struct mm_struct *mm; -#ifdef CONFIG_MMU_GATHER_TABLE_FREE struct mmu_table_batch *batch; -#endif unsigned long start; unsigned long end; diff --git a/mm/Kconfig b/mm/Kconfig index c1ddf59c0d71a8..bc7befafb47b57 100644 --- a/mm/Kconfig +++ b/mm/Kconfig @@ -1465,7 +1465,7 @@ config HAVE_ARCH_TLB_REMOVE_TABLE config PT_RECLAIM def_bool y - depends on MMU_GATHER_RCU_TABLE_FREE && !HAVE_ARCH_TLB_REMOVE_TABLE + depends on MMU && !HAVE_ARCH_TLB_REMOVE_TABLE help Try to reclaim empty user page table pages in paths other than munmap and exit_mmap path. diff --git a/mm/gup.c b/mm/gup.c index a4036c02e2137f..c2dfcb4744bc3f 100644 --- a/mm/gup.c +++ b/mm/gup.c @@ -2700,8 +2700,9 @@ EXPORT_SYMBOL(get_user_pages_unlocked); * Before activating this code, please be aware that the following assumptions * are currently made: * - * *) Either MMU_GATHER_RCU_TABLE_FREE is enabled, and tlb_remove_table() is used to - * free pages containing page tables or TLB flushing requires IPI broadcast. + * *) tlb_remove_table() is used to free pages containing page tables, with + * the free deferred until an RCU grace period has elapsed (see + * mm/mmu_gather.c). * * *) ptes can be read atomically by the architecture. * diff --git a/mm/mmu_gather.c b/mm/mmu_gather.c index 3985d856de7f9b..2a72a9686773a3 100644 --- a/mm/mmu_gather.c +++ b/mm/mmu_gather.c @@ -218,8 +218,6 @@ bool __tlb_remove_page_size(struct mmu_gather *tlb, struct page *page, int page_ #endif /* MMU_GATHER_NO_GATHER */ -#ifdef CONFIG_MMU_GATHER_TABLE_FREE - static void __tlb_remove_table_free(struct mmu_table_batch *batch) { int i; @@ -230,10 +228,8 @@ static void __tlb_remove_table_free(struct mmu_table_batch *batch) free_page((unsigned long)batch); } -#ifdef CONFIG_MMU_GATHER_RCU_TABLE_FREE - /* - * Semi RCU freeing of the page directories. + * RCU freeing of the page directories. * * This is needed by some architectures to implement software pagetable walkers. * @@ -259,13 +255,13 @@ static void __tlb_remove_table_free(struct mmu_table_batch *batch) * means. * * What we do is batch the freed directory pages (tables) and RCU free them. - * We use the sched RCU variant, as that guarantees that IRQ/preempt disabling - * holds off grace periods. + * Disabling IRQs or preemption holds off RCU grace periods, so this protects + * both rcu_read_lock() and IRQ-disabling walkers. * * However, in order to batch these pages we need to allocate storage, this * allocation is deep inside the MM code and can thus easily fail on memory - * pressure. To guarantee progress we fall back to single table freeing, see - * the implementation of tlb_remove_table_one(). + * pressure. To guarantee progress we fall back to single table freeing, which + * is also RCU-deferred - see the implementation of tlb_remove_table_one(). * */ @@ -315,15 +311,6 @@ void tlb_remove_table_sync_rcu(void) synchronize_rcu(); } -#else /* !CONFIG_MMU_GATHER_RCU_TABLE_FREE */ - -static void tlb_remove_table_free(struct mmu_table_batch *batch) -{ - __tlb_remove_table_free(batch); -} - -#endif /* CONFIG_MMU_GATHER_RCU_TABLE_FREE */ - /* * If we want tlb_remove_table() to imply TLB invalidates. */ @@ -403,13 +390,6 @@ static inline void tlb_table_init(struct mmu_gather *tlb) tlb->batch = NULL; } -#else /* !CONFIG_MMU_GATHER_TABLE_FREE */ - -static inline void tlb_table_flush(struct mmu_gather *tlb) { } -static inline void tlb_table_init(struct mmu_gather *tlb) { } - -#endif /* CONFIG_MMU_GATHER_TABLE_FREE */ - static void tlb_flush_mmu_free(struct mmu_gather *tlb) { tlb_table_flush(tlb); From f898f5559c093885e15609e1855e5a5486f54266 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 25 Sep 2026 21:09:47 +0100 Subject: [PATCH 0840/1352] mm: change the contract for free_pgtables(), update docs Now that page tables are freed after an RCU grace period, it is safe for read-only page table walkers to walk page table ranges that are being concurrently torn down, provided the mm is kept alive via mmgrab(). It is however unsafe for writers to do so, as they must obtain an appropriate lock to do so safely. Update the pte_offset_map_lock()'s comment block to reflect this. Similarly update the process addresses documentation. Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-12-31e91065fea4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Kiryl Shutsemau (Meta) Acked-by: David Hildenbrand (Arm) Cc: Zi Yan Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Guo Ren Cc: Brian Cain Cc: Geert Uytterhoeven Cc: Dinh Nguyen Cc: Simon Schuster Cc: Jonas Bonn Cc: Stefan Kristiansson Cc: Stafford Horne Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti Cc: Russell King Cc: Vineet Gupta Cc: Michal Simek Cc: Chris Zankel Cc: Max Filippov Cc: Will Deacon Cc: Aneesh Kumar K.V Cc: Nicholas Piggin Cc: Peter Zijlstra Cc: David S. Miller Cc: Andreas Larsson Cc: Richard Henderson Cc: Matt Turner Cc: Magnus Lindholm Cc: Catalin Marinas Cc: Mark Rutland Cc: Huacai Chen Cc: WANG Xuerui Cc: Thomas Bogendoerfer Cc: James Bottomley Cc: Helge Deller Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Christophe Leroy Cc: Heiko Carstens Cc: Vasily Gorbik Cc: Alexander Gordeev Cc: Christian Borntraeger Cc: Sven Schnelle Cc: Richard Weinberger Cc: Anton Ivanov Cc: Johannes Berg Cc: Thomas Gleixner Cc: Ingo Molnar Cc: Borislav Petkov Cc: Dave Hansen Cc: H. Peter Anvin Cc: Arnd Bergmann Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Cc: Yoshinori Sato Cc: Shakeel Butt Cc: Jonathan Corbet Cc: Randy Dunlap Cc: Hugh Dickins Cc: Qi Zheng --- Documentation/mm/process_addrs.rst | 11 +++++++++++ mm/pgtable-generic.c | 15 +++++++++++---- 2 files changed, 22 insertions(+), 4 deletions(-) diff --git a/Documentation/mm/process_addrs.rst b/Documentation/mm/process_addrs.rst index a7296f251799cb..78231995e49012 100644 --- a/Documentation/mm/process_addrs.rst +++ b/Documentation/mm/process_addrs.rst @@ -537,6 +537,17 @@ We establish basic locking rules when interacting with page tables: * When changing a page table entry the page table lock for that page table **must** be held, except if you can safely assume nobody can access the page tables concurrently (such as on invocation of :c:func:`!free_pgtables`). +* Page tables may be *walked* under RCU alone, as page tables are freed only + after an RCU grace period has elapsed. However, any entry found must be + revalidated after the page table lock is taken (such as the + :c:func:`!pmd_same` recheck performed by :c:func:`!pte_offset_map_lock`) + before it is acted upon. Changing an entry requires the page table lock + and one of the locks that excludes teardown (any one of the mmap, VMA or + rmap locks). +* When traversing page tables under RCU alone it is important to take care + when operating upon leaf entries - if the value is operated upon (for + instance getting the folio associated with a PTE) an appropriate lock must + be taken to prevent concurrent modification. * Reads from and writes to page table entries must be *appropriately* atomic. See the section on atomicity below for details. * Populating previously empty entries requires that the mmap or VMA locks are diff --git a/mm/pgtable-generic.c b/mm/pgtable-generic.c index b45e891d1193fd..26643d76bfb00e 100644 --- a/mm/pgtable-generic.c +++ b/mm/pgtable-generic.c @@ -397,10 +397,17 @@ pte_t *pte_offset_map_rw_nolock(struct mm_struct *mm, pmd_t *pmd, * Note: "RO" / "RW" expresses the intended semantics, not that the *kmap* will * be read-only/read-write protected. * - * Note that free_pgtables(), used after unmapping detached vmas, or when - * exiting the whole mm, does not take page table lock before freeing a page - * table, and may not use RCU at all: "outsiders" like khugepaged should avoid - * pte_offset_map() and co once the vma is detached from mm or mm_users is zero. + * Note that free_pgtables(), used after unmapping detached vmas or when exiting + * the whole mm, does not take a page table lock before freeing a page table. + * + * As page table freeing itself is RCU-safe, page table readers can safely run + * concurrently with page table teardown. + * + * However, writers CANNOT as, without a lock being held, nothing prevents + * concurrent teardown. + * + * Also note that the PGD itself is freed at mmdrop() time, not under RCU - so + * the walker must keep the mm alive either by pinning the mm or the VMA. */ pte_t *pte_offset_map_lock(struct mm_struct *mm, pmd_t *pmd, unsigned long addr, spinlock_t **ptlp) From b2ccead9cdae1d0559acb2b9b29292568a2bb4ca Mon Sep 17 00:00:00 2001 From: Bo Zhang Date: Tue, 8 Sep 2026 14:26:49 +0800 Subject: [PATCH 0841/1352] mm: vmscan: avoid anon scanning for GFP_NOIO with low swapcache We have observed some cases where memory is allocated with GFP_NOIO, so we cannot reclaim any anon folios unless they are in swapcache. We can end up spending more than 150 ms looping in `shrink_folio_list()` scanning non-swapcache folios without reclaiming a single folio. This is pure overhead. This is particularly true on systems using zRAM, where swapcache is relatively rare. So let's check whether anon reclaim is allowed by GFP_IO and whether there is enough swapcache to make it worthwhile. If the swapcache is extremely low, we're essentially searching for a needle in a haystack, so let's avoid scanning anon in the first place. On Android this is triggered by dm-verity hash-block reads through dm-bufio, which legitimately use GFP_NOIO because they run underneath the IO path: verity_verify_io -> verity_hash_for_block -> verity_verify_level -> dm_bufio_read_with_ioprio -> new_read -> __bufio_new -> alloc_buffer gfp: GFP_NOIO | __GFP_NORETRY | __GFP_NOMEMALLOC | __GFP_NOWARN Such a reclaimer can land on a memcg with a large, unswapped anon LRU and a tiny file LRU (e.g. inactive_anon ~335 MB vs inactive_file ~4 MB, with negligible swapcache). shrink_lruvec() then keeps feeding that huge anon list into shrink_folio_list() - ~2400 shrink_folio_list() calls, ~93,000 anon folios scanned - where every folio is kept because it needs IO. The 150+ ms above is one such single shrink_lruvec() pass (not accumulated across a reclaim cycle), and it reclaims nothing; the actual progress comes entirely from the file side. Aging anon alongside file does have some value for a later __GFP_IO reclaimer, so it is not strictly pure overhead. But that aging is only deferred, not lost: kswapd and other __GFP_IO reclaimers still walk and age anon. Spending ~168 ms aging memory that this context cannot reclaim is not a worthwhile trade-off in a latency-sensitive path. To stay conservative, this only skips anon when the swapcache is really tiny - below 1/64 of the anon LRU - i.e. when essentially no anon on the list can be reclaimed without IO. Whenever there is a meaningful amount of swapcached anon, the normal path is used and anon is scanned and aged as before. Note this only addresses the traditional active/inactive LRU. MGLRU selects anon vs file scanning in its own path and is not covered here; fixing the MGLRU case is left as a TODO. Link: https://lore.kernel.org/20260908062649.1045883-1-zhangbo56@xiaomi.com Signed-off-by: Bo Zhang Signed-off-by: Andrew Morton Reviewed-by: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Shakeel Butt --- mm/vmscan.c | 62 ++++++++++++++++++++++++++++++++++++++++++++++++----- 1 file changed, 57 insertions(+), 5 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 4059d130078960..8cbf562e625203 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -362,20 +362,72 @@ static bool can_demote(int nid, struct scan_control *sc, return !nodes_empty(allowed_mask); } +#ifdef CONFIG_SWAP +static inline bool reclaimable_anon_is_low(struct mem_cgroup *memcg, + int nid, struct scan_control *sc) +{ + pg_data_t *pgdat = NODE_DATA(nid); + unsigned long anon_pages, swapcache; + + /* + * A GFP_NOIO reclaimer can only reclaim anon that is already in the + * swapcache (adding anon to the swapcache needs IO). When swapcache is + * far below the anon LRU, scanning anon reclaims nothing and only burns + * CPU. The 1/64 threshold keeps this to the case where anon is + * effectively unreclaimable. + */ + if (!sc || (sc->gfp_mask & __GFP_IO)) + return false; + + /* + * FIXME: MGLRU doesn't fully respect can_reclaim_anon_pages() for the + * scanning type, so only apply this to the traditional LRU for now. + */ + if (lru_gen_enabled()) + return false; + + if (memcg) { + struct lruvec *lruvec = mem_cgroup_lruvec(memcg, pgdat); + + anon_pages = lruvec_page_state(lruvec, NR_INACTIVE_ANON) + + lruvec_page_state(lruvec, NR_ACTIVE_ANON); + swapcache = lruvec_page_state(lruvec, NR_SWAPCACHE); + } else { + anon_pages = node_page_state(pgdat, NR_INACTIVE_ANON) + + node_page_state(pgdat, NR_ACTIVE_ANON); + swapcache = node_page_state(pgdat, NR_SWAPCACHE); + } + + return swapcache < (anon_pages >> 6); +} +#else +static inline bool reclaimable_anon_is_low(struct mem_cgroup *memcg, + int nid, struct scan_control *sc) +{ + return true; +} +#endif /* CONFIG_SWAP */ + static inline bool can_reclaim_anon_pages(struct mem_cgroup *memcg, int nid, struct scan_control *sc) { if (memcg == NULL) { /* - * For non-memcg reclaim, is there - * space in any swap device? + * For non-memcg reclaim, is there space in any swap device? + * And under GFP_NOIO, is there enough swapcached anon to make + * scanning anon worthwhile? */ - if (get_nr_swap_pages() > 0) + if (get_nr_swap_pages() > 0 && + !reclaimable_anon_is_low(memcg, nid, sc)) return true; } else { - /* Is the memcg below its swap limit? */ - if (mem_cgroup_get_nr_swap_pages(memcg) > 0) + /* + * Is the memcg below its swap limit, and under GFP_NOIO does + * it have enough swapcached anon to make scanning worthwhile? + */ + if (mem_cgroup_get_nr_swap_pages(memcg) > 0 && + !reclaimable_anon_is_low(memcg, nid, sc)) return true; } From c542ea74e4b84831f7162b10d80d207ae55ccefd Mon Sep 17 00:00:00 2001 From: Baolin Wang Date: Mon, 14 Sep 2026 19:29:22 +0800 Subject: [PATCH 0842/1352] mm: mglru: clear the reference counter for rejected folios As per the comment on LRU_REFS_FLAGS, when accessed folios are promoted to a new generation, LRU_REFS_FLAGS should be cleared so that the reference counter can start over. For folios rejected by shrink_folio_list(), we clear LRU_REFS_FLAGS and set the PG_active flag when lru_gen_folio_seq() would place them in the oldest generation. That's fine. But for rejected folios where lru_gen_folio_seq() returns a generation other than the oldest one (which can be treated as a promotion), we do not clear LRU_REFS_FLAGS. This can violate the promotion mechanism. And this means the rejected folio enters the new generation with stale, inflated tier bits, which can inflate reference counts and distort eviction statistics for these rejected folios. Fix this by clearing LRU_REFS_FLAGS for rejected folios. Of course, I need to evaluate the impact of the changes, which mainly falls into 3 cases: 1. When lru_gen_folio_seq() returns the oldest generation for rejected folios, there are no logic changes, and they will be put back into the 2nd youngest generation. 2. For rejected folios with PG_active set by shrink_folio_list(), we only clear the LRU_REFS_FLAGS and do not change the generation. 3. For rejected folios with PG_referenced set, the original code would put them back into the 2nd oldest generation. After this patch, we will put them back into the 2nd youngest generation. I think case 3 is also reasonable, before commmit 6cbdd9726fb5 ("mm/mglru: use folio_mark_accessed to replace folio_set_active"), a rejected referenced folio was also put back to the 2nd youngest gen. Meanwhile, I didn't see any noticeable performance impact on my 32-core Arm machine when running 'make -j32' to build the kernel inside a 3G-limited memcg with either zram or NVMe swap. Link: https://lore.kernel.org/7384df363c12e4acdaa2e0428420cd8eed320ee7.1789384831.git.baolin.wang@linux.alibaba.com Signed-off-by: Baolin Wang Signed-off-by: Andrew Morton Reviewed-by: Baoquan He Reviewed-by: Kairui Song Reviewed-by: Barry Song Cc: Axel Rasmussen Cc: David Hildenbrand Cc: Johannes Weiner Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie --- mm/vmscan.c | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 8cbf562e625203..a49da82ed4797a 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -5075,11 +5075,15 @@ static int evict_folios(unsigned long nr_to_scan, struct lruvec *lruvec, continue; } - /* don't add rejected folios to the oldest generation */ - if (lru_gen_folio_seq(lruvec, folio, false) == min_seq[type]) { - folio_set_lru_refs(folio, 0); + /* + * See the comments on LRU_REFS_FLAGS. + * + * The rejected folios are never added to the oldest generation, + * so this effectively promotes them by at least one generation. + */ + folio_set_lru_refs(folio, 0); + if (lru_gen_folio_seq(lruvec, folio, false) == min_seq[type]) folio_set_active(folio); - } } move_folios_to_lru(&list); From 07734bc296c147908d44e556b55350be5e7c9fb1 Mon Sep 17 00:00:00 2001 From: Anastasios Papagiannis Date: Wed, 9 Sep 2026 09:42:31 +0300 Subject: [PATCH 0843/1352] mm/nommu: reject wrapping ranges in access_remote_vm() The NOMMU implementation of access_process_vm() rejects address ranges whose end wraps around, but access_remote_vm() bypasses this check even though both functions delegate to __access_remote_vm(). Move the wraparound check into __access_remote_vm() so it applies to both entry points. This is originally reported in [1]. Link: https://lore.kernel.org/20260909064231.18693-1-tasos.papagiannnis@gmail.com Fixes: f55f199b7d76 ("NOMMU: implement access_remote_vm") Signed-off-by: Anastasios Papagiannis Signed-off-by: Andrew Morton Closes: https://lore.kernel.org/bpf/4ef240a5bea36ff84df9589671367832860795159386a4c8fba546a0fa8b786f@mail.kernel.org/ [1] Reviewed-by: Lorenzo Stoakes (ARM) Cc: Liam R. Howlett Cc: Hajime Tazaki Cc: --- mm/nommu.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/mm/nommu.c b/mm/nommu.c index 498e01ee40b056..ed44510e37707d 100644 --- a/mm/nommu.c +++ b/mm/nommu.c @@ -1674,6 +1674,9 @@ static int __access_remote_vm(struct mm_struct *mm, unsigned long addr, struct vm_area_struct *vma; int write = gup_flags & FOLL_WRITE; + if (addr + len < addr) + return 0; + if (mmap_read_lock_killable(mm)) return 0; @@ -1727,9 +1730,6 @@ int access_process_vm(struct task_struct *tsk, unsigned long addr, void *buf, in { struct mm_struct *mm; - if (addr + len < addr) - return 0; - mm = get_task_mm(tsk); if (!mm) return 0; From 185633bf5728c03a5e5694b022607f3cdb48d271 Mon Sep 17 00:00:00 2001 From: Youngjun Park Date: Thu, 10 Sep 2026 01:15:51 +0900 Subject: [PATCH 0844/1352] mm/swap: fix stale comment on swap_info_struct::cluster_info Patch series "mm/swap: skip empty clusters in the swapoff scan", v4. Speed up swapoff and reduce scanning stalls on large, mostly empty swap devices by skipping empty clusters. Reduce swapoff time on a 1 TiB device from 158 ms to 94 ms after filling 128 GiB, and from 392 ms to 73 ms after filling 512 GiB; expect little benefit when substantial swap data remains. find_next_to_unuse() walks a swap device one offset at a time. Slot state now lives in a per cluster swap table, so patch 2 dismisses an empty cluster with one counter read instead of SWAPFILE_CLUSTER table reads. Patch 1 is an unrelated one line comment fix noticed on the way. A debug test confirmed the skip path runs, and swapoff completed under load with no DEBUG_VM or lockdep splats. This patch (of 2): setup_swap_clusters_info() allocates cluster_info for every swap area, not only for SSDs. Link: https://lore.kernel.org/20260909161552.2335971-1-youngjun.park@lge.com Link: https://lore.kernel.org/20260909161552.2335971-2-youngjun.park@lge.com Signed-off-by: Youngjun Park Signed-off-by: Andrew Morton Acked-by: Kairui Song Reviewed-by: Barry Song Cc: Baoquan He Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham --- include/linux/swap.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/include/linux/swap.h b/include/linux/swap.h index a37ad8375e011e..61005501888c53 100644 --- a/include/linux/swap.h +++ b/include/linux/swap.h @@ -240,7 +240,7 @@ struct swap_info_struct { struct plist_node list; /* entry in swap_active_head */ signed char type; /* strange name for an index */ unsigned int max; /* size of this swap device */ - struct swap_cluster_info *cluster_info; /* cluster info. Only for SSD */ + struct swap_cluster_info *cluster_info; /* array, one entry per cluster */ struct list_head free_clusters; /* free clusters list */ struct list_head full_clusters; /* full clusters list */ struct list_head nonfull_clusters[SWAP_NR_ORDERS]; From 17b750b24cad0d99a7d5b8def0a8eac018dad49a Mon Sep 17 00:00:00 2001 From: Youngjun Park Date: Thu, 10 Sep 2026 01:15:52 +0900 Subject: [PATCH 0845/1352] mm/swap: scan by cluster in find_next_to_unuse() find_next_to_unuse() walks every offset from 0 to si->max, and swapoff restarts that walk on each retry, so the cost scales with the size of the device rather than with the few slots the shmem and mmlist passes could not free. It has caused stalls before. The flat walk predates the swap table. Slot state now lives in a per cluster table, and wait_for_allocation() stops all allocation before try_to_unuse() runs, so a cluster that holds no slot in use stays that way. Skip such a cluster instead of reading all of its entries. Fill a 1 TiB swap up to some amount, then swapoff. What is left sits at the top of what was filled, so every slot below it is free. Medians over 11 pairs at 32 and 128 GiB, 3 pairs at 256 and 512. filled swapoff old new 32 GiB 92.4ms 66.3ms 128 GiB 157.6ms 94.4ms 256 GiB 209.7ms 63.8ms 512 GiB 391.8ms 73.4ms old grows with how much was filled, new does not. In the ordinary case swap still holds real data and swapoff spends its time reading it back. it is tested 4 GiB on an 8 GiB device, where the scan is 1.4% of try_to_unuse(), and there is no difference either way. Commit dc644a073769 ("mm: add three more cond_resched() in swapoff") answered those stalls with a cond_resched() every 256 offsets. A walk bounded by one cluster no longer needs that counter. The loop now runs at most SWAPFILE_CLUSTER times before it returns or reschedules, the same bound swap_reclaim_full_clusters() already scans between cond_resched() calls. The scan end is clamped to si->max, so the walk stops there rather than running into the masked tail of the last cluster. ci->count is read without ci->lock, so READ_ONCE() marks the read for KCSAN. Allocation is already stopped, so the count can only drop, and a slot stops being counted only after its folio has left the swap cache. An empty cluster therefore holds nothing for try_to_unuse() to act on. Link: https://lore.kernel.org/20260909161552.2335971-3-youngjun.park@lge.com Signed-off-by: Youngjun Park Signed-off-by: Andrew Morton Reviewed-by: Barry Song Acked-by: Kairui Song Reviewed-by: Baoquan He Reviewed-by: Nhat Pham Cc: Chris Li Cc: Kemeng Shi --- mm/swapfile.c | 43 ++++++++++++++++++++++++++++++------------- 1 file changed, 30 insertions(+), 13 deletions(-) diff --git a/mm/swapfile.c b/mm/swapfile.c index 48d3cd40defdb4..a8118f095f4d9a 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -370,8 +370,6 @@ static void discard_swap_cluster(struct swap_info_struct *si, } } -#define LATENCY_LIMIT 256 - static inline bool cluster_is_empty(struct swap_cluster_info *info) { return info->count == 0; @@ -2726,7 +2724,9 @@ static int unuse_mm(struct mm_struct *mm, unsigned int type) static unsigned int find_next_to_unuse(struct swap_info_struct *si, unsigned int prev) { - unsigned int i; + struct swap_cluster_info *ci; + unsigned long i, end; + unsigned int ci_off; unsigned long swp_tb; /* @@ -2735,19 +2735,36 @@ static unsigned int find_next_to_unuse(struct swap_info_struct *si, * hits are okay, and sys_swapoff() has already prevented new * allocations from this area (while holding swap_lock). */ - for (i = prev + 1; i < si->max; i++) { - swp_tb = swap_table_get(__swap_offset_to_cluster(si, i), - i % SWAPFILE_CLUSTER); - if (!swp_tb_is_null(swp_tb) && !swp_tb_is_bad(swp_tb)) - break; - if ((i % LATENCY_LIMIT) == 0) + i = prev + 1; + while (i < si->max) { + ci = __swap_offset_to_cluster(si, i); + end = min_t(unsigned long, + ALIGN_DOWN(i, SWAPFILE_CLUSTER) + SWAPFILE_CLUSTER, + si->max); + + /* + * An empty cluster has no slot in use, so skip it whole. + * A slot is uncounted only after its folio left the swap + * cache, so there is nothing here for try_to_unuse() to act on. + * Count only drops here, so a READ_ONCE() without ci->lock is + * enough, unlike in every other cluster_is_empty() caller. + */ + if (!READ_ONCE(ci->count)) { + i = end; cond_resched(); - } + continue; + } - if (i == si->max) - i = 0; + ci_off = i % SWAPFILE_CLUSTER; + for (; i < end; ci_off++, i++) { + swp_tb = swap_table_get(ci, ci_off); + if (!swp_tb_is_null(swp_tb) && !swp_tb_is_bad(swp_tb)) + return i; + } + cond_resched(); + } - return i; + return 0; } static int try_to_unuse(unsigned int type) From cba4d6e5f7c2aba5f4d6de63ce6c9c46d1ec39da Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 9 Sep 2026 07:04:03 -0700 Subject: [PATCH 0846/1352] mm/damon/vaddr: support prep_probes Patch series "mm/damon/vaddr: support {prep,apply}_probes". DAMON supports data attributes monitoring. However, only the physical address space operation set (paddr) is supporting it. Add the support to the virtual address space operation set (vaddr). Patch 1 adds prep_probes support to vaddr. Patch 2 moves probe filter handling code in paddr.c that can be reused by vaddr to ops-common.c. Patch 3 adds minimum apply_probes support to vaddr. Patch 4 extends the support for hugetlb. Patch 5 extends the support for pgidle_unset filter. Test ==== I confirmed it can capture ~48 mb working set of masim in vaddr mode, like below. First, start masim [1] to access ~48 mb memory at a time, in the background. $ ./masim/masim.py run \ --config_file ./masim/configs/stairs-50mb.cfg \ --repeat 10 --quiet & Note that the config says the working set is 50mb. It is 50 million bytes, so ~48 MiB. Start traditional access monitoring of masim's virtual address space using damo [2]. $ sudo ./damo/damo start $(pidof masim) Confirm it can capture the ~48 MiB working set as the 4-th region on the snapshot. $ sudo ./damo/damo report access heatmap: 11111111334[...]3000000000000000000000000000000000000000489999997777777743333333555556[...]8 # min/max temperatures: -1,150,000,000, 80,009,499, column size: 6.963 MiB intervals: sample 5 ms aggr 100 ms (max access hz 200) 0 addr 85.355 TiB size 55.703 MiB access 0 hz age 9.500 s 1 addr 85.355 TiB size 18.984 MiB access 0 hz age 6.700 s 2 addr 127.183 TiB size 278.516 MiB access 0 hz age 11.500 s 3 addr 127.183 TiB size 7.570 MiB access 0 hz age 900 ms 4 addr 127.183 TiB size 48.133 MiB access 190 hz age 800 ms 5 addr 127.183 TiB size 54.977 MiB access 0 hz age 1.700 s 6 addr 127.183 TiB size 55.113 MiB access 0 hz age 6.700 s 7 addr 127.183 TiB size 37.902 MiB access 0 hz age 3.900 s 8 addr 127.990 TiB size 120.000 KiB access 0 hz age 11.400 s 9 addr 127.990 TiB size 8.000 KiB access 70 hz age 0 ns 10 addr 127.990 TiB size 4.000 KiB access 0 hz age 11.200 s memory bw estimate: 8.931 GiB per second total size: 557.027 MiB record DAMON intervals: sample 5 ms, aggr 100 ms Stop access monitoring and start probe-only mode access monitoring. $ sudo ./damo/damo stop $ sudo ./damo/damo start $(pidof masim) --probe_prep set_pgidle \ --probe_filter allow pgidle_unset --probe_weight 1 Confirm it can also capture the ~48 MiB working set as the 11-th region on the snapshot. $ sudo ./damo/damo report attrs heatmap: 00000000113[...]40000000000000000000000000000000000000000000000001111114999999533333336[...]6 # min/max temperatures: -840,000,000, 330,002,000, column size: 6.961 MiB probe prep: set_pgidle, filter: allow pgidle_unset (weight: 1) intervals: sample 5 ms aggr 100 ms (max probe hits 20) # size address age probe_hits 0 8.000 KiB 127.990 TiB 8.700 s 0 1 120.000 KiB 127.990 TiB 8.600 s 0 2 55.543 MiB 127.183 TiB 8.400 s 0 3 110.008 MiB 127.183 TiB 8.300 s 0 4 55.352 MiB 127.183 TiB 8.100 s 0 5 55.605 MiB 127.183 TiB 8 s 0 6 55.691 MiB 85.355 TiB 7.900 s 0 7 54.430 MiB 127.183 TiB 7.900 s 0 8 50.391 MiB 127.183 TiB 6.900 s 0 9 18.879 MiB 85.355 TiB 6 s 0 10 52.844 MiB 127.183 TiB 3.500 s 0 11 48.039 MiB 127.183 TiB 3.300 s 20 12 4.000 KiB 127.990 TiB 8.500 s 20 memory bw estimate: 0 B per second total size: 556.910 MiB record DAMON intervals: sample 5 ms, aggr 100 ms This patch (of 5): DAMON virtual address space operation set (vaddr) is not supporting prep_probes. Add the support. Link: https://lore.kernel.org/20260909140408.104699-2-sj@kernel.org Link: https://github.com/sjp38/masim [1] Link: https://github.com/damonitor/damo [2] Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Gutierrez Asier --- mm/damon/vaddr.c | 40 ++++++++++++++++++++++++++++++++++++++++ 1 file changed, 40 insertions(+) diff --git a/mm/damon/vaddr.c b/mm/damon/vaddr.c index af9e1b82454cc2..20f4784f178173 100644 --- a/mm/damon/vaddr.c +++ b/mm/damon/vaddr.c @@ -483,6 +483,45 @@ static unsigned int damon_va_check_accesses(struct damon_ctx *ctx) return max_nr_accesses; } +static void damon_va_prep_probe_region(struct damon_ctx *ctx, + struct mm_struct *mm, struct damon_region *r, + struct damon_probe *probe) +{ + struct damon_prep *p; + + damon_for_each_prep(p, probe) { + switch (p->action) { + case DAMON_PREP_SET_PGIDLE: + damon_va_mkold(mm, r->sampling_addr); + break; + default: + break; + } + } +} + +static void damon_va_prep_probes(struct damon_ctx *ctx, bool set_samples) +{ + struct damon_target *t; + struct mm_struct *mm; + struct damon_region *r; + struct damon_probe *p; + + damon_for_each_target(t, ctx) { + mm = damon_get_mm(t); + if (!mm) + continue; + damon_for_each_region(r, t) { + if (set_samples) + r->sampling_addr = damon_rand(ctx, r->ar.start, + r->ar.end); + damon_for_each_probe(p, ctx) + damon_va_prep_probe_region(ctx, mm, r, p); + } + mmput(mm); + } +} + static bool damos_va_filter_young_match(struct damos_filter *filter, struct folio *folio, struct vm_area_struct *vma, unsigned long addr, pte_t *ptep, pmd_t *pmdp) @@ -911,6 +950,7 @@ static int __init damon_va_initcall(void) .update = damon_va_update, .prepare_access_checks = damon_va_prepare_access_checks, .check_accesses = damon_va_check_accesses, + .prep_probes = damon_va_prep_probes, .target_valid = damon_va_target_valid, .cleanup_target = damon_va_cleanup_target, .apply_scheme = damon_va_apply_scheme, From 139c37814c7b486928410fd7288e9f68ab46ee22 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 9 Sep 2026 07:04:04 -0700 Subject: [PATCH 0847/1352] mm/damon/paddr: move probe filter handling to ops-common Probe filter matching logic for anon and memcg type filters in DAMON's physical address space operation set (paddr) can be reused by virtual address space operation set (vaddr) in future. Prepare the future by moving the code to ops-common.c Link: https://lore.kernel.org/20260909140408.104699-3-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Gutierrez Asier --- mm/damon/ops-common.c | 32 ++++++++++++++++++++++++++++++++ mm/damon/ops-common.h | 2 ++ mm/damon/paddr.c | 23 +---------------------- 3 files changed, 35 insertions(+), 22 deletions(-) diff --git a/mm/damon/ops-common.c b/mm/damon/ops-common.c index acf8f216c51cc8..c36cc39cd2c707 100644 --- a/mm/damon/ops-common.c +++ b/mm/damon/ops-common.c @@ -531,3 +531,35 @@ bool damos_ops_has_filter(struct damos *s) return true; return false; } + +bool damon_ops_filter_match(struct damon_filter *filter, struct folio *folio) +{ + bool matched = false; + struct mem_cgroup *memcg; + + switch (filter->type) { + case DAMON_FILTER_TYPE_ANON: + if (!folio) { + matched = false; + break; + } + matched = folio_test_anon(folio); + break; + case DAMON_FILTER_TYPE_MEMCG: + if (!folio) { + matched = false; + break; + } + rcu_read_lock(); + memcg = folio_memcg_check(folio); + if (!memcg) + matched = false; + else + matched = filter->memcg_id == mem_cgroup_id(memcg); + rcu_read_unlock(); + break; + default: + break; + } + return matched == filter->matching; +} diff --git a/mm/damon/ops-common.h b/mm/damon/ops-common.h index 172f0f17c4a84d..ee27058251005b 100644 --- a/mm/damon/ops-common.h +++ b/mm/damon/ops-common.h @@ -31,3 +31,5 @@ bool damos_folio_filter_match(struct damos_filter *filter, struct folio *folio); unsigned long damon_migrate_pages(struct list_head *folio_list, int target_nid); bool damos_ops_has_filter(struct damos *s); + +bool damon_ops_filter_match(struct damon_filter *filter, struct folio *folio); diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c index d2173a448d0b0c..d7c81829445ba1 100644 --- a/mm/damon/paddr.c +++ b/mm/damon/paddr.c @@ -143,29 +143,8 @@ static bool damon_pa_filter_match(struct damon_filter *filter, struct folio *folio) { bool matched = false; - struct mem_cgroup *memcg; switch (filter->type) { - case DAMON_FILTER_TYPE_ANON: - if (!folio) { - matched = false; - break; - } - matched = folio_test_anon(folio); - break; - case DAMON_FILTER_TYPE_MEMCG: - if (!folio) { - matched = false; - break; - } - rcu_read_lock(); - memcg = folio_memcg_check(folio); - if (!memcg) - matched = false; - else - matched = filter->memcg_id == mem_cgroup_id(memcg); - rcu_read_unlock(); - break; case DAMON_FILTER_TYPE_PGIDLE_UNSET: if (!folio) matched = false; @@ -173,7 +152,7 @@ static bool damon_pa_filter_match(struct damon_filter *filter, matched = damon_folio_young(folio); break; default: - break; + return damon_ops_filter_match(filter, folio); } return matched == filter->matching; } From 320366851c6913d6a85c6032043db86f0f7e3ac0 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 9 Sep 2026 07:04:05 -0700 Subject: [PATCH 0848/1352] mm/damon/vaddr: support apply_probe DAMON virtual address space operation set (vaddr) is not supporting apply_probe. Add a minimum support. Do not support hugetlb pages and PGIDLE_UNSET filter type for simplicity. Link: https://lore.kernel.org/20260909140408.104699-4-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Gutierrez Asier --- mm/damon/vaddr.c | 123 +++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 123 insertions(+) diff --git a/mm/damon/vaddr.c b/mm/damon/vaddr.c index 20f4784f178173..7c548ec0cf6b5b 100644 --- a/mm/damon/vaddr.c +++ b/mm/damon/vaddr.c @@ -522,6 +522,128 @@ static void damon_va_prep_probes(struct damon_ctx *ctx, bool set_samples) } } +static bool damon_va_filter_pass(struct folio *folio, struct damon_probe *p) +{ + struct damon_filter *f; + bool pass = true; + + damon_for_each_filter(f, p) { + if (damon_ops_filter_match(f, folio)) { + pass = f->allow; + break; + } + pass = !f->allow; + } + return pass; +} + +struct damon_va_probe_walk_private { + struct damon_ctx *ctx; + struct damon_region *r; +}; + +static void damon_va_probe_folio(struct damon_ctx *ctx, + struct damon_region *r, struct folio *folio) +{ + struct damon_probe *probe; + int i = 0; + + damon_for_each_probe(probe, ctx) { + if (damon_va_filter_pass(folio, probe)) + r->probe_hits[i]++; + i++; + } +} + +static int damon_va_probe_pmd_entry(pmd_t *pmd, unsigned long addr, + unsigned long next, struct mm_walk *walk) +{ + pte_t *pte; + pte_t ptent; + spinlock_t *ptl; + struct folio *folio; + struct damon_va_probe_walk_private *priv = walk->private; + +#ifdef CONFIG_TRANSPARENT_HUGEPAGE + ptl = pmd_trans_huge_lock(pmd, walk->vma); + if (ptl) { + pmd_t pmde = pmdp_get(pmd); + + if (!pmd_present(pmde)) + goto huge_out; + folio = vm_normal_folio_pmd(walk->vma, addr, pmde); + if (!folio) + goto huge_out; + damon_va_probe_folio(priv->ctx, priv->r, folio); + +huge_out: + spin_unlock(ptl); + return 0; + } +#endif /* CONFIG_TRANSPARENT_HUGEPAGE */ + + pte = pte_offset_map_lock(walk->mm, pmd, addr, &ptl); + if (!pte) + return 0; + ptent = ptep_get(pte); + if (!pte_present(ptent)) + goto out; + folio = vm_normal_folio(walk->vma, addr, ptent); + if (!folio) + goto out; + damon_va_probe_folio(priv->ctx, priv->r, folio); + +out: + pte_unmap_unlock(pte, ptl); + return 0; +} + +static void __damon_va_apply_probes(struct damon_ctx *ctx, + struct mm_struct *mm, struct damon_region *r) +{ + struct damon_va_probe_walk_private arg = { + .ctx = ctx, + .r = r, + }; + struct mm_walk_ops damon_probe_walk_ops = { + .pmd_entry = damon_va_probe_pmd_entry, + .hugetlb_entry = NULL, + }; + unsigned long addr = r->sampling_addr; + + if (!mm) + return; + + damon_va_walk_page_range(mm, addr, addr + 1, &damon_probe_walk_ops, + &arg); +} + +static unsigned int damon_va_apply_probes(struct damon_ctx *ctx, + bool set_samples, bool return_max_wsum) +{ + struct damon_target *t; + struct mm_struct *mm; + struct damon_region *r; + unsigned int max_wsum = 0; + + damon_for_each_target(t, ctx) { + mm = damon_get_mm(t); + damon_for_each_region(r, t) { + if (set_samples) + r->sampling_addr = damon_rand(ctx, r->ar.start, + r->ar.end); + __damon_va_apply_probes(ctx, mm, r); + if (return_max_wsum) + max_wsum = max(damon_probe_hits_wsum(r, false, + ctx), max_wsum); + } + if (mm) + mmput(mm); + } + + return max_wsum; +} + static bool damos_va_filter_young_match(struct damos_filter *filter, struct folio *folio, struct vm_area_struct *vma, unsigned long addr, pte_t *ptep, pmd_t *pmdp) @@ -951,6 +1073,7 @@ static int __init damon_va_initcall(void) .prepare_access_checks = damon_va_prepare_access_checks, .check_accesses = damon_va_check_accesses, .prep_probes = damon_va_prep_probes, + .apply_probes = damon_va_apply_probes, .target_valid = damon_va_target_valid, .cleanup_target = damon_va_cleanup_target, .apply_scheme = damon_va_apply_scheme, From ef79343f0847cdbef5e6957b42fe80ea4251743a Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 9 Sep 2026 07:04:06 -0700 Subject: [PATCH 0849/1352] mm/damon/vaddr: extend apply_probes() for hugetlb DAMON virtual address space operation set(vaddr) does not support hugetlb pages in apply_probes. Extend it for hugetlb pages. Link: https://lore.kernel.org/20260909140408.104699-5-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Gutierrez Asier --- mm/damon/vaddr.c | 30 +++++++++++++++++++++++++++++- 1 file changed, 29 insertions(+), 1 deletion(-) diff --git a/mm/damon/vaddr.c b/mm/damon/vaddr.c index 7c548ec0cf6b5b..45239f05e113b2 100644 --- a/mm/damon/vaddr.c +++ b/mm/damon/vaddr.c @@ -598,6 +598,34 @@ static int damon_va_probe_pmd_entry(pmd_t *pmd, unsigned long addr, return 0; } +#ifdef CONFIG_HUGETLB_PAGE +static int damon_va_probe_hugetlb_entry(pte_t *pte, unsigned long hmask, + unsigned long addr, unsigned long end, struct mm_walk *walk) +{ + struct damon_va_probe_walk_private *priv = walk->private; + struct hstate *h = hstate_vma(walk->vma); + struct folio *folio; + spinlock_t *ptl; + pte_t entry; + + ptl = huge_pte_lock(h, walk->mm, pte); + entry = huge_ptep_get(walk->mm, addr, pte); + if (!pte_present(entry)) + goto out; + + folio = pfn_folio(pte_pfn(entry)); + folio_get(folio); + damon_va_probe_folio(priv->ctx, priv->r, folio); + folio_put(folio); + +out: + spin_unlock(ptl); + return 0; +} +#else +#define damon_va_probe_hugetlb_entry NULL +#endif /* CONFIG_HUGETLB_PAGE */ + static void __damon_va_apply_probes(struct damon_ctx *ctx, struct mm_struct *mm, struct damon_region *r) { @@ -607,7 +635,7 @@ static void __damon_va_apply_probes(struct damon_ctx *ctx, }; struct mm_walk_ops damon_probe_walk_ops = { .pmd_entry = damon_va_probe_pmd_entry, - .hugetlb_entry = NULL, + .hugetlb_entry = damon_va_probe_hugetlb_entry, }; unsigned long addr = r->sampling_addr; From ccf03cdd0e9ee3ccd2b10b1f19a7c5362f8c1930 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 9 Sep 2026 07:04:07 -0700 Subject: [PATCH 0850/1352] mm/damon/vaddr: support pgidle_unset probe filter type DAMON virtual address space operation set (vaddr) does not support pgidle_unset probe filter type. Add the support. Link: https://lore.kernel.org/20260909140408.104699-6-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Gutierrez Asier --- mm/damon/vaddr.c | 55 ++++++++++++++++++++++++++++++++++++++++++------ 1 file changed, 48 insertions(+), 7 deletions(-) diff --git a/mm/damon/vaddr.c b/mm/damon/vaddr.c index 45239f05e113b2..9a38dc89a156ef 100644 --- a/mm/damon/vaddr.c +++ b/mm/damon/vaddr.c @@ -522,13 +522,49 @@ static void damon_va_prep_probes(struct damon_ctx *ctx, bool set_samples) } } -static bool damon_va_filter_pass(struct folio *folio, struct damon_probe *p) +static bool damon_va_young_addr(struct folio *folio, pte_t *pte, pmd_t *pmd, + struct mm_struct *mm, unsigned long addr) +{ + bool young = false; + + if (pte) + young = pte_young(*pte); + else if (pmd) + young = pmd_young(*pmd); + young = young || !folio_test_idle(folio) || + mmu_notifier_test_young(mm, addr); + return young; +} + +static bool damon_va_filter_match(struct damon_filter *filter, + struct folio *folio, pte_t *pte, pmd_t *pmd, + struct mm_struct *mm, unsigned long addr) +{ + bool matched = false; + + switch (filter->type) { + case DAMON_FILTER_TYPE_PGIDLE_UNSET: + if (!folio) + matched = false; + else + matched = damon_va_young_addr(folio, pte, pmd, mm, + addr); + break; + default: + return damon_ops_filter_match(filter, folio); + } + return matched == filter->matching; +} + +static bool damon_va_filter_pass(struct folio *folio, struct damon_probe *p, + pte_t *pte, pmd_t *pmd, struct mm_struct *mm, + unsigned long addr) { struct damon_filter *f; bool pass = true; damon_for_each_filter(f, p) { - if (damon_ops_filter_match(f, folio)) { + if (damon_va_filter_match(f, folio, pte, pmd, mm, addr)) { pass = f->allow; break; } @@ -543,13 +579,15 @@ struct damon_va_probe_walk_private { }; static void damon_va_probe_folio(struct damon_ctx *ctx, - struct damon_region *r, struct folio *folio) + struct damon_region *r, struct folio *folio, + pte_t *pte, pmd_t *pmd, struct mm_struct *mm) { struct damon_probe *probe; int i = 0; damon_for_each_probe(probe, ctx) { - if (damon_va_filter_pass(folio, probe)) + if (damon_va_filter_pass(folio, probe, pte, pmd, mm, + r->sampling_addr)) r->probe_hits[i]++; i++; } @@ -574,7 +612,8 @@ static int damon_va_probe_pmd_entry(pmd_t *pmd, unsigned long addr, folio = vm_normal_folio_pmd(walk->vma, addr, pmde); if (!folio) goto huge_out; - damon_va_probe_folio(priv->ctx, priv->r, folio); + damon_va_probe_folio(priv->ctx, priv->r, folio, NULL, &pmde, + walk->vma->vm_mm); huge_out: spin_unlock(ptl); @@ -591,7 +630,8 @@ static int damon_va_probe_pmd_entry(pmd_t *pmd, unsigned long addr, folio = vm_normal_folio(walk->vma, addr, ptent); if (!folio) goto out; - damon_va_probe_folio(priv->ctx, priv->r, folio); + damon_va_probe_folio(priv->ctx, priv->r, folio, &ptent, NULL, + walk->vma->vm_mm); out: pte_unmap_unlock(pte, ptl); @@ -615,7 +655,8 @@ static int damon_va_probe_hugetlb_entry(pte_t *pte, unsigned long hmask, folio = pfn_folio(pte_pfn(entry)); folio_get(folio); - damon_va_probe_folio(priv->ctx, priv->r, folio); + damon_va_probe_folio(priv->ctx, priv->r, folio, &entry, NULL, + walk->vma->vm_mm); folio_put(folio); out: From 5b029a83d3d103622a54b6b2375870822ab25af5 Mon Sep 17 00:00:00 2001 From: Usama Arif Date: Mon, 7 Sep 2026 09:19:38 -0700 Subject: [PATCH 0851/1352] mm: zswap: don't fail a large-folio swapin whose range is not in zswap thp_swapin_suitable_orders() and shmem_swap_alloc_folio() sample zswap_never_enabled() to decide whether a swapin may use a large folio. zswap_load() samples the same one-way static key again once the read reaches it. Nothing serialises the two reads, and in between the task allocates and pins a high-order folio, which can sleep. If zswap is enabled for the first time in that window, a large folio that was correctly permitted reaches zswap_load(), which rejects every large folio with -EINVAL. swap_read_folio() treats anything other than -ENOENT as "zswap handled it" and skips the backing-device read, so the folio comes back unlocked and not uptodate: SIGBUS for an anonymous fault, -EIO for shmem. The data is intact on the swap device - it was written there before zswap was ever enabled - and the not-uptodate folio stays in the swap cache, so every retry of the fault fails the same way. With panic_on_warn the WARN takes the machine down rather than the task. Scan the range instead of rejecting the folio. The caller has pinned every slot before issuing the read, so zswap cannot start a store or a writeback into the range and the scan is stable. If nothing in the range is in zswap it is all on the backing device: return -ENOENT and let swap_read_folio() read it. A range that does have a slot in zswap is still refused, because zswap stores large folios as order-0 entries and cannot reconstruct one. That stays reachable - a slot shared with another task can be stored inside the same window - and refusing is correct, since the alternative is returning the stale device copy. Report it as -EIO rather than -EINVAL: the request is valid, zswap just cannot serve it. The only caller distinguishes -ENOENT from everything else, so that part is a documentation fix. Link: https://lore.kernel.org/20260907161938.1932355-1-usama.arif@linux.dev Fixes: 242d12c98174 ("mm: support large folios swap-in for sync io devices") Co-developed-by: Alexandre Ghiti Signed-off-by: Alexandre Ghiti Signed-off-by: Usama Arif Signed-off-by: Andrew Morton Acked-by: Yosry Ahmed Acked-by: Nhat Pham Cc: Chengming Zhou Cc: Johannes Weiner --- mm/zswap.c | 58 ++++++++++++++++++++++++++++++++++++++++-------------- 1 file changed, 43 insertions(+), 15 deletions(-) diff --git a/mm/zswap.c b/mm/zswap.c index 3a6f8901764641..b5411ab164809d 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -1555,6 +1555,32 @@ bool zswap_store(struct folio *folio) return ret; } +/** + * zswap_is_present() - is any slot in [entry, entry + nr) in zswap? + * @entry: base swap entry of the range + * @nr: number of contiguous slots to check + * + * Context: The caller must keep the range pinned, otherwise the answer can + * change under it. + * Return: true if at least one slot in the range is in zswap. + */ +static bool zswap_is_present(swp_entry_t entry, unsigned int nr) +{ + pgoff_t offset = swp_offset(entry); + struct xarray *tree = swap_zswap_tree(entry); + unsigned long index = offset; + + /* + * A pinned range is at most SWAPFILE_CLUSTER slots and is aligned to + * its own size, so one tree covers all of it and a single lookup is + * enough. Scanning only part of the range would report a false + * "absent" and let the caller read a stale copy from the device. + */ + BUILD_BUG_ON(SWAPFILE_CLUSTER > ZSWAP_ADDRESS_SPACE_PAGES); + + return xa_find(tree, &index, offset + nr - 1, XA_PRESENT); +} + /** * zswap_load() - load a folio from zswap * @folio: folio to load @@ -1562,15 +1588,12 @@ bool zswap_store(struct folio *folio) * Return: 0 on success, with the folio unlocked and marked up-to-date, or one * of the following error codes: * - * -EIO: if the swapped out content was in zswap, but could not be loaded - * into the page due to a decompression failure. The folio is unlocked, but - * NOT marked up-to-date, so that an IO error is emitted (e.g. do_swap_page() - * will SIGBUS). - * - * -EINVAL: if the swapped out content was in zswap, but the page belongs - * to a large folio, which is not supported by zswap. The folio is unlocked, - * but NOT marked up-to-date, so that an IO error is emitted (e.g. - * do_swap_page() will SIGBUS). + * -EIO: if the swapped out content was in zswap but could not be handed + * back, either because decompression failed or because a slot in a + * large-folio range is still in zswap and zswap cannot reconstruct a large + * folio from per-page entries. The folio is unlocked, but NOT marked + * up-to-date, so that an IO error is emitted (e.g. do_swap_page() will + * SIGBUS). * * -ENOENT: if the swapped out content was not in zswap. The folio remains * locked on return. @@ -1589,13 +1612,18 @@ int zswap_load(struct folio *folio) return -ENOENT; /* - * Large folios should not be swapped in while zswap is being used, as - * they are not properly handled. Zswap does not properly load large - * folios, and a large folio may only be partially in zswap. + * A large folio can legitimately reach zswap_load() with its whole + * range on the backing device, so scan the range rather than rejecting + * it outright. The caller has pinned every slot, so zswap cannot start + * a store or a writeback into the range while we look. */ - if (WARN_ON_ONCE(folio_test_large(folio))) { - folio_unlock(folio); - return -EINVAL; + if (folio_test_large(folio)) { + if (WARN_ON_ONCE(zswap_is_present(swp, + folio_nr_pages(folio)))) { + folio_unlock(folio); + return -EIO; + } + return -ENOENT; } entry = xa_load(tree, offset); From 46b99728fd709f5abfe55f18cfe42616da7d6b05 Mon Sep 17 00:00:00 2001 From: Jinmeng Zhou Date: Mon, 7 Sep 2026 21:20:55 +0800 Subject: [PATCH 0852/1352] mm/hugetlb: fix subpool minimum reservation rollback When a reservation request is partially covered by a subpool minimum and the remaining global reservation fails, the error path first calls hugepage_subpool_put_pages() for the subpool-backed portion. It removes the failed global portion from used_hpages only afterwards. hugepage_subpool_put_pages() uses used_hpages to decide whether rsv_hpages should be restored. Since used_hpages still includes the global portion, it can remain at or above min_hpages and prevent that restoration. It then reports the subpool reservation as releasable, causing hugetlb_acct_memory() to incorrectly decrement h->resv_huge_pages. This was reproduced with four 2 MB huge pages and a hugetlbfs mount with size=10M,min_size=8M. After a successful three-page reservation, a two-page reservation which needed one subpool page and one global page failed with -ENOMEM. HugePages_Rsvd incorrectly dropped from four to three even though the subpool minimum was still four pages. Roll back the failed global portion from used_hpages first, so that hugepage_subpool_put_pages() evaluates the minimum reservation against the current usage and returns the correct global adjustment. Link: https://lore.kernel.org/20260907132055.26696-1-zhoujinmeng@bytedance.com Fixes: a833a693a490 ("mm: hugetlb: fix incorrect fallback for subpool") Signed-off-by: Jinmeng Zhou Signed-off-by: Andrew Morton Acked-by: Muchun Song Tested-by: Ackerley Tng Cc: David Hildenbrand Cc: Ma Wupeng Cc: Oscar Salvador Cc: --- mm/hugetlb.c | 18 +++++++++--------- 1 file changed, 9 insertions(+), 9 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index afffaa3d2d3741..5e05711f352d86 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -6865,15 +6865,6 @@ long hugetlb_reserve_pages(struct inode *inode, out_put_pages: spool_resv = chg - gbl_reserve; - if (spool_resv) { - /* put sub pool's reservation back, chg - gbl_reserve */ - gbl_resv = hugepage_subpool_put_pages(spool, spool_resv); - /* - * subpool's reserved pages can not be put back due to race, - * return to hstate. - */ - hugetlb_acct_memory(h, -gbl_resv); - } /* Restore used_hpages for pages that failed global reservation */ if (gbl_reserve && spool) { unsigned long flags; @@ -6883,6 +6874,15 @@ long hugetlb_reserve_pages(struct inode *inode, spool->used_hpages -= gbl_reserve; unlock_or_release_subpool(spool, flags); } + if (spool_resv) { + /* put sub pool's reservation back, chg - gbl_reserve */ + gbl_resv = hugepage_subpool_put_pages(spool, spool_resv); + /* + * subpool's reserved pages can not be put back due to race, + * return to hstate. + */ + hugetlb_acct_memory(h, -gbl_resv); + } out_uncharge_cgroup: hugetlb_cgroup_uncharge_cgroup_rsvd(hstate_index(h), chg * pages_per_huge_page(h), h_cg); From c3220566a86a86b540cb5746176d0786d855a903 Mon Sep 17 00:00:00 2001 From: Longlong Xia Date: Tue, 8 Sep 2026 09:28:01 +0800 Subject: [PATCH 0853/1352] mm/zswap: publish the initial pool with list_add_rcu() zswap_setup() publishes the pool on the zswap_pools list with a plain list_add(), but the list is walked by concurrent RCU readers holding nothing but rcu_read_lock() through zswap_total_pages(), e.g. /proc/meminfo and the shrinker count path. CPU 0 (writer) CPU 1 (reader) -------------- -------------- zswap_pool_create(): pool->zs_pool = zs_create_pool(); (1) list_add() -> __list_add(): WRITE_ONCE(zswap_pools.next, &pool->list); (2) zswap_total_pages(): pool = READ_ONCE( (a) zswap_pools.next); zs_get_total_pages( (b) pool->zs_pool); If (2) becomes visible to CPU 1 before (1), CPU 1 finds the pool at (a) but dereferences a wild pointer at (b). Publish the node with list_add_rcu(). Link: https://lore.kernel.org/20260908012801.1864430-1-xialonglong2025@163.com Fixes: 91cdcd8d624b ("mm: zswap: optimize zswap pool size tracking") Signed-off-by: Longlong Xia Signed-off-by: Andrew Morton Reported-by: Sashiko Closes: https://sashiko.dev/#/patchset/20260906133601.3563324-1-xialonglong2025%40163.com Acked-by: Yosry Ahmed Acked-by: Nhat Pham Assisted-by: Zcode:GLM-5.3 Cc: Chengming Zhou Cc: Johannes Weiner Cc: --- mm/zswap.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/zswap.c b/mm/zswap.c index b5411ab164809d..2a95aedc08fcca 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -1815,7 +1815,7 @@ static int zswap_setup(void) pool = __zswap_pool_create_fallback(); if (pool) { pr_info("loaded using pool %s\n", pool->tfm_name); - list_add(&pool->list, &zswap_pools); + list_add_rcu(&pool->list, &zswap_pools); zswap_has_pool = true; static_branch_enable(&zswap_ever_enabled); } else { From ee598ed0b7753a2b8632e391ff19b941c00b7732 Mon Sep 17 00:00:00 2001 From: Bingfang Guo Date: Thu, 10 Sep 2026 11:46:58 +0800 Subject: [PATCH 0854/1352] mm/memcg: clear folio memcg after changing per memcg stats I notice extremely high swapcached count in the per memcg level memory.stat when running tests with cgroupv1 setup by swapping pages in and out. It seems that the counter never gets decreased so the value is rather useless and confusing to users reading it. So I think fixing it so that the value can reflect the actual swapcache usage correctly could be helpful. __memcg1_swapout() transfers the memsw charge of a folio to its swap entry and clears folio->memcg_data as part of that. In the vmscan swapout path it runs before __swap_cache_del_folio(), which then decrements the swapcache stats through lruvec_stat_mod_folio(). Since folio->memcg_data has already been cleared, folio_memcg() returns NULL and the NR_SWAPCACHE decrement only updates the node-level counter instead of the memcg's lruvec, leaking the per-memcg swapcache count. Users using swaps will read totally meaningless swapcache value from per memcg memory.stat, like 200GB of swapcache on a 64GB setup, which is quite confusing and may trigger monitoring alerts, if any. Move the __memcg1_swapout() call into __swap_cache_del_folio(), after the NR_FILE_PAGES and NR_SWAPCACHE updates but before __swap_cache_do_del_folio() removes the folio from the swap cache. This keeps the stats attributed to the folio's memcg while still recording the swap cgroup with a valid folio->swap. Add a swapout parameter so the plain swap_cache_del_folio() path is left unchanged. Link: https://lore.kernel.org/20260910-memcg-swapcache-stats-fix-v5-1-033f510ba748@tencent.com Fixes: 2732acda82c9 ("mm, swap: use swap cache as the swap in synchronize layer") Signed-off-by: Bingfang Guo Signed-off-by: Andrew Morton Acked-by: Kairui Song Cc: Axel Rasmussen Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kemeng Shi Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Nhat Pham Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie Cc: --- mm/swap.h | 6 ++++-- mm/swap_state.c | 13 ++++++++++--- mm/vmscan.c | 3 +-- 3 files changed, 15 insertions(+), 7 deletions(-) diff --git a/mm/swap.h b/mm/swap.h index 0b5d507739bcb6..b3b54c28929a19 100644 --- a/mm/swap.h +++ b/mm/swap.h @@ -319,7 +319,8 @@ struct folio *swap_cache_alloc_folio(swp_entry_t target_entry, gfp_t gfp_mask, void __swap_cache_add_folio(struct swap_cluster_info *ci, struct folio *folio, swp_entry_t entry); void __swap_cache_del_folio(struct swap_cluster_info *ci, - struct folio *folio, swp_entry_t entry, void *shadow); + struct folio *folio, swp_entry_t entry, void *shadow, + bool swapout); void __swap_cache_replace_folio(struct swap_cluster_info *ci, struct folio *old, struct folio *new); @@ -452,7 +453,8 @@ static inline void swap_cache_del_folio(struct folio *folio) } static inline void __swap_cache_del_folio(struct swap_cluster_info *ci, - struct folio *folio, swp_entry_t entry, void *shadow) + struct folio *folio, swp_entry_t entry, void *shadow, + bool swapout) { } diff --git a/mm/swap_state.c b/mm/swap_state.c index 305877e1f4d7bf..625c185a1ca4d5 100644 --- a/mm/swap_state.c +++ b/mm/swap_state.c @@ -306,21 +306,28 @@ static void __swap_cache_do_del_folio(struct swap_cluster_info *ci, * @folio: The folio. * @entry: The first swap entry that the folio corresponds to. * @shadow: shadow value to be filled in the swap cache. + * @swapout: whether this folio is being reclaimed after swapout. * * Removes a folio from the swap cache and fills a shadow in place. * This won't put the folio's refcount. The caller has to do that. * * Context: Caller must ensure the folio is locked and in the swap cache * using the index of @entry, and lock the cluster that holds the entries. + * If @swapout is set, the folio should be in reclaim path and IRQs + * should be disabled. */ void __swap_cache_del_folio(struct swap_cluster_info *ci, struct folio *folio, - swp_entry_t entry, void *shadow) + swp_entry_t entry, void *shadow, bool swapout) { unsigned long nr_pages = folio_nr_pages(folio); - __swap_cache_do_del_folio(ci, folio, entry, shadow); node_stat_mod_folio(folio, NR_FILE_PAGES, -nr_pages); lruvec_stat_mod_folio(folio, NR_SWAPCACHE, -nr_pages); + + if (swapout) + __memcg1_swapout(folio, ci); + + __swap_cache_do_del_folio(ci, folio, entry, shadow); } /** @@ -339,7 +346,7 @@ void swap_cache_del_folio(struct folio *folio) swp_entry_t entry = folio->swap; ci = swap_cluster_lock(__swap_entry_to_info(entry), swp_offset(entry)); - __swap_cache_del_folio(ci, folio, entry, NULL); + __swap_cache_del_folio(ci, folio, entry, NULL, false); swap_cluster_unlock(ci); folio_ref_sub(folio, folio_nr_pages(folio)); diff --git a/mm/vmscan.c b/mm/vmscan.c index a49da82ed4797a..8e9c73dcd19bb3 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -807,8 +807,7 @@ static int __remove_mapping(struct address_space *mapping, struct folio *folio, if (reclaimed && !mapping_exiting(mapping)) shadow = workingset_eviction(folio, target_memcg); - __memcg1_swapout(folio, ci); - __swap_cache_del_folio(ci, folio, swap, shadow); + __swap_cache_del_folio(ci, folio, swap, shadow, true); swap_cluster_unlock_irq(ci); } else { void (*free_folio)(struct folio *); From 8ce7fed40d20858c2942fb52556a933ecb81f3f1 Mon Sep 17 00:00:00 2001 From: Tianyi Chen Date: Thu, 10 Sep 2026 20:56:44 +0800 Subject: [PATCH 0855/1352] selftests/mm: reject invalid test selections before running tests Patch series "selftests/mm: Validate selections and scope memfd_secret setup", v3. run_vmtests.sh can reach test setup after invalid options or category selections. Its memfd_secret preparation can also change ptrace_scope when that category was not selected. Patch 1 rejects invalid selections before setup. Patch 2 gates memfd_secret preparation on category selection and executable presence. This patch (of 2): getopts reports unknown options and missing arguments, but run_vmtests.sh ignores its error result and continues with test setup. An empty -t argument also falls back to the default selection, while unknown category names can silently select no tests and still reach setup code. Exit on getopts errors and validate category names against the existing list in usage() before any test setup. Reject empty and whitespace-only selections, and normalize category separators so validation and execution agree. Initialize the default selection before parsing options so only -t changes the selection. Link: https://lore.kernel.org/20260910125645.285866-1-diannaaav@gmail.com Link: https://lore.kernel.org/20260910125645.285866-2-diannaaav@gmail.com Fixes: 85463321e726 ("selftests/vm: enable running select groups of tests") Signed-off-by: Tianyi Chen Signed-off-by: Andrew Morton Assisted-by: LLM Cc: David Hildenbrand (Arm) Cc: Joel Savitz Cc: Shuah Khan --- tools/testing/selftests/mm/run_vmtests.sh | 26 +++++++++++++++++++---- 1 file changed, 22 insertions(+), 4 deletions(-) diff --git a/tools/testing/selftests/mm/run_vmtests.sh b/tools/testing/selftests/mm/run_vmtests.sh index d09f9f6a384ee7..9e62ab4c677520 100755 --- a/tools/testing/selftests/mm/run_vmtests.sh +++ b/tools/testing/selftests/mm/run_vmtests.sh @@ -96,26 +96,44 @@ separated by spaces: example: ./run_vmtests.sh -t "hmm mmap ksm" EOF - exit 0 } RUN_ALL=false RUN_DESTRUCTIVE=false TAP_PREFIX="# " +VM_SELFTEST_ITEMS="default" + while getopts "aht:nd" OPT; do case ${OPT} in "a") RUN_ALL=true ;; - "h") usage ;; + "h") usage; exit 0 ;; "t") VM_SELFTEST_ITEMS=${OPTARG} ;; "n") TAP_PREFIX= ;; "d") RUN_DESTRUCTIVE=true ;; + "?") exit 1 ;; esac done shift $((OPTIND -1)) -# default behavior: run all tests -VM_SELFTEST_ITEMS=${VM_SELFTEST_ITEMS:-default} +# Normalize whitespace so validation and test_selected() use the same names. +read -r -a selected_categories <<< "${VM_SELFTEST_ITEMS//$'\n'/ }" +VM_SELFTEST_ITEMS="${selected_categories[*]}" +if [ -z "$VM_SELFTEST_ITEMS" ]; then + echo "No test categories specified" >&2 + exit 1 +fi + +if [ "$VM_SELFTEST_ITEMS" != "default" ]; then + # Keep the documented category list as the source of valid names. + valid_categories=$(usage | sed -n 's/^- //p') + for category in "${selected_categories[@]}"; do + if ! grep -Fxq -- "$category" <<< "$valid_categories"; then + echo "Unknown test category: $category" >&2 + exit 1 + fi + done +fi test_selected() { if [ "$VM_SELFTEST_ITEMS" == "default" ]; then From 5c8f8bcd8df7b0a8444a975a9e4710976904d002 Mon Sep 17 00:00:00 2001 From: Tianyi Chen Date: Thu, 10 Sep 2026 20:56:45 +0800 Subject: [PATCH 0856/1352] selftests/mm: only prepare ptrace_scope when memfd_secret is selected The memfd_secret setup clears ptrace_scope whenever its test binary is executable, even when a different category was selected. run_test() filters the test invocation, but it does not protect the preceding setup. Check the category selection before entering the memfd_secret block so running unrelated categories does not change ptrace_scope. Keep the existing executable check and the behavior when memfd_secret is selected. Link: https://lore.kernel.org/20260910125645.285866-3-diannaaav@gmail.com Signed-off-by: Tianyi Chen Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Joel Savitz Cc: Shuah Khan --- tools/testing/selftests/mm/run_vmtests.sh | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tools/testing/selftests/mm/run_vmtests.sh b/tools/testing/selftests/mm/run_vmtests.sh index 9e62ab4c677520..9bbef9410ccc35 100755 --- a/tools/testing/selftests/mm/run_vmtests.sh +++ b/tools/testing/selftests/mm/run_vmtests.sh @@ -370,7 +370,7 @@ CATEGORY="process_madv" run_test ./process_madv CATEGORY="vma_merge" run_test ./merge -if [ -x ./memfd_secret ] +if test_selected "memfd_secret" && [ -x ./memfd_secret ] then if [ -f /proc/sys/kernel/yama/ptrace_scope ]; then (echo 0 > /proc/sys/kernel/yama/ptrace_scope 2>&1) | tap_prefix From a862bc7171d85ec93945eac92afab9e8edc7e0b5 Mon Sep 17 00:00:00 2001 From: Kefeng Wang Date: Thu, 10 Sep 2026 20:35:42 +0800 Subject: [PATCH 0857/1352] mm: zswap: convert zswap_invalidate() to take a range Patch series "mm: zswap: optimize zswap invalidate and store", v3. This series range-ifies zswap_invalidate() to skip xarray lookups when zswap is unused, and reuses it in zswap_store() to eliminate redundant per-slot lookups and open-coded logic. This patch (of 3): zswap_invalidate() takes a swp_entry_t only to unpack it right back into type and offset, and both callers already have those values in hand. Pass type, offset, and nr_entries directly so swap_range_free() can invalidate an entire range in one call instead of looping in the caller. Link: https://lore.kernel.org/20260910123544.818146-1-wangkefeng.wang@huawei.com Link: https://lore.kernel.org/20260910123544.818146-2-wangkefeng.wang@huawei.com Signed-off-by: Kefeng Wang Signed-off-by: Andrew Morton Suggested-by: Johannes Weiner Acked-by: Yosry Ahmed Reviewed-by: Johannes Weiner Cc: Chengming Zhou Cc: Kairui Song Cc: Nhat Pham Cc: Yosry Ahmed --- include/linux/zswap.h | 8 ++++++-- mm/swapfile.c | 4 +--- mm/zswap.c | 29 +++++++++++++++++++---------- 3 files changed, 26 insertions(+), 15 deletions(-) diff --git a/include/linux/zswap.h b/include/linux/zswap.h index 30c193a1207e16..df6cafbe95dc0c 100644 --- a/include/linux/zswap.h +++ b/include/linux/zswap.h @@ -27,7 +27,7 @@ struct zswap_lruvec_state { unsigned long zswap_total_pages(void); bool zswap_store(struct folio *folio); int zswap_load(struct folio *folio); -void zswap_invalidate(swp_entry_t swp); +void zswap_invalidate(int type, pgoff_t offset, unsigned long nr_entries); int zswap_swapon(int type, unsigned long nr_pages); void zswap_swapoff(int type); void zswap_memcg_offline_cleanup(struct mem_cgroup *memcg); @@ -49,7 +49,11 @@ static inline int zswap_load(struct folio *folio) return -ENOENT; } -static inline void zswap_invalidate(swp_entry_t swp) {} +static inline void zswap_invalidate(int type, pgoff_t offset, + unsigned long nr_entries) +{ +} + static inline int zswap_swapon(int type, unsigned long nr_pages) { return 0; diff --git a/mm/swapfile.c b/mm/swapfile.c index a8118f095f4d9a..2c263563b70ebc 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -1316,10 +1316,8 @@ static void swap_range_free(struct swap_info_struct *si, unsigned long offset, { unsigned long end = offset + nr_entries - 1; void (*swap_slot_free_notify)(struct block_device *, unsigned long); - unsigned int i; - for (i = 0; i < nr_entries; i++) - zswap_invalidate(swp_entry(si->type, offset + i)); + zswap_invalidate(si->type, offset, nr_entries); if (si->flags & SWP_BLKDEV) swap_slot_free_notify = diff --git a/mm/zswap.c b/mm/zswap.c index 2a95aedc08fcca..9153bdfdd5df37 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -228,10 +228,15 @@ static bool zswap_has_pool; /* One swap address space for each 64M swap space */ #define ZSWAP_ADDRESS_SPACE_SHIFT 14 #define ZSWAP_ADDRESS_SPACE_PAGES (1 << ZSWAP_ADDRESS_SPACE_SHIFT) + +static inline struct xarray *zswap_tree(int type, pgoff_t offset) +{ + return &zswap_trees[type][offset >> ZSWAP_ADDRESS_SPACE_SHIFT]; +} + static inline struct xarray *swap_zswap_tree(swp_entry_t swp) { - return &zswap_trees[swp_type(swp)][swp_offset(swp) - >> ZSWAP_ADDRESS_SPACE_SHIFT]; + return zswap_tree(swp_type(swp), swp_offset(swp)); } #define zswap_pool_debug(msg, p) \ @@ -1656,18 +1661,22 @@ int zswap_load(struct folio *folio) return 0; } -void zswap_invalidate(swp_entry_t swp) +void zswap_invalidate(int type, pgoff_t offset, unsigned long nr_entries) { - pgoff_t offset = swp_offset(swp); - struct xarray *tree = swap_zswap_tree(swp); struct zswap_entry *entry; + struct xarray *tree; + unsigned long i; - if (xa_empty(tree)) - return; + for (i = 0; i < nr_entries; i++) { + tree = zswap_tree(type, offset + i); - entry = xa_erase(tree, offset); - if (entry) - zswap_entry_free(entry); + if (xa_empty(tree)) + continue; + + entry = xa_erase(tree, offset + i); + if (entry) + zswap_entry_free(entry); + } } int zswap_swapon(int type, unsigned long nr_pages) From 936b9056d7d10b91fa06eeec25a48dc7a644eabb Mon Sep 17 00:00:00 2001 From: Kefeng Wang Date: Thu, 10 Sep 2026 20:35:43 +0800 Subject: [PATCH 0858/1352] mm: zswap: skip xarray walk in zswap_invalidate() when zswap is unused zswap_invalidate() still walks the per-area xarray even when zswap has never been enabled. Add a zswap_never_enabled() to skip it. Link: https://lore.kernel.org/20260910123544.818146-3-wangkefeng.wang@huawei.com Signed-off-by: Kefeng Wang Signed-off-by: Andrew Morton Acked-by: Yosry Ahmed Reviewed-by: Johannes Weiner Cc: Chengming Zhou Cc: Kairui Song Cc: Nhat Pham Cc: Yosry Ahmed --- mm/zswap.c | 3 +++ 1 file changed, 3 insertions(+) diff --git a/mm/zswap.c b/mm/zswap.c index 9153bdfdd5df37..bf86651d746499 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -1667,6 +1667,9 @@ void zswap_invalidate(int type, pgoff_t offset, unsigned long nr_entries) struct xarray *tree; unsigned long i; + if (zswap_never_enabled()) + return; + for (i = 0; i < nr_entries; i++) { tree = zswap_tree(type, offset + i); From 65f2a8a12545850b57fc93c244ed8b63e725775d Mon Sep 17 00:00:00 2001 From: Kefeng Wang Date: Thu, 10 Sep 2026 20:35:44 +0800 Subject: [PATCH 0859/1352] mm: zswap: reuse zswap_invalidate() in zswap_store() Reuse zswap_invalidate() in zswap_store() check_old path. Its zswap_never_enabled() and xa_empty() guards avoid redundant xarray lookups and deduplicate the per-slot free logic. Link: https://lore.kernel.org/20260910123544.818146-4-wangkefeng.wang@huawei.com Signed-off-by: Kefeng Wang Signed-off-by: Andrew Morton Acked-by: Yosry Ahmed Reviewed-by: Johannes Weiner Cc: Chengming Zhou Cc: Kairui Song Cc: Nhat Pham Cc: Yosry Ahmed --- mm/zswap.c | 15 ++------------- 1 file changed, 2 insertions(+), 13 deletions(-) diff --git a/mm/zswap.c b/mm/zswap.c index bf86651d746499..d0b6c229b6169d 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -1543,19 +1543,8 @@ bool zswap_store(struct folio *folio) * offsets corresponding to each page of the folio. Otherwise, * writeback could overwrite the new data in the swapfile. */ - if (!ret) { - unsigned type = swp_type(swp); - pgoff_t offset = swp_offset(swp); - struct zswap_entry *entry; - struct xarray *tree; - - for (index = 0; index < nr_pages; ++index) { - tree = swap_zswap_tree(swp_entry(type, offset + index)); - entry = xa_erase(tree, offset + index); - if (entry) - zswap_entry_free(entry); - } - } + if (!ret) + zswap_invalidate(swp_type(swp), swp_offset(swp), nr_pages); return ret; } From 662f723acb1132104e0946b9232e8ab587bba545 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:15 +0100 Subject: [PATCH 0860/1352] mm/khugepaged: drop redundant mm_struct pin in madvise_collapse() Patch series "mm/collapse: separate a collapse from its callers", v4. There is no line between the collapse engine and the callers that ask for a collapse. khugepaged.c holds both, and they reach into each other. - Sixteen tests through the collapse path read cc->is_khugepaged to work out what they are allowed to do, when every one of those decisions was made by the caller before it asked. - collapse_single_pmd() does both halves of a collapse behind one call and drops mmap_lock somewhere in the middle. Which of its paths dropped it is not something a caller can see, so it hands back a bool and the caller keeps track. - MADV_COLLAPSE's implementation -- the walk over the user's range, the per-PMD loop, the errno translation -- sits in khugepaged.c, which is the daemon's file. So: draw the line. State what a caller allows in a policy, split the call in two with the lock as the boundary, and move the syscall to madvise.c. What the engine offers is then three calls, with the lock state written down against each, and a policy the caller fills for itself: collapse_control_init(cc) once, before the first table collapse_policy_*(&cc->policy) what this caller allows collapse_scan_pmd(vma, addr, ...) per table, under mmap_lock collapse_run_pmd(mm, addr, ...) when a scan found work, no mmap_lock The engine stays in khugepaged.c for now; what changes is that it has an interface, and that neither half has to ask about the other. madvise.c gains the operation it should have had all along. This patch (of 13): madvise_collapse() holds an mmgrab() reference across its work. It is redundant. Every caller already holds mm_users: - madvise(2) works on current->mm, which lives as long as the task is in the syscall; - process_madvise(2) reaches a remote mm through mm_access(), which takes an mm_users reference and holds it until the syscall returns; - io_uring passes current->mm; - DAMON takes one with get_task_mm() and drops it after the call. Drop the mmgrab()/mmdrop() pair. Link: https://lore.kernel.org/20260928100630.21870-1-kirill@shutemov.name Link: https://lore.kernel.org/20260928100630.21870-2-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Zi Yan Reviewed-by: Baolin Wang Assisted-by: LLM Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- mm/khugepaged.c | 2 -- 1 file changed, 2 deletions(-) diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 85ea095906fb70..7096dbf09b031b 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -3252,7 +3252,6 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start, cc->is_khugepaged = false; cc->progress = 0; - mmgrab(mm); lru_add_drain_all(); for (addr = hstart; addr < hend; addr += HPAGE_PMD_SIZE) { @@ -3308,7 +3307,6 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start, } out_nolock: mmap_assert_locked(mm); - mmdrop(mm); kfree(cc); return thps == ((hend - hstart) >> HPAGE_PMD_SHIFT) ? 0 From d8e8de586bde581da85e72d61452e30043456563 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:16 +0100 Subject: [PATCH 0861/1352] mm/khugepaged: count collapses where khugepaged makes them collapse_single_pmd() bumps khugepaged_pages_collapsed for its caller, and tests cc->is_khugepaged to know whether it should: the counter belongs to the daemon, and MADV_COLLAPSE must not touch it. The daemon sees every result of every collapse it asks for, so it can keep its own counter without the shared path testing who called. Link: https://lore.kernel.org/20260928100630.21870-3-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Zi Yan Reviewed-by: Baolin Wang Assisted-by: LLM Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- mm/khugepaged.c | 11 ++++------- 1 file changed, 4 insertions(+), 7 deletions(-) diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 7096dbf09b031b..ec1023032b8090 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -2838,10 +2838,8 @@ static enum scan_result collapse_single_pmd(unsigned long addr, mmap_assert_locked(mm); - if (vma_is_anonymous(vma)) { - result = collapse_scan_pmd(mm, vma, addr, lock_dropped, cc); - goto end; - } + if (vma_is_anonymous(vma)) + return collapse_scan_pmd(mm, vma, addr, lock_dropped, cc); file = get_file(vma->vm_file); pgoff = linear_page_index(vma, addr); @@ -2877,9 +2875,6 @@ static enum scan_result collapse_single_pmd(unsigned long addr, result = SCAN_SUCCEED; mmap_read_unlock(mm); } -end: - if (cc->is_khugepaged && result == SCAN_SUCCEED) - ++khugepaged_pages_collapsed; return result; } @@ -2956,6 +2951,8 @@ static void collapse_scan_mm_slot(unsigned int progress_max, *result = collapse_single_pmd(khugepaged_scan.address, vma, &lock_dropped, cc); + if (*result == SCAN_SUCCEED) + khugepaged_pages_collapsed++; /* move to next address */ khugepaged_scan.address += HPAGE_PMD_SIZE; if (lock_dropped) From c7734acf1c978f92be2c3d4bfd220bc910335d94 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:17 +0100 Subject: [PATCH 0862/1352] mm/khugepaged: rename mthp_present_ptes bitmap to eligible_ptes The name says less than the bit means. A set bit means not only that the PTE is present, but also that it passed the other checks: uffd, lazyfree, anonymity, sharing. The PTE can be considered a collapse source. mthp_collapse() then reads the bitmap starting at the PMD order, so the bitmap is not specific to mTHP either. Name it for what a set bit means, and update the comments that named it. No functional change. Link: https://lore.kernel.org/20260928100630.21870-4-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Zi Yan Reviewed-by: Baolin Wang Assisted-by: LLM Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- mm/khugepaged.c | 32 ++++++++++++++++---------------- 1 file changed, 16 insertions(+), 16 deletions(-) diff --git a/mm/khugepaged.c b/mm/khugepaged.c index ec1023032b8090..07ae6c2e4fcf35 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -115,8 +115,8 @@ struct collapse_control { /* nodemask for allocation fallback */ nodemask_t alloc_nmask; - /* Each bit represents a single occupied (!none/zero) page. */ - DECLARE_BITMAP(mthp_present_ptes, MAX_PTRS_PER_PTE); + /* Each bit marks a PTE the scan accepted as a collapse source */ + DECLARE_BITMAP(eligible_ptes, MAX_PTRS_PER_PTE); }; /** @@ -627,7 +627,7 @@ static void collapse_control_init_scan(struct collapse_control *cc) { memset(cc->node_load, 0, sizeof(cc->node_load)); nodes_clear(cc->alloc_nmask); - bitmap_zero(cc->mthp_present_ptes, MAX_PTRS_PER_PTE); + bitmap_zero(cc->eligible_ptes, MAX_PTRS_PER_PTE); } static void release_pte_folio(struct folio *folio) @@ -1512,15 +1512,15 @@ static unsigned int max_order_from_offset(unsigned int offset) * mthp_collapse() consumes the bitmap that is generated during * collapse_scan_pmd() to determine what regions and mTHP orders fit best. * - * Each bit in cc->mthp_present_ptes represents a single occupied (!none/zero) - * page. We start at the PMD order and check if it is eligible for collapse; + * Each bit in cc->eligible_ptes marks a PTE the scan accepted as a collapse + * source. We start at the PMD order and check if it is eligible for collapse; * if not, we check the left and right halves of the PTE page table we are * examining at a lower order. * - * For each of these, we determine how many PTE entries are occupied in the - * range of PTE entries we propose to collapse, then we compare this to a - * threshold number of PTE entries which would need to be occupied for a - * collapse to be permitted at that order (accounting for max_ptes_none). + * For each of these, we count the eligible PTEs in the range we propose to + * collapse, then we compare this to the number of eligible PTEs the range + * would need for a collapse to be permitted at that order (accounting for + * max_ptes_none). * * If a collapse is permitted, we attempt to collapse the PTE range into a * mTHP. @@ -1529,7 +1529,7 @@ static enum scan_result mthp_collapse(struct mm_struct *mm, unsigned long address, int referenced, int unmapped, struct collapse_control *cc, unsigned long enabled_orders) { - unsigned int nr_occupied_ptes, nr_ptes, max_ptes_none; + unsigned int nr_eligible_ptes, nr_ptes, max_ptes_none; enum scan_result last_result = SCAN_FAIL; int collapsed = 0; bool alloc_failed = false; @@ -1544,18 +1544,18 @@ static enum scan_result mthp_collapse(struct mm_struct *mm, goto next_order; max_ptes_none = collapse_max_ptes_none(cc, NULL, order); - nr_occupied_ptes = bitmap_weight_from(cc->mthp_present_ptes, offset, + nr_eligible_ptes = bitmap_weight_from(cc->eligible_ptes, offset, offset + nr_ptes); /* * Swap PTEs accepted during the scan are counted in @unmapped, - * not in the present-PTE bitmap. Account them for the PMD-order + * not in cc->eligible_ptes. Account them for the PMD-order * candidate. */ if (is_pmd_order(order)) - nr_occupied_ptes += unmapped; + nr_eligible_ptes += unmapped; - if (nr_occupied_ptes >= nr_ptes - max_ptes_none) { + if (nr_eligible_ptes >= nr_ptes - max_ptes_none) { enum scan_result ret; collapse_address = address + offset * PAGE_SIZE; @@ -1761,8 +1761,8 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, } } - /* Set bit for occupied pages */ - __set_bit(i, cc->mthp_present_ptes); + /* The scan accepted this PTE as a collapse source */ + __set_bit(i, cc->eligible_ptes); /* * Record which node the original page is from and save this * information to cc->node_load[]. From 3b28f9956d000632b4729edf67f1bb94afe01a7e Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:18 +0100 Subject: [PATCH 0863/1352] mm/collapse: add collapse.h for the collapse interface khugepaged.c holds both the users of collapse and the machinery that performs it. The daemon's scan loop, the sysfs tunables, MADV_COLLAPSE's entry point and the collapse itself all sit in one file and reach into each other freely. Nothing marks where a user ends and the engine begins. Start drawing that line. Add mm/collapse.h for what the two sides have to agree on: - enum scan_result - what the engine hands back; - struct collapse_control - the state a request carries. And two constants move with them: - KHUGEPAGED_MAX_PTES_LIMIT -> COLLAPSE_MAX_PTES_LIMIT; - KHUGEPAGED_MIN_MTHP_ORDER -> COLLAPSE_MIN_MTHP_ORDER. Neither is a fact about the daemon, so both lose the KHUGEPAGED_ prefix. No functional change. Link: https://lore.kernel.org/20260928100630.21870-5-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Zi Yan Reviewed-by: Baolin Wang Assisted-by: LLM Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- MAINTAINERS | 1 + mm/collapse.h | 64 +++++++++++++++++++++++++++++++++++++++ mm/khugepaged.c | 79 ++++++++----------------------------------------- 3 files changed, 78 insertions(+), 66 deletions(-) create mode 100644 mm/collapse.h diff --git a/MAINTAINERS b/MAINTAINERS index ca76ab10fe68ed..fe1d70ed5100b5 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -17448,6 +17448,7 @@ F: Documentation/admin-guide/mm/transhuge.rst F: include/linux/huge_mm.h F: include/linux/khugepaged.h F: include/trace/events/huge_memory.h +F: mm/collapse.h F: mm/huge_memory.c F: mm/khugepaged.c F: mm/mm_slot.h diff --git a/mm/collapse.h b/mm/collapse.h new file mode 100644 index 00000000000000..b115034d90187a --- /dev/null +++ b/mm/collapse.h @@ -0,0 +1,64 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +#ifndef __MM_COLLAPSE_H +#define __MM_COLLAPSE_H + +#include +#include +#include +#include + +#define COLLAPSE_MAX_PTES_LIMIT (HPAGE_PMD_NR - 1) +#define COLLAPSE_MIN_MTHP_ORDER 2 + +enum scan_result { + SCAN_FAIL, + SCAN_SUCCEED, + SCAN_NO_PTE_TABLE, + SCAN_PMD_MAPPED, + SCAN_EXCEED_NONE_PTE, + SCAN_EXCEED_SWAP_PTE, + SCAN_EXCEED_SHARED_PTE, + SCAN_PTE_NON_PRESENT, + SCAN_PTE_UFFD, + SCAN_PTE_MAPPED_HUGEPAGE, + SCAN_LACK_REFERENCED_PAGE, + SCAN_PAGE_NULL, + SCAN_SCAN_ABORT, + SCAN_PAGE_COUNT, + SCAN_PAGE_LRU, + SCAN_PAGE_LOCK, + SCAN_PAGE_ANON, + SCAN_PAGE_LAZYFREE, + SCAN_PAGE_COMPOUND, + SCAN_ANY_PROCESS, + SCAN_VMA_NULL, + SCAN_VMA_CHECK, + SCAN_ADDRESS_RANGE, + SCAN_DEL_PAGE_LRU, + SCAN_ALLOC_HUGE_PAGE_FAIL, + SCAN_CGROUP_CHARGE_FAIL, + SCAN_TRUNCATED, + SCAN_PAGE_HAS_PRIVATE, + SCAN_STORE_FAILED, + SCAN_COPY_MC, + SCAN_PAGE_FILLED, + SCAN_PAGE_DIRTY_OR_WRITEBACK, +}; + +struct collapse_control { + bool is_khugepaged; + + /* Num pages scanned per node */ + u32 node_load[MAX_NUMNODES]; + + /* Num pages scanned (see khugepaged_pages_to_scan) */ + unsigned int progress; + + /* nodemask for allocation fallback */ + nodemask_t alloc_nmask; + + /* Each bit marks a PTE the scan accepted as a collapse source */ + DECLARE_BITMAP(eligible_ptes, MAX_PTRS_PER_PTE); +}; + +#endif /* __MM_COLLAPSE_H */ diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 07ae6c2e4fcf35..6c7ac9ebcd4b45 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -26,44 +26,10 @@ #include #include +#include "collapse.h" #include "internal.h" -#include "page_alloc.h" #include "mm_slot.h" - -enum scan_result { - SCAN_FAIL, - SCAN_SUCCEED, - SCAN_NO_PTE_TABLE, - SCAN_PMD_MAPPED, - SCAN_EXCEED_NONE_PTE, - SCAN_EXCEED_SWAP_PTE, - SCAN_EXCEED_SHARED_PTE, - SCAN_PTE_NON_PRESENT, - SCAN_PTE_UFFD, - SCAN_PTE_MAPPED_HUGEPAGE, - SCAN_LACK_REFERENCED_PAGE, - SCAN_PAGE_NULL, - SCAN_SCAN_ABORT, - SCAN_PAGE_COUNT, - SCAN_PAGE_LRU, - SCAN_PAGE_LOCK, - SCAN_PAGE_ANON, - SCAN_PAGE_LAZYFREE, - SCAN_PAGE_COMPOUND, - SCAN_ANY_PROCESS, - SCAN_VMA_NULL, - SCAN_VMA_CHECK, - SCAN_ADDRESS_RANGE, - SCAN_DEL_PAGE_LRU, - SCAN_ALLOC_HUGE_PAGE_FAIL, - SCAN_CGROUP_CHARGE_FAIL, - SCAN_TRUNCATED, - SCAN_PAGE_HAS_PRIVATE, - SCAN_STORE_FAILED, - SCAN_COPY_MC, - SCAN_PAGE_FILLED, - SCAN_PAGE_DIRTY_OR_WRITEBACK, -}; +#include "page_alloc.h" #define CREATE_TRACE_POINTS #include @@ -91,7 +57,6 @@ static DECLARE_WAIT_QUEUE_HEAD(khugepaged_wait); * * Note that these are only respected if collapse was initiated by khugepaged. */ -#define KHUGEPAGED_MAX_PTES_LIMIT (HPAGE_PMD_NR - 1) unsigned int khugepaged_max_ptes_none __read_mostly; static unsigned int khugepaged_max_ptes_swap __read_mostly; static unsigned int khugepaged_max_ptes_shared __read_mostly; @@ -101,24 +66,6 @@ static DEFINE_READ_MOSTLY_HASHTABLE(mm_slots_hash, MM_SLOTS_HASH_BITS); static struct kmem_cache *mm_slot_cache __ro_after_init; -#define KHUGEPAGED_MIN_MTHP_ORDER 2 - -struct collapse_control { - bool is_khugepaged; - - /* Num pages scanned per node */ - u32 node_load[MAX_NUMNODES]; - - /* Num pages scanned (see khugepaged_pages_to_scan) */ - unsigned int progress; - - /* nodemask for allocation fallback */ - nodemask_t alloc_nmask; - - /* Each bit marks a PTE the scan accepted as a collapse source */ - DECLARE_BITMAP(eligible_ptes, MAX_PTRS_PER_PTE); -}; - /** * struct khugepaged_scan - cursor for scanning * @mm_head: the head of the mm list to scan @@ -267,7 +214,7 @@ static ssize_t max_ptes_none_store(struct kobject *kobj, unsigned long max_ptes_none; err = kstrtoul(buf, 10, &max_ptes_none); - if (err || max_ptes_none > KHUGEPAGED_MAX_PTES_LIMIT) + if (err || max_ptes_none > COLLAPSE_MAX_PTES_LIMIT) return -EINVAL; khugepaged_max_ptes_none = max_ptes_none; @@ -292,7 +239,7 @@ static ssize_t max_ptes_swap_store(struct kobject *kobj, unsigned long max_ptes_swap; err = kstrtoul(buf, 10, &max_ptes_swap); - if (err || max_ptes_swap > KHUGEPAGED_MAX_PTES_LIMIT) + if (err || max_ptes_swap > COLLAPSE_MAX_PTES_LIMIT) return -EINVAL; khugepaged_max_ptes_swap = max_ptes_swap; @@ -318,7 +265,7 @@ static ssize_t max_ptes_shared_store(struct kobject *kobj, unsigned long max_ptes_shared; err = kstrtoul(buf, 10, &max_ptes_shared); - if (err || max_ptes_shared > KHUGEPAGED_MAX_PTES_LIMIT) + if (err || max_ptes_shared > COLLAPSE_MAX_PTES_LIMIT) return -EINVAL; khugepaged_max_ptes_shared = max_ptes_shared; @@ -378,19 +325,19 @@ static unsigned int collapse_max_ptes_none(struct collapse_control *cc, if (is_pmd_order(order)) return max_ptes_none; /* - * for mTHP collapse with the sysctl value set to KHUGEPAGED_MAX_PTES_LIMIT, + * for mTHP collapse with the sysctl value set to COLLAPSE_MAX_PTES_LIMIT, * scale the maximum number of PTEs to the order of the collapse. */ - if (max_ptes_none == KHUGEPAGED_MAX_PTES_LIMIT) + if (max_ptes_none == COLLAPSE_MAX_PTES_LIMIT) return (1 << order) - 1; /* - * For mTHP collapse of values other than 0 or KHUGEPAGED_MAX_PTES_LIMIT, + * For mTHP collapse of values other than 0 or COLLAPSE_MAX_PTES_LIMIT, * emit a warning and return 0. */ if (max_ptes_none) pr_warn_once("mTHP collapse does not support max_ptes_none" " values other than 0 or %u, defaulting to 0.\n", - KHUGEPAGED_MAX_PTES_LIMIT); + COLLAPSE_MAX_PTES_LIMIT); return 0; } @@ -476,7 +423,7 @@ int __init khugepaged_init(void) return -ENOMEM; khugepaged_pages_to_scan = HPAGE_PMD_NR * 8; - khugepaged_max_ptes_none = KHUGEPAGED_MAX_PTES_LIMIT; + khugepaged_max_ptes_none = COLLAPSE_MAX_PTES_LIMIT; khugepaged_max_ptes_swap = HPAGE_PMD_NR / 8; khugepaged_max_ptes_shared = HPAGE_PMD_NR / 2; @@ -1601,8 +1548,8 @@ static enum scan_result mthp_collapse(struct mm_struct *mm, * any smaller order enabled. When at the smallest order * we must always move to the next offset. */ - if (order > KHUGEPAGED_MIN_MTHP_ORDER && - (enabled_orders & GENMASK(order - 1, 0))) { + if (order > COLLAPSE_MIN_MTHP_ORDER && + (enabled_orders & GENMASK(order - 1, 0))) { order--; continue; } @@ -1666,7 +1613,7 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, * is then checked again in mthp_collapse() for each attempted order. */ if (enabled_orders != BIT(HPAGE_PMD_ORDER)) - max_ptes_none = KHUGEPAGED_MAX_PTES_LIMIT; + max_ptes_none = COLLAPSE_MAX_PTES_LIMIT; pte = pte_offset_map_lock(mm, pmd, start_addr, &ptl); if (!pte) { From 3f1613973d5bddbf63b723c59880d71b41b72459 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:19 +0100 Subject: [PATCH 0864/1352] mm/collapse: state what a collapse may do in the policy Tests scattered through the collapse path decide what a collapse is allowed to do by asking whether khugepaged started it. Between them they settle: - which VMAs are eligible, and how hard to try for a folio; - how many empty, swapped-out or shared PTEs a window may contain, and whether a sub-PMD window is held to a stricter rule than a PMD; - whether a range has to look used, and whether a MADV_FREE'd page is left alone; - whether the PMD is mapped as part of the request, and whether dirty pages are worth writing back and retrying. None of those is a fact about khugepaged. Each is something the caller decided before asking, and the collapse code should not have to look up who called to find out. Add struct collapse_policy for the caller to fill: khugepaged from its own settings, MADV_COLLAPSE from the fact that a user asked explicitly. Every test becomes a read of a field, and cc->is_khugepaged goes, having no reader left. The PTE limits come as two sets, one for a PMD-sized window and one for anything smaller, so that the helpers pick a set for the order and read it. khugepaged takes no swapped-out or shared PTE into a sub-PMD window, and empty PTEs only when the knob says all or nothing; the warning for a knob value in between moves to where khugepaged fills its policy. The fields only one side reads say which: anon_ or file_. David Hildenbrand asked for both. khugepaged fills the policy once per scan pass, MADV_COLLAPSE once per call. That is the one change in behaviour. The max_ptes_* limits and the defrag setting behind the allocation mask are sampled once per pass rather than on every table. A table scanned early in a pass and one scanned late are then treated alike. collapse_file() also drops a NULL check on the collapse_control. It has one call site, reached only from collapse_single_pmd(), which dereferences cc unconditionally, so the check was already dead. Link: https://lore.kernel.org/20260928100630.21870-6-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Zi Yan Reviewed-by: Baolin Wang Tested-by: Baolin Wang Assisted-by: LLM Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- mm/collapse.h | 41 +++++++++++++- mm/khugepaged.c | 147 +++++++++++++++++++++++++----------------------- 2 files changed, 118 insertions(+), 70 deletions(-) diff --git a/mm/collapse.h b/mm/collapse.h index b115034d90187a..dcd11707195518 100644 --- a/mm/collapse.h +++ b/mm/collapse.h @@ -45,8 +45,47 @@ enum scan_result { SCAN_PAGE_DIRTY_OR_WRITEBACK, }; +/* How many PTEs of a window may be missing, swapped out or shared */ +struct collapse_limits { + /* Counted over a PMD-sized window; HPAGE_PMD_NR means "no limit" */ + unsigned int max_ptes_none; + unsigned int max_ptes_swap; + unsigned int max_ptes_shared; +}; + +/* What a collapse is allowed to do, decided by the caller that asks for it */ +struct collapse_policy { + /* Limits for a PMD-sized window */ + struct collapse_limits pmd; + + /* + * Limits for a smaller window. Its max_ptes_none is either 0 or + * COLLAPSE_MAX_PTES_LIMIT, the latter meaning all but one PTE of the + * window whatever its order; any other value counts as 0. + */ + struct collapse_limits sub_pmd; + + /* Leave clean lazyfree folios to reclaim rather than collapse them */ + bool anon_skip_lazyfree; + + /* Refuse an anonymous range with no sign of use */ + bool anon_require_referenced; + + /* Map the PMD over a file collapse instead of leaving it to a fault */ + bool file_install_pmd; + + /* Write dirty pages back and retry once instead of refusing them */ + bool file_writeback_dirty; + + /* How hard to try for a destination folio */ + gfp_t gfp; + + /* Which VMAs are eligible, as thp_vma_allowable_orders() spells it */ + enum tva_type tva_type; +}; + struct collapse_control { - bool is_khugepaged; + struct collapse_policy policy; /* Num pages scanned per node */ u32 node_load[MAX_NUMNODES]; diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 6c7ac9ebcd4b45..35b282ed39ff1d 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -314,30 +314,17 @@ static bool pte_none_or_zero(pte_t pte) static unsigned int collapse_max_ptes_none(struct collapse_control *cc, struct vm_area_struct *vma, unsigned int order) { - const unsigned int max_ptes_none = khugepaged_max_ptes_none; + unsigned int max_ptes_none; if (vma && userfaultfd_armed(vma)) return 0; - /* for MADV_COLLAPSE, allow any empty/shared zeropage PTEs */ - if (!cc->is_khugepaged) - return HPAGE_PMD_NR; - /* for PMD collapse, respect the user defined maximum */ if (is_pmd_order(order)) - return max_ptes_none; - /* - * for mTHP collapse with the sysctl value set to COLLAPSE_MAX_PTES_LIMIT, - * scale the maximum number of PTEs to the order of the collapse. - */ + return cc->policy.pmd.max_ptes_none; + + /* Below PMD order: all but one PTE of the window, or none */ + max_ptes_none = cc->policy.sub_pmd.max_ptes_none; if (max_ptes_none == COLLAPSE_MAX_PTES_LIMIT) return (1 << order) - 1; - /* - * For mTHP collapse of values other than 0 or COLLAPSE_MAX_PTES_LIMIT, - * emit a warning and return 0. - */ - if (max_ptes_none) - pr_warn_once("mTHP collapse does not support max_ptes_none" - " values other than 0 or %u, defaulting to 0.\n", - COLLAPSE_MAX_PTES_LIMIT); return 0; } @@ -353,20 +340,9 @@ static unsigned int collapse_max_ptes_none(struct collapse_control *cc, static unsigned int collapse_max_ptes_shared(struct collapse_control *cc, unsigned int order) { - /* - * For MADV_COLLAPSE, do not restrict the number of PTEs that map shared - * anonymous pages. - */ - if (!cc->is_khugepaged) - return HPAGE_PMD_NR; - /* - * for mTHP collapse do not allow collapsing anonymous memory pages that - * are shared between processes. - */ - if (!is_pmd_order(order)) - return 0; - /* for PMD collapse, respect the user defined maximum */ - return khugepaged_max_ptes_shared; + if (is_pmd_order(order)) + return cc->policy.pmd.max_ptes_shared; + return cc->policy.sub_pmd.max_ptes_shared; } /** @@ -381,17 +357,9 @@ static unsigned int collapse_max_ptes_shared(struct collapse_control *cc, static unsigned int collapse_max_ptes_swap(struct collapse_control *cc, unsigned int order) { - /* - * For MADV_COLLAPSE, do not restrict the number PTEs entries or - * pagecache entries that are non-present. - */ - if (!cc->is_khugepaged) - return HPAGE_PMD_NR; - /* for mTHP collapse do not allow any non-present PTEs or pagecache entries */ - if (!is_pmd_order(order)) - return 0; - /* for PMD collapse, respect the user defined maximum */ - return khugepaged_max_ptes_swap; + if (is_pmd_order(order)) + return cc->policy.pmd.max_ptes_swap; + return cc->policy.sub_pmd.max_ptes_swap; } int hugepage_madvise(struct vm_area_struct *vma, @@ -678,7 +646,7 @@ static enum scan_result __collapse_huge_page_isolate(struct vm_area_struct *vma, * If the vma has the VM_DROPPABLE flag, the collapse will * preserve the lazyfree property without needing to skip. */ - if (cc->is_khugepaged && !(vma->vm_flags & VM_DROPPABLE) && + if (cc->policy.anon_skip_lazyfree && !(vma->vm_flags & VM_DROPPABLE) && folio_test_lazyfree(folio) && !pte_dirty(pteval)) { result = SCAN_PAGE_LAZYFREE; goto out; @@ -767,12 +735,12 @@ static enum scan_result __collapse_huge_page_isolate(struct vm_area_struct *vma, if (folio_test_large(folio)) list_add_tail(&folio->lru, compound_pagelist); next: - if (cc->is_khugepaged && + if (cc->policy.anon_require_referenced && folio_pte_referenced(folio, vma, addr, pteval)) referenced++; } - if (unlikely(cc->is_khugepaged && !referenced)) { + if (unlikely(cc->policy.anon_require_referenced && !referenced)) { result = SCAN_LACK_REFERENCED_PAGE; } else { result = SCAN_SUCCEED; @@ -938,9 +906,7 @@ static void khugepaged_alloc_sleep(void) remove_wait_queue(&khugepaged_wait, &wait); } -static struct collapse_control khugepaged_collapse_control = { - .is_khugepaged = true, -}; +static struct collapse_control khugepaged_collapse_control; static bool collapse_scan_abort(int nid, struct collapse_control *cc) { @@ -976,6 +942,52 @@ static inline gfp_t alloc_hugepage_khugepaged_gfpmask(void) return khugepaged_defrag() ? GFP_TRANSHUGE : GFP_TRANSHUGE_LIGHT; } +/* khugepaged collapses on its own initiative, so it obeys its own settings */ +static void collapse_policy_khugepaged(struct collapse_policy *p) +{ + p->pmd.max_ptes_none = READ_ONCE(khugepaged_max_ptes_none); + p->pmd.max_ptes_swap = READ_ONCE(khugepaged_max_ptes_swap); + p->pmd.max_ptes_shared = READ_ONCE(khugepaged_max_ptes_shared); + + /* + * A sub-PMD window takes no swapped-out and no shared PTE: reading + * pages back or breaking CoW is not worth it for an mTHP. Empty PTEs + * it takes all or nothing, since anything in between would let one + * collapse feed the next. + */ + p->sub_pmd.max_ptes_none = p->pmd.max_ptes_none; + if (p->sub_pmd.max_ptes_none && + p->sub_pmd.max_ptes_none != COLLAPSE_MAX_PTES_LIMIT) + pr_warn_once("mTHP collapse does not support max_ptes_none values other than 0 or %u, defaulting to 0.\n", + COLLAPSE_MAX_PTES_LIMIT); + p->sub_pmd.max_ptes_swap = 0; + p->sub_pmd.max_ptes_shared = 0; + + p->anon_skip_lazyfree = true; + p->anon_require_referenced = true; + p->file_install_pmd = false; + p->file_writeback_dirty = false; + p->gfp = alloc_hugepage_khugepaged_gfpmask(); + p->tva_type = TVA_KHUGEPAGED; +} + +/* MADV_COLLAPSE was asked for explicitly, so it is not held to those */ +static void collapse_policy_madvise(struct collapse_policy *p) +{ + p->pmd.max_ptes_none = HPAGE_PMD_NR; + p->pmd.max_ptes_swap = HPAGE_PMD_NR; + p->pmd.max_ptes_shared = HPAGE_PMD_NR; + /* Never read: MADV_COLLAPSE collapses to PMD order only */ + p->sub_pmd = p->pmd; + + p->anon_skip_lazyfree = false; + p->anon_require_referenced = false; + p->file_install_pmd = true; + p->file_writeback_dirty = true; + p->gfp = GFP_TRANSHUGE; + p->tva_type = TVA_FORCED_COLLAPSE; +} + #ifdef CONFIG_NUMA static int collapse_find_target_node(struct collapse_control *cc) { @@ -1013,8 +1025,7 @@ static enum scan_result hugepage_vma_revalidate(struct mm_struct *mm, unsigned l struct collapse_control *cc, unsigned int order) { struct vm_area_struct *vma; - enum tva_type type = cc->is_khugepaged ? TVA_KHUGEPAGED : - TVA_FORCED_COLLAPSE; + enum tva_type type = cc->policy.tva_type; if (unlikely(collapse_test_exit_or_disable(mm))) return SCAN_ANY_PROCESS; @@ -1197,8 +1208,7 @@ static enum scan_result __collapse_huge_page_swapin(struct mm_struct *mm, static enum scan_result alloc_charge_folio(struct folio **foliop, struct mm_struct *mm, struct collapse_control *cc, unsigned int order) { - gfp_t gfp = (cc->is_khugepaged ? alloc_hugepage_khugepaged_gfpmask() : - GFP_TRANSHUGE); + gfp_t gfp = cc->policy.gfp; int node = collapse_find_target_node(cc); struct folio *folio; @@ -1581,7 +1591,7 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, const unsigned int max_ptes_shared = collapse_max_ptes_shared(cc, HPAGE_PMD_ORDER); const unsigned int max_ptes_swap = collapse_max_ptes_swap(cc, HPAGE_PMD_ORDER); unsigned int max_ptes_none = collapse_max_ptes_none(cc, vma, HPAGE_PMD_ORDER); - enum tva_type tva_flags = cc->is_khugepaged ? TVA_KHUGEPAGED : TVA_FORCED_COLLAPSE; + enum tva_type tva_flags = cc->policy.tva_type; pmd_t *pmd; pte_t *pte, *_pte, pteval; int i; @@ -1681,7 +1691,7 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, * If the vma has the VM_DROPPABLE flag, the collapse will * preserve the lazyfree property without needing to skip. */ - if (cc->is_khugepaged && !(vma->vm_flags & VM_DROPPABLE) && + if (cc->policy.anon_skip_lazyfree && !(vma->vm_flags & VM_DROPPABLE) && folio_test_lazyfree(folio) && !pte_dirty(pteval)) { result = SCAN_PAGE_LAZYFREE; failed_pfn = folio_pfn(folio); @@ -1747,13 +1757,13 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, goto out_unmap; } - if (cc->is_khugepaged && + if (cc->policy.anon_require_referenced && folio_pte_referenced(folio, vma, addr, pteval)) referenced++; } - if (cc->is_khugepaged && - (!referenced || - (unmapped && referenced < HPAGE_PMD_NR / 2))) { + if (cc->policy.anon_require_referenced && + (!referenced || + (unmapped && referenced < HPAGE_PMD_NR / 2))) { result = SCAN_LACK_REFERENCED_PAGE; } else { result = SCAN_SUCCEED; @@ -2605,11 +2615,11 @@ static enum scan_result collapse_file(struct mm_struct *mm, unsigned long addr, xas_unlock_irq(&xas); /* - * Remove pte page tables, so we can re-fault the page as huge. - * If MADV_COLLAPSE, adjust result to call try_collapse_pte_mapped_thp(). + * Remove pte page tables, so we can re-fault the page as huge. A + * caller that wants the PMD mapped now is told to go and do that. */ retract_page_tables(mapping, start); - if (cc && !cc->is_khugepaged) + if (cc->policy.file_install_pmd) result = SCAN_PTE_MAPPED_HUGEPAGE; folio_unlock(new_folio); @@ -2796,11 +2806,8 @@ static enum scan_result collapse_single_pmd(unsigned long addr, retry: result = collapse_scan_file(mm, addr, file, pgoff, cc); - /* - * For MADV_COLLAPSE, when encountering dirty pages, try to writeback, - * then retry the collapse one time. - */ - if (!cc->is_khugepaged && result == SCAN_PAGE_DIRTY_OR_WRITEBACK && + /* Dirty pages are worth a writeback and one more try, if asked for */ + if (cc->policy.file_writeback_dirty && result == SCAN_PAGE_DIRTY_OR_WRITEBACK && !triggered_wb && mapping_can_writeback(file->f_mapping)) { const loff_t lstart = (loff_t)pgoff << PAGE_SHIFT; const loff_t lend = lstart + HPAGE_PMD_SIZE - 1; @@ -2817,7 +2824,7 @@ static enum scan_result collapse_single_pmd(unsigned long addr, result = SCAN_ANY_PROCESS; else result = try_collapse_pte_mapped_thp(mm, addr, - !cc->is_khugepaged); + cc->policy.file_install_pmd); if (result == SCAN_PMD_MAPPED) result = SCAN_SUCCEED; mmap_read_unlock(mm); @@ -2966,6 +2973,8 @@ static void khugepaged_do_scan(struct collapse_control *cc) lru_add_drain_all(); + collapse_policy_khugepaged(&cc->policy); + cc->progress = 0; while (true) { cond_resched(); @@ -3193,7 +3202,7 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start, cc = kmalloc_obj(*cc); if (!cc) return -ENOMEM; - cc->is_khugepaged = false; + collapse_policy_madvise(&cc->policy); cc->progress = 0; lru_add_drain_all(); From 378e6d915aa4011eea927455778aa18109bbb753 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:20 +0100 Subject: [PATCH 0865/1352] mm/collapse: drop the collapse_possible() wrapper collapse_possible() only forwards to collapse_possible_orders() and turns its mask into a bool. Its three callers can test the mask themselves. No functional change. Link: https://lore.kernel.org/20260928100630.21870-7-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Zi Yan Reviewed-by: Baolin Wang Assisted-by: LLM Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- mm/khugepaged.c | 15 +++++---------- 1 file changed, 5 insertions(+), 10 deletions(-) diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 35b282ed39ff1d..63e999ff1fc51f 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -494,17 +494,11 @@ static unsigned long collapse_possible_orders(struct vm_area_struct *vma, return thp_vma_allowable_orders(vma, vm_flags, tva_flags, orders); } -static bool collapse_possible(struct vm_area_struct *vma, - vm_flags_t vm_flags, enum tva_type tva_flags) -{ - return collapse_possible_orders(vma, vm_flags, tva_flags); -} - void khugepaged_enter_vma(struct vm_area_struct *vma, vm_flags_t vm_flags) { - if (!mm_flags_test(MMF_VM_HUGEPAGE, vma->vm_mm) && hugepage_enabled() - && collapse_possible(vma, vm_flags, TVA_KHUGEPAGED)) + if (!mm_flags_test(MMF_VM_HUGEPAGE, vma->vm_mm) && hugepage_enabled() && + collapse_possible_orders(vma, vm_flags, TVA_KHUGEPAGED)) __khugepaged_enter(vma->vm_mm); } @@ -2878,7 +2872,8 @@ static void collapse_scan_mm_slot(unsigned int progress_max, cc->progress++; break; } - if (!collapse_possible(vma, vma->vm_flags, TVA_KHUGEPAGED)) { + if (!collapse_possible_orders(vma, vma->vm_flags, + TVA_KHUGEPAGED)) { cc->progress++; continue; } @@ -3190,7 +3185,7 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start, BUG_ON(vma->vm_start > start); BUG_ON(vma->vm_end < end); - if (!collapse_possible(vma, vma->vm_flags, TVA_FORCED_COLLAPSE)) + if (!collapse_possible_orders(vma, vma->vm_flags, TVA_FORCED_COLLAPSE)) return -EINVAL; hstart = ALIGN(start, HPAGE_PMD_SIZE); From 7010a488a77a438038322104e6f28140b1cee401 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:21 +0100 Subject: [PATCH 0866/1352] mm/collapse: name the per-table scan reset for what it resets collapse_control_init_scan() resets what one scan accumulates: the node load, the allocation nodemask and the eligible-PTE bitmap. It runs before every PTE table a scan is given, not once per control, so its name points at the wrong thing. Call it collapse_scan_reset(). Preparation for giving a control a real init and release, run once each for a whole series of scans. Two names a word apart would then stand for two different jobs. No functional change. Link: https://lore.kernel.org/20260928100630.21870-8-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Zi Yan Reviewed-by: Baolin Wang Assisted-by: LLM Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- mm/khugepaged.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 63e999ff1fc51f..a43c4d1b2ef6dd 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -532,7 +532,7 @@ void __khugepaged_exit(struct mm_struct *mm) } } -static void collapse_control_init_scan(struct collapse_control *cc) +static void collapse_scan_reset(struct collapse_control *cc) { memset(cc->node_load, 0, sizeof(cc->node_load)); nodes_clear(cc->alloc_nmask); @@ -1607,7 +1607,7 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, goto out; } - collapse_control_init_scan(cc); + collapse_scan_reset(cc); enabled_orders = collapse_possible_orders(vma, vma->vm_flags, tva_flags); @@ -2678,7 +2678,7 @@ static enum scan_result collapse_scan_file(struct mm_struct *mm, present = 0; swap = 0; - collapse_control_init_scan(cc); + collapse_scan_reset(cc); rcu_read_lock(); xas_for_each(&xas, folio, start + HPAGE_PMD_NR - 1) { if (xas_retry(&xas, folio)) From e6f88c38691ed21c88a6d7cad233f972f75a7f8e Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:22 +0100 Subject: [PATCH 0867/1352] mm/collapse: call collapse_file() from collapse_single_pmd() collapse_scan_file() reads the page cache to decide whether a table is worth collapsing and, when it is, calls collapse_file() itself. The caller cannot get between the decision and the collapse. Move the collapse_file() call up into collapse_single_pmd(), so the scan stops at the decision. Two things change with it. The writeback retry re-runs collapse_file() alone instead of rescanning first; collapse_file() repeats the scan's checks under the page cache lock anyway. And mm_khugepaged_scan_file fires before the collapse, so for an accepted table its status reads SCAN_SUCCEED; what the collapse made of the table is for mm_khugepaged_collapse_file to report. Preparation for splitting a collapse into a scan under mmap_lock and a run without it. The file scan has to stop where the anonymous one will. Link: https://lore.kernel.org/20260928100630.21870-9-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Acked-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- mm/khugepaged.c | 21 +++++++++++++-------- 1 file changed, 13 insertions(+), 8 deletions(-) diff --git a/mm/khugepaged.c b/mm/khugepaged.c index a43c4d1b2ef6dd..e6a89609650b76 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -2760,13 +2760,9 @@ static enum scan_result collapse_scan_file(struct mm_struct *mm, else cc->progress += HPAGE_PMD_NR; - if (result == SCAN_SUCCEED) { - if (present < HPAGE_PMD_NR - max_ptes_none) { - result = SCAN_EXCEED_NONE_PTE; - count_vm_event(THP_SCAN_EXCEED_NONE_PTE); - } else { - result = collapse_file(mm, addr, file, start, cc); - } + if (result == SCAN_SUCCEED && present < HPAGE_PMD_NR - max_ptes_none) { + result = SCAN_EXCEED_NONE_PTE; + count_vm_event(THP_SCAN_EXCEED_NONE_PTE); } trace_mm_khugepaged_scan_file(mm, failed_pfn, file, present, swap, result); @@ -2797,8 +2793,16 @@ static enum scan_result collapse_single_pmd(unsigned long addr, mmap_read_unlock(mm); *lock_dropped = true; -retry: + + /* + * SCAN_PTE_MAPPED_HUGEPAGE is work too: the page cache already holds + * the PMD folio, and only the PTE table is left to retract. + */ result = collapse_scan_file(mm, addr, file, pgoff, cc); + if (result != SCAN_SUCCEED) + goto put; +retry: + result = collapse_file(mm, addr, file, pgoff, cc); /* Dirty pages are worth a writeback and one more try, if asked for */ if (cc->policy.file_writeback_dirty && result == SCAN_PAGE_DIRTY_OR_WRITEBACK && @@ -2810,6 +2814,7 @@ static enum scan_result collapse_single_pmd(unsigned long addr, triggered_wb = true; goto retry; } +put: fput(file); if (result == SCAN_PTE_MAPPED_HUGEPAGE) { From 852a54ed1c5d7e284e1cb718fc211ed0ad67dc46 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:23 +0100 Subject: [PATCH 0868/1352] mm/collapse: separate scanning a PTE table from collapsing it A collapse is two jobs. One reads a PTE table under mmap_lock and decides whether the range is worth collapsing. The other allocates, isolates, copies and flushes, and wants the lock given up first. collapse_single_pmd() did both, so the boundary between them was somewhere in the middle of a function. Give each half its own function: - collapse_scan_pmd() scans one table. The anonymous scan that used to carry that name keeps its body as collapse_scan_anon_pmd(), and collapse_scan_pmd() is now the entry that picks the anonymous or the file side. - collapse_run_pmd() does the collapse the scan asked for, and is handed what the scan returned. SCAN_SUCCEED means there is something to collapse. SCAN_PTE_MAPPED_HUGEPAGE means the page cache already holds the PMD folio and only the PTE table is left to retract. Both are work for the run; anything else is why there is nothing to do. collapse_single_pmd() is now the two of them with the mmap_lock drop in between, so its callers see what they saw before. collapse_control_init() sets a control up before its first scan. What the scan found and the run needs travels in collapse_control. For an anonymous table that is the orders and the referenced and swapped-out counts, which mthp_collapse() and collapse_huge_page() now read from there instead of taking as arguments. For a file it is the file itself and the offset in it: a file collapse works on the page cache and never sees a VMA, so the scan takes the reference while it still has one and the run gives it back. The file scan moves under mmap_lock with the anonymous one, where before the lock was given up first. The lock is now held over the page cache walk, an RCU walk over one table's worth of slots with no PTL, and taken fewer times. collapse_scan_mm_slot() ends its walk whenever the lock was dropped, so a refused file table used to cost khugepaged an unlock, a trip back through khugepaged_do_scan(), a relock and a VMA lookup. Now only a table that goes on to be collapsed does. Link: https://lore.kernel.org/20260928100630.21870-10-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: Zi Yan Assisted-by: LLM Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- mm/collapse.h | 14 +++++ mm/khugepaged.c | 141 ++++++++++++++++++++++++++++++++---------------- 2 files changed, 108 insertions(+), 47 deletions(-) diff --git a/mm/collapse.h b/mm/collapse.h index dcd11707195518..ca7b367c89cb39 100644 --- a/mm/collapse.h +++ b/mm/collapse.h @@ -98,6 +98,20 @@ struct collapse_control { /* Each bit marks a PTE the scan accepted as a collapse source */ DECLARE_BITMAP(eligible_ptes, MAX_PTRS_PER_PTE); + + /* + * What a scan found and the run after it needs. Live only between the + * two, and read by nobody else. + * + * The file side takes a reference while it still has the VMA, since a + * file collapse works on the page cache and never sees one; the run is + * what gives it back. + */ + unsigned long scan_orders; + int scan_referenced; + int scan_unmapped; + struct file *scan_file; + pgoff_t scan_pgoff; }; #endif /* __MM_COLLAPSE_H */ diff --git a/mm/khugepaged.c b/mm/khugepaged.c index e6a89609650b76..87bbba6ce59ab9 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -1252,8 +1252,8 @@ static pgtable_t alloc_deposit_pte_table(struct mm_struct *mm) * while allocating a THP, as that could trigger direct reclaim/compaction. * Note that the VMA must be rechecked after grabbing the mmap_lock again. */ -static enum scan_result collapse_huge_page(struct mm_struct *mm, unsigned long start_addr, - int referenced, int unmapped, struct collapse_control *cc, +static enum scan_result collapse_huge_page(struct mm_struct *mm, + unsigned long start_addr, struct collapse_control *cc, unsigned int order) { const unsigned long pmd_addr = start_addr & HPAGE_PMD_MASK; @@ -1300,14 +1300,14 @@ static enum scan_result collapse_huge_page(struct mm_struct *mm, unsigned long s goto out_nolock; } - if (unmapped) { + if (cc->scan_unmapped) { /* * __collapse_huge_page_swapin() will return with mmap_lock * released when it fails. So we jump out_nolock directly in * that case. Continuing to collapse causes inconsistency. */ result = __collapse_huge_page_swapin(mm, vma, start_addr, pmd, - referenced, order); + cc->scan_referenced, order); if (result != SCAN_SUCCEED) goto out_nolock; } @@ -1476,9 +1476,8 @@ static unsigned int max_order_from_offset(unsigned int offset) * If a collapse is permitted, we attempt to collapse the PTE range into a * mTHP. */ -static enum scan_result mthp_collapse(struct mm_struct *mm, - unsigned long address, int referenced, int unmapped, - struct collapse_control *cc, unsigned long enabled_orders) +static enum scan_result mthp_collapse(struct mm_struct *mm, unsigned long address, + struct collapse_control *cc) { unsigned int nr_eligible_ptes, nr_ptes, max_ptes_none; enum scan_result last_result = SCAN_FAIL; @@ -1491,7 +1490,7 @@ static enum scan_result mthp_collapse(struct mm_struct *mm, while (offset < HPAGE_PMD_NR) { nr_ptes = 1UL << order; - if (!test_bit(order, &enabled_orders)) + if (!test_bit(order, &cc->scan_orders)) goto next_order; max_ptes_none = collapse_max_ptes_none(cc, NULL, order); @@ -1499,19 +1498,18 @@ static enum scan_result mthp_collapse(struct mm_struct *mm, offset + nr_ptes); /* - * Swap PTEs accepted during the scan are counted in @unmapped, - * not in cc->eligible_ptes. Account them for the PMD-order - * candidate. + * Swap PTEs accepted during the scan are counted in + * cc->scan_unmapped, not in cc->eligible_ptes. Account them for + * the PMD-order candidate. */ if (is_pmd_order(order)) - nr_eligible_ptes += unmapped; + nr_eligible_ptes += cc->scan_unmapped; if (nr_eligible_ptes >= nr_ptes - max_ptes_none) { enum scan_result ret; collapse_address = address + offset * PAGE_SIZE; - ret = collapse_huge_page(mm, collapse_address, referenced, - unmapped, cc, order); + ret = collapse_huge_page(mm, collapse_address, cc, order); switch (ret) { /* Cases where we continue to next collapse candidate */ @@ -1553,7 +1551,7 @@ static enum scan_result mthp_collapse(struct mm_struct *mm, * we must always move to the next offset. */ if (order > COLLAPSE_MIN_MTHP_ORDER && - (enabled_orders & GENMASK(order - 1, 0))) { + (cc->scan_orders & GENMASK(order - 1, 0))) { order--; continue; } @@ -1578,14 +1576,14 @@ static enum scan_result mthp_collapse(struct mm_struct *mm, return last_result; } -static enum scan_result collapse_scan_pmd(struct mm_struct *mm, - struct vm_area_struct *vma, unsigned long start_addr, - bool *lock_dropped, struct collapse_control *cc) +static enum scan_result collapse_scan_anon_pmd(struct vm_area_struct *vma, + unsigned long start_addr, struct collapse_control *cc) { const unsigned int max_ptes_shared = collapse_max_ptes_shared(cc, HPAGE_PMD_ORDER); const unsigned int max_ptes_swap = collapse_max_ptes_swap(cc, HPAGE_PMD_ORDER); unsigned int max_ptes_none = collapse_max_ptes_none(cc, vma, HPAGE_PMD_ORDER); enum tva_type tva_flags = cc->policy.tva_type; + struct mm_struct *mm = vma->vm_mm; pmd_t *pmd; pte_t *pte, *_pte, pteval; int i; @@ -1765,12 +1763,9 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, out_unmap: pte_unmap_unlock(pte, ptl); if (result == SCAN_SUCCEED) { - /* collapse_huge_page() expects the lock to be dropped before calling */ - mmap_read_unlock(mm); - result = mthp_collapse(mm, start_addr, referenced, - unmapped, cc, enabled_orders); - /* mmap_lock was released above, set lock_dropped */ - *lock_dropped = true; + cc->scan_orders = enabled_orders; + cc->scan_referenced = referenced; + cc->scan_unmapped = unmapped; } out: trace_mm_khugepaged_scan_pmd(mm, failed_pfn, referenced, @@ -2765,41 +2760,67 @@ static enum scan_result collapse_scan_file(struct mm_struct *mm, count_vm_event(THP_SCAN_EXCEED_NONE_PTE); } - trace_mm_khugepaged_scan_file(mm, failed_pfn, file, present, swap, result); + trace_mm_khugepaged_scan_file(mm, failed_pfn, file, present, swap, + result); return result; } -/* - * Try to collapse a single PMD starting at a PMD aligned addr, and return - * the results. - */ -static enum scan_result collapse_single_pmd(unsigned long addr, - struct vm_area_struct *vma, bool *lock_dropped, - struct collapse_control *cc) +static void collapse_control_init(struct collapse_control *cc) +{ + cc->progress = 0; + cc->scan_file = NULL; +} + +static enum scan_result collapse_scan_pmd(struct vm_area_struct *vma, + unsigned long addr, struct collapse_control *cc) { - struct mm_struct *mm = vma->vm_mm; - bool triggered_wb = false; enum scan_result result; - struct file *file; pgoff_t pgoff; - mmap_assert_locked(mm); + mmap_assert_locked(vma->vm_mm); + /* Whatever the last scan found has to have been run by now */ + if (WARN_ON_ONCE(cc->scan_file)) { + fput(cc->scan_file); + cc->scan_file = NULL; + } if (vma_is_anonymous(vma)) - return collapse_scan_pmd(mm, vma, addr, lock_dropped, cc); + return collapse_scan_anon_pmd(vma, addr, cc); - file = get_file(vma->vm_file); pgoff = linear_page_index(vma, addr); - - mmap_read_unlock(mm); - *lock_dropped = true; - + result = collapse_scan_file(vma->vm_mm, addr, vma->vm_file, pgoff, cc); /* * SCAN_PTE_MAPPED_HUGEPAGE is work too: the page cache already holds - * the PMD folio, and only the PTE table is left to retract. + * the PMD folio, and retracting the PTE table is the run's job. */ - result = collapse_scan_file(mm, addr, file, pgoff, cc); - if (result != SCAN_SUCCEED) + if (result != SCAN_SUCCEED && result != SCAN_PTE_MAPPED_HUGEPAGE) + return result; + + /* + * A file collapse works on the page cache and never sees a VMA, so take + * what it needs from this one while it is still here. + */ + cc->scan_file = get_file(vma->vm_file); + cc->scan_pgoff = pgoff; + return result; +} + +static enum scan_result collapse_run_pmd(struct mm_struct *mm, + unsigned long addr, enum scan_result result, + struct collapse_control *cc) +{ + struct file *file = cc->scan_file; + bool triggered_wb = false; + pgoff_t pgoff; + + if (!file) + return mthp_collapse(mm, addr, cc); + + cc->scan_file = NULL; + pgoff = cc->scan_pgoff; + + /* The scan found the PMD folio in place: nothing to collapse */ + if (result == SCAN_PTE_MAPPED_HUGEPAGE) goto put; retry: result = collapse_file(mm, addr, file, pgoff, cc); @@ -2817,6 +2838,10 @@ static enum scan_result collapse_single_pmd(unsigned long addr, put: fput(file); + /* + * A PMD folio is in the page cache, whether the collapse just put it + * there or found it: retract the PTE table, and map the PMD if asked. + */ if (result == SCAN_PTE_MAPPED_HUGEPAGE) { mmap_read_lock(mm); if (collapse_test_exit_or_disable(mm)) @@ -2831,6 +2856,28 @@ static enum scan_result collapse_single_pmd(unsigned long addr, return result; } +/* + * Try to collapse a single PMD starting at a PMD aligned addr, and return + * the results. + */ +static enum scan_result collapse_single_pmd(unsigned long addr, + struct vm_area_struct *vma, bool *lock_dropped, + struct collapse_control *cc) +{ + struct mm_struct *mm = vma->vm_mm; + enum scan_result result; + + result = collapse_scan_pmd(vma, addr, cc); + if (result != SCAN_SUCCEED && result != SCAN_PTE_MAPPED_HUGEPAGE) + return result; + + /* The collapse takes its own locks, so give this up */ + mmap_read_unlock(mm); + *lock_dropped = true; + + return collapse_run_pmd(mm, addr, result, cc); +} + static void collapse_scan_mm_slot(unsigned int progress_max, enum scan_result *result, struct collapse_control *cc) __releases(&khugepaged_mm_lock) @@ -2973,9 +3020,9 @@ static void khugepaged_do_scan(struct collapse_control *cc) lru_add_drain_all(); + collapse_control_init(cc); collapse_policy_khugepaged(&cc->policy); - cc->progress = 0; while (true) { cond_resched(); @@ -3202,8 +3249,8 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start, cc = kmalloc_obj(*cc); if (!cc) return -ENOMEM; + collapse_control_init(cc); collapse_policy_madvise(&cc->policy); - cc->progress = 0; lru_add_drain_all(); From 629bb91745ffba1e2e021dd36fa90e1ae9e14298 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:24 +0100 Subject: [PATCH 0869/1352] mm/collapse: open-code collapse_single_pmd() in its two callers A scan and a collapse want different things from mmap_lock. The scan reads one PTE table under the lock the caller holds, refuses most of the time, and the caller moves on to the next table without letting go. The collapse allocates, may sleep in writeback and takes the lock for write itself, so the lock it is handed is of no use to it. collapse_single_pmd() kept that boundary inside itself. It dropped the lock on some paths and not others, and reported which by way of a bool its callers had to carry along and then act on. Both callers already act on a drop, khugepaged by ending its walk and madvise_collapse() by looking its VMA up again. The code that has to know is not the code that does it. Open-code it in the two callers. Each scans under the lock it already holds and, when the scan found work, gives the lock up before running the collapse. The scan then has one rule, called locked and returning locked, and the run another, called unlocked. Nothing is left to report, so khugepaged's lock_dropped and madvise_collapse()'s mmap_unlocked both go. The engine never touches a lock it did not take, and how a caller locks its scan is the caller's business alone. khugepaged's walk carries on to the next table while the scan keeps refusing, and ends once a collapse has taken the lock from under it. madvise_collapse() re-finds its VMA after a collapse, which it did before, and now uses a NULL vma to say that it has to. It still reports the drop to its own caller, from the line that does it. Preparation for moving madvise_collapse() out of khugepaged.c: what it needs from the engine is then two calls with one lock rule each. The lock is given up and taken again at the same points as before. No functional change. Link: https://lore.kernel.org/20260928100630.21870-11-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Baolin Wang Assisted-by: LLM Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- mm/khugepaged.c | 103 +++++++++++++++++++++++------------------------- 1 file changed, 50 insertions(+), 53 deletions(-) diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 87bbba6ce59ab9..1b35778335bead 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -2856,28 +2856,6 @@ static enum scan_result collapse_run_pmd(struct mm_struct *mm, return result; } -/* - * Try to collapse a single PMD starting at a PMD aligned addr, and return - * the results. - */ -static enum scan_result collapse_single_pmd(unsigned long addr, - struct vm_area_struct *vma, bool *lock_dropped, - struct collapse_control *cc) -{ - struct mm_struct *mm = vma->vm_mm; - enum scan_result result; - - result = collapse_scan_pmd(vma, addr, cc); - if (result != SCAN_SUCCEED && result != SCAN_PTE_MAPPED_HUGEPAGE) - return result; - - /* The collapse takes its own locks, so give this up */ - mmap_read_unlock(mm); - *lock_dropped = true; - - return collapse_run_pmd(mm, addr, result, cc); -} - static void collapse_scan_mm_slot(unsigned int progress_max, enum scan_result *result, struct collapse_control *cc) __releases(&khugepaged_mm_lock) @@ -2940,7 +2918,7 @@ static void collapse_scan_mm_slot(unsigned int progress_max, VM_BUG_ON(khugepaged_scan.address & ~HPAGE_PMD_MASK); while (khugepaged_scan.address < hend) { - bool lock_dropped = false; + unsigned long addr; cond_resched(); if (unlikely(collapse_test_exit_or_disable(mm))) @@ -2950,23 +2928,30 @@ static void collapse_scan_mm_slot(unsigned int progress_max, khugepaged_scan.address + HPAGE_PMD_SIZE > hend); - *result = collapse_single_pmd(khugepaged_scan.address, - vma, &lock_dropped, cc); - if (*result == SCAN_SUCCEED) - khugepaged_pages_collapsed++; + addr = khugepaged_scan.address; /* move to next address */ khugepaged_scan.address += HPAGE_PMD_SIZE; - if (lock_dropped) - /* - * We released mmap_lock so break loop. Note - * that we drop mmap_lock before all hugepage - * allocations, so if allocation fails, we are - * guaranteed to break here and report the - * correct result back to caller. - */ - goto breakouterloop_mmap_lock; - if (cc->progress >= progress_max) - goto breakouterloop; + + *result = collapse_scan_pmd(vma, addr, cc); + /* Nothing to do here, and the lock is still ours */ + if (*result != SCAN_SUCCEED && + *result != SCAN_PTE_MAPPED_HUGEPAGE) { + if (cc->progress >= progress_max) + goto breakouterloop; + continue; + } + + /* + * A collapse takes its own locks and is slow enough + * that a writer should not wait behind it, so give the + * lock up. That ends this walk: vma and the mm are + * whatever the collapse leaves them. + */ + mmap_read_unlock(mm); + *result = collapse_run_pmd(mm, addr, *result, cc); + if (*result == SCAN_SUCCEED) + khugepaged_pages_collapsed++; + goto breakouterloop_mmap_lock; } } breakouterloop: @@ -3232,7 +3217,6 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start, unsigned long hstart, hend, addr; enum scan_result last_fail = SCAN_FAIL; int thps = 0; - bool mmap_unlocked = false; BUG_ON(vma->vm_start > start); BUG_ON(vma->vm_end < end); @@ -3255,25 +3239,40 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start, lru_add_drain_all(); for (addr = hstart; addr < hend; addr += HPAGE_PMD_SIZE) { - enum scan_result result = SCAN_FAIL; + struct vm_area_struct *found; + enum scan_result result; - if (mmap_unlocked) { + /* + * A collapse gives the lock up, so the VMA has to be found + * again after one: it can shrink while nothing is held. A scan + * that finds nothing to collapse leaves the lock alone, so a + * range that is already collapsed walks on without relocking. + */ + if (!vma) { cond_resched(); mmap_read_lock(mm); - mmap_unlocked = false; - *lock_dropped = true; - result = hugepage_vma_revalidate(mm, addr, false, &vma, + result = hugepage_vma_revalidate(mm, addr, false, &found, cc, HPAGE_PMD_ORDER); if (result != SCAN_SUCCEED) { last_fail = result; - goto out_nolock; + goto out_locked; } - + vma = found; hend = min(hend, vma->vm_end & HPAGE_PMD_MASK); } - result = collapse_single_pmd(addr, vma, &mmap_unlocked, cc); + result = collapse_scan_pmd(vma, addr, cc); + /* Nothing to do here, and the lock is still ours */ + if (result != SCAN_SUCCEED && result != SCAN_PTE_MAPPED_HUGEPAGE) + goto tally; + /* The collapse takes its own locks, so give this up */ + mmap_read_unlock(mm); + *lock_dropped = true; + vma = NULL; + + result = collapse_run_pmd(mm, addr, result, cc); +tally: switch (result) { case SCAN_SUCCEED: case SCAN_PMD_MAPPED: @@ -3295,17 +3294,15 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start, default: last_fail = result; /* Other error, exit */ - goto out_maybelock; + goto out; } } -out_maybelock: +out: /* Caller expects us to hold mmap_lock on return */ - if (mmap_unlocked) { - *lock_dropped = true; + if (!vma) mmap_read_lock(mm); - } -out_nolock: +out_locked: mmap_assert_locked(mm); kfree(cc); From 9f49b5e1dc8ab0c9dac8557732861a82bb7143a7 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:25 +0100 Subject: [PATCH 0870/1352] mm/collapse: work out the orders a VMA allows once per VMA The scan asked collapse_possible_orders() for every PTE table, for an answer that is a property of the VMA. Both callers walk a VMA a table at a time, so let them work it out once and pass the mask in. It is only good while the lock that produced it is held, so madvise_collapse() takes it again after every collapse. The mask is then sampled once per VMA rather than once per table. A thp enabled knob written during a walk takes effect one VMA later, and cannot widen a collapse: hugepage_vma_revalidate() tests the order again under the lock the collapse retakes. Link: https://lore.kernel.org/20260928100630.21870-12-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Baolin Wang Assisted-by: LLM Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- mm/khugepaged.c | 32 ++++++++++++++++++-------------- 1 file changed, 18 insertions(+), 14 deletions(-) diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 1b35778335bead..c961b8d121f572 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -1577,12 +1577,12 @@ static enum scan_result mthp_collapse(struct mm_struct *mm, unsigned long addres } static enum scan_result collapse_scan_anon_pmd(struct vm_area_struct *vma, - unsigned long start_addr, struct collapse_control *cc) + unsigned long start_addr, struct collapse_control *cc, + unsigned long enabled_orders) { const unsigned int max_ptes_shared = collapse_max_ptes_shared(cc, HPAGE_PMD_ORDER); const unsigned int max_ptes_swap = collapse_max_ptes_swap(cc, HPAGE_PMD_ORDER); unsigned int max_ptes_none = collapse_max_ptes_none(cc, vma, HPAGE_PMD_ORDER); - enum tva_type tva_flags = cc->policy.tva_type; struct mm_struct *mm = vma->vm_mm; pmd_t *pmd; pte_t *pte, *_pte, pteval; @@ -1593,7 +1593,6 @@ static enum scan_result collapse_scan_anon_pmd(struct vm_area_struct *vma, struct folio *folio = NULL; unsigned long failed_pfn = -1; unsigned long addr; - unsigned long enabled_orders; spinlock_t *ptl; int node = NUMA_NO_NODE, unmapped = 0; @@ -1607,8 +1606,6 @@ static enum scan_result collapse_scan_anon_pmd(struct vm_area_struct *vma, collapse_scan_reset(cc); - enabled_orders = collapse_possible_orders(vma, vma->vm_flags, tva_flags); - /* * If PMD is the only enabled order, enforce max_ptes_none, otherwise * scan all pages to populate the bitmap for mTHP collapse. The bitmap @@ -2772,7 +2769,8 @@ static void collapse_control_init(struct collapse_control *cc) } static enum scan_result collapse_scan_pmd(struct vm_area_struct *vma, - unsigned long addr, struct collapse_control *cc) + unsigned long addr, struct collapse_control *cc, + unsigned long orders) { enum scan_result result; pgoff_t pgoff; @@ -2785,7 +2783,7 @@ static enum scan_result collapse_scan_pmd(struct vm_area_struct *vma, } if (vma_is_anonymous(vma)) - return collapse_scan_anon_pmd(vma, addr, cc); + return collapse_scan_anon_pmd(vma, addr, cc, orders); pgoff = linear_page_index(vma, addr); result = collapse_scan_file(vma->vm_mm, addr, vma->vm_file, pgoff, cc); @@ -2895,15 +2893,17 @@ static void collapse_scan_mm_slot(unsigned int progress_max, vma_iter_init(&vmi, mm, khugepaged_scan.address); for_each_vma(vmi, vma) { - unsigned long hstart, hend; + unsigned long hstart, hend, orders; cond_resched(); if (unlikely(collapse_test_exit_or_disable(mm))) { cc->progress++; break; } - if (!collapse_possible_orders(vma, vma->vm_flags, - TVA_KHUGEPAGED)) { + /* One mask for the whole VMA */ + orders = collapse_possible_orders(vma, vma->vm_flags, + TVA_KHUGEPAGED); + if (!orders) { cc->progress++; continue; } @@ -2932,7 +2932,7 @@ static void collapse_scan_mm_slot(unsigned int progress_max, /* move to next address */ khugepaged_scan.address += HPAGE_PMD_SIZE; - *result = collapse_scan_pmd(vma, addr, cc); + *result = collapse_scan_pmd(vma, addr, cc, orders); /* Nothing to do here, and the lock is still ours */ if (*result != SCAN_SUCCEED && *result != SCAN_PTE_MAPPED_HUGEPAGE) { @@ -3214,14 +3214,16 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start, { struct collapse_control *cc; struct mm_struct *mm = vma->vm_mm; - unsigned long hstart, hend, addr; + unsigned long hstart, hend, addr, orders; enum scan_result last_fail = SCAN_FAIL; int thps = 0; BUG_ON(vma->vm_start > start); BUG_ON(vma->vm_end < end); - if (!collapse_possible_orders(vma, vma->vm_flags, TVA_FORCED_COLLAPSE)) + orders = collapse_possible_orders(vma, vma->vm_flags, + TVA_FORCED_COLLAPSE); + if (!orders) return -EINVAL; hstart = ALIGN(start, HPAGE_PMD_SIZE); @@ -3259,9 +3261,11 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start, } vma = found; hend = min(hend, vma->vm_end & HPAGE_PMD_MASK); + orders = collapse_possible_orders(vma, vma->vm_flags, + TVA_FORCED_COLLAPSE); } - result = collapse_scan_pmd(vma, addr, cc); + result = collapse_scan_pmd(vma, addr, cc, orders); /* Nothing to do here, and the lock is still ours */ if (result != SCAN_SUCCEED && result != SCAN_PTE_MAPPED_HUGEPAGE) goto tally; From d61162d14e65f8627c37e46ac6a8d845c21d71eb Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:26 +0100 Subject: [PATCH 0871/1352] mm/collapse: declare the collapse interface in collapse.h A collapse takes three calls: - collapse_control_init() - set up the control a caller carries; - collapse_scan_pmd() - scan one PTE table, under mmap_lock; - collapse_run_pmd() - collapse what the scan found, no mmap_lock. All three are static in khugepaged.c, as are collapse_possible_orders(), which says what a VMA allows, and the revalidate a caller needs once a collapse has given the mmap_lock up. No other file can ask for a collapse without them. Declare them in collapse.h, with a comment stating the order they are called in and who holds the lock over each step. Each function gets a kerneldoc comment where it is defined: what it takes, what it does, and the lock state on entry and exit. hugepage_vma_revalidate() becomes collapse_vma_revalidate(): it is part of what a collapse offers now, not a helper of the daemon. Preparation for implementing MADV_COLLAPSE in madvise.c. No functional change. Link: https://lore.kernel.org/20260928100630.21870-13-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: Zi Yan Assisted-by: LLM Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- mm/collapse.h | 45 ++++++++++++++++++++++++++ mm/khugepaged.c | 86 ++++++++++++++++++++++++++++++++++++++++--------- 2 files changed, 115 insertions(+), 16 deletions(-) diff --git a/mm/collapse.h b/mm/collapse.h index ca7b367c89cb39..b19351fbe19b94 100644 --- a/mm/collapse.h +++ b/mm/collapse.h @@ -114,4 +114,49 @@ struct collapse_control { pgoff_t scan_pgoff; }; +/* Which orders a VMA may collapse to, zero when it may not collapse at all */ +unsigned long collapse_possible_orders(struct vm_area_struct *vma, + vm_flags_t vm_flags, enum tva_type tva_flags); + +/* + * A caller states what it allows in cc->policy and then hands over one PTE + * table's worth of a VMA at a time: + * + * collapse_control_init(cc) once, before the first table + * collapse_scan_pmd(vma, addr, ...) per table + * collapse_run_pmd(mm, addr, result, cc) when a scan found work + * + * The caller holds mmap_lock for reading over the scan and passes an address + * within @vma, aligned to the PTE table to scan. + * + * The scan returns with that lock still held. Almost every table it is + * offered has nothing to collapse, so a caller walks a whole VMA under the one + * lock it took to get there. SCAN_SUCCEED means there is + * something to collapse. SCAN_PTE_MAPPED_HUGEPAGE means the page cache + * already holds the PMD folio and only the PTE table is left to retract. + * Both are work for the run, which is handed what the scan returned; anything + * else is why there is nothing to do. + * + * The run is called without the lock and returns without it, taking what it + * needs in between: what it does -- allocate, isolate, copy, flush -- is slow + * enough that a writer would wait behind it. The caller gives the lock up + * first, and with it @vma and anything derived under it, so a caller carrying + * on has to look up again with collapse_vma_revalidate(). The run revalidates + * for itself rather than trusting what the scan saw. + * + * A scan that found something has to be run: the file side takes a reference on + * the file while it still has the VMA to take it from, and the run is what + * gives it back. + */ +void collapse_control_init(struct collapse_control *cc); +enum scan_result collapse_scan_pmd(struct vm_area_struct *vma, + unsigned long addr, struct collapse_control *cc, + unsigned long orders); +enum scan_result collapse_run_pmd(struct mm_struct *mm, unsigned long addr, + enum scan_result result, struct collapse_control *cc); +enum scan_result collapse_vma_revalidate(struct mm_struct *mm, + unsigned long address, bool expect_anon, + struct vm_area_struct **vmap, struct collapse_control *cc, + unsigned int order); + #endif /* __MM_COLLAPSE_H */ diff --git a/mm/khugepaged.c b/mm/khugepaged.c index c961b8d121f572..86b591ec2c492b 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -476,11 +476,18 @@ void __khugepaged_enter(struct mm_struct *mm) wake_up_interruptible(&khugepaged_wait); } -/* - * Check what orders are possible based on the vma and collapse type. - * This is used to determine if mTHP collapse is a viable option. +/** + * collapse_possible_orders - which orders a VMA may collapse to + * @vma: the VMA + * @vm_flags: its flags, passed separately where they are about to change + * @tva_flags: who is asking, as thp_vma_allowable_orders() spells it + * + * khugepaged may collapse anonymous memory to any enabled order; everything + * else collapses to PMD order only. + * + * Return: the orders as a bitmask, zero when the VMA may not collapse at all. */ -static unsigned long collapse_possible_orders(struct vm_area_struct *vma, +unsigned long collapse_possible_orders(struct vm_area_struct *vma, vm_flags_t vm_flags, enum tva_type tva_flags) { unsigned long orders; @@ -1008,13 +1015,22 @@ static int collapse_find_target_node(struct collapse_control *cc) } #endif -/* - * If mmap_lock temporarily dropped, revalidate vma - * after taking the mmap_lock again. - * Returns enum scan_result value. +/** + * collapse_vma_revalidate - look a VMA up again after mmap_lock was dropped + * @mm: the mm + * @address: an address within the PTE table being collapsed + * @expect_anon: the collapse started on an anonymous VMA + * @vmap: the VMA found, if any + * @cc: the control, for the policy that says who is asking + * @order: the order the collapse is going for + * + * Called with mmap_lock held, for reading or writing, once it has been given up + * and taken back. The VMA has to span the whole PMD whatever @order is; with + * @expect_anon it also has to be anonymous and have an anon_vma. + * + * Return: SCAN_SUCCEED, or why a collapse of @order at @address is off. */ - -static enum scan_result hugepage_vma_revalidate(struct mm_struct *mm, unsigned long address, +enum scan_result collapse_vma_revalidate(struct mm_struct *mm, unsigned long address, bool expect_anon, struct vm_area_struct **vmap, struct collapse_control *cc, unsigned int order) { @@ -1287,7 +1303,7 @@ static enum scan_result collapse_huge_page(struct mm_struct *mm, } mmap_read_lock(mm); - result = hugepage_vma_revalidate(mm, pmd_addr, /*expect_anon=*/ true, + result = collapse_vma_revalidate(mm, pmd_addr, /*expect_anon=*/ true, &vma, cc, order); if (result != SCAN_SUCCEED) { mmap_read_unlock(mm); @@ -1322,7 +1338,7 @@ static enum scan_result collapse_huge_page(struct mm_struct *mm, * mmap_lock. */ mmap_write_lock(mm); - result = hugepage_vma_revalidate(mm, pmd_addr, /*expect_anon=*/ true, + result = collapse_vma_revalidate(mm, pmd_addr, /*expect_anon=*/ true, &vma, cc, order); if (result != SCAN_SUCCEED) goto out_up_write; @@ -2762,13 +2778,36 @@ static enum scan_result collapse_scan_file(struct mm_struct *mm, return result; } -static void collapse_control_init(struct collapse_control *cc) +/** + * collapse_control_init - set up a control before its first scan + * @cc: the control the caller carries across its scans + * + * cc->policy is the caller's to fill. + */ +void collapse_control_init(struct collapse_control *cc) { cc->progress = 0; cc->scan_file = NULL; } -static enum scan_result collapse_scan_pmd(struct vm_area_struct *vma, +/** + * collapse_scan_pmd - scan one PTE table for a collapse candidate + * @vma: the VMA the table belongs to + * @addr: start of the table, PMD aligned + * @cc: the caller's control + * @orders: the orders the caller allows for @vma + * + * Called with mmap_lock held for reading and returns with it still held. + * Almost every table it is offered has nothing to collapse, so a caller walks + * a whole VMA under the one lock it took to get there. + * + * Return: SCAN_SUCCEED when there is something to collapse; + * SCAN_PTE_MAPPED_HUGEPAGE when the page cache already holds the PMD folio and + * only the PTE table is left to retract. Both are work for collapse_run_pmd(), + * which is handed what the scan returned. Anything else is why there is + * nothing to do. + */ +enum scan_result collapse_scan_pmd(struct vm_area_struct *vma, unsigned long addr, struct collapse_control *cc, unsigned long orders) { @@ -2803,7 +2842,22 @@ static enum scan_result collapse_scan_pmd(struct vm_area_struct *vma, return result; } -static enum scan_result collapse_run_pmd(struct mm_struct *mm, +/** + * collapse_run_pmd - collapse the table a scan found work in + * @mm: the mm + * @addr: start of the table, as given to the scan + * @result: what the scan returned + * @cc: the control the scan ran with + * + * Called without mmap_lock and returns without it, taking what it needs in + * between: what it does -- allocate, isolate, copy, flush -- is slow enough + * that a writer would wait behind it. The caller gives the lock up first, + * and with it the VMA and anything derived under it. The run revalidates for + * itself rather than trusting what the scan saw. + * + * Return: what the collapse made of the table. + */ +enum scan_result collapse_run_pmd(struct mm_struct *mm, unsigned long addr, enum scan_result result, struct collapse_control *cc) { @@ -3253,7 +3307,7 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start, if (!vma) { cond_resched(); mmap_read_lock(mm); - result = hugepage_vma_revalidate(mm, addr, false, &found, + result = collapse_vma_revalidate(mm, addr, false, &found, cc, HPAGE_PMD_ORDER); if (result != SCAN_SUCCEED) { last_fail = result; From 461132911db59fde691a45427260c6ca1865a276 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:27 +0100 Subject: [PATCH 0872/1352] mm/collapse: implement MADV_COLLAPSE in madvise.c MADV_COLLAPSE is a madvise operation, but its implementation sat in khugepaged.c. The daemon's file therefore also held a syscall's worth of code that has nothing to do with the daemon: the walk over the user's range, the per-PMD loop, and the errno translation. Move it to madvise.c, among the operations it belongs with, along with the errno map and the policy it states for itself. It takes a struct madvise_behavior like every one of those operations, which is where the range, the VMA and the lock-dropped flag it used to be handed separately already live. It stays a caller of the interface khugepaged uses, so nothing about the collapse changes. The !CONFIG_TRANSPARENT_HUGEPAGE stub moves in with it. Link: https://lore.kernel.org/20260928100630.21870-14-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: Zi Yan Assisted-by: LLM Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- include/linux/huge_mm.h | 9 --- mm/khugepaged.c | 159 +------------------------------------ mm/madvise.c | 170 +++++++++++++++++++++++++++++++++++++++- 3 files changed, 170 insertions(+), 168 deletions(-) diff --git a/include/linux/huge_mm.h b/include/linux/huge_mm.h index c745f7ad22987f..8ca0fa3be2acb1 100644 --- a/include/linux/huge_mm.h +++ b/include/linux/huge_mm.h @@ -510,8 +510,6 @@ change_huge_pud(struct mmu_gather *tlb, struct vm_area_struct *vma, int hugepage_madvise(struct vm_area_struct *vma, vm_flags_t *vm_flags, int advice); -int madvise_collapse(struct vm_area_struct *vma, unsigned long start, - unsigned long end, bool *lock_dropped); void vma_adjust_trans_huge(struct vm_area_struct *vma, unsigned long start, unsigned long end, struct vm_area_struct *next); spinlock_t *__pmd_trans_huge_lock(pmd_t *pmd, struct vm_area_struct *vma); @@ -715,13 +713,6 @@ static inline int hugepage_madvise(struct vm_area_struct *vma, return -EINVAL; } -static inline int madvise_collapse(struct vm_area_struct *vma, - unsigned long start, - unsigned long end, bool *lock_dropped) -{ - return -EINVAL; -} - static inline void vma_adjust_trans_huge(struct vm_area_struct *vma, unsigned long start, unsigned long end, diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 86b591ec2c492b..8a5c7f38096ef7 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -972,23 +972,6 @@ static void collapse_policy_khugepaged(struct collapse_policy *p) p->tva_type = TVA_KHUGEPAGED; } -/* MADV_COLLAPSE was asked for explicitly, so it is not held to those */ -static void collapse_policy_madvise(struct collapse_policy *p) -{ - p->pmd.max_ptes_none = HPAGE_PMD_NR; - p->pmd.max_ptes_swap = HPAGE_PMD_NR; - p->pmd.max_ptes_shared = HPAGE_PMD_NR; - /* Never read: MADV_COLLAPSE collapses to PMD order only */ - p->sub_pmd = p->pmd; - - p->anon_skip_lazyfree = false; - p->anon_require_referenced = false; - p->file_install_pmd = true; - p->file_writeback_dirty = true; - p->gfp = GFP_TRANSHUGE; - p->tva_type = TVA_FORCED_COLLAPSE; -} - #ifdef CONFIG_NUMA static int collapse_find_target_node(struct collapse_control *cc) { @@ -2857,9 +2840,8 @@ enum scan_result collapse_scan_pmd(struct vm_area_struct *vma, * * Return: what the collapse made of the table. */ -enum scan_result collapse_run_pmd(struct mm_struct *mm, - unsigned long addr, enum scan_result result, - struct collapse_control *cc) +enum scan_result collapse_run_pmd(struct mm_struct *mm, unsigned long addr, + enum scan_result result, struct collapse_control *cc) { struct file *file = cc->scan_file; bool triggered_wb = false; @@ -3230,140 +3212,3 @@ bool current_is_khugepaged(void) { return kthread_func(current) == khugepaged; } - -static int madvise_collapse_errno(enum scan_result r) -{ - /* - * MADV_COLLAPSE breaks from existing madvise(2) conventions to provide - * actionable feedback to caller, so they may take an appropriate - * fallback measure depending on the nature of the failure. - */ - switch (r) { - case SCAN_ALLOC_HUGE_PAGE_FAIL: - return -ENOMEM; - case SCAN_CGROUP_CHARGE_FAIL: - case SCAN_EXCEED_NONE_PTE: - return -EBUSY; - /* Resource temporary unavailable - trying again might succeed */ - case SCAN_PAGE_COUNT: - case SCAN_PAGE_LOCK: - case SCAN_PAGE_LRU: - case SCAN_DEL_PAGE_LRU: - case SCAN_PAGE_FILLED: - case SCAN_PAGE_HAS_PRIVATE: - case SCAN_PAGE_DIRTY_OR_WRITEBACK: - return -EAGAIN; - /* - * Other: Trying again likely not to succeed / error intrinsic to - * specified memory range. khugepaged likely won't be able to collapse - * either. - */ - default: - return -EINVAL; - } -} - -int madvise_collapse(struct vm_area_struct *vma, unsigned long start, - unsigned long end, bool *lock_dropped) -{ - struct collapse_control *cc; - struct mm_struct *mm = vma->vm_mm; - unsigned long hstart, hend, addr, orders; - enum scan_result last_fail = SCAN_FAIL; - int thps = 0; - - BUG_ON(vma->vm_start > start); - BUG_ON(vma->vm_end < end); - - orders = collapse_possible_orders(vma, vma->vm_flags, - TVA_FORCED_COLLAPSE); - if (!orders) - return -EINVAL; - - hstart = ALIGN(start, HPAGE_PMD_SIZE); - hend = ALIGN_DOWN(end, HPAGE_PMD_SIZE); - - if (hstart >= hend) - return 0; - - cc = kmalloc_obj(*cc); - if (!cc) - return -ENOMEM; - collapse_control_init(cc); - collapse_policy_madvise(&cc->policy); - - lru_add_drain_all(); - - for (addr = hstart; addr < hend; addr += HPAGE_PMD_SIZE) { - struct vm_area_struct *found; - enum scan_result result; - - /* - * A collapse gives the lock up, so the VMA has to be found - * again after one: it can shrink while nothing is held. A scan - * that finds nothing to collapse leaves the lock alone, so a - * range that is already collapsed walks on without relocking. - */ - if (!vma) { - cond_resched(); - mmap_read_lock(mm); - result = collapse_vma_revalidate(mm, addr, false, &found, - cc, HPAGE_PMD_ORDER); - if (result != SCAN_SUCCEED) { - last_fail = result; - goto out_locked; - } - vma = found; - hend = min(hend, vma->vm_end & HPAGE_PMD_MASK); - orders = collapse_possible_orders(vma, vma->vm_flags, - TVA_FORCED_COLLAPSE); - } - - result = collapse_scan_pmd(vma, addr, cc, orders); - /* Nothing to do here, and the lock is still ours */ - if (result != SCAN_SUCCEED && result != SCAN_PTE_MAPPED_HUGEPAGE) - goto tally; - - /* The collapse takes its own locks, so give this up */ - mmap_read_unlock(mm); - *lock_dropped = true; - vma = NULL; - - result = collapse_run_pmd(mm, addr, result, cc); -tally: - switch (result) { - case SCAN_SUCCEED: - case SCAN_PMD_MAPPED: - ++thps; - break; - /* Whitelisted set of results where continuing OK */ - case SCAN_NO_PTE_TABLE: - case SCAN_PTE_NON_PRESENT: - case SCAN_PTE_UFFD: - case SCAN_LACK_REFERENCED_PAGE: - case SCAN_PAGE_NULL: - case SCAN_PAGE_COUNT: - case SCAN_PAGE_LOCK: - case SCAN_PAGE_COMPOUND: - case SCAN_PAGE_LRU: - case SCAN_DEL_PAGE_LRU: - last_fail = result; - break; - default: - last_fail = result; - /* Other error, exit */ - goto out; - } - } - -out: - /* Caller expects us to hold mmap_lock on return */ - if (!vma) - mmap_read_lock(mm); -out_locked: - mmap_assert_locked(mm); - kfree(cc); - - return thps == ((hend - hstart) >> HPAGE_PMD_SHIFT) ? 0 - : madvise_collapse_errno(last_fail); -} diff --git a/mm/madvise.c b/mm/madvise.c index 963337f93a7a11..acb5215e2f079f 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -38,6 +38,7 @@ #include "internal.h" #include "swap.h" +#include "collapse.h" #define __MADV_SET_ANON_VMA_NAME (-1) @@ -906,6 +907,172 @@ bool madvise_dontneed_free_valid_vma(struct madvise_behavior *madv_behavior) return true; } +#ifdef CONFIG_TRANSPARENT_HUGEPAGE + +/* MADV_COLLAPSE was asked for explicitly, so it is not held to those */ +static void collapse_policy_madvise(struct collapse_policy *p) +{ + p->pmd.max_ptes_none = HPAGE_PMD_NR; + p->pmd.max_ptes_swap = HPAGE_PMD_NR; + p->pmd.max_ptes_shared = HPAGE_PMD_NR; + /* Never read: MADV_COLLAPSE collapses to PMD order only */ + p->sub_pmd = p->pmd; + + p->anon_skip_lazyfree = false; + p->anon_require_referenced = false; + p->file_install_pmd = true; + p->file_writeback_dirty = true; + p->gfp = GFP_TRANSHUGE; + p->tva_type = TVA_FORCED_COLLAPSE; +} + +static int madvise_collapse_errno(enum scan_result r) +{ + /* + * MADV_COLLAPSE breaks from existing madvise(2) conventions to provide + * actionable feedback to caller, so they may take an appropriate + * fallback measure depending on the nature of the failure. + */ + switch (r) { + case SCAN_ALLOC_HUGE_PAGE_FAIL: + return -ENOMEM; + case SCAN_CGROUP_CHARGE_FAIL: + case SCAN_EXCEED_NONE_PTE: + return -EBUSY; + /* Resource temporary unavailable - trying again might succeed */ + case SCAN_PAGE_COUNT: + case SCAN_PAGE_LOCK: + case SCAN_PAGE_LRU: + case SCAN_DEL_PAGE_LRU: + case SCAN_PAGE_FILLED: + case SCAN_PAGE_HAS_PRIVATE: + case SCAN_PAGE_DIRTY_OR_WRITEBACK: + return -EAGAIN; + /* + * Other: Trying again likely not to succeed / error intrinsic to + * specified memory range. khugepaged likely won't be able to collapse + * either. + */ + default: + return -EINVAL; + } +} + +static int madvise_collapse(struct madvise_behavior *madv_behavior) +{ + struct madvise_behavior_range *range = &madv_behavior->range; + struct vm_area_struct *vma = madv_behavior->vma; + struct mm_struct *mm = madv_behavior->mm; + struct collapse_control *cc; + unsigned long hstart, hend, addr, orders; + enum scan_result last_fail = SCAN_FAIL; + int thps = 0; + + BUG_ON(vma->vm_start > range->start); + BUG_ON(vma->vm_end < range->end); + + orders = collapse_possible_orders(vma, vma->vm_flags, + TVA_FORCED_COLLAPSE); + if (!orders) + return -EINVAL; + + hstart = ALIGN(range->start, HPAGE_PMD_SIZE); + hend = ALIGN_DOWN(range->end, HPAGE_PMD_SIZE); + + if (hstart >= hend) + return 0; + + cc = kmalloc_obj(*cc); + if (!cc) + return -ENOMEM; + collapse_control_init(cc); + collapse_policy_madvise(&cc->policy); + + lru_add_drain_all(); + + for (addr = hstart; addr < hend; addr += HPAGE_PMD_SIZE) { + struct vm_area_struct *found; + enum scan_result result; + + /* + * A collapse gives the lock up, so the VMA has to be found + * again after one: it can shrink while nothing is held. A scan + * that finds nothing to collapse leaves the lock alone, so a + * range that is already collapsed walks on without relocking. + */ + if (!vma) { + cond_resched(); + mmap_read_lock(mm); + result = collapse_vma_revalidate(mm, addr, false, &found, + cc, HPAGE_PMD_ORDER); + if (result != SCAN_SUCCEED) { + last_fail = result; + goto out_locked; + } + vma = found; + hend = min(hend, vma->vm_end & HPAGE_PMD_MASK); + orders = collapse_possible_orders(vma, vma->vm_flags, + TVA_FORCED_COLLAPSE); + } + + result = collapse_scan_pmd(vma, addr, cc, orders); + /* Nothing to do here, and the lock is still ours */ + if (result != SCAN_SUCCEED && result != SCAN_PTE_MAPPED_HUGEPAGE) + goto tally; + + /* The collapse takes its own locks, so give this up */ + mmap_read_unlock(mm); + mark_mmap_lock_dropped(madv_behavior); + vma = NULL; + + result = collapse_run_pmd(mm, addr, result, cc); +tally: + switch (result) { + case SCAN_SUCCEED: + case SCAN_PMD_MAPPED: + ++thps; + break; + /* Whitelisted set of results where continuing OK */ + case SCAN_NO_PTE_TABLE: + case SCAN_PTE_NON_PRESENT: + case SCAN_PTE_UFFD: + case SCAN_LACK_REFERENCED_PAGE: + case SCAN_PAGE_NULL: + case SCAN_PAGE_COUNT: + case SCAN_PAGE_LOCK: + case SCAN_PAGE_COMPOUND: + case SCAN_PAGE_LRU: + case SCAN_DEL_PAGE_LRU: + last_fail = result; + break; + default: + last_fail = result; + /* Other error, exit */ + goto out; + } + } + +out: + /* Caller expects us to hold mmap_lock on return */ + if (!vma) + mmap_read_lock(mm); +out_locked: + mmap_assert_locked(mm); + kfree(cc); + + return thps == ((hend - hstart) >> HPAGE_PMD_SHIFT) ? 0 + : madvise_collapse_errno(last_fail); +} + +#else /* CONFIG_TRANSPARENT_HUGEPAGE */ + +static int madvise_collapse(struct madvise_behavior *madv_behavior) +{ + return -EINVAL; +} + +#endif /* CONFIG_TRANSPARENT_HUGEPAGE */ + static long madvise_dontneed_free(struct madvise_behavior *madv_behavior) { struct mm_struct *mm = madv_behavior->mm; @@ -1373,8 +1540,7 @@ static int madvise_vma_behavior(struct madvise_behavior *madv_behavior) case MADV_DONTNEED_LOCKED: return madvise_dontneed_free(madv_behavior); case MADV_COLLAPSE: - return madvise_collapse(vma, range->start, range->end, - &madv_behavior->lock_dropped); + return madvise_collapse(madv_behavior); case MADV_GUARD_INSTALL: return madvise_guard_install(madv_behavior); case MADV_GUARD_REMOVE: From c0140bbd438673508ab84b1be64dcdf5bcc31beb Mon Sep 17 00:00:00 2001 From: Huaisheng Ye Date: Wed, 9 Sep 2026 15:46:42 +0800 Subject: [PATCH 0873/1352] mm/hugetlb: account for allowed nodes when gathering surplus pages Hugetlb reservations are accounted globally, but hugetlb_acct_memory() also verifies that the current cpuset and MPOL_BIND policy contain enough free huge pages to add a new reservation. gather_surplus_pages() calculates its allocation shortfall from the global free and reserved counters. If the global pool has enough free pages, but those pages reside outside the nodes allowed by the task, it allocates no surplus pages. The subsequent allowed_mems_nr() check then rejects the reservation and mmap() fails with ENOMEM, even when nr_overcommit_hugepages permits allocating surplus pages on the allowed nodes. Calculate both the global shortfall and the shortfall within the allowed nodes, and allocate the larger of the two. Include surplus pages allocated outside hugetlb_lock in both calculations when rechecking after reacquiring the lock. These pages are constrained by alloc_nodemask, so they satisfy both shortages. Easy way to reproduce this issue with 2+ NUMA nodes system: # echo 0 > /sys/kernel/mm/hugepages/hugepages-2048kB/nr_hugepage # echo 3 > /sys/devices/system/node/node0/hugepages/hugepages-2048kB/nr_hugepages # echo 1 > /sys/kernel/mm/hugepages/hugepages-2048kB/nr_overcommit_hugepages # cd tools/testing/selftests/mm # numactl --membind=1 ./hugetlb-mmap 2 21 TAP version 13 # [INFO] detected hugetlb page size: 2048 KiB # [INFO] detected hugetlb page size: 1048576 KiB # 2048 kB hugepages 1..2 # Mapping 2 Mbytes Bail out! mmap: Cannot allocate memory (12) # Planned tests != run tests (2 != 0) # Totals: pass:0 fail:0 xfail:0 xpass:0 skip:0 error:0 This fixes hugetlb mappings when, for example, a task runs with MPOL_BIND on Node 1 while the existing free huge pages are on Node 0. Similar issue also could be found in ltp if the free pages of global pool reside outside the nodes allowed by the application. # cd ltp/testcases/kernel/mem/hugetlb/hugemmap/ # numactl --cpunodebind=0 --membind=1 ./hugemmap10 Link: https://lore.kernel.org/20260909074642.7308-1-yehuaisheng@open-hieco.net Fixes: e4e574b767ba ("hugetlb: Try to grow hugetlb pool for MAP_SHARED mappings") Signed-off-by: Huaisheng Ye Signed-off-by: Andrew Morton Acked-by: Muchun Song Cc: David Hildenbrand Cc: Oscar Salvador --- mm/hugetlb.c | 21 +++++++++++++++++---- 1 file changed, 17 insertions(+), 4 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 5e05711f352d86..fd00141b089a98 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -125,6 +125,7 @@ struct mutex *hugetlb_fault_mutex_table __ro_after_init; /* Forward declaration */ static int hugetlb_acct_memory(struct hstate *h, long delta); +static unsigned int allowed_mems_nr(struct hstate *h); static void hugetlb_vma_lock_free(struct vm_area_struct *vma); static void hugetlb_vma_lock_alloc(struct vm_area_struct *vma); static void __hugetlb_vma_unlock_write_free(struct vm_area_struct *vma); @@ -2252,6 +2253,19 @@ static nodemask_t *policy_mbind_nodemask(gfp_t gfp) return NULL; } +/* + * Reservations are globally accounted, but they must also be backed by free + * pages on nodes allowed by the current cpuset and MPOL_BIND policy. + */ +static long surplus_pages_needed(struct hstate *h, long delta, long allocated) +{ + long global_free = (long)h->free_huge_pages + allocated; + long allowed_free = (long)allowed_mems_nr(h) + allocated; + + return max((long)h->resv_huge_pages + delta - global_free, + delta - allowed_free); +} + /* * Increase the hugetlb pool such that it can accommodate a reservation * of size 'delta'. @@ -2274,7 +2288,7 @@ static int gather_surplus_pages(struct hstate *h, long delta) alloc_nodemask = cpuset_current_mems_allowed; lockdep_assert_held(&hugetlb_lock); - needed = (h->resv_huge_pages + delta) - h->free_huge_pages; + needed = surplus_pages_needed(h, delta, 0); if (needed <= 0) { h->resv_huge_pages += delta; return 0; @@ -2305,11 +2319,10 @@ static int gather_surplus_pages(struct hstate *h, long delta) /* * After retaking hugetlb_lock, we need to recalculate 'needed' - * because either resv_huge_pages or free_huge_pages may have changed. + * because either resv_huge_pages or the free page counts may have changed. */ spin_lock_irq(&hugetlb_lock); - needed = (h->resv_huge_pages + delta) - - (h->free_huge_pages + allocated); + needed = surplus_pages_needed(h, delta, allocated); if (needed > 0) { if (alloc_ok) goto retry; From e18cf847d8abb60b8f95c6f1dede5458109d7279 Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Thu, 10 Sep 2026 14:44:15 +0800 Subject: [PATCH 0874/1352] selftests/mm: fix ptrace PEEKDATA check in memfd_secret test try_ptrace() treats PTRACE_PEEKDATA return value as a boolean check. A successful read returns non-zero data (memory filled with 0x55), causing the test to incorrectly report PASS when secret memory protection is broken. Check the return value against -1 instead. The test should only pass when PTRACE_PEEKDATA fails, which means secret memory protection works. Link: https://lore.kernel.org/20260910064415.71623-1-hongfu.li@linux.dev Fixes: 76fe17ef588a ("secretmem: test: add basic selftest for memfd_secret(2)") Signed-off-by: Hongfu Li Signed-off-by: Andrew Morton Acked-by: Lorenzo Stoakes (ARM) Acked-by: Mike Rapoport (Microsoft) Cc: David Hildenbrand Cc: James Bottomley Cc: Liam R. Howlett Cc: Michal Hocko Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/mm/memfd_secret.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/tools/testing/selftests/mm/memfd_secret.c b/tools/testing/selftests/mm/memfd_secret.c index c55d84c5e613a4..aef774be87f14a 100644 --- a/tools/testing/selftests/mm/memfd_secret.c +++ b/tools/testing/selftests/mm/memfd_secret.c @@ -145,7 +145,8 @@ static void try_ptrace(int fd, int pipefd[2]) exit(KSFT_FAIL); } - if (ptrace(PTRACE_PEEKDATA, ppid, mem, 0)) + /* PEEKDATA on secret memory must fail, else protection is broken. */ + if (ptrace(PTRACE_PEEKDATA, ppid, mem, 0) == -1) exit(KSFT_PASS); exit(KSFT_FAIL); From bbf347e70912f1d4a0d07f307611507cecaccdf7 Mon Sep 17 00:00:00 2001 From: Qiqi Liu Date: Tue, 8 Sep 2026 18:23:56 +0800 Subject: [PATCH 0875/1352] mm: page_alloc: add missing hooks to bulk allocation path The bulk allocation path in alloc_pages_bulk_noprof() currently misses trace_mm_page_alloc() and kmsan_alloc_page() calls, leaving bulk-allocated pages invisible to ftrace/BPF/perf and leaving KMSAN shadow memory stale. Add both calls in the bulk loop to match the standard allocation path, placing them before set_page_refcounted() for consistency. The gfp mask passed to kmsan_alloc_page() is stripped of __GFP_RECLAIM because the bulk loop runs under the PCP spinlock, and KMSAN's stack depot allocation must not sleep. Both are no-ops when their respective features are disabled, so there is no overhead in production kernels. Link: https://lore.kernel.org/20260908102356.344075-1-liuqiqi@kylinos.cn Signed-off-by: Qiqi Liu Signed-off-by: Andrew Morton Suggested-by: Gregory Price Reviewed-by: Vlastimil Babka (SUSE) Reviewed-by: Gregory Price (Meta) Reviewed-by: Zi Yan Cc: Brendan Jackman Cc: Johannes Weiner Cc: Michal Hocko Cc: Suren Baghdasaryan --- mm/page_alloc.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/mm/page_alloc.c b/mm/page_alloc.c index f2eea5e7637cd1..4fb62c6fc3e432 100644 --- a/mm/page_alloc.c +++ b/mm/page_alloc.c @@ -5303,6 +5303,8 @@ unsigned long alloc_pages_bulk_noprof(gfp_t gfp, int preferred_nid, nr_account++; prep_new_page(page, 0, gfp, ALLOC_DEFAULT); + trace_mm_page_alloc(page, 0, gfp, ac.migratetype); + kmsan_alloc_page(page, 0, gfp & ~__GFP_RECLAIM); set_page_refcounted(page); page_array[nr_populated++] = page; } From eae1981311059b71b669d517e0d0f3c063f21a23 Mon Sep 17 00:00:00 2001 From: Sergey Senozhatsky Date: Mon, 7 Sep 2026 19:57:28 +0900 Subject: [PATCH 0876/1352] zram: convert to SG-list zsmalloc object read API Patch series "zsmallc: remove old object read API". zram remains the only user of old zsmalloc object read API. This series removes the old API and converts zram to use the new SG-list based API. This patch (of 2): zram remains the last user of old zsmalloc object read API, that performed linearisation on the zsmalloc side. There is a new SG-list API, that has a bunch of benefits. Switch zram to SG-list zsmalloc object read API. Link: https://lore.kernel.org/20260907105739.1793316-1-senozhatsky@chromium.org Link: https://lore.kernel.org/20260907105739.1793316-2-senozhatsky@chromium.org Signed-off-by: Sergey Senozhatsky Signed-off-by: Andrew Morton Cc: Minchan Kim Cc: Nhat Pham --- drivers/block/zram/zcomp.c | 23 ++++++++++++-- drivers/block/zram/zcomp.h | 4 ++- drivers/block/zram/zram_drv.c | 56 ++++++++++++++++++++--------------- 3 files changed, 55 insertions(+), 28 deletions(-) diff --git a/drivers/block/zram/zcomp.c b/drivers/block/zram/zcomp.c index 974c4691887e6f..028e2f0f587e3f 100644 --- a/drivers/block/zram/zcomp.c +++ b/drivers/block/zram/zcomp.c @@ -7,6 +7,8 @@ #include #include #include +#include +#include #include #include @@ -158,17 +160,32 @@ int zcomp_compress(struct zcomp *comp, struct zcomp_strm *zstrm, } int zcomp_decompress(struct zcomp *comp, struct zcomp_strm *zstrm, - const void *src, unsigned int src_len, void *dst) + struct scatterlist *sg, unsigned int src_len, void *dst) { struct zcomp_req req = { - .src = src, .dst = dst, .src_len = src_len, .dst_len = PAGE_SIZE, }; + void *src = NULL; + int ret; might_sleep(); - return comp->ops->decompress(comp->params, &zstrm->ctx, &req); + + if (sg_is_last(sg)) { + /* the object is contained within one page, read it in-place */ + src = kmap_local_page(sg_page(sg)); + req.src = src + sg->offset; + } else { + /* the object spans two pages, linearize it into local copy */ + sg_copy_to_buffer(sg, 2, zstrm->local_copy, src_len); + req.src = zstrm->local_copy; + } + + ret = comp->ops->decompress(comp->params, &zstrm->ctx, &req); + if (src) + kunmap_local(src); + return ret; } int zcomp_cpu_up_prepare(unsigned int cpu, struct hlist_node *node) diff --git a/drivers/block/zram/zcomp.h b/drivers/block/zram/zcomp.h index 81a0f3f6ff4871..f38fd31f9e4e02 100644 --- a/drivers/block/zram/zcomp.h +++ b/drivers/block/zram/zcomp.h @@ -5,6 +5,8 @@ #include +struct scatterlist; + #define ZCOMP_PARAM_NOT_SET INT_MIN struct deflate_params { @@ -91,6 +93,6 @@ void zcomp_stream_put(struct zcomp_strm *zstrm); int zcomp_compress(struct zcomp *comp, struct zcomp_strm *zstrm, const void *src, unsigned int *dst_len); int zcomp_decompress(struct zcomp *comp, struct zcomp_strm *zstrm, - const void *src, unsigned int src_len, void *dst); + struct scatterlist *sg, unsigned int src_len, void *dst); #endif /* _ZCOMP_H_ */ diff --git a/drivers/block/zram/zram_drv.c b/drivers/block/zram/zram_drv.c index 2359eaa6f53184..024402438bd1d4 100644 --- a/drivers/block/zram/zram_drv.c +++ b/drivers/block/zram/zram_drv.c @@ -32,6 +32,7 @@ #include #include #include +#include #include #include @@ -1347,9 +1348,9 @@ static int decompress_bdev_page(struct zram *zram, struct page *page, unsigned long index) { struct zcomp_strm *zstrm; + struct scatterlist sg[1]; unsigned int size; int ret, prio; - void *src; slot_lock(zram, index); /* Since slot was unlocked we need to make sure it's still ZRAM_WB */ @@ -1368,13 +1369,18 @@ static int decompress_bdev_page(struct zram *zram, struct page *page, size = get_slot_size(zram, index); prio = get_slot_comp_priority(zram, index); + sg_init_table(sg, 1); + sg_set_page(sg, page, size, 0); + zstrm = zcomp_stream_get(zram->comps[prio]); - src = kmap_local_page(page); - ret = zcomp_decompress(zram->comps[prio], zstrm, src, size, + ret = zcomp_decompress(zram->comps[prio], zstrm, sg, size, zstrm->local_copy); - if (!ret) - copy_page(src, zstrm->local_copy); - kunmap_local(src); + if (!ret) { + void *dst = kmap_local_page(page); + + copy_page(dst, zstrm->local_copy); + kunmap_local(dst); + } zcomp_stream_put(zstrm); slot_unlock(zram, index); @@ -2086,15 +2092,19 @@ static int read_same_filled_page(struct zram *zram, struct page *page, static int read_incompressible_page(struct zram *zram, struct page *page, unsigned long index) { + struct scatterlist sg[2]; unsigned long handle; void *src, *dst; handle = get_slot_handle(zram, index); - src = zs_obj_read_begin(zram->mem_pool, handle, PAGE_SIZE, NULL); + zs_obj_read_sg_begin(zram->mem_pool, handle, sg, PAGE_SIZE); + /* an incompressible object never spans two pages */ + src = kmap_local_page(sg_page(sg)); dst = kmap_local_page(page); copy_page(dst, src); kunmap_local(dst); - zs_obj_read_end(zram->mem_pool, handle, PAGE_SIZE, src); + kunmap_local(src); + zs_obj_read_sg_end(zram->mem_pool, handle); return 0; } @@ -2103,9 +2113,10 @@ static int read_compressed_page(struct zram *zram, struct page *page, unsigned long index) { struct zcomp_strm *zstrm; + struct scatterlist sg[2]; unsigned long handle; unsigned int size; - void *src, *dst; + void *dst; int ret, prio; handle = get_slot_handle(zram, index); @@ -2113,12 +2124,11 @@ static int read_compressed_page(struct zram *zram, struct page *page, prio = get_slot_comp_priority(zram, index); zstrm = zcomp_stream_get(zram->comps[prio]); - src = zs_obj_read_begin(zram->mem_pool, handle, size, - zstrm->local_copy); + zs_obj_read_sg_begin(zram->mem_pool, handle, sg, size); dst = kmap_local_page(page); - ret = zcomp_decompress(zram->comps[prio], zstrm, src, size, dst); + ret = zcomp_decompress(zram->comps[prio], zstrm, sg, size, dst); kunmap_local(dst); - zs_obj_read_end(zram->mem_pool, handle, size, src); + zs_obj_read_sg_end(zram->mem_pool, handle); zcomp_stream_put(zstrm); return ret; @@ -2128,25 +2138,23 @@ static int read_compressed_page(struct zram *zram, struct page *page, static int read_from_zspool_raw(struct zram *zram, struct page *page, unsigned long index) { - struct zcomp_strm *zstrm; + struct scatterlist sg[2]; unsigned long handle; unsigned int size; - void *src; + void *dst; handle = get_slot_handle(zram, index); size = get_slot_size(zram, index); /* - * We need to get stream just for ->local_copy buffer, in - * case if object spans two physical pages. No decompression - * takes place here, as we read raw compressed data. + * No decompression takes place here, we copy out raw compressed + * data directly into the destination page. */ - zstrm = zcomp_stream_get(zram->comps[ZRAM_PRIMARY_COMP]); - src = zs_obj_read_begin(zram->mem_pool, handle, size, - zstrm->local_copy); - memcpy_to_page(page, 0, src, size); - zs_obj_read_end(zram->mem_pool, handle, size, src); - zcomp_stream_put(zstrm); + zs_obj_read_sg_begin(zram->mem_pool, handle, sg, size); + dst = kmap_local_page(page); + sg_copy_to_buffer(sg, sg_nents(sg), dst, size); + kunmap_local(dst); + zs_obj_read_sg_end(zram->mem_pool, handle); memzero_page(page, size, PAGE_SIZE - size); From 86d92ad3c6f67d69790ecc40a5bdc18800eb46c1 Mon Sep 17 00:00:00 2001 From: Sergey Senozhatsky Date: Mon, 7 Sep 2026 19:57:29 +0900 Subject: [PATCH 0877/1352] zsmalloc: remove old object read API There are no users left of the old API, all have switched to SG-list object read API. Remove it. Link: https://lore.kernel.org/20260907105739.1793316-3-senozhatsky@chromium.org Signed-off-by: Sergey Senozhatsky Signed-off-by: Andrew Morton Cc: Minchan Kim Cc: Nhat Pham --- include/linux/zsmalloc.h | 4 --- mm/zsmalloc.c | 77 ---------------------------------------- 2 files changed, 81 deletions(-) diff --git a/include/linux/zsmalloc.h b/include/linux/zsmalloc.h index 478410c880b1fe..5b7298a5026b8d 100644 --- a/include/linux/zsmalloc.h +++ b/include/linux/zsmalloc.h @@ -40,10 +40,6 @@ unsigned int zs_lookup_class_index(struct zs_pool *pool, unsigned int size); void zs_pool_stats(struct zs_pool *pool, struct zs_pool_stats *stats); -void *zs_obj_read_begin(struct zs_pool *pool, unsigned long handle, - size_t mem_len, void *local_copy); -void zs_obj_read_end(struct zs_pool *pool, unsigned long handle, - size_t mem_len, void *handle_mem); void zs_obj_read_sg_begin(struct zs_pool *pool, unsigned long handle, struct scatterlist *sg, size_t mem_len); void zs_obj_read_sg_end(struct zs_pool *pool, unsigned long handle); diff --git a/mm/zsmalloc.c b/mm/zsmalloc.c index 825022a7a328fe..11be37c4317189 100644 --- a/mm/zsmalloc.c +++ b/mm/zsmalloc.c @@ -1134,83 +1134,6 @@ unsigned long zs_get_total_pages(struct zs_pool *pool) } EXPORT_SYMBOL_GPL(zs_get_total_pages); -void *zs_obj_read_begin(struct zs_pool *pool, unsigned long handle, - size_t mem_len, void *local_copy) -{ - struct zspage *zspage; - struct zpdesc *zpdesc; - unsigned long obj, off; - unsigned int obj_idx; - struct size_class *class; - void *addr; - - /* Guarantee we can get zspage from handle safely */ - read_lock(&pool->lock); - obj = handle_to_obj(handle); - obj_to_location(obj, &zpdesc, &obj_idx); - zspage = get_zspage(zpdesc); - - /* Make sure migration doesn't move any pages in this zspage */ - zspage_read_lock(zspage); - read_unlock(&pool->lock); - - class = zspage_class(pool, zspage); - off = offset_in_page(class->size * obj_idx); - - if (!ZsHugePage(zspage)) - off += ZS_HANDLE_SIZE; - - if (off + mem_len <= PAGE_SIZE) { - /* this object is contained entirely within a page */ - addr = kmap_local_zpdesc(zpdesc); - addr += off; - } else { - size_t sizes[2]; - - /* this object spans two pages */ - sizes[0] = PAGE_SIZE - off; - sizes[1] = mem_len - sizes[0]; - addr = local_copy; - - memcpy_from_page(addr, zpdesc_page(zpdesc), - off, sizes[0]); - zpdesc = get_next_zpdesc(zpdesc); - memcpy_from_page(addr + sizes[0], - zpdesc_page(zpdesc), - 0, sizes[1]); - } - - return addr; -} -EXPORT_SYMBOL_GPL(zs_obj_read_begin); - -void zs_obj_read_end(struct zs_pool *pool, unsigned long handle, - size_t mem_len, void *handle_mem) -{ - struct zspage *zspage; - struct zpdesc *zpdesc; - unsigned long obj, off; - unsigned int obj_idx; - struct size_class *class; - - obj = handle_to_obj(handle); - obj_to_location(obj, &zpdesc, &obj_idx); - zspage = get_zspage(zpdesc); - class = zspage_class(pool, zspage); - off = offset_in_page(class->size * obj_idx); - - if (!ZsHugePage(zspage)) - off += ZS_HANDLE_SIZE; - - if (off + mem_len <= PAGE_SIZE) { - handle_mem -= off; - kunmap_local(handle_mem); - } - - zspage_read_unlock(zspage); -} -EXPORT_SYMBOL_GPL(zs_obj_read_end); - void zs_obj_read_sg_begin(struct zs_pool *pool, unsigned long handle, struct scatterlist *sg, size_t mem_len) { From a31016f2b8197db9f4078543ee739f472a92cfcf Mon Sep 17 00:00:00 2001 From: Kemeng Shi Date: Mon, 7 Sep 2026 17:13:53 +0800 Subject: [PATCH 0878/1352] mm, swap: fix potential NULL dereference when trying a sleep table allocation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Patch series "mm, swap: some random fixes and cleanups", v3. This series contains some random fixes and cleanups. More details can be found in respective patches. This patch (of 4): The root cause of this issue is because multi-tables are updated in non atomic context. To be more specific, the issue could be triggerred as following: swap_alloc_fast swap_cluster_populate() /* Try a sleep allocation */ spin_unlock(&ci->lock); swap_cluster_alloc_table() rcu_assign_pointer(ci->table, table); ci = swap_cluster_lock(si, offset) cluster_is_usable(ci, order) if (!cluster_table_is_alloced(ci)) // ok alloc_swap_scan_cluster() cluster_scan_range() __swap_table_get() /* free table when more table allocation fails */ ci->memcg_table = kzalloc_obj(*ci->memcg_table, gfp); if (!ci->memcg_table) swap_cluster_free_table() rcu_assign_pointer(ci->table, NULL); table = rcu_dereference_check(ci->table, lockdep_is_held(&ci->lock)); atomic_long_read(&table[off]); // NULL dereference Since memory order guarantee between ci->table, as well as between ci->table and ci->zero_bitmap, fix the issue by making tables visible at the end of swap_cluster_populate(). Current memory order guarantee is as following: On write side: rcu_assign_pointer(ci->table, table) will offer release to ensure zero_bitmap and memcg_table visible before ci->table. On read side: folio_alloc_swap swap_alloc_fast/swap_alloc_slow /* ci->table: protected by cluster lock */ swap_cluster_lock cluster_is_usable ... __swap_table_set ... swap_cluster_unlock mem_cgroup_try_charge_swap ... /* memcg_table: protected by cluster lock */ swap_cluster_get_and_lock __swap_cgroup_set swap_cluster_unlock swap_writeout swap_zeromap_folio_set /* zero_bitmap: protected by cluster lock */ swap_cluster_get_and_lock __swap_table_set_zero swap_cluster_unlock Link: https://lore.kernel.org/20260907091356.53026-1-shikemeng@huaweicloud.com Link: https://lore.kernel.org/20260907091356.53026-2-shikemeng@huaweicloud.com Fixes: b197d41462c2 ("mm/memcg, swap: store cgroup id in cluster table directly") Signed-off-by: Kemeng Shi Signed-off-by: Andrew Morton Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: Nhat Pham Cc: Kairui Song Cc: Luiz Capitulino Cc: Youngjun Park --- mm/swapfile.c | 30 ++++++++++++++++++++---------- 1 file changed, 20 insertions(+), 10 deletions(-) diff --git a/mm/swapfile.c b/mm/swapfile.c index 2c263563b70ebc..8df8b2c2e5b405 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -416,6 +416,17 @@ static void swap_cluster_free_table_folio_rcu_cb(struct rcu_head *head) folio_put(folio); } +static void swap_cluster_free_count_table(struct swap_table *table) +{ + if (!SWP_TABLE_USE_PAGE) { + kmem_cache_free(swap_table_cachep, table); + return; + } + + call_rcu(&(folio_page(virt_to_folio(table), 0)->rcu_head), + swap_cluster_free_table_folio_rcu_cb); +} + static void swap_cluster_free_table(struct swap_cluster_info *ci) { struct swap_table *table; @@ -435,13 +446,7 @@ static void swap_cluster_free_table(struct swap_cluster_info *ci) return; rcu_assign_pointer(ci->table, NULL); - if (!SWP_TABLE_USE_PAGE) { - kmem_cache_free(swap_table_cachep, table); - return; - } - - call_rcu(&(folio_page(virt_to_folio(table), 0)->rcu_head), - swap_cluster_free_table_folio_rcu_cb); + swap_cluster_free_count_table(table); } static int swap_cluster_alloc_table(struct swap_cluster_info *ci, gfp_t gfp) @@ -464,14 +469,12 @@ static int swap_cluster_alloc_table(struct swap_cluster_info *ci, gfp_t gfp) if (!table) return -ENOMEM; - rcu_assign_pointer(ci->table, table); - #ifdef CONFIG_MEMCG if (!mem_cgroup_disabled()) { VM_WARN_ON_ONCE(ci->memcg_table); ci->memcg_table = kzalloc_obj(*ci->memcg_table, gfp); if (!ci->memcg_table) { - swap_cluster_free_table(ci); + swap_cluster_free_count_table(table); return -ENOMEM; } } @@ -482,9 +485,16 @@ static int swap_cluster_alloc_table(struct swap_cluster_info *ci, gfp_t gfp) ci->zero_bitmap = bitmap_zalloc(SWAPFILE_CLUSTER, gfp); if (!ci->zero_bitmap) { swap_cluster_free_table(ci); + swap_cluster_free_count_table(table); return -ENOMEM; } #endif + + /* + * Make tables visible to cluster_is_usable() after everything is + * ready. + */ + rcu_assign_pointer(ci->table, table); return 0; } From 50331251bef2fc2ce428c61f01634c205270b8d7 Mon Sep 17 00:00:00 2001 From: Kemeng Shi Date: Mon, 7 Sep 2026 17:13:54 +0800 Subject: [PATCH 0879/1352] mm, swap: move setup_swap_clusters_info() after SWP_SOLIDSTATE initialization In setup_swap_clusters_info(), SWP_SOLIDSTATE is used to decide global_cluster allocation. Move setup_swap_clusters_info() after SWP_SOLIDSTATE initialization to avoid unneeded global_cluster allocation. Link: https://lore.kernel.org/20260907091356.53026-3-shikemeng@huaweicloud.com Fixes: 451c6326105b ("mm, swap: clean up swapon process and locking") Signed-off-by: Kemeng Shi Signed-off-by: Andrew Morton Reviewed-by: Luiz Capitulino Acked-by: Kairui Song Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: Nhat Pham Cc: Youngjun Park --- mm/swapfile.c | 13 ++++++++----- 1 file changed, 8 insertions(+), 5 deletions(-) diff --git a/mm/swapfile.c b/mm/swapfile.c index 8df8b2c2e5b405..40ceab21ed5b8d 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -3763,11 +3763,6 @@ SYSCALL_DEFINE2(swapon, const char __user *, specialfile, int, swap_flags) maxpages = si->max; - /* Set up the swap cluster info */ - error = setup_swap_clusters_info(si, swap_header, maxpages); - if (error) - goto bad_swap_unlock_inode; - if (si->bdev && bdev_stable_writes(si->bdev)) si->flags |= SWP_STABLE_WRITES; @@ -3781,6 +3776,14 @@ SYSCALL_DEFINE2(swapon, const char __user *, specialfile, int, swap_flags) inced_nr_rotate_swap = true; } + /* + * Set up the swap cluster info after SWP_ flags handling as + * setup_swap_clusters_info() checks SWP_SOLIDSTATE. + */ + error = setup_swap_clusters_info(si, swap_header, maxpages); + if (error) + goto bad_swap_unlock_inode; + if ((swap_flags & SWAP_FLAG_DISCARD) && si->bdev && bdev_max_discard_sectors(si->bdev)) { /* From e222c6bbc313f7e8fffafb8cade7aaa42d630f49 Mon Sep 17 00:00:00 2001 From: Kemeng Shi Date: Mon, 7 Sep 2026 17:13:55 +0800 Subject: [PATCH 0880/1352] mm, swap: return early from swap_extend_table_try_free() on first non-zero entry Return immediately when the first non-zero swap count is found as any non-zero swap count prevents freeing extend_table and further iteration is pointless. Link: https://lore.kernel.org/20260907091356.53026-4-shikemeng@huaweicloud.com Signed-off-by: Kemeng Shi Signed-off-by: Andrew Morton Reviewed-by: Youngjun Park Acked-by: Kairui Song Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: Luiz Capitulino Cc: Nhat Pham --- mm/swapfile.c | 9 +++------ 1 file changed, 3 insertions(+), 6 deletions(-) diff --git a/mm/swapfile.c b/mm/swapfile.c index 40ceab21ed5b8d..7b35ca90776a8a 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -1523,20 +1523,17 @@ int swap_retry_table_alloc(swp_entry_t entry, gfp_t gfp) static void swap_extend_table_try_free(struct swap_cluster_info *ci) { unsigned long i; - bool can_free = true; if (!ci->extend_table) return; for (i = 0; i < SWAPFILE_CLUSTER; i++) { if (ci->extend_table[i]) - can_free = false; + return; } - if (can_free) { - kfree(ci->extend_table); - ci->extend_table = NULL; - } + kfree(ci->extend_table); + ci->extend_table = NULL; } /* Decrease the swap count of one slot, without freeing it */ From 91676c5560e6cf679b16f037d4b4736bc94f996f Mon Sep 17 00:00:00 2001 From: Kemeng Shi Date: Mon, 7 Sep 2026 17:13:56 +0800 Subject: [PATCH 0881/1352] mm, swap: remove unneeded swap_extend_table_try_free() in swap_dup_entries_cluster() Since commit 0475fde0f68de ("mm, swap: avoid leaving unused extend table after alloc race"), extend table is always allocated when any swap count reach MAX - 1 and is always freed when swap count decrease to MAX - 1 or MAX - 2 with cluster lock held. So the extend table will always be freed properly when decrease swap count in __swap_cluster_put_entry(). So swap_extend_table_try_free() outside of __swap_cluster_put_entry() is unneeded and can be removed. Link: https://lore.kernel.org/20260907091356.53026-5-shikemeng@huaweicloud.com Signed-off-by: Kemeng Shi Signed-off-by: Andrew Morton Reviewed-by: Youngjun Park Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: Kairui Song Cc: Luiz Capitulino Cc: Nhat Pham --- mm/swapfile.c | 1 - 1 file changed, 1 deletion(-) diff --git a/mm/swapfile.c b/mm/swapfile.c index 7b35ca90776a8a..2cd0d0ba966c38 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -1725,7 +1725,6 @@ static int swap_dup_entries_cluster(struct swap_info_struct *si, failed: while (ci_off-- > ci_start) __swap_cluster_put_entry(ci, ci_off); - swap_extend_table_try_free(ci); swap_cluster_unlock(ci); return err; } From 8e36dac81c3d168e1fc93dc80f061669cb280608 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Mon, 7 Sep 2026 09:36:54 +0300 Subject: [PATCH 0882/1352] docs/core-api: memory-allocation: clarify when to use kzalloc_obj and kzalloc Make it abundantly clear that in the most cases objects should be allocated with kzalloc_obj() and buffers should be allocated with kzalloc(). Link: https://lore.kernel.org/20260907063654.2248617-1-rppt@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Acked-by: Vlastimil Babka (SUSE) Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: David Hildenbrand (Arm) Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Michal Hocko Cc: Randy Dunlap Cc: SeongJae Park Cc: Suren Baghdasaryan --- Documentation/core-api/memory-allocation.rst | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/Documentation/core-api/memory-allocation.rst b/Documentation/core-api/memory-allocation.rst index 823f7fa57429b7..648222b6c85763 100644 --- a/Documentation/core-api/memory-allocation.rst +++ b/Documentation/core-api/memory-allocation.rst @@ -23,12 +23,14 @@ answer, although very likely you should use kzalloc_obj(); -or +if you need memory for an object and :: kzalloc(, GFP_KERNEL); +if you need memory for a buffer. + Of course there are cases when other allocation APIs and different GFP flags must be used. From a599b26cef550bf45903556a66cf70f3e4aa3da6 Mon Sep 17 00:00:00 2001 From: Ridong Chen Date: Mon, 7 Sep 2026 10:54:44 +0800 Subject: [PATCH 0883/1352] mm/page_counter: avoid integer overflow in effective_protection() Patch series "mm/mglru: fix ineffective memory protection for non-kswapd reclaim", v4. For MGLRU, memory.min/low is not honored during non-kswapd global reclaim (global direct reclaim and root-level memory.reclaim), because these paths shrink memcgs using stale protection (emin/elow). Patch 2 is the actual fix. Patch 1 is a prerequisite: an integer overflow in effective_protection(), spotted by the sashiko review tool, which patch 2's new caller would also be exposed to. This patch (of 2): effective_protection() scales a parent's protection by a ratio of page counts, e.g. for recursive protection: (parent_effective - siblings_protected) * (usage - protected) / (parent_usage - siblings_protected) The multiply is done at unsigned long width before dividing. On systems with >= 16TB RAM the product can exceed 2^64 and wrap, giving a bogus protection value and silently breaking memory.min/low enforcement. Use mul_u64_u64_div_u64() to multiply in a 128-bit intermediate. Because usage and parent_usage are not read atomically (a child is charged before its parent), usage - protected can briefly exceed the divisor, making the quotient overflow 64 bits and trap (#DE on x86). Cap it so the ratio stays <= 1. Reported by the sashiko review tool [1]. Link: https://lore.kernel.org/20260907025445.1836238-1-ridong.chen@linux.dev Link: https://lore.kernel.org/20260907025445.1836238-2-ridong.chen@linux.dev Link: https://sashiko.dev/#/patchset/20260826133054.88529-1-ridong.chen@linux.dev?part=1 [1] Fixes: bc50bcc6e00b ("mm: memcontrol: clean up and document effective low/min calculations") Fixes: 8a931f801340 ("mm: memcontrol: recursive memory.low protection") Reviewed-by: Barry Song Reviewed-by: Johannes Weiner Signed-off-by: Ridong Chen Signed-off-by: Andrew Morton Assisted-by: Claude:claude-opus-4-8 Cc: Axel Rasmussen Cc: Chris Down Cc: David Hildenbrand Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Cc: Shakeel Butt Cc: Tejun Heo Cc: Wei Xu Cc: Yuanchu Xie Cc: Yu Zhao Cc: --- mm/page_counter.c | 21 +++++++++++++++------ 1 file changed, 15 insertions(+), 6 deletions(-) diff --git a/mm/page_counter.c b/mm/page_counter.c index 450543f4b318b6..98322803941a70 100644 --- a/mm/page_counter.c +++ b/mm/page_counter.c @@ -8,6 +8,7 @@ #include #include #include +#include #include #include #include @@ -376,7 +377,8 @@ static unsigned long effective_protection(unsigned long usage, * otherwise get a smaller chunk than what they claimed. */ if (siblings_protected > parent_effective) - return protected * parent_effective / siblings_protected; + return mul_u64_u64_div_u64(protected, parent_effective, + siblings_protected); /* * Ok, utilized protection of all children is within what the @@ -417,13 +419,20 @@ static unsigned long effective_protection(unsigned long usage, if (parent_effective > siblings_protected && parent_usage > siblings_protected && usage > protected) { - unsigned long unclaimed; + unsigned long parent_unclaimed, parent_unprotected, unprotected; - unclaimed = parent_effective - siblings_protected; - unclaimed *= usage - protected; - unclaimed /= parent_usage - siblings_protected; + parent_unclaimed = parent_effective - siblings_protected; + parent_unprotected = parent_usage - siblings_protected; - ep += unclaimed; + /* + * The usages aren't read atomically, so a child can transiently + * appear to use more than its parent, making the ratio exceed 1 + * and the quotient overflow 64 bits (#DE on x86). Cap it. + */ + unprotected = min(usage - protected, parent_unprotected); + + ep += mul_u64_u64_div_u64(parent_unclaimed, unprotected, + parent_unprotected); } return ep; From 305666211eb30d8def49de9deef43da14aa6b21a Mon Sep 17 00:00:00 2001 From: Ridong Chen Date: Mon, 7 Sep 2026 10:54:45 +0800 Subject: [PATCH 0884/1352] mm/mglru: fix ineffective memory protection for non-kswapd reclaim For MGLRU, memory.min/low is not honored during global proactive reclaim (writing to the root memory.reclaim) and global direct reclaim, because these paths shrink memcgs using stale protection (emin/elow). It can be reproduced as follows: # echo 7 > /sys/kernel/mm/lru_gen/enabled # cd /sys/fs/cgroup # mkdir -p a/b # echo 100M > a/memory.min # echo +memory > a/cgroup.subtree_control # echo 100M > a/b/memory.min # echo $$ > a/b/cgroup.procs # dd if=/dev/zero of=/tmp/testfile bs=1M count=200 # cat a/b/memory.current 222650368 # echo 500M > memory.reclaim -bash: echo: write error: Resource temporarily unavailable # cat a/b/memory.current 6070272 memory.min is 100M, yet reclaim drops a/b down to 6M, breaking the protection. The traditional LRU path is not affected because shrink_node() calls mem_cgroup_calculate_protection() for each memcg it visits during a top-down tree walk. Commit 30d77b7eef01 ("mm/mglru: fix ineffective protection calculation") moved the protection computation into lru_gen_age_node(), which only runs for kswapd. Non-kswapd global reclaim reaches shrink_one() through lru_gen_shrink_node() -> shrink_many() without any protection computation, so emin/elow are whatever a previous kswapd run left behind - or zero if kswapd never ran on this node. Relying on a prior kswapd pass is not correct either: a memcg's emin/elow are derived from its ancestors' memory.min/low settings and from children_min_usage, both of which change over time, so emin/elow go stale even after kswapd has run and must be recomputed at the point of reclaim. Introduce mem_cgroup_calculate_protection_path() which computes emin/elow along the root-to-target path only, by iterating through the cgroup ancestors array top-down. This avoids the full tree traversal that would be needed with mem_cgroup_calculate_protection(), limiting the cost to O(depth) per memcg - typically 3-5 levels. Call it from shrink_one() for the non-kswapd path so that each memcg about to be shrunk has correct protection values. Link: https://lore.kernel.org/20260907025445.1836238-3-ridong.chen@linux.dev Fixes: e4dde56cd208 ("mm: multi-gen LRU: per-node lru_gen_folio lists") Signed-off-by: Ridong Chen Signed-off-by: Andrew Morton Reviewed-by: Barry Song Reviewed-by: Johannes Weiner Assisted-by: Claude:claude-opus-4-8 Cc: Axel Rasmussen Cc: Chris Down Cc: David Hildenbrand Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Cc: Shakeel Butt Cc: Tejun Heo Cc: Wei Xu Cc: Yuanchu Xie Cc: Yu Zhao Cc: --- include/linux/memcontrol.h | 10 +++++++++ mm/memcontrol.c | 45 ++++++++++++++++++++++++++++++++++++++ mm/vmscan.c | 8 ++++++- 3 files changed, 62 insertions(+), 1 deletion(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 4f720791a31c95..46bf724cae7af9 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -1921,6 +1921,16 @@ static inline bool memcg_is_dying(struct mem_cgroup *memcg) } #endif /* CONFIG_MEMCG */ +#if defined(CONFIG_MEMCG) && defined(CONFIG_LRU_GEN) +void mem_cgroup_calculate_protection_path(struct mem_cgroup *root, + struct mem_cgroup *memcg); +#else +static inline void mem_cgroup_calculate_protection_path(struct mem_cgroup *root, + struct mem_cgroup *memcg) +{ +} +#endif + #if defined(CONFIG_MEMCG) && defined(CONFIG_ZSWAP) bool obj_cgroup_may_zswap(struct obj_cgroup *objcg); void obj_cgroup_charge_zswap(struct obj_cgroup *objcg, size_t size); diff --git a/mm/memcontrol.c b/mm/memcontrol.c index cf53d4ac7ecc18..791e536efaebe8 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -5267,6 +5267,51 @@ void mem_cgroup_calculate_protection(struct mem_cgroup *root, page_counter_calculate_protection(&root->memory, &memcg->memory, recursive_protection); } +#ifdef CONFIG_LRU_GEN +/** + * mem_cgroup_calculate_protection_path - compute protection along a path + * @root: the top ancestor of the sub-tree being checked (NULL for root_mem_cgroup) + * @memcg: the target memory cgroup + * + * Walk the ancestor path from @root down to @memcg and compute the effective + * protection at each level. This is safe for isolated queries because it + * ensures parents are computed before children. + */ +void mem_cgroup_calculate_protection_path(struct mem_cgroup *root, + struct mem_cgroup *memcg) +{ + bool recursive_protection = + cgrp_dfl_root.flags & CGRP_ROOT_MEMORY_RECURSIVE_PROT; + struct cgroup *cg; + int root_level, i; + + if (mem_cgroup_disabled()) + return; + + if (!root) + root = root_mem_cgroup; + + if (memcg == root) + return; + + root_level = root->css.cgroup->level; + cg = memcg->css.cgroup; + + rcu_read_lock(); + for (i = root_level + 1; i <= cg->level; i++) { + struct mem_cgroup *cur; + + cur = mem_cgroup_from_css(cgroup_css(cg->ancestors[i], + &memory_cgrp_subsys)); + if (cur) + page_counter_calculate_protection(&root->memory, + &cur->memory, + recursive_protection); + } + rcu_read_unlock(); +} +#endif /* CONFIG_LRU_GEN */ + static int charge_memcg(struct folio *folio, struct mem_cgroup *memcg, gfp_t gfp) { diff --git a/mm/vmscan.c b/mm/vmscan.c index 8e9c73dcd19bb3..80041e2b8049c7 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -5251,7 +5251,13 @@ static int shrink_one(struct lruvec *lruvec, struct scan_control *sc) struct mem_cgroup *memcg = lruvec_memcg(lruvec); struct pglist_data *pgdat = lruvec_pgdat(lruvec); - /* lru_gen_age_node() called mem_cgroup_calculate_protection() */ + /* + * For kswapd, mem_cgroup_calculate_protection() has already + * been called during the top-down cgroup traversal. + */ + if (!current_is_kswapd()) + mem_cgroup_calculate_protection_path(NULL, memcg); + if (mem_cgroup_below_min(NULL, memcg)) return MEMCG_LRU_YOUNG; From d9cd889564a021f2e25ca8120b26deffd7944219 Mon Sep 17 00:00:00 2001 From: Tianyi Chen Date: Tue, 8 Sep 2026 17:55:15 +0800 Subject: [PATCH 0885/1352] tools/testing/vma: cover hole filling through __mmap_region() The mmap tests extend existing mappings one neighbor at a time, while merge tests construct merge state directly. Neither exercises filling a hole between compatible mappings through the mmap setup and completion path. Fill a gap through __mmap_region() and require both neighbors to merge into one VMA. Repeat with only the new mapping's execute permission set and require three separate VMAs. Check boundaries, permissions, page offsets, map_count and cleanup. Link: https://lore.kernel.org/178886112560.138404.17741948638665342936.vma-v2@tychen.cc Signed-off-by: Tianyi Chen Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Assisted-by: Codex:GPT-6 Cc: Jann Horn Cc: Liam R. Howlett Cc: Pedro Falcato Cc: Vlastimil Babka --- tools/testing/vma/tests/mmap.c | 68 ++++++++++++++++++++++++++++++++++ 1 file changed, 68 insertions(+) diff --git a/tools/testing/vma/tests/mmap.c b/tools/testing/vma/tests/mmap.c index fa73faff226263..53e4abe6a63e6c 100644 --- a/tools/testing/vma/tests/mmap.c +++ b/tools/testing/vma/tests/mmap.c @@ -45,6 +45,72 @@ static bool test_mmap_region_basic(void) return true; } +static bool mmap_region_fill_hole(bool merge) +{ + const vma_flags_t vma_flags = mk_vma_flags(VMA_READ_BIT, VMA_WRITE_BIT, + VMA_MAYREAD_BIT, VMA_MAYWRITE_BIT, VMA_MAYEXEC_BIT); + vma_flags_t middle_flags = vma_flags; + struct mm_struct mm = {}; + struct vm_area_struct *vma; + unsigned long addr; + int count = 0; + VMA_ITERATOR(vmi, &mm, 0); + + current->mm = &mm; + if (!merge) + vma_flags_set(&middle_flags, VMA_EXEC_BIT); + + /* Map at 0x300000, length 0x3000. */ + addr = __mmap_region(NULL, 0x300000, 0x3000, vma_flags, 0x300, NULL); + ASSERT_EQ(addr, 0x300000); + + /* Map at 0x306000, length 0x3000, leaving a hole. */ + addr = __mmap_region(NULL, 0x306000, 0x3000, vma_flags, 0x306, NULL); + ASSERT_EQ(addr, 0x306000); + ASSERT_EQ(mm.map_count, 2); + + /* Map at 0x303000, length 0x3000, filling the hole. */ + addr = __mmap_region(NULL, 0x303000, 0x3000, middle_flags, 0x303, NULL); + ASSERT_EQ(addr, 0x303000); + ASSERT_EQ(mm.map_count, merge ? 1 : 3); + + vma_iter_set(&vmi, 0); + for_each_vma(vmi, vma) { + const unsigned long start = 0x300000 + count * 0x3000; + const unsigned long end = merge ? 0x309000 : start + 0x3000; + /* Only the middle VMA in the non-merge case has VMA_EXEC. */ + const bool is_middle_vma = count == 1; + const bool expect_exec_vma = is_middle_vma && !merge; + + ASSERT_EQ(vma->vm_start, start); + ASSERT_EQ(vma->vm_end, end); + ASSERT_EQ(vma_start_pgoff(vma), start >> PAGE_SHIFT); + ASSERT_EQ(vma_start_anon_pgoff(vma), start >> PAGE_SHIFT); + + ASSERT_TRUE(vma_test_all(vma, VMA_READ_BIT, VMA_WRITE_BIT, + VMA_MAYREAD_BIT, VMA_MAYWRITE_BIT, + VMA_MAYEXEC_BIT)); + ASSERT_EQ(vma_test(vma, VMA_EXEC_BIT), expect_exec_vma); + + count++; + } + + ASSERT_EQ(count, mm.map_count); + + ASSERT_EQ(cleanup_mm(&mm, &vmi), count); + return true; +} + +static bool test_mmap_region_fill_hole_merge(void) +{ + return mmap_region_fill_hole(true); +} + +static bool test_mmap_region_fill_hole_flags_mismatch(void) +{ + return mmap_region_fill_hole(false); +} + static bool test_pure_anon_dev_zero(void) { const vma_flags_t vma_flags = mk_vma_flags(VMA_READ_BIT, VMA_WRITE_BIT, @@ -84,5 +150,7 @@ static bool test_pure_anon_dev_zero(void) static void run_mmap_tests(int *num_tests, int *num_fail) { TEST(mmap_region_basic); + TEST(mmap_region_fill_hole_merge); + TEST(mmap_region_fill_hole_flags_mismatch); TEST(pure_anon_dev_zero); } From 045fdaa232e5677eb253c0ecaf7b5a748e1194b0 Mon Sep 17 00:00:00 2001 From: Longlong Xia Date: Sun, 6 Sep 2026 21:59:38 +0800 Subject: [PATCH 0886/1352] mm/zswap: enable zswap_ever_enabled in zswap_pool_create() If zswap is enabled by default at boot and pool creation fails, then a pool is later created by updating the compressor, data written to zswap is corrupted on swapin. Enable the static key in zswap_pool_create(), covering boot-time and runtime pool creation with a single site. Verified with fault injection on a stock kernel (compressor builtin, CONFIG_ZSWAP_DEFAULT_ON=n): 1. Boot with zswap.enabled=1; pool creation fails, init completes pool-less (static key off). 2. Echo an available compressor name to zswap.compressor. 3. Enable zswap. 4. madvise(MADV_PAGEOUT) a pattern-verified 512 MiB region, then fault it back in and verify. Step 4 reads back 131072/131072 zeroed pages (zswpin=0, zswpout=131072) without this patch; all pages intact (zswpin=131072) with it. Link: https://lore.kernel.org/20260906135938.3568108-1-xialonglong2025@163.com Fixes: 2d4d2b1cfb85 ("mm: zswap: add zswap_never_enabled()") Signed-off-by: Longlong Xia Signed-off-by: Andrew Morton Suggested-by: Yosry Ahmed Acked-by: Yosry Ahmed Assisted-by: Zcode:GLM-5.3 Cc: Chengming Zhou Cc: Johannes Weiner Cc: Nhat Pham Cc: --- mm/zswap.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/mm/zswap.c b/mm/zswap.c index d0b6c229b6169d..cfaedc85dfedb3 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -329,6 +329,8 @@ static struct zswap_pool *zswap_pool_create(char *compressor) zswap_pool_debug("created", pool); + static_branch_enable(&zswap_ever_enabled); + return pool; ref_fail: @@ -1818,7 +1820,6 @@ static int zswap_setup(void) pr_info("loaded using pool %s\n", pool->tfm_name); list_add_rcu(&pool->list, &zswap_pools); zswap_has_pool = true; - static_branch_enable(&zswap_ever_enabled); } else { pr_err("pool creation failed\n"); zswap_enabled = false; From f7f58e6714519235e48d65b6ba7267877b8ac668 Mon Sep 17 00:00:00 2001 From: Jianyue Wu Date: Sun, 6 Sep 2026 15:47:17 +0800 Subject: [PATCH 0887/1352] mm/zswap: release retired pools via queue_rcu_work() instead of synchronize_rcu() Patch series "mm/zswap: shrink zswap_entry via a pool id", v6. Every stored page has a struct zswap_entry, so its size is pure per-page overhead. On 64-bit it is currently 56 bytes, of which 8 bytes are a pointer to the owning zswap_pool. Only a handful of pools are ever live: a new pool is created only when the compressor is (re)set, and pools are reused across compressor switches. A list cannot look a pool up by id. An allocating xarray can, which lets each zswap_entry store a u8 instead of a pointer. This series: 1. Releases retired pools with queue_rcu_work() instead of a worker calling synchronize_rcu(), so the release worker no longer blocks on an RCU grace period. 2. Replaces the zswap_pools list with an allocating xarray (XA_FLAGS_ALLOC1 | XA_FLAGS_LOCK_BH) and a separate RCU-protected current-pool pointer, giving each pool a stable small id. Ids start at 1. The reserved id 0 is never allocated, so looking it up resolves to NULL. The table grows as needed up to 255 live pools (u8 pool_idx), not a fixed slot array. 3. Stores that u8 pool id in each zswap_entry instead of the pool pointer. The u8 fits in padding after the bool referenced field. On 64-bit that shrinks the entry from 56 to 48 bytes, which fits 73 to 85 objects in a 4K slab (~2MiB of metadata saved per 1GiB of data held in zswap). The 255-id cap counts every pool still in the xarray. Switching compressor kills the old pool, but that pool stays in the table until its last entry drops the pool's ref, so a draining pool still occupies an id. The id is reused only after xa_erase. Switching back to a compressor whose pool is still in the table resurrects it instead of allocating a new id. If every id is occupied, creating a pool for another compressor fails and the switch is rejected. Pool table locking: xa_for_each() walks and the current-pool pointer use an explicit rcu_read_lock(), because xa_for_each()'s own RCU does not span the loop body. xa_load() takes RCU around the lookup itself, so zswap_entry_pool() needs no extra rcu_read_lock(). The returned pool stays valid because a live entry pins it via percpu_ref, so the id cannot be reused under it. xa_lock is taken only in xa_alloc_bh() and xa_erase_bh(). Benchmark (x86_64, compressor=lzo, MADV_PAGEOUT store + fault-in load): - e2e store+load median latency: no measurable regression vs baseline at matched stored_delta Each store, free, and decompress looks up the pool with xa_load() instead of following a pointer. With only a handful of live pools the xarray walk is short. This patch (of 3): When a pool's last reference is dropped, __zswap_pool_empty() removes it from the pool list and schedules __zswap_pool_release(), which calls synchronize_rcu() to wait for readers before tearing the pool down. synchronize_rcu() is a synchronous, potentially long wait. Replace it with queue_rcu_work(): __zswap_pool_empty() hands the pool to queue_rcu_work(), which waits for a grace period asynchronously and then runs __zswap_pool_release() from a worker for the sleepable teardown (__zswap_pool_empty() can run in atomic context and must not block). The grace-period guarantee is unchanged; the retirement path just no longer blocks on it. Link: https://lore.kernel.org/20260906-shrink_zswap_entry_v6-v6-0-ac4cf61565fb@gmail.com Link: https://lore.kernel.org/20260906-shrink_zswap_entry_v6-v6-1-ac4cf61565fb@gmail.com Signed-off-by: Jianyue Wu Signed-off-by: Andrew Morton Suggested-by: Yosry Ahmed Acked-by: Yosry Ahmed Cc: Chengming Zhou Cc: Chris Li Cc: Johannes Weiner Cc: Nhat Pham --- mm/zswap.c | 12 +++++------- 1 file changed, 5 insertions(+), 7 deletions(-) diff --git a/mm/zswap.c b/mm/zswap.c index cfaedc85dfedb3..526327266e7621 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -155,7 +155,7 @@ struct zswap_pool { struct crypto_acomp_ctx __percpu *acomp_ctx; struct percpu_ref ref; struct list_head list; - struct work_struct release_work; + struct rcu_work release_rwork; struct hlist_node node; char tfm_name[CRYPTO_MAX_ALG_NAME]; }; @@ -386,10 +386,8 @@ static void zswap_pool_destroy(struct zswap_pool *pool) static void __zswap_pool_release(struct work_struct *work) { - struct zswap_pool *pool = container_of(work, typeof(*pool), - release_work); - - synchronize_rcu(); + struct zswap_pool *pool = container_of(to_rcu_work(work), + typeof(*pool), release_rwork); /* nobody should have been able to get a ref... */ WARN_ON(!percpu_ref_is_zero(&pool->ref)); @@ -413,8 +411,8 @@ static void __zswap_pool_empty(struct percpu_ref *ref) list_del_rcu(&pool->list); - INIT_WORK(&pool->release_work, __zswap_pool_release); - schedule_work(&pool->release_work); + INIT_RCU_WORK(&pool->release_rwork, __zswap_pool_release); + queue_rcu_work(system_percpu_wq, &pool->release_rwork); spin_unlock_bh(&zswap_pools_lock); } From e0990478dfd6ee58be4ae9675f8f245277cddc04 Mon Sep 17 00:00:00 2001 From: Jianyue Wu Date: Sun, 6 Sep 2026 15:47:18 +0800 Subject: [PATCH 0888/1352] mm/zswap: replace the zswap_pools list with an allocating xarray Originally zswap kept its pools on an RCU list whose head also served as the current pool. Convert the pool table to an allocating xarray keyed by a small integer id, and track the current pool with a separate RCU-protected pointer. The xarray gives each pool a stable id for a later zswap_entry shrink. XA_FLAGS_ALLOC1 starts ids at 1, so id 0 remains reserved. The id range is bounded by ZSWAP_MAX_POOL_ID because the later entry field is a u8. Keep compressor switching close to the previous flow: look up an existing pool with xa_for_each(), resurrect it if reused, or create a new one. zswap_pool_create() allocates the pool's id and publishes it into the xarray as its final step, so the create call either fully publishes or fully unwinds on failure. Publishing makes the pool live, so a caller that later fails (e.g. param_set_charp()) must still kill the pool to erase it from the xarray. Compressor switches update zswap_current_pool with rcu_assign_pointer(), serialized by the module parameter lock, so no xa_lock is needed for that update. The pool walk above is lockless under RCU. xa_lock is taken only to allocate (xa_alloc_bh()) and erase (xa_erase_bh()) xarray entries. Link: https://lore.kernel.org/20260906-shrink_zswap_entry_v6-v6-2-ac4cf61565fb@gmail.com Signed-off-by: Jianyue Wu Signed-off-by: Andrew Morton Suggested-by: Nhat Pham Suggested-by: Yosry Ahmed Suggested-by: Johannes Weiner Acked-by: Yosry Ahmed Cc: Chengming Zhou Cc: Chris Li --- mm/zswap.c | 122 +++++++++++++++++++++++++++++------------------------ 1 file changed, 66 insertions(+), 56 deletions(-) diff --git a/mm/zswap.c b/mm/zswap.c index 526327266e7621..8b6b1dce79480e 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -34,6 +34,7 @@ #include #include #include +#include #include #include @@ -154,12 +155,23 @@ struct zswap_pool { struct zs_pool *zs_pool; struct crypto_acomp_ctx __percpu *acomp_ctx; struct percpu_ref ref; - struct list_head list; struct rcu_work release_rwork; struct hlist_node node; + u8 idx; char tfm_name[CRYPTO_MAX_ALG_NAME]; }; +/* + * Live pools keyed by id (1..ZSWAP_MAX_POOL_ID). XA_FLAGS_ALLOC1 keeps id 0 + * reserved so it is never handed to a live pool. XA_FLAGS_LOCK_BH makes the + * xa_lock softirq-safe: it is taken from __zswap_pool_empty(), which runs from + * a percpu_ref release callback in softirq context. + */ +#define ZSWAP_FIRST_POOL_ID 1 +#define ZSWAP_MAX_POOL_ID U8_MAX +static DEFINE_XARRAY_FLAGS(zswap_pools, XA_FLAGS_ALLOC1 | XA_FLAGS_LOCK_BH); +static struct zswap_pool __rcu *zswap_current_pool; + /* Global LRU lists shared by all zswap pools. */ static struct list_lru zswap_list_lru; @@ -200,10 +212,6 @@ struct zswap_entry { static struct xarray *zswap_trees[MAX_SWAPFILES]; static unsigned int nr_zswap_trees[MAX_SWAPFILES]; -/* RCU-protected iteration */ -static LIST_HEAD(zswap_pools); -/* protects zswap_pools list modification */ -static DEFINE_SPINLOCK(zswap_pools_lock); /* pool counter to provide unique names to zsmalloc */ static atomic_t zswap_pools_count = ATOMIC_INIT(0); @@ -280,6 +288,7 @@ static struct zswap_pool *zswap_pool_create(char *compressor) struct zswap_pool *pool; char name[38]; /* 'zswap' + 32 char (max) num + \0 */ int ret, cpu; + u32 id; if (!zswap_has_pool && !strcmp(compressor, ZSWAP_PARAM_UNSET)) return NULL; @@ -325,7 +334,22 @@ static struct zswap_pool *zswap_pool_create(char *compressor) PERCPU_REF_ALLOW_REINIT, GFP_KERNEL); if (ret) goto ref_fail; - INIT_LIST_HEAD(&pool->list); + + /* + * Publish only after the pool is fully built, so lockless walkers + * never see a half-initialized pool. The _bh variant pairs with the + * softirq-context xa_lock taken in __zswap_pool_empty(). + */ + ret = xa_alloc_bh(&zswap_pools, &id, pool, + XA_LIMIT(ZSWAP_FIRST_POOL_ID, ZSWAP_MAX_POOL_ID), + GFP_KERNEL); + if (ret) { + if (ret == -EBUSY) + pr_err("cannot allocate pool id (max %d live pools)\n", + ZSWAP_MAX_POOL_ID - ZSWAP_FIRST_POOL_ID + 1); + goto xa_fail; + } + pool->idx = id; zswap_pool_debug("created", pool); @@ -333,6 +357,8 @@ static struct zswap_pool *zswap_pool_create(char *compressor) return pool; +xa_fail: + percpu_ref_exit(&pool->ref); ref_fail: cpuhp_state_remove_instance(CPUHP_MM_ZSWP_POOL_PREPARE, &pool->node); @@ -393,28 +419,22 @@ static void __zswap_pool_release(struct work_struct *work) WARN_ON(!percpu_ref_is_zero(&pool->ref)); percpu_ref_exit(&pool->ref); - /* pool is now off zswap_pools list and has no references. */ + /* The pool is no longer in zswap_pools and has no references. */ zswap_pool_destroy(pool); } -static struct zswap_pool *zswap_pool_current(void); - static void __zswap_pool_empty(struct percpu_ref *ref) { struct zswap_pool *pool; pool = container_of(ref, typeof(*pool), ref); - spin_lock_bh(&zswap_pools_lock); - - WARN_ON(pool == zswap_pool_current()); + WARN_ON(pool == rcu_access_pointer(zswap_current_pool)); - list_del_rcu(&pool->list); + xa_erase_bh(&zswap_pools, pool->idx); INIT_RCU_WORK(&pool->release_rwork, __zswap_pool_release); queue_rcu_work(system_percpu_wq, &pool->release_rwork); - - spin_unlock_bh(&zswap_pools_lock); } static int __must_check zswap_pool_tryget(struct zswap_pool *pool) @@ -440,20 +460,13 @@ static struct zswap_pool *__zswap_pool_current(void) { struct zswap_pool *pool; - pool = list_first_or_null_rcu(&zswap_pools, typeof(*pool), list); + pool = rcu_dereference(zswap_current_pool); WARN_ONCE(!pool && zswap_has_pool, "%s: no page storage pool!\n", __func__); return pool; } -static struct zswap_pool *zswap_pool_current(void) -{ - assert_spin_locked(&zswap_pools_lock); - - return __zswap_pool_current(); -} - static struct zswap_pool *zswap_pool_current_get(void) { struct zswap_pool *pool; @@ -469,23 +482,28 @@ static struct zswap_pool *zswap_pool_current_get(void) return pool; } -/* type and compressor must be null-terminated */ +/* compressor must be null-terminated */ static struct zswap_pool *zswap_pool_find_get(char *compressor) { struct zswap_pool *pool; + unsigned long id; - assert_spin_locked(&zswap_pools_lock); - - list_for_each_entry_rcu(pool, &zswap_pools, list) { + /* + * __zswap_pool_empty() can erase from zswap_pools in softirq while we + * walk. rcu_read_lock() keeps the walk consistent and each pool alive + * across tryget(). xa_for_each()'s own RCU does not span the loop body. + */ + rcu_read_lock(); + xa_for_each(&zswap_pools, id, pool) { if (strcmp(pool->tfm_name, compressor)) continue; /* if we can't get it, it's about to be destroyed */ - if (!zswap_pool_tryget(pool)) - continue; - return pool; + if (zswap_pool_tryget(pool)) + break; } + rcu_read_unlock(); - return NULL; + return pool; } static unsigned long zswap_max_pages(void) @@ -502,9 +520,14 @@ unsigned long zswap_total_pages(void) { struct zswap_pool *pool; unsigned long total = 0; + unsigned long id; + /* + * rcu_read_lock() keeps each pool alive across zs_get_total_pages(). + * xa_for_each()'s own RCU does not span the loop body. + */ rcu_read_lock(); - list_for_each_entry_rcu(pool, &zswap_pools, list) + xa_for_each(&zswap_pools, id, pool) total += zs_get_total_pages(pool->zs_pool); rcu_read_unlock(); @@ -561,20 +584,13 @@ static int zswap_compressor_param_set(const char *val, const struct kernel_param return -ENOENT; } - spin_lock_bh(&zswap_pools_lock); - pool = zswap_pool_find_get(s); - if (pool) { + if (!pool) { + pool = zswap_pool_create(s); + } else { zswap_pool_debug("using existing", pool); - WARN_ON(pool == zswap_pool_current()); - list_del_rcu(&pool->list); - } - - spin_unlock_bh(&zswap_pools_lock); + WARN_ON(pool == rcu_access_pointer(zswap_current_pool)); - if (!pool) - pool = zswap_pool_create(s); - else { /* * Restore the initial ref dropped by percpu_ref_kill() * when the pool was decommissioned and switch it again @@ -591,24 +607,18 @@ static int zswap_compressor_param_set(const char *val, const struct kernel_param else ret = -EINVAL; - spin_lock_bh(&zswap_pools_lock); - + /* + * Compressor switches are serialized by the kernel param lock, so this + * is the only writer of zswap_current_pool: no xa_lock needed. + */ if (!ret) { - put_pool = zswap_pool_current(); - list_add_rcu(&pool->list, &zswap_pools); + put_pool = rcu_access_pointer(zswap_current_pool); + rcu_assign_pointer(zswap_current_pool, pool); zswap_has_pool = true; } else if (pool) { - /* - * Add the possibly pre-existing pool to the end of the pools - * list; if it's new (and empty) then it'll be removed and - * destroyed by the put after we drop the lock - */ - list_add_tail_rcu(&pool->list, &zswap_pools); put_pool = pool; } - spin_unlock_bh(&zswap_pools_lock); - /* * Drop the ref from either the old current pool, * or the new pool we failed to add @@ -1816,7 +1826,7 @@ static int zswap_setup(void) pool = __zswap_pool_create_fallback(); if (pool) { pr_info("loaded using pool %s\n", pool->tfm_name); - list_add_rcu(&pool->list, &zswap_pools); + rcu_assign_pointer(zswap_current_pool, pool); zswap_has_pool = true; } else { pr_err("pool creation failed\n"); From 5bb7a2a89cdb1341858cf9f5656edebb7b58e9bf Mon Sep 17 00:00:00 2001 From: Jianyue Wu Date: Sun, 6 Sep 2026 15:47:19 +0800 Subject: [PATCH 0889/1352] mm/zswap: reference the pool by id to shrink struct zswap_entry struct zswap_entry is one allocation per stored page, so its size is pure overhead. It currently embeds an 8-byte pool pointer, even though the live pools now sit in an allocating xarray keyed by a small integer id that fits in a u8. Replace the per-entry pool pointer with that u8 id and resolve it through the xarray with xa_load(). xa_load() does its own RCU-protected lookup, so the caller needs no rcu_read_lock() section of its own. The resolved pool stays valid because a live entry pins it via percpu_ref (taken in zswap_store_page()), so its id cannot be reused. A live entry never uses the reserved id 0, so a zeroed id resolves to NULL and trips a WARN rather than aliasing a live pool. The u8 fits in the padding after the bool referenced field, shrinking the entry from 56 to 48 bytes on 64-bit. This raises objs_per_slab from 73 to 85 and saves about 2MiB of metadata per 1GiB of data held in zswap. Link: https://lore.kernel.org/20260906-shrink_zswap_entry_v6-v6-3-ac4cf61565fb@gmail.com Signed-off-by: Jianyue Wu Signed-off-by: Andrew Morton Suggested-by: Chris Li Acked-by: Yosry Ahmed Cc: Chengming Zhou Cc: Johannes Weiner Cc: Nhat Pham --- mm/zswap.c | 37 ++++++++++++++++++++++++++++++------- 1 file changed, 30 insertions(+), 7 deletions(-) diff --git a/mm/zswap.c b/mm/zswap.c index 8b6b1dce79480e..ff7c6742af50f5 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -194,7 +194,7 @@ static struct shrinker *zswap_shrinker; * writeback logic. The entry is only reclaimed by the writeback * logic if referenced is unset. See comments in the shrinker * section for context. - * pool - the zswap_pool the entry's data is in + * pool_idx - id of the zswap_pool that the entry's data is in. * handle - zsmalloc allocation handle that stores the compressed page data * objcg - the obj_cgroup that the compressed memory is charged to * lru - handle to the pool's lru used to evict pages. @@ -203,12 +203,22 @@ struct zswap_entry { swp_entry_t swpentry; unsigned int length; bool referenced; - struct zswap_pool *pool; + u8 pool_idx; unsigned long handle; struct obj_cgroup *objcg; struct list_head lru; }; +/* + * No RCU section is needed around the returned pointer: a stored entry pins + * its pool via percpu_ref (taken in zswap_store_page()), so the id cannot be + * reused under us. Callers WARN and handle a NULL from a corrupt pool_idx. + */ +static struct zswap_pool *zswap_entry_pool(struct zswap_entry *entry) +{ + return xa_load(&zswap_pools, entry->pool_idx); +} + static struct xarray *zswap_trees[MAX_SWAPFILES]; static unsigned int nr_zswap_trees[MAX_SWAPFILES]; @@ -766,9 +776,13 @@ static void zswap_entry_cache_free(struct zswap_entry *entry) */ static void zswap_entry_free(struct zswap_entry *entry) { + struct zswap_pool *pool = zswap_entry_pool(entry); + zswap_lru_del(entry); - zs_free(entry->pool->zs_pool, entry->handle); - zswap_pool_put(entry->pool); + if (!WARN_ON_ONCE(!pool)) { + zs_free(pool->zs_pool, entry->handle); + zswap_pool_put(pool); + } if (entry->objcg) { obj_cgroup_uncharge_zswap(entry->objcg, entry->length); obj_cgroup_put(entry->objcg); @@ -924,12 +938,15 @@ static bool zswap_compress(struct folio *folio, long index, static bool zswap_decompress(struct zswap_entry *entry, struct folio *folio) { - struct zswap_pool *pool = entry->pool; + struct zswap_pool *pool = zswap_entry_pool(entry); struct scatterlist input[2]; /* zsmalloc returns an SG list 1-2 entries */ struct scatterlist output; struct crypto_acomp_ctx *acomp_ctx; int ret = 0, dlen; + if (WARN_ON_ONCE(!pool)) + return false; + acomp_ctx = raw_cpu_ptr(pool->acomp_ctx); mutex_lock(&acomp_ctx->mutex); zs_obj_read_sg_begin(pool->zs_pool, entry->handle, input, entry->length); @@ -965,7 +982,7 @@ static bool zswap_decompress(struct zswap_entry *entry, struct folio *folio) pr_alert_ratelimited("Decompression error from zswap (%d:%lu %s %u->%d)\n", swp_type(entry->swpentry), swp_offset(entry->swpentry), - entry->pool->tfm_name, + pool->tfm_name, entry->length, dlen); return false; } @@ -1423,6 +1440,13 @@ static bool zswap_store_page(struct folio *folio, long index, if (!zswap_compress(folio, index, entry, pool)) goto compress_failed; + /* + * Set pool_idx before the xa_store() below publishes the entry, or a + * concurrent reader could resolve a stale pool_idx left by slab reuse + * to an unrelated live pool. + */ + entry->pool_idx = pool->idx; + old = xa_store(swap_zswap_tree(page_swpentry), swp_offset(page_swpentry), entry, GFP_KERNEL); @@ -1468,7 +1492,6 @@ static bool zswap_store_page(struct folio *folio, long index, * The publishing order matters to prevent writeback from seeing * an incoherent entry. */ - entry->pool = pool; entry->swpentry = page_swpentry; entry->objcg = objcg; entry->referenced = true; From a511358cd03589e6748dda2a2c75241a06713c5d Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 10 Sep 2026 07:22:28 -0700 Subject: [PATCH 0890/1352] mm/damon/api: introduce DAMON_FILTER_TYPE_PGIDLE_SET Patch series "mm/damon: introduce pgidle_set probe filter type". Users can do effective access monitoring with DAMON, using set_pgidle probe prep operation and pgidle_unset allowing probe filter. In the setup, high probe_hits means the region was accessed frequently. Zero probe_hits means the region was not accessed. The filter, however, matches only the memory that can show the idleness. Regions of zero probe_hits may contain memory that was just unable to show the idleness. On virtual address space based monitoring, for example, non-present pages are reported as cold. It is truly cold, but if the purpose is to make some operations like reclaiming or compressing the data, this is not really useful. Introduce a new probe filter type, pgidle_set, that confirms the page is marked as idle. Using it instead of pgidle_unset, users can find cold pages that can effectively be processed. Patch 1 introduces the pgidle_set probe filter type to DAMON kernel API. Patches 2 and 3 add support for it on physical and virtual address space operation sets, respectively. Patch 4 adds support for it on the DAMON sysfs interface. Finally, patch 5 updates the documentation for the new filter type. This patch (of 5): DAMON_FILTER_TYPE_PGIDLE_UNSET matches only memory that was able to confirm if it is marked as idle. If the memory cannot be marked as idle, it simply doesn't match. Hence, finding memory that was marked as idle with only DAMON_FILTER_TYPE_PGIDLE_UNSET is impossible. Introduce a new filter for the purpose, DAMON_FILTER_TYPE_PGIDLE_SET. Link: https://lore.kernel.org/20260910142234.171562-1-sj@kernel.org Link: https://lore.kernel.org/20260910142234.171562-2-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/damon.h | 2 ++ 1 file changed, 2 insertions(+) diff --git a/include/linux/damon.h b/include/linux/damon.h index 871d26adf6ae57..16800b3dc379b9 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -774,11 +774,13 @@ struct damon_prep { * @DAMON_FILTER_TYPE_ANON: Anonymous pages. * @DAMON_FILTER_TYPE_MEMCG: Specific memcg's pages. * @DAMON_FILTER_TYPE_PGIDLE_UNSET: Pgidle is unset. + * @DAMON_FILTER_TYPE_PGIDLE_SET: Pgidle is set. */ enum damon_filter_type { DAMON_FILTER_TYPE_ANON, DAMON_FILTER_TYPE_MEMCG, DAMON_FILTER_TYPE_PGIDLE_UNSET, + DAMON_FILTER_TYPE_PGIDLE_SET, }; /** From 31e37c22782342bca180712c634e77e321c76606 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 10 Sep 2026 07:22:29 -0700 Subject: [PATCH 0891/1352] mm/damon/paddr: support DAMON_FILTER_TYPE_PGIDLE_SET Implement DAMON_FILTER_TYPE_PGIDLE_SET support on the physical address space DAMON operation set (paddr). Link: https://lore.kernel.org/20260910142234.171562-3-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/paddr.c | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c index d7c81829445ba1..79195026b903a6 100644 --- a/mm/damon/paddr.c +++ b/mm/damon/paddr.c @@ -151,6 +151,12 @@ static bool damon_pa_filter_match(struct damon_filter *filter, else matched = damon_folio_young(folio); break; + case DAMON_FILTER_TYPE_PGIDLE_SET: + if (!folio) + matched = false; + else + matched = damon_folio_young(folio) == false; + break; default: return damon_ops_filter_match(filter, folio); } From 174051f9324170578be31f1c50962a95e7b9525e Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 10 Sep 2026 07:22:30 -0700 Subject: [PATCH 0892/1352] mm/damon/vaddr: support DAMON_FILTER_TYPE_PGIDLE_SET Implement DAMON_FILTER_TYPE_PGIDLE_SET support on the virtual address space DAMON operation set (vaddr). Link: https://lore.kernel.org/20260910142234.171562-4-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/vaddr.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/mm/damon/vaddr.c b/mm/damon/vaddr.c index 9a38dc89a156ef..7063356370c34b 100644 --- a/mm/damon/vaddr.c +++ b/mm/damon/vaddr.c @@ -550,6 +550,13 @@ static bool damon_va_filter_match(struct damon_filter *filter, matched = damon_va_young_addr(folio, pte, pmd, mm, addr); break; + case DAMON_FILTER_TYPE_PGIDLE_SET: + if (!folio) + matched = false; + else + matched = !damon_va_young_addr(folio, pte, pmd, mm, + addr); + break; default: return damon_ops_filter_match(filter, folio); } From 90ad761c7d56ab46a06726ce66d0ff2e2481cb1a Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 10 Sep 2026 07:22:31 -0700 Subject: [PATCH 0893/1352] mm/damon/sysfs: support DAMON_FILTER_TYPE_PGIDLE_SET Extend DAMON sysfs interface to let users setup DAMON_FILTER_TYPE_PGIDLE_SET probe filter. Link: https://lore.kernel.org/20260910142234.171562-5-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index b576e97cbfdb8a..51fa506c879b02 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -1001,6 +1001,10 @@ damon_sysfs_filter_type_names[] = { .type = DAMON_FILTER_TYPE_PGIDLE_UNSET, .name = "pgidle_unset", }, + { + .type = DAMON_FILTER_TYPE_PGIDLE_SET, + .name = "pgidle_set", + }, }; static ssize_t type_show(struct kobject *kobj, From 059a9fc87f1ac71b8af3431a082ba27ecbb9bef0 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 10 Sep 2026 07:22:32 -0700 Subject: [PATCH 0894/1352] Docs/mm/damon/design: update for pgidle_set probe filter Update DAMON design document for the newly added pgidle_set probe filter type. Link: https://lore.kernel.org/20260910142234.171562-6-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/mm/damon/design.rst | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index aac84de261aa8a..22b785cd11dfec 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -299,6 +299,7 @@ filter types. Currently below filter types are supported. - ``memcg``: Same to that for DAMOS filters. - ``pgidle_unset``: Matches if the page for the memory is marked as not access-idle. +- ``pgidle_set``: Matches if the page for the memory is marked as access-idle. If such probes are registered, DAMON executes the probes for each region's sampling memory when it does the access :ref:`sampling @@ -314,7 +315,7 @@ actions are registered, DAMON applies the actions to each region's sampling memory before starting the next sampling interval. Currently only one action, ``set_pgidle`` is supported. The action marks the page for the probing target memory as access-idle. This can be useful to be used together with -``pgidle_unset`` probe filter. +``pgidle_unset`` or ``pgidle_set`` probe filter. This is a sampling based mechanism. Hence, it is lightweight but the output may include some measurement errors. The output should be used with good From 16fc80dccb87394087270098f730117c729771c9 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Wed, 30 Sep 2026 22:06:16 +0800 Subject: [PATCH 0895/1352] mm/sparse-vmemmap: factor out shared vmemmap tail page allocation Patch series "mm: Switch device DAX to section-based vmemmap optimization", v6. This series is split out from the earlier, larger series "mm: Generalize HVO for HugeTLB and device DAX" [1]. While the parent series generalizes vmemmap optimization across HugeTLB and device DAX, this subset addresses a single, self-contained step: switching device DAX to the section-based sparse-vmemmap optimization infrastructure introduced for HugeTLB. After the HugeTLB conversion, optimized vmemmap state is described by the memory section and the sparse-vmemmap population path can allocate or reuse shared tail vmemmap pages based on that metadata. Device DAX still uses the older DAX-specific population model, including a separate tail vmemmap page reservation and architecture-specific logic to locate or populate reusable tail pages. This series makes device DAX use the same section-based model. Device DAX records the compound page order from pgmap->vmemmap_shift in section metadata before vmemmap population, uses the common per-zone shared tail vmemmap page, and drops the extra reserved tail page. The powerpc radix path is updated to use the same shared tail-page helper, so the generic and powerpc DAX paths follow the same reservation model. The first patches prepare the shared infrastructure by factoring out shared tail-page allocation, allocating the per-zone shared tail-page array dynamically, and introducing a generic CONFIG_VMEMMAP_OPTIMIZATION symbol. The middle patches move device DAX onto that infrastructure by recording the device DAX compound page order in memory-section metadata, using that metadata to back generic device DAX mappings with the common per-zone shared tail page, exposing the shared helpers so the powerpc radix path can use the same model, and dropping the extra DAX-only tail page reservation and the now-unused section accounting arguments. The final patch updates the documentation for the new DAX layout. This is intended to be the third smaller step toward the broader HVO generalization. The wider HVO consolidation between HugeTLB and device DAX is left for follow-up series. This patch (of 12): HugeTLB and sparse-vmemmap each have their own helper to allocate the shared vmemmap tail page used by vmemmap optimization. Factor that logic into a common vmemmap_shared_tail_page() helper. It allocates the page through vmemmap_alloc_block(), initializes the tail struct pages, and uses cmpxchg() to install the per-zone shared page. This removes duplicate allocation logic while handling both early boot and runtime allocation through the same helper. Link: https://lore.kernel.org/20260930140627.57431-1-songmuchun@bytedance.com Link: https://lore.kernel.org/20260930140627.57431-2-songmuchun@bytedance.com Link: https://lore.kernel.org/20260513130542.35604-1-songmuchun@bytedance.com/ [1] Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Acked-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Ritesh Harjani (IBM) Cc: Shrikanth Hegde Cc: Randy Dunlap Cc: Lance Yang --- mm/hugetlb_vmemmap.c | 29 +----------------- mm/sparse-vmemmap.c | 70 ++++++++++++++++++++------------------------ mm/sparse.h | 3 ++ 3 files changed, 36 insertions(+), 66 deletions(-) diff --git a/mm/hugetlb_vmemmap.c b/mm/hugetlb_vmemmap.c index f977d0a7e00274..76765c97ff68b4 100644 --- a/mm/hugetlb_vmemmap.c +++ b/mm/hugetlb_vmemmap.c @@ -19,7 +19,6 @@ #include #include "hugetlb_vmemmap.h" #include "sparse.h" -#include "internal.h" /** * struct vmemmap_remap_walk - walk vmemmap page table @@ -493,32 +492,6 @@ static bool vmemmap_should_optimize_folio(const struct hstate *h, struct folio * return true; } -static struct page *vmemmap_get_tail(unsigned int order, struct zone *zone) -{ - const unsigned int idx = order - VMEMMAP_OPTIMIZATION_MIN_ORDER; - struct page *tail, *p; - int node = zone_to_nid(zone); - - tail = READ_ONCE(zone->vmemmap_tails[idx]); - if (likely(tail)) - return tail; - - tail = alloc_pages_node(node, GFP_KERNEL | __GFP_ZERO, 0); - if (!tail) - return NULL; - - p = page_to_virt(tail); - for (int i = 0; i < PAGE_SIZE / sizeof(struct page); i++) - init_compound_tail(p + i, NULL, order, zone); - - if (cmpxchg(&zone->vmemmap_tails[idx], NULL, tail)) { - __free_page(tail); - tail = READ_ONCE(zone->vmemmap_tails[idx]); - } - - return tail; -} - static int __hugetlb_vmemmap_optimize_folio(const struct hstate *h, struct folio *folio, struct list_head *vmemmap_pages, @@ -535,7 +508,7 @@ static int __hugetlb_vmemmap_optimize_folio(const struct hstate *h, return ret; nid = folio_nid(folio); - vmemmap_tail = vmemmap_get_tail(h->order, folio_zone(folio)); + vmemmap_tail = vmemmap_shared_tail_page(h->order, folio_zone(folio)); if (!vmemmap_tail) return -ENOMEM; diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index f22d815d7af04e..9af349581fbd1b 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -42,27 +42,13 @@ #include "mm_init.h" #include "sparse.h" -/* - * Allocate a block of memory to be used to back the virtual memory map - * or to back the page tables that are used to create the mapping. - * Uses the main allocators if they are available, else bootmem. - */ - -static void * __ref __earlyonly_bootmem_alloc(int node, - unsigned long size, - unsigned long align, - unsigned long goal) -{ - return memmap_alloc(size, align, goal, node, false); -} - -void * __meminit vmemmap_alloc_block(unsigned long size, int node) +void __ref *vmemmap_alloc_block(unsigned long size, int node) { /* If the main allocator is up use that, fallback to bootmem. */ if (slab_is_available()) { gfp_t gfp_mask = GFP_KERNEL|__GFP_RETRY_MAYFAIL|__GFP_NOWARN; int order = get_order(size); - static bool warned __meminitdata; + static bool warned; struct page *page; page = alloc_pages_node(node, gfp_mask, order); @@ -76,8 +62,7 @@ void * __meminit vmemmap_alloc_block(unsigned long size, int node) } return NULL; } else - return __earlyonly_bootmem_alloc(node, size, size, - __pa(MAX_DMA_ADDRESS)); + return memmap_alloc(size, size, __pa(MAX_DMA_ADDRESS), node, false); } static void * __meminit altmap_alloc_block_buf(unsigned long size, @@ -185,34 +170,43 @@ static void * __meminit vmemmap_alloc_block_zero(unsigned long size, int node) } #ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP -static __meminit struct page *vmemmap_get_tail(unsigned int order, struct zone *zone) +struct page __ref *vmemmap_shared_tail_page(unsigned int order, struct zone *zone) { - struct page *p, *tail; - unsigned int idx; - int node = zone_to_nid(zone); + const unsigned int idx = order - VMEMMAP_OPTIMIZATION_MIN_ORDER; + struct page *page; + void *addr; - if (WARN_ON_ONCE(order < VMEMMAP_OPTIMIZATION_MIN_ORDER)) - return NULL; - if (WARN_ON_ONCE(order > MAX_FOLIO_ORDER)) + if (WARN_ON_ONCE(idx >= VMEMMAP_OPTIMIZATION_NR_ORDERS)) return NULL; - idx = order - VMEMMAP_OPTIMIZATION_MIN_ORDER; - tail = zone->vmemmap_tails[idx]; - if (tail) - return tail; - p = vmemmap_alloc_block_zero(PAGE_SIZE, node); - if (!p) + page = READ_ONCE(zone->vmemmap_tails[idx]); + if (page) + return page; + + addr = vmemmap_alloc_block(PAGE_SIZE, zone_to_nid(zone)); + if (!addr) return NULL; - for (int i = 0; i < PAGE_SIZE / sizeof(struct page); i++) - init_compound_tail(p + i, NULL, order, zone); - tail = virt_to_page(p); - zone->vmemmap_tails[idx] = tail; + for (int i = 0; i < PAGE_SIZE / sizeof(struct page); i++) { + page = (struct page *)addr + i; + mm_zero_struct_page(page); + init_compound_tail(page, NULL, order, zone); + } - return tail; + page = virt_to_page(addr); + if (cmpxchg(&zone->vmemmap_tails[idx], NULL, page) != NULL) { + if (slab_is_available()) + __free_page(page); + else + memblock_free(addr, PAGE_SIZE); + page = READ_ONCE(zone->vmemmap_tails[idx]); + } + + return page; } #else -static inline struct page *vmemmap_get_tail(unsigned int order, struct zone *zone) +static inline struct page *vmemmap_shared_tail_page(unsigned int order, + struct zone *zone) { return NULL; } @@ -229,7 +223,7 @@ static __meminit void *vmemmap_alloc_pte(unsigned long pfn, int node, return vmemmap_alloc_block_buf(PAGE_SIZE, node, altmap); zone = pfn_to_zone(pfn, node); - page = vmemmap_get_tail(order, zone); + page = vmemmap_shared_tail_page(order, zone); if (!page) return NULL; diff --git a/mm/sparse.h b/mm/sparse.h index d3a71ef4fad0fe..6e7aaeaa559471 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -142,6 +142,9 @@ static inline void sparse_sections_init(void) {} * mm/sparse-vmemmap.c */ #ifdef CONFIG_SPARSEMEM_VMEMMAP +#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP +struct page *vmemmap_shared_tail_page(unsigned int order, struct zone *zone); +#endif void sparse_init_subsection_map(void); int section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, struct vmem_altmap *altmap, struct dev_pagemap *pgmap); From 3514caff3840f6195e3aa8b796ab745143939f2a Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Wed, 30 Sep 2026 22:06:17 +0800 Subject: [PATCH 0896/1352] mm/sparse-vmemmap: allocate shared tail page array dynamically Commit 622026e87c40 ("mm/hugetlb: remove fake head pages") added the per-zone vmemmap_tails array. Its size depends on MAX_FOLIO_ORDER, which had been moved to mmzone.h in preparation for the array. PUD_ORDER is defined by linux/pgtable.h, which cannot be included from mmzone.h without creating an include cycle. It was therefore open-coded as PUD_SHIFT - PAGE_SHIFT. This removed the dependency on PUD_ORDER, but not the underlying dependency on architecture page-table definitions. PUD_SHIFT is generally provided by architecture page-table headers, which are not guaranteed to have been included when mmzone.h is parsed. The dependency remained hidden because vmemmap_tails was originally guarded by CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP. Under that condition, MAX_FOLIO_ORDER resolves to either MAX_PAGE_ORDER or the fixed HugeTLB limit, rather than the PUD_SHIFT-based definition. Device DAX, however, does not require CONFIG_HUGETLB_PAGE. When it is converted to use section-based vmemmap optimization, MAX_FOLIO_ORDER can resolve to PUD_SHIFT - PAGE_SHIFT while it is being used to size vmemmap_tails. This would make struct zone depend on architecture page-table definitions being available when mmzone.h is parsed. Replace the embedded array with a pointer and allocate it on first use. This moves the order-count evaluation into sparse-vmemmap.c, after the architecture page-table definitions are available, and removes the dependency from mmzone.h. Removing the compile-time array also removes the original reason for keeping MAX_FOLIO_ORDER and the vmemmap optimization sizing definitions in mmzone.h. Follow-up cleanups can place each definition in the header owned by its respective subsystem. Link: https://lore.kernel.org/20260930140627.57431-3-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Mike Rapoport Cc: Qi Zheng Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Ritesh Harjani (IBM) Cc: Shrikanth Hegde Cc: Randy Dunlap Cc: Lance Yang --- include/linux/mmzone.h | 7 +------ mm/sparse-vmemmap.c | 39 +++++++++++++++++++++++++++++++++++---- 2 files changed, 36 insertions(+), 10 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index acd94cecc0d399..68807ff7f9465e 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -113,11 +113,6 @@ (VMEMMAP_OPTIMIZATION_PAGES * PAGE_SIZE / sizeof(struct page)) #define VMEMMAP_OPTIMIZATION_MIN_ORDER (ilog2(VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES) + 1) -#define __VMEMMAP_OPTIMIZATION_NR_ORDERS \ - (MAX_FOLIO_ORDER - VMEMMAP_OPTIMIZATION_MIN_ORDER + 1) -#define VMEMMAP_OPTIMIZATION_NR_ORDERS \ - (__VMEMMAP_OPTIMIZATION_NR_ORDERS > 0 ? __VMEMMAP_OPTIMIZATION_NR_ORDERS : 0) - enum migratetype { MIGRATE_UNMOVABLE, MIGRATE_MOVABLE, @@ -1156,7 +1151,7 @@ struct zone { atomic_long_t vm_stat[NR_VM_ZONE_STAT_ITEMS]; atomic_long_t vm_numa_event[NR_VM_NUMA_EVENT_ITEMS]; #ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP - struct page *vmemmap_tails[VMEMMAP_OPTIMIZATION_NR_ORDERS]; + struct page **vmemmap_tails; #endif } ____cacheline_internodealigned_in_smp; diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 9af349581fbd1b..0ace48268095f5 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -170,16 +170,47 @@ static void * __meminit vmemmap_alloc_block_zero(unsigned long size, int node) } #ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP +#define VMEMMAP_OPTIMIZATION_NR_ORDERS (MAX_FOLIO_ORDER - VMEMMAP_OPTIMIZATION_MIN_ORDER + 1) + +static __ref struct page **vmemmap_tails(struct zone *zone) +{ + const size_t size = array_size(VMEMMAP_OPTIMIZATION_NR_ORDERS, + sizeof(*zone->vmemmap_tails)); + struct page **pages = READ_ONCE(zone->vmemmap_tails); + + if (pages) + return pages; + + pages = slab_is_available() ? kzalloc_objs(*pages, VMEMMAP_OPTIMIZATION_NR_ORDERS) : + memblock_alloc(size, __alignof__(*pages)); + if (!pages) + return NULL; + + if (cmpxchg(&zone->vmemmap_tails, NULL, pages) != NULL) { + if (slab_is_available()) + kfree(pages); + else + memblock_free(pages, size); + pages = READ_ONCE(zone->vmemmap_tails); + } + + return pages; +} + struct page __ref *vmemmap_shared_tail_page(unsigned int order, struct zone *zone) { const unsigned int idx = order - VMEMMAP_OPTIMIZATION_MIN_ORDER; - struct page *page; + struct page *page, **pages; void *addr; if (WARN_ON_ONCE(idx >= VMEMMAP_OPTIMIZATION_NR_ORDERS)) return NULL; - page = READ_ONCE(zone->vmemmap_tails[idx]); + pages = vmemmap_tails(zone); + if (!pages) + return NULL; + + page = READ_ONCE(pages[idx]); if (page) return page; @@ -194,12 +225,12 @@ struct page __ref *vmemmap_shared_tail_page(unsigned int order, struct zone *zon } page = virt_to_page(addr); - if (cmpxchg(&zone->vmemmap_tails[idx], NULL, page) != NULL) { + if (cmpxchg(&pages[idx], NULL, page) != NULL) { if (slab_is_available()) __free_page(page); else memblock_free(addr, PAGE_SIZE); - page = READ_ONCE(zone->vmemmap_tails[idx]); + page = READ_ONCE(pages[idx]); } return page; From ee74f9143d3ba1ae0fdf68e16e661479b7ea6f60 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Wed, 30 Sep 2026 22:06:18 +0800 Subject: [PATCH 0897/1352] mm/sparse-vmemmap: introduce CONFIG_VMEMMAP_OPTIMIZATION The section-based vmemmap optimization infrastructure is guarded by CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP, but it can also be used by ZONE_DEVICE users that set dev_pagemap::vmemmap_shift. Introduce CONFIG_VMEMMAP_OPTIMIZATION as a common config for the shared infrastructure. Select the new option from HUGETLB_PAGE_OPTIMIZE_VMEMMAP and from ZONE_DEVICE when the architecture opts in to DAX vmemmap optimization, and use it to guard the generic sparse-vmemmap state and helpers. Link: https://lore.kernel.org/20260930140627.57431-4-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Acked-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Ritesh Harjani (IBM) Cc: Shrikanth Hegde Cc: Randy Dunlap Cc: Lance Yang --- arch/x86/entry/vdso/vdso32/fake_32bit_build.h | 2 +- fs/Kconfig | 1 + include/linux/mm.h | 3 +++ include/linux/mmzone.h | 10 +++++----- include/linux/page-flags.h | 5 ++--- mm/Kconfig | 5 +++++ mm/sparse-vmemmap.c | 2 +- mm/sparse.h | 6 +++--- 8 files changed, 21 insertions(+), 13 deletions(-) diff --git a/arch/x86/entry/vdso/vdso32/fake_32bit_build.h b/arch/x86/entry/vdso/vdso32/fake_32bit_build.h index bc3e549795c3f3..72a92cb9b53d3b 100644 --- a/arch/x86/entry/vdso/vdso32/fake_32bit_build.h +++ b/arch/x86/entry/vdso/vdso32/fake_32bit_build.h @@ -11,7 +11,7 @@ #undef CONFIG_PGTABLE_LEVELS #undef CONFIG_ILLEGAL_POINTER_VALUE #undef CONFIG_SPARSEMEM_VMEMMAP -#undef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP +#undef CONFIG_VMEMMAP_OPTIMIZATION #undef CONFIG_NR_CPUS #undef CONFIG_PARAVIRT_XXL diff --git a/fs/Kconfig b/fs/Kconfig index d1c210c6508f0a..1454b7fe9641fd 100644 --- a/fs/Kconfig +++ b/fs/Kconfig @@ -278,6 +278,7 @@ config HUGETLB_PAGE_OPTIMIZE_VMEMMAP def_bool HUGETLB_PAGE depends on ARCH_WANT_OPTIMIZE_HUGETLB_VMEMMAP depends on SPARSEMEM_VMEMMAP + select VMEMMAP_OPTIMIZATION config HUGETLB_PMD_PAGE_TABLE_SHARING def_bool HUGETLB_PAGE diff --git a/include/linux/mm.h b/include/linux/mm.h index c49ef99b4413b4..070ce27e9cd3cd 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -5175,6 +5175,9 @@ static inline bool __vmemmap_can_optimize(struct vmem_altmap *altmap, unsigned long nr_pages; unsigned long nr_vmemmap_pages; + if (!IS_ENABLED(CONFIG_VMEMMAP_OPTIMIZATION)) + return false; + if (!pgmap || !is_power_of_2(sizeof(struct page))) return false; diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 68807ff7f9465e..ee9cbaaa63f4f2 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -102,9 +102,9 @@ * * HVO which is only active if the size of struct page is a power of 2. */ -#define MAX_FOLIO_VMEMMAP_ALIGN \ - (IS_ENABLED(CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP) && \ - is_power_of_2(sizeof(struct page)) ? \ +#define MAX_FOLIO_VMEMMAP_ALIGN \ + (IS_ENABLED(CONFIG_VMEMMAP_OPTIMIZATION) && \ + is_power_of_2(sizeof(struct page)) ? \ MAX_FOLIO_NR_PAGES * sizeof(struct page) : 0) /* The number of retained vmemmap pages with HVO enabled. */ @@ -1150,7 +1150,7 @@ struct zone { /* Zone statistics */ atomic_long_t vm_stat[NR_VM_ZONE_STAT_ITEMS]; atomic_long_t vm_numa_event[NR_VM_NUMA_EVENT_ITEMS]; -#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP +#ifdef CONFIG_VMEMMAP_OPTIMIZATION struct page **vmemmap_tails; #endif } ____cacheline_internodealigned_in_smp; @@ -2014,7 +2014,7 @@ struct mem_section { unsigned long section_mem_map; struct mem_section_usage *usage; -#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP +#ifdef CONFIG_VMEMMAP_OPTIMIZATION /* * Normally, sections hold regular (order-0) pages. However, for * sections with HVO enabled, this tracks the compound page order diff --git a/include/linux/page-flags.h b/include/linux/page-flags.h index 86dd0470da1173..7080a6a1a79e72 100644 --- a/include/linux/page-flags.h +++ b/include/linux/page-flags.h @@ -208,14 +208,13 @@ enum pageflags { static __always_inline bool compound_info_has_mask(void) { /* - * Limit mask usage to HugeTLB vmemmap optimization (HVO) where it - * makes a difference. + * Limit mask usage to HVO where it makes a difference. * * The approach with mask would work in the wider set of conditions, * but it requires validating that struct pages are naturally aligned * for all orders up to the MAX_FOLIO_ORDER, which can be tricky. */ - if (!IS_ENABLED(CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP)) + if (!IS_ENABLED(CONFIG_VMEMMAP_OPTIMIZATION)) return false; return is_power_of_2(sizeof(struct page)); diff --git a/mm/Kconfig b/mm/Kconfig index bc7befafb47b57..30170a936f1fc0 100644 --- a/mm/Kconfig +++ b/mm/Kconfig @@ -461,6 +461,10 @@ config SPARSEMEM_VMEMMAP pfn_to_page and page_to_pfn operations. This is the most efficient option when sufficient kernel resources are available. +config VMEMMAP_OPTIMIZATION + bool + depends on SPARSEMEM_VMEMMAP + # # Select this config option from the architecture Kconfig, if it is preferred # to enable the feature of HugeTLB/dev_dax vmemmap optimization. @@ -1220,6 +1224,7 @@ config ZONE_DMA32 config ZONE_DEVICE bool "Device memory (pmem, HMM, etc...) hotplug support" depends on MEMORY_HOTREMOVE + select VMEMMAP_OPTIMIZATION if ARCH_WANT_OPTIMIZE_DAX_VMEMMAP select XARRAY_MULTI help diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 0ace48268095f5..6916a36907781e 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -169,7 +169,7 @@ static void * __meminit vmemmap_alloc_block_zero(unsigned long size, int node) return p; } -#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP +#ifdef CONFIG_VMEMMAP_OPTIMIZATION #define VMEMMAP_OPTIMIZATION_NR_ORDERS (MAX_FOLIO_ORDER - VMEMMAP_OPTIMIZATION_MIN_ORDER + 1) static __ref struct page **vmemmap_tails(struct zone *zone) diff --git a/mm/sparse.h b/mm/sparse.h index 6e7aaeaa559471..326ad43bb5c37a 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -10,7 +10,7 @@ #include -#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP +#ifdef CONFIG_VMEMMAP_OPTIMIZATION static inline unsigned int section_compound_order(const struct mem_section *section) { return section->compound_page_order; @@ -75,7 +75,7 @@ static inline bool vmemmap_optimizable_pfn(unsigned long pfn) static inline bool vmemmap_optimizable_order(unsigned int order) { - if (!IS_ENABLED(CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP)) + if (!IS_ENABLED(CONFIG_VMEMMAP_OPTIMIZATION)) return false; if (!is_power_of_2(sizeof(struct page))) @@ -142,7 +142,7 @@ static inline void sparse_sections_init(void) {} * mm/sparse-vmemmap.c */ #ifdef CONFIG_SPARSEMEM_VMEMMAP -#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP +#ifdef CONFIG_VMEMMAP_OPTIMIZATION struct page *vmemmap_shared_tail_page(unsigned int order, struct zone *zone); #endif void sparse_init_subsection_map(void); From 7c4e12007d0fcaec0773d95045d3afadb74cc6c8 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Wed, 30 Sep 2026 22:06:19 +0800 Subject: [PATCH 0898/1352] mm/sparse-vmemmap: open-code init_compound_tail() init_compound_tail() is only used by vmemmap_shared_tail_page(), where the shared tail page setup intentionally passes NULL as the compound head. Keeping this helper in mm/internal.h exposes that special case to the rest of the MM code and can make the NULL head argument look generally valid. Open-code the initialization at the only call site so the special-case use stays local to sparse vmemmap optimization. No functional change intended. Link: https://lore.kernel.org/20260930140627.57431-5-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Acked-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Ritesh Harjani (IBM) Cc: Shrikanth Hegde Cc: Randy Dunlap Cc: Lance Yang --- mm/internal.h | 9 --------- mm/sparse-vmemmap.c | 5 ++++- 2 files changed, 4 insertions(+), 10 deletions(-) diff --git a/mm/internal.h b/mm/internal.h index da14c56fb24e11..0dca33db068f6b 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -786,15 +786,6 @@ static inline void prep_compound_tail(struct page *tail, VM_WARN_ON_ONCE(tail->private); } -static inline void init_compound_tail(struct page *tail, - const struct page *head, unsigned int order, struct zone *zone) -{ - atomic_set(&tail->_mapcount, -1); - set_page_node(tail, zone_to_nid(zone)); - set_page_zone(tail, zone_idx(zone)); - prep_compound_tail(tail, head, order); -} - #if defined CONFIG_COMPACTION || defined CONFIG_CMA /* diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 6916a36907781e..f77e6e1e5a8bf0 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -221,7 +221,10 @@ struct page __ref *vmemmap_shared_tail_page(unsigned int order, struct zone *zon for (int i = 0; i < PAGE_SIZE / sizeof(struct page); i++) { page = (struct page *)addr + i; mm_zero_struct_page(page); - init_compound_tail(page, NULL, order, zone); + atomic_set(&page->_mapcount, -1); + set_page_node(page, zone_to_nid(zone)); + set_page_zone(page, zone_idx(zone)); + prep_compound_tail(page, NULL, order); } page = virt_to_page(addr); From 75b78e0d8f0b07c8e7f08aed5d654f384b31de2c Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Wed, 30 Sep 2026 22:06:20 +0800 Subject: [PATCH 0899/1352] mm/sparse-vmemmap: prepare DAX vmemmap population for compound page orders Device DAX still uses vmemmap_populate_compound_pages() to populate its compound-page vmemmap mappings. That helper allocates the head and first tail vmemmap pages explicitly, then reuses the first tail page for the remaining tail page mappings. Device DAX is being moved to the section-based vmemmap optimization infrastructure, but it cannot switch to the generic section-based population path yet. Once a later patch records the DAX compound page order in section metadata, DAX head and first-tail PFNs can look optimizable to the generic helpers as well. Rename the existing VMEMMAP_POPULATE_PAGEREF flag to VMEMMAP_POPULATE_DAX and pass it through all paths in vmemmap_populate_compound_pages() during this transition. This keeps DAX head/first-tail allocations on the normal vmemmap allocation path, while preserving the existing page reference for reused DAX tail mappings. Link: https://lore.kernel.org/20260930140627.57431-6-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Cc: David Hildenbrand Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Mike Rapoport Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Ritesh Harjani (IBM) Cc: Shrikanth Hegde Cc: Randy Dunlap Cc: Lance Yang --- mm/sparse-vmemmap.c | 27 +++++++++++++++------------ 1 file changed, 15 insertions(+), 12 deletions(-) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index f77e6e1e5a8bf0..b4b72f233bf656 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -35,8 +35,8 @@ /* * Flags for vmemmap_populate_range and friends. */ -/* Get a ref on the head page struct page, for ZONE_DEVICE compound pages */ -#define VMEMMAP_POPULATE_PAGEREF 0x0001 +/* Vmemmap population for ZONE_DEVICE compound pages */ +#define VMEMMAP_POPULATE_DAX 0x0001 #include "internal.h" #include "mm_init.h" @@ -247,13 +247,17 @@ static inline struct page *vmemmap_shared_tail_page(unsigned int order, #endif static __meminit void *vmemmap_alloc_pte(unsigned long pfn, int node, - struct vmem_altmap *altmap) + struct vmem_altmap *altmap, unsigned long flags) { struct zone *zone; struct page *page; const unsigned int order = pfn_to_section_compound_order(pfn); - if (!vmemmap_optimizable_pfn(pfn)) + /* + * Device DAX still relies on vmemmap_populate_compound_pages() for + * head/first-tail allocation and tail-page reuse. + */ + if (!vmemmap_optimizable_pfn(pfn) || flags & VMEMMAP_POPULATE_DAX) return vmemmap_alloc_block_buf(PAGE_SIZE, node, altmap); zone = pfn_to_zone(pfn, node); @@ -275,7 +279,7 @@ static pte_t * __meminit vmemmap_pte_populate(pmd_t *pmd, unsigned long addr, in pte_t entry; if (ptpfn == (unsigned long)-1) { - void *p = vmemmap_alloc_pte(pfn, node, altmap); + void *p = vmemmap_alloc_pte(pfn, node, altmap, flags); if (!p) return NULL; @@ -290,7 +294,7 @@ static pte_t * __meminit vmemmap_pte_populate(pmd_t *pmd, unsigned long addr, in * and through vmemmap_populate_compound_pages() when * slab is available. */ - if (flags & VMEMMAP_POPULATE_PAGEREF) + if (flags & VMEMMAP_POPULATE_DAX) get_page(pfn_to_page(ptpfn)); } entry = pfn_pte(ptpfn, PAGE_KERNEL); @@ -547,6 +551,7 @@ static int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, unsigned long end, int node, struct dev_pagemap *pgmap) { + const unsigned long flags = VMEMMAP_POPULATE_DAX; unsigned long size, addr; pte_t *pte; int rc; @@ -561,8 +566,7 @@ static int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, * with just tail struct pages. */ return vmemmap_populate_range(start, end, node, NULL, - pte_pfn(ptep_get(pte)), - VMEMMAP_POPULATE_PAGEREF); + pte_pfn(ptep_get(pte)), flags); } size = min(end - start, pgmap_vmemmap_nr(pgmap) * sizeof(struct page)); @@ -570,13 +574,13 @@ static int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, unsigned long next, last = addr + size; /* Populate the head page vmemmap page */ - pte = vmemmap_populate_address(addr, node, NULL, -1, 0); + pte = vmemmap_populate_address(addr, node, NULL, -1, flags); if (!pte) return -ENOMEM; /* Populate the tail pages vmemmap page */ next = addr + PAGE_SIZE; - pte = vmemmap_populate_address(next, node, NULL, -1, 0); + pte = vmemmap_populate_address(next, node, NULL, -1, flags); if (!pte) return -ENOMEM; @@ -586,8 +590,7 @@ static int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, */ next += PAGE_SIZE; rc = vmemmap_populate_range(next, last, node, NULL, - pte_pfn(ptep_get(pte)), - VMEMMAP_POPULATE_PAGEREF); + pte_pfn(ptep_get(pte)), flags); if (rc) return -ENOMEM; } From d3c41859f45dc15fbb743782b75a19e656d17afa Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Wed, 30 Sep 2026 22:06:21 +0800 Subject: [PATCH 0900/1352] mm/sparse-vmemmap: set compound page order for device DAX Device DAX can use vmemmap optimization only when a full section is populated with a compound-page geometry. Record that geometry as the compound page order in section metadata before populating the section, so later vmemmap accounting and population decisions can use the section state directly. Clear the compound page order when the section becomes empty again. Also reject partial additions to a section that already has optimized vmemmap mappings. compound_nr_pages() determines how many struct pages to initialize with a section as the smallest granularity. A section therefore cannot safely mix optimized and ordinary vmemmap layouts. Partial additions continue to use ordinary vmemmap population, so they do not save vmemmap memory. Such additions are uncommon, and the lost saving is negligible. Link: https://lore.kernel.org/20260930140627.57431-7-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Acked-by: David Hildenbrand (Arm) Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Mike Rapoport Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Ritesh Harjani (IBM) Cc: Shrikanth Hegde Cc: Randy Dunlap Cc: Lance Yang --- mm/mm_init.c | 15 +++++---------- mm/sparse-vmemmap.c | 16 ++++++++++++---- 2 files changed, 17 insertions(+), 14 deletions(-) diff --git a/mm/mm_init.c b/mm/mm_init.c index 97e0158d2aca5b..efffa8609b8592 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -1049,16 +1049,11 @@ static void zone_device_page_init_from_template(struct page *page, * of an altmap. See vmemmap_populate_compound_pages(). */ static inline unsigned long compound_nr_pages(unsigned long pfn, - struct vmem_altmap *altmap, struct dev_pagemap *pgmap) { - /* - * If DAX memory is hot-plugged into an unoccupied subsection - * of an early section, the unoptimized boot memmap is reused. - * See section_activate(). - */ - if (early_section(__pfn_to_section(pfn)) || - !vmemmap_can_optimize(altmap, pgmap)) + const struct mem_section *ms = __pfn_to_section(pfn); + + if (!section_vmemmap_optimizable(ms)) return pgmap_vmemmap_nr(pgmap); return VMEMMAP_RESERVE_NR * (PAGE_SIZE / sizeof(struct page)); @@ -1144,7 +1139,7 @@ void __ref memmap_init_zone_device(struct zone *zone, memcpy(&template, page, sizeof(*page)); if (pfns_per_compound != 1) memmap_init_compound(page, pfn, zone_idx, nid, pgmap, - compound_nr_pages(pfn, altmap, pgmap)); + compound_nr_pages(pfn, pgmap)); pfn += pfns_per_compound; /* Initialize the remaining head pages from template. */ @@ -1160,7 +1155,7 @@ void __ref memmap_init_zone_device(struct zone *zone, continue; memmap_init_compound(page, pfn, zone_idx, nid, pgmap, - compound_nr_pages(pfn, altmap, pgmap)); + compound_nr_pages(pfn, pgmap)); } pageblock_migratetype_init_range(start_pfn, nr_pages, MIGRATE_MOVABLE, diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index b4b72f233bf656..26be355aaa3713 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -135,14 +135,14 @@ int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages struct vmem_altmap *altmap, struct dev_pagemap *pgmap) { const struct mem_section *ms = __pfn_to_section(pfn); - const int order = pgmap ? pgmap->vmemmap_shift : section_compound_order(ms); + const int order = section_compound_order(ms); const int vmemmap_pages = pgmap ? VMEMMAP_RESERVE_NR : VMEMMAP_OPTIMIZATION_PAGES; const unsigned long pages_per_compound = 1UL << order; VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SUBSECTION)); VM_WARN_ON_ONCE(nr_pages > PAGES_PER_SECTION); - if (!vmemmap_can_optimize(altmap, pgmap) && !section_vmemmap_optimizable(ms)) + if (!section_vmemmap_optimizable(ms)) return DIV_ROUND_UP(nr_pages * sizeof(struct page), PAGE_SIZE); if (order < PFN_SECTION_SHIFT) { @@ -612,7 +612,7 @@ struct page * __meminit __populate_section_memmap(unsigned long pfn, !IS_ALIGNED(nr_pages, PAGES_PER_SUBSECTION))) return NULL; - if (vmemmap_can_optimize(altmap, pgmap)) + if (pgmap && section_vmemmap_optimizable(__pfn_to_section(pfn))) r = vmemmap_populate_compound_pages(pfn, start, end, nid, pgmap); else r = vmemmap_populate(start, end, nid, altmap); @@ -831,8 +831,10 @@ static void section_deactivate(unsigned long pfn, unsigned long nr_pages, else if (memmap) free_map_bootmem(memmap); - if (empty) + if (empty) { ms->section_mem_map = (unsigned long)NULL; + section_set_compound_order(ms, 0); + } } static struct page * __meminit section_activate(int nid, unsigned long pfn, @@ -842,8 +844,13 @@ static struct page * __meminit section_activate(int nid, unsigned long pfn, struct mem_section *ms = __pfn_to_section(pfn); struct mem_section_usage *usage = NULL; struct page *memmap; + unsigned int order; int rc; + order = vmemmap_can_optimize(altmap, pgmap) ? pgmap->vmemmap_shift : 0; + if (nr_pages < PAGES_PER_SECTION && section_compound_order(ms)) + return ERR_PTR(-EOPNOTSUPP); + if (!ms->usage) { usage = kzalloc(mem_section_usage_size(), GFP_KERNEL); if (!usage) @@ -869,6 +876,7 @@ static struct page * __meminit section_activate(int nid, unsigned long pfn, if (nr_pages < PAGES_PER_SECTION && early_section(ms)) return pfn_to_page(pfn); + section_set_compound_order_range(pfn, nr_pages, order); memmap = populate_section_memmap(pfn, nr_pages, nid, altmap, pgmap); if (!memmap) { section_deactivate(pfn, nr_pages, altmap, pgmap); From 0bc78003583487fc1531d9d65033ed4294457949 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Wed, 30 Sep 2026 22:06:22 +0800 Subject: [PATCH 0901/1352] mm/sparse-vmemmap: switch device DAX to shared tail vmemmap pages HugeTLB vmemmap optimization now uses per-zone shared tail vmemmap pages. Device DAX has not been switched to that mechanism yet. Switch device DAX to vmemmap_shared_tail_page() as well. This aligns DAX with HugeTLB by using the common per-zone shared tail vmemmap page. The optimization is enabled only for DEV-DAX through pgmap->vmemmap_shift, which supplies the compound page order recorded in section metadata before vmemmap population. Unlike FS-DAX, DEV-DAX does not modify tail struct pages, so sharing them is safe. Since the shared tail page can now back ZONE_DEVICE vmemmap mappings, initialize its entries with PG_reserved for device zones. Also skip poisoning vmemmap-optimizable sections while their struct pages may be shared. Each PTE mapping the shared device DAX tail page takes a page reference. A sufficiently large range could therefore cycle the reference count back to zero if population were allowed to continue after it became non-positive. Use try_get_page() so further mappings fail at that point. The section population error path tears down mappings created for the failed section, while the warning makes this currently impractical limit visible. Link: https://lore.kernel.org/20260930140627.57431-8-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Cc: David Hildenbrand Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Mike Rapoport Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Ritesh Harjani (IBM) Cc: Shrikanth Hegde Cc: Randy Dunlap Cc: Lance Yang --- include/linux/mmzone.h | 10 +++++++ mm/memory_hotplug.c | 6 ++-- mm/sparse-vmemmap.c | 63 +++++++++++++++++------------------------- 3 files changed, 40 insertions(+), 39 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index ee9cbaaa63f4f2..cd68c1904c9118 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -2143,11 +2143,21 @@ static inline int online_device_section(const struct mem_section *section) return section && ((section->section_mem_map & flags) == flags); } + +static inline struct zone *device_zone(int nid) +{ + return &NODE_DATA(nid)->node_zones[ZONE_DEVICE]; +} #else static inline int online_device_section(const struct mem_section *section) { return 0; } + +static inline struct zone *device_zone(int nid) +{ + return NULL; +} #endif static inline int online_section_nr(unsigned long nr) diff --git a/mm/memory_hotplug.c b/mm/memory_hotplug.c index b428da66d279c0..d7a59167bec43c 100644 --- a/mm/memory_hotplug.c +++ b/mm/memory_hotplug.c @@ -43,6 +43,7 @@ #include "mm_init.h" #include "page_alloc.h" #include "shuffle.h" +#include "sparse.h" enum { MEMMAP_ON_MEMORY_DISABLE = 0, @@ -554,8 +555,9 @@ void remove_pfn_range_from_zone(struct zone *zone, /* Select all remaining pages up to the next section boundary */ cur_nr_pages = min(end_pfn - pfn, SECTION_ALIGN_UP(pfn + 1) - pfn); - page_init_poison(pfn_to_page(pfn), - sizeof(struct page) * cur_nr_pages); + if (!section_vmemmap_optimizable(__pfn_to_section(pfn))) + page_init_poison(pfn_to_page(pfn), + sizeof(struct page) * cur_nr_pages); } /* diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 26be355aaa3713..d40a2f5b5fca3a 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -225,6 +225,8 @@ struct page __ref *vmemmap_shared_tail_page(unsigned int order, struct zone *zon set_page_node(page, zone_to_nid(zone)); set_page_zone(page, zone_idx(zone)); prep_compound_tail(page, NULL, order); + if (zone_is_zone_device(zone)) + __SetPageReserved(page); } page = virt_to_page(addr); @@ -288,14 +290,18 @@ static pte_t * __meminit vmemmap_pte_populate(pmd_t *pmd, unsigned long addr, in /* * When a PTE/PMD entry is freed from the init_mm * there's a free_pages() call to this page allocated - * above. Thus this get_page() is paired with the + * above. Thus this try_get_page() is paired with the * put_page_testzero() on the freeing path. * This can only called by certain ZONE_DEVICE path, * and through vmemmap_populate_compound_pages() when * slab is available. + * + * Use try_get_page() to prevent the shared page refcount + * from overflowing. */ - if (flags & VMEMMAP_POPULATE_DAX) - get_page(pfn_to_page(ptpfn)); + if ((flags & VMEMMAP_POPULATE_DAX) && + !try_get_page(pfn_to_page(ptpfn))) + return NULL; } entry = pfn_pte(ptpfn, PAGE_KERNEL); set_pte_at(&init_mm, addr, pte, entry); @@ -529,47 +535,27 @@ static bool __meminit reuse_compound_section(unsigned long start_pfn, return !IS_ALIGNED(offset, nr_pages) && nr_pages > PAGES_PER_SUBSECTION; } -static pte_t * __meminit compound_section_tail_page(unsigned long addr) -{ - pte_t *pte; - - addr -= PAGE_SIZE; - - /* - * Assuming sections are populated sequentially, the previous section's - * page data can be reused. - */ - pte = pte_offset_kernel(pmd_off_k(addr), addr); - if (!pte) - return NULL; - - return pte; -} - static int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, unsigned long start, unsigned long end, int node, struct dev_pagemap *pgmap) { const unsigned long flags = VMEMMAP_POPULATE_DAX; + const unsigned int order = pfn_to_section_compound_order(start_pfn); unsigned long size, addr; pte_t *pte; + struct page *page; int rc; - if (reuse_compound_section(start_pfn, pgmap)) { - pte = compound_section_tail_page(start); - if (!pte) - return -ENOMEM; + page = vmemmap_shared_tail_page(order, device_zone(node)); + if (!page) + return -ENOMEM; - /* - * Reuse the page that was populated in the prior iteration - * with just tail struct pages. - */ + if (reuse_compound_section(start_pfn, pgmap)) return vmemmap_populate_range(start, end, node, NULL, - pte_pfn(ptep_get(pte)), flags); - } + page_to_pfn(page), flags); - size = min(end - start, pgmap_vmemmap_nr(pgmap) * sizeof(struct page)); + size = min(end - start, (1UL << order) * sizeof(struct page)); for (addr = start; addr < end; addr += size) { unsigned long next, last = addr + size; @@ -585,12 +571,12 @@ static int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, return -ENOMEM; /* - * Reuse the previous page for the rest of tail pages + * Reuse the shared page for the rest of tail pages * See layout diagram in Documentation/mm/vmemmap_dedup.rst */ next += PAGE_SIZE; rc = vmemmap_populate_range(next, last, node, NULL, - pte_pfn(ptep_get(pte)), flags); + page_to_pfn(page), flags); if (rc) return -ENOMEM; } @@ -922,13 +908,16 @@ int __meminit sparse_add_section(int nid, unsigned long start_pfn, if (IS_ERR(memmap)) return PTR_ERR(memmap); + ms = __nr_to_section(section_nr); /* - * Poison uninitialized struct pages in order to catch invalid flags - * combinations. + * Poison uninitialized struct pages to catch invalid flag combinations. + * + * Tail struct pages in a vmemmap-optimized section are initialized and + * shared during vmemmap population, so they must not be overwritten here. */ - page_init_poison(memmap, sizeof(struct page) * nr_pages); + if (!section_vmemmap_optimizable(ms)) + page_init_poison(memmap, sizeof(struct page) * nr_pages); - ms = __nr_to_section(section_nr); __section_mark_present(ms, section_nr); /* Align memmap to section boundary in the subsection case */ From c3003fd49d9494c7fadc438d602535485351ef68 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Wed, 30 Sep 2026 23:07:48 +0800 Subject: [PATCH 0902/1352] fixup! mm/sparse-vmemmap: switch device DAX to shared tail vmemmap pages prep_compound_tail() makes PageCompound() true. Calling __SetPageReserved() afterwards violates the PF_NO_COMPOUND policy and triggers VM_BUG_ON_PGFLAGS() when CONFIG_DEBUG_VM_PGFLAGS is enabled. Set PG_reserved before preparing the compound tail. Link: https://lore.kernel.org/20260930150748.1134516-1-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Mike Rapoport Cc: Qi Zheng Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Ritesh Harjani (IBM) Cc: Shrikanth Hegde Cc: Randy Dunlap Cc: Lance Yang --- mm/sparse-vmemmap.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index d40a2f5b5fca3a..060342ac48827d 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -224,9 +224,9 @@ struct page __ref *vmemmap_shared_tail_page(unsigned int order, struct zone *zon atomic_set(&page->_mapcount, -1); set_page_node(page, zone_to_nid(zone)); set_page_zone(page, zone_idx(zone)); - prep_compound_tail(page, NULL, order); if (zone_is_zone_device(zone)) __SetPageReserved(page); + prep_compound_tail(page, NULL, order); } page = virt_to_page(addr); From 6f7b8caa2b0c9d84a71bbf09693c6eec46b954ee Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Wed, 30 Sep 2026 22:06:23 +0800 Subject: [PATCH 0903/1352] mm/sparse-vmemmap: move vmemmap optimization helpers to a public header The vmemmap optimization helpers currently live in mm/sparse.h, which is an internal MM header. That works for MM code, but prevents powerpc from using the same interfaces without including a private header. Move the declarations and inline helpers to vmemmap-optimization.h. This is a preparatory change for powerpc, which has its own vmemmap optimization implementation and needs to use the common vmemmap optimization interfaces from architecture code. Link: https://lore.kernel.org/20260930140627.57431-9-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Acked-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Ritesh Harjani (IBM) Cc: Shrikanth Hegde Cc: Randy Dunlap Cc: Lance Yang --- MAINTAINERS | 1 + arch/loongarch/include/asm/pgtable.h | 1 + arch/riscv/mm/init.c | 1 + include/linux/mmzone.h | 17 ----- include/linux/vmemmap-optimization.h | 109 +++++++++++++++++++++++++++ mm/hugetlb.c | 2 +- mm/hugetlb_vmemmap.c | 2 +- mm/sparse.h | 78 +------------------ 8 files changed, 115 insertions(+), 96 deletions(-) create mode 100644 include/linux/vmemmap-optimization.h diff --git a/MAINTAINERS b/MAINTAINERS index fe1d70ed5100b5..4689a021006002 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -12097,6 +12097,7 @@ F: Documentation/mm/hugetlbfs_reserv.rst F: Documentation/mm/vmemmap_dedup.rst F: fs/hugetlbfs/ F: include/linux/hugetlb.h +F: include/linux/vmemmap-optimization.h F: include/trace/events/hugetlbfs.h F: mm/hugetlb.c F: mm/hugetlb_cgroup.c diff --git a/arch/loongarch/include/asm/pgtable.h b/arch/loongarch/include/asm/pgtable.h index cf29a4c8ac593a..f876031351319a 100644 --- a/arch/loongarch/include/asm/pgtable.h +++ b/arch/loongarch/include/asm/pgtable.h @@ -72,6 +72,7 @@ #include #include +#include #include #include diff --git a/arch/riscv/mm/init.c b/arch/riscv/mm/init.c index fb37b0b67efec6..857f9a55039ce7 100644 --- a/arch/riscv/mm/init.c +++ b/arch/riscv/mm/init.c @@ -22,6 +22,7 @@ #include #include #include +#include #include #include diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index cd68c1904c9118..65de3bb13eb3ae 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -96,23 +96,6 @@ #define MAX_FOLIO_NR_PAGES (1UL << MAX_FOLIO_ORDER) -/* - * HugeTLB Vmemmap Optimization (HVO) requires struct pages of the head page to - * be naturally aligned with regard to the folio size. - * - * HVO which is only active if the size of struct page is a power of 2. - */ -#define MAX_FOLIO_VMEMMAP_ALIGN \ - (IS_ENABLED(CONFIG_VMEMMAP_OPTIMIZATION) && \ - is_power_of_2(sizeof(struct page)) ? \ - MAX_FOLIO_NR_PAGES * sizeof(struct page) : 0) - -/* The number of retained vmemmap pages with HVO enabled. */ -#define VMEMMAP_OPTIMIZATION_PAGES 1 -#define VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES \ - (VMEMMAP_OPTIMIZATION_PAGES * PAGE_SIZE / sizeof(struct page)) -#define VMEMMAP_OPTIMIZATION_MIN_ORDER (ilog2(VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES) + 1) - enum migratetype { MIGRATE_UNMOVABLE, MIGRATE_MOVABLE, diff --git a/include/linux/vmemmap-optimization.h b/include/linux/vmemmap-optimization.h new file mode 100644 index 00000000000000..bd0974b262a4d1 --- /dev/null +++ b/include/linux/vmemmap-optimization.h @@ -0,0 +1,109 @@ +/* SPDX-License-Identifier: GPL-2.0-or-later */ +/* + * vmemmap-optimization.h + * + * Generic vmemmap optimization declarations. + * + * Author: Muchun Song + */ +#ifndef _LINUX_VMEMMAP_OPTIMIZATION_H +#define _LINUX_VMEMMAP_OPTIMIZATION_H + +#include +#include +#include +#include + +/* + * HugeTLB Vmemmap Optimization (HVO) requires struct pages of the head page to + * be naturally aligned with regard to the folio size. + * + * HVO which is only active if the size of struct page is a power of 2. + */ +#define MAX_FOLIO_VMEMMAP_ALIGN \ + (IS_ENABLED(CONFIG_VMEMMAP_OPTIMIZATION) && \ + is_power_of_2(sizeof(struct page)) ? \ + MAX_FOLIO_NR_PAGES * sizeof(struct page) : 0) + +/* The number of retained vmemmap pages with HVO enabled. */ +#define VMEMMAP_OPTIMIZATION_PAGES 1 +#define VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES \ + (VMEMMAP_OPTIMIZATION_PAGES * PAGE_SIZE / sizeof(struct page)) +#define VMEMMAP_OPTIMIZATION_MIN_ORDER (ilog2(VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES) + 1) + +#ifdef CONFIG_VMEMMAP_OPTIMIZATION +static inline unsigned int section_compound_order(const struct mem_section *section) +{ + return section->compound_page_order; +} + +static inline void section_set_compound_order(struct mem_section *section, + unsigned int order) +{ + VM_WARN_ON(section_compound_order(section) && order && + section_compound_order(section) != order); + section->compound_page_order = order; +} + +static inline void section_set_compound_order_range(unsigned long pfn, + unsigned long nr_pages, unsigned int order) +{ + unsigned long section_nr = pfn_to_section_nr(pfn); + + if (!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SECTION)) + return; + + for (unsigned long i = 0; i < nr_pages / PAGES_PER_SECTION; i++) + section_set_compound_order(__nr_to_section(section_nr + i), order); +} + +static inline unsigned int pfn_to_section_compound_order(unsigned long pfn) +{ + return section_compound_order(__pfn_to_section(pfn)); +} + +struct page *vmemmap_shared_tail_page(unsigned int order, struct zone *zone); +#else +static inline unsigned int section_compound_order(const struct mem_section *section) +{ + return 0; +} + +static inline void section_set_compound_order(struct mem_section *section, + unsigned int order) +{ +} + +static inline void section_set_compound_order_range(unsigned long pfn, + unsigned long nr_pages, unsigned int order) +{ +} + +static inline unsigned int pfn_to_section_compound_order(unsigned long pfn) +{ + return 0; +} +#endif /* CONFIG_VMEMMAP_OPTIMIZATION */ + +static inline bool vmemmap_optimizable_pfn(unsigned long pfn) +{ + const unsigned int order = pfn_to_section_compound_order(pfn); + const unsigned long nr_pages = 1UL << order; + + if (!is_power_of_2(sizeof(struct page))) + return false; + + return (pfn & (nr_pages - 1)) >= VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES; +} + +static inline bool vmemmap_optimizable_order(unsigned int order) +{ + if (!IS_ENABLED(CONFIG_VMEMMAP_OPTIMIZATION)) + return false; + + if (!is_power_of_2(sizeof(struct page))) + return false; + + return order >= VMEMMAP_OPTIMIZATION_MIN_ORDER; +} +#endif /* _LINUX_VMEMMAP_OPTIMIZATION_H */ diff --git a/mm/hugetlb.c b/mm/hugetlb.c index fd00141b089a98..2003439ea13c6f 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -38,6 +38,7 @@ #include #include #include +#include #include #include @@ -52,7 +53,6 @@ #include "hugetlb_cma.h" #include "hugetlb_internal.h" #include "mm_init.h" -#include "sparse.h" #include #define HUGE_BOOTMEM_ZONES_VALID BIT(0) diff --git a/mm/hugetlb_vmemmap.c b/mm/hugetlb_vmemmap.c index 76765c97ff68b4..0057fa2a16a891 100644 --- a/mm/hugetlb_vmemmap.c +++ b/mm/hugetlb_vmemmap.c @@ -15,10 +15,10 @@ #include #include #include +#include #include #include "hugetlb_vmemmap.h" -#include "sparse.h" /** * struct vmemmap_remap_walk - walk vmemmap page table diff --git a/mm/sparse.h b/mm/sparse.h index 326ad43bb5c37a..a5111087ee3a3d 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -9,80 +9,7 @@ #define __MM_SPARSE_H #include - -#ifdef CONFIG_VMEMMAP_OPTIMIZATION -static inline unsigned int section_compound_order(const struct mem_section *section) -{ - return section->compound_page_order; -} - -static inline void section_set_compound_order(struct mem_section *section, - unsigned int order) -{ - VM_WARN_ON(section_compound_order(section) && order && - section_compound_order(section) != order); - section->compound_page_order = order; -} - -static inline void section_set_compound_order_range(unsigned long pfn, - unsigned long nr_pages, unsigned int order) -{ - unsigned long section_nr = pfn_to_section_nr(pfn); - - if (!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SECTION)) - return; - - for (unsigned long i = 0; i < nr_pages / PAGES_PER_SECTION; i++) - section_set_compound_order(__nr_to_section(section_nr + i), order); -} - -static inline unsigned int pfn_to_section_compound_order(unsigned long pfn) -{ - return section_compound_order(__pfn_to_section(pfn)); -} -#else -static inline unsigned int section_compound_order(const struct mem_section *section) -{ - return 0; -} - -static inline void section_set_compound_order(struct mem_section *section, - unsigned int order) -{ -} - -static inline void section_set_compound_order_range(unsigned long pfn, - unsigned long nr_pages, unsigned int order) -{ -} - -static inline unsigned int pfn_to_section_compound_order(unsigned long pfn) -{ - return 0; -} -#endif - -static inline bool vmemmap_optimizable_pfn(unsigned long pfn) -{ - const unsigned int order = pfn_to_section_compound_order(pfn); - const unsigned long nr_pages = 1UL << order; - - if (!is_power_of_2(sizeof(struct page))) - return false; - - return (pfn & (nr_pages - 1)) >= VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES; -} - -static inline bool vmemmap_optimizable_order(unsigned int order) -{ - if (!IS_ENABLED(CONFIG_VMEMMAP_OPTIMIZATION)) - return false; - - if (!is_power_of_2(sizeof(struct page))) - return false; - - return order >= VMEMMAP_OPTIMIZATION_MIN_ORDER; -} +#include /* * mm/sparse.c @@ -142,9 +69,6 @@ static inline void sparse_sections_init(void) {} * mm/sparse-vmemmap.c */ #ifdef CONFIG_SPARSEMEM_VMEMMAP -#ifdef CONFIG_VMEMMAP_OPTIMIZATION -struct page *vmemmap_shared_tail_page(unsigned int order, struct zone *zone); -#endif void sparse_init_subsection_map(void); int section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, struct vmem_altmap *altmap, struct dev_pagemap *pgmap); From 32a18ca098b6779a4dcf69e8d56450323d03e246 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Wed, 30 Sep 2026 22:06:24 +0800 Subject: [PATCH 0904/1352] powerpc/mm: switch device DAX to shared tail vmemmap pages The powerpc radix compound vmemmap population path still finds a reusable tail page by walking the vmemmap page tables. Switch it to the common vmemmap_shared_tail_page() helper instead, so it can use the shared vmemmap page directly to simplify the code. This removes the powerpc-specific tail-page lookup and its fallback path and aligns the device DAX vmemmap optimization path with HugeTLB. Link: https://lore.kernel.org/20260930140627.57431-10-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Mike Rapoport Cc: Qi Zheng Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Ritesh Harjani (IBM) Cc: Shrikanth Hegde Cc: Randy Dunlap Cc: Lance Yang --- arch/powerpc/mm/book3s64/radix_pgtable.c | 80 +++--------------------- include/linux/vmemmap-optimization.h | 6 ++ mm/sparse-vmemmap.c | 6 -- 3 files changed, 15 insertions(+), 77 deletions(-) diff --git a/arch/powerpc/mm/book3s64/radix_pgtable.c b/arch/powerpc/mm/book3s64/radix_pgtable.c index cf692b2b5f7bc5..ee068f24a79f07 100644 --- a/arch/powerpc/mm/book3s64/radix_pgtable.c +++ b/arch/powerpc/mm/book3s64/radix_pgtable.c @@ -19,6 +19,7 @@ #include #include #include +#include #include #include @@ -1250,59 +1251,6 @@ static pte_t * __meminit radix__vmemmap_populate_address(unsigned long addr, int return pte; } -static pte_t * __meminit vmemmap_compound_tail_page(unsigned long addr, - unsigned long pfn_offset, int node) -{ - pgd_t *pgd; - p4d_t *p4d; - pud_t *pud; - pmd_t *pmd; - pte_t *pte; - unsigned long map_addr; - - /* the second vmemmap page which we use for duplication */ - map_addr = addr - pfn_offset * sizeof(struct page) + PAGE_SIZE; - pgd = pgd_offset_k(map_addr); - p4d = p4d_offset(pgd, map_addr); - pud = vmemmap_pud_alloc(p4d, node, map_addr); - if (!pud) - return NULL; - pmd = vmemmap_pmd_alloc(pud, node, map_addr); - if (!pmd) - return NULL; - if (pmd_leaf(*pmd)) - /* - * The second page is mapped as a hugepage due to a nearby request. - * Force our mapping to page size without deduplication - */ - return NULL; - pte = vmemmap_pte_alloc(pmd, node, map_addr); - if (!pte) - return NULL; - /* - * Check if there exist a mapping to the left - */ - if (pte_none(*pte)) { - /* - * Populate the head page vmemmap page. - * It can fall in different pmd, hence - * vmemmap_populate_address() - */ - pte = radix__vmemmap_populate_address(map_addr - PAGE_SIZE, node, NULL, NULL); - if (!pte) - return NULL; - /* - * Populate the tail pages vmemmap page - */ - pte = radix__vmemmap_pte_populate(pmd, map_addr, node, NULL, NULL); - if (!pte) - return NULL; - vmemmap_verify(pte, node, map_addr, map_addr + PAGE_SIZE); - return pte; - } - return pte; -} - int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, unsigned long start, unsigned long end, int node, @@ -1320,6 +1268,12 @@ int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, pud_t *pud; pmd_t *pmd; pte_t *pte; + struct page *tail_page; + unsigned int order = pfn_to_section_compound_order(start_pfn); + + tail_page = vmemmap_shared_tail_page(order, device_zone(node)); + if (!tail_page) + return -ENOMEM; for (addr = start; addr < end; addr = next) { @@ -1349,10 +1303,9 @@ int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, next = addr + PAGE_SIZE; continue; } else { - unsigned long nr_pages = pgmap_vmemmap_nr(pgmap); + unsigned long nr_pages = 1UL << order; unsigned long addr_pfn = page_to_pfn((struct page *)addr); unsigned long pfn_offset = addr_pfn - ALIGN_DOWN(addr_pfn, nr_pages); - pte_t *tail_page_pte; /* * if the address is aligned to huge page size it is the @@ -1377,23 +1330,8 @@ int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, next = addr + 2 * PAGE_SIZE; continue; } - /* - * get the 2nd mapping details - * Also create it if that doesn't exist - */ - tail_page_pte = vmemmap_compound_tail_page(addr, pfn_offset, node); - if (!tail_page_pte) { - - pte = radix__vmemmap_pte_populate(pmd, addr, node, NULL, NULL); - if (!pte) - return -ENOMEM; - vmemmap_verify(pte, node, addr, addr + PAGE_SIZE); - - next = addr + PAGE_SIZE; - continue; - } - pte = radix__vmemmap_pte_populate(pmd, addr, node, NULL, pte_page(*tail_page_pte)); + pte = radix__vmemmap_pte_populate(pmd, addr, node, NULL, tail_page); if (!pte) return -ENOMEM; vmemmap_verify(pte, node, addr, addr + PAGE_SIZE); diff --git a/include/linux/vmemmap-optimization.h b/include/linux/vmemmap-optimization.h index bd0974b262a4d1..fa9e9abd66560f 100644 --- a/include/linux/vmemmap-optimization.h +++ b/include/linux/vmemmap-optimization.h @@ -83,6 +83,12 @@ static inline unsigned int pfn_to_section_compound_order(unsigned long pfn) { return 0; } + +static inline struct page *vmemmap_shared_tail_page(unsigned int order, + struct zone *zone) +{ + return NULL; +} #endif /* CONFIG_VMEMMAP_OPTIMIZATION */ static inline bool vmemmap_optimizable_pfn(unsigned long pfn) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 060342ac48827d..14f99c129b381f 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -240,12 +240,6 @@ struct page __ref *vmemmap_shared_tail_page(unsigned int order, struct zone *zon return page; } -#else -static inline struct page *vmemmap_shared_tail_page(unsigned int order, - struct zone *zone) -{ - return NULL; -} #endif static __meminit void *vmemmap_alloc_pte(unsigned long pfn, int node, From b49957b595d540dcfd4a3332032924c69f31f1a8 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Wed, 30 Sep 2026 23:28:26 +0800 Subject: [PATCH 0905/1352] fixup! powerpc/mm: switch device DAX to shared tail vmemmap pages The PowerPC radix path takes one reference for every PTE that maps the shared tail page. Unlike the generic path, get_page() lets the counter continue after it becomes non-positive and eventually cycle to zero. Use try_get_page() so vmemmap population fails before the shared page reference count can cycle, matching the generic implementation. Link: https://lore.kernel.org/20260930152826.76084-1-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Mike Rapoport Cc: Qi Zheng Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Ritesh Harjani (IBM) Cc: Shrikanth Hegde Cc: Randy Dunlap Cc: Lance Yang --- arch/powerpc/mm/book3s64/radix_pgtable.c | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/arch/powerpc/mm/book3s64/radix_pgtable.c b/arch/powerpc/mm/book3s64/radix_pgtable.c index ee068f24a79f07..0e778d17240c33 100644 --- a/arch/powerpc/mm/book3s64/radix_pgtable.c +++ b/arch/powerpc/mm/book3s64/radix_pgtable.c @@ -1042,13 +1042,17 @@ static pte_t * __meminit radix__vmemmap_pte_populate(pmd_t *pmdp, unsigned long /* * When a PTE/PMD entry is freed from the init_mm * there's a free_pages() call to this page allocated - * above. Thus this get_page() is paired with the + * above. Thus this try_get_page() is paired with the * put_page_testzero() on the freeing path. * This can only called by certain ZONE_DEVICE path, * and through vmemmap_populate_compound_pages() when * slab is available. + * + * Use try_get_page() to prevent the shared page refcount + * from overflowing. */ - get_page(reuse); + if (!try_get_page(reuse)) + return NULL; p = page_to_virt(reuse); pr_debug("Tail page reuse vmemmap mapping\n"); } From 074aced9cc5d1c91e77e30228ed5a2a81ad96f72 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Wed, 30 Sep 2026 22:06:25 +0800 Subject: [PATCH 0906/1352] mm/sparse-vmemmap: drop the extra tail page from device DAX reservation The device DAX vmemmap population still reserves one extra tail vmemmap page after the head page. Drop that extra reservation and let the shared tail page cover all tail vmemmap pages after the head page, so DAX follows the same reservation model as HugeTLB. This reduces the reserved vmemmap pages for optimized DAX mappings to one and removes the now-unneeded first-tail population from the generic and powerpc paths to simplify the code as well. Link: https://lore.kernel.org/20260930140627.57431-11-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Acked-by: David Hildenbrand (Arm) Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Mike Rapoport Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Ritesh Harjani (IBM) Cc: Shrikanth Hegde Cc: Randy Dunlap Cc: Lance Yang --- arch/powerpc/mm/book3s64/radix_pgtable.c | 46 ++---------------------- include/linux/mm.h | 4 +-- mm/mm_init.c | 2 +- mm/sparse-vmemmap.c | 13 ++----- 4 files changed, 8 insertions(+), 57 deletions(-) diff --git a/arch/powerpc/mm/book3s64/radix_pgtable.c b/arch/powerpc/mm/book3s64/radix_pgtable.c index 0e778d17240c33..a3332f32ffb25e 100644 --- a/arch/powerpc/mm/book3s64/radix_pgtable.c +++ b/arch/powerpc/mm/book3s64/radix_pgtable.c @@ -1222,39 +1222,6 @@ int __meminit radix__vmemmap_populate(unsigned long start, unsigned long end, in return 0; } -static pte_t * __meminit radix__vmemmap_populate_address(unsigned long addr, int node, - struct vmem_altmap *altmap, - struct page *reuse) -{ - pgd_t *pgd; - p4d_t *p4d; - pud_t *pud; - pmd_t *pmd; - pte_t *pte; - - pgd = pgd_offset_k(addr); - p4d = p4d_offset(pgd, addr); - pud = vmemmap_pud_alloc(p4d, node, addr); - if (!pud) - return NULL; - pmd = vmemmap_pmd_alloc(pud, node, addr); - if (!pmd) - return NULL; - if (pmd_leaf(*pmd)) - /* - * The second page is mapped as a hugepage due to a nearby request. - * Force our mapping to page size without deduplication - */ - return NULL; - pte = vmemmap_pte_alloc(pmd, node, addr); - if (!pte) - return NULL; - radix__vmemmap_pte_populate(pmd, addr, node, NULL, NULL); - vmemmap_verify(pte, node, addr, addr + PAGE_SIZE); - - return pte; -} - int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, unsigned long start, unsigned long end, int node, @@ -1301,7 +1268,7 @@ int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, if (!pte_none(*pte)) { /* * This could be because we already have a compound - * page whose VMEMMAP_RESERVE_NR pages were mapped and + * page whose retained vmemmap page was mapped and * this request fall in those pages. */ next = addr + PAGE_SIZE; @@ -1322,16 +1289,7 @@ int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, return -ENOMEM; vmemmap_verify(pte, node, addr, addr + PAGE_SIZE); - /* - * Populate the tail pages vmemmap page - * It can fall in different pmd, hence - * vmemmap_populate_address() - */ - pte = radix__vmemmap_populate_address(addr + PAGE_SIZE, node, NULL, NULL); - if (!pte) - return -ENOMEM; - - next = addr + 2 * PAGE_SIZE; + next = addr + PAGE_SIZE; continue; } diff --git a/include/linux/mm.h b/include/linux/mm.h index 070ce27e9cd3cd..30a3365bca8271 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -38,6 +38,7 @@ #include #include #include +#include struct mempolicy; struct anon_vma; @@ -5167,7 +5168,6 @@ static inline void vmem_altmap_free(struct vmem_altmap *altmap, } #endif -#define VMEMMAP_RESERVE_NR 2 #ifdef CONFIG_ARCH_WANT_OPTIMIZE_DAX_VMEMMAP static inline bool __vmemmap_can_optimize(struct vmem_altmap *altmap, struct dev_pagemap *pgmap) @@ -5187,7 +5187,7 @@ static inline bool __vmemmap_can_optimize(struct vmem_altmap *altmap, * For vmemmap optimization with DAX we need minimum 2 vmemmap * pages. See layout diagram in Documentation/mm/vmemmap_dedup.rst */ - return !altmap && (nr_vmemmap_pages > VMEMMAP_RESERVE_NR); + return !altmap && (nr_vmemmap_pages > VMEMMAP_OPTIMIZATION_PAGES); } /* * If we don't have an architecture override, use the generic rule diff --git a/mm/mm_init.c b/mm/mm_init.c index efffa8609b8592..56bb4567a49405 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -1056,7 +1056,7 @@ static inline unsigned long compound_nr_pages(unsigned long pfn, if (!section_vmemmap_optimizable(ms)) return pgmap_vmemmap_nr(pgmap); - return VMEMMAP_RESERVE_NR * (PAGE_SIZE / sizeof(struct page)); + return VMEMMAP_OPTIMIZATION_PAGES * (PAGE_SIZE / sizeof(struct page)); } static void __ref memmap_init_compound(struct page *head, diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 14f99c129b381f..0f4c2a8930309e 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -136,7 +136,6 @@ int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages { const struct mem_section *ms = __pfn_to_section(pfn); const int order = section_compound_order(ms); - const int vmemmap_pages = pgmap ? VMEMMAP_RESERVE_NR : VMEMMAP_OPTIMIZATION_PAGES; const unsigned long pages_per_compound = 1UL << order; VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SUBSECTION)); @@ -147,13 +146,13 @@ int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages if (order < PFN_SECTION_SHIFT) { VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, pages_per_compound)); - return vmemmap_pages * nr_pages / pages_per_compound; + return VMEMMAP_OPTIMIZATION_PAGES * nr_pages / pages_per_compound; } VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SECTION)); if (IS_ALIGNED(pfn, pages_per_compound)) - return vmemmap_pages; + return VMEMMAP_OPTIMIZATION_PAGES; return 0; } @@ -558,17 +557,11 @@ static int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, if (!pte) return -ENOMEM; - /* Populate the tail pages vmemmap page */ - next = addr + PAGE_SIZE; - pte = vmemmap_populate_address(next, node, NULL, -1, flags); - if (!pte) - return -ENOMEM; - /* * Reuse the shared page for the rest of tail pages * See layout diagram in Documentation/mm/vmemmap_dedup.rst */ - next += PAGE_SIZE; + next = addr + PAGE_SIZE; rc = vmemmap_populate_range(next, last, node, NULL, page_to_pfn(page), flags); if (rc) From 2b265534717eda8e69ee5ac09e5adf6e8019e85f Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Wed, 30 Sep 2026 22:06:26 +0800 Subject: [PATCH 0907/1352] mm/sparse-vmemmap: drop unused section_nr_vmemmap_pages() arguments section_nr_vmemmap_pages() no longer uses the altmap or pgmap arguments, so drop them from the helper and its callers. Link: https://lore.kernel.org/20260930140627.57431-12-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Acked-by: David Hildenbrand (Arm) Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Mike Rapoport Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Ritesh Harjani (IBM) Cc: Shrikanth Hegde Cc: Randy Dunlap Cc: Lance Yang --- mm/sparse-vmemmap.c | 10 ++++------ mm/sparse.c | 3 +-- mm/sparse.h | 6 ++---- 3 files changed, 7 insertions(+), 12 deletions(-) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 0f4c2a8930309e..23b31bc4bf94ab 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -131,8 +131,7 @@ void __meminit vmemmap_verify(pte_t *pte, int node, start, end - 1); } -int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, - struct vmem_altmap *altmap, struct dev_pagemap *pgmap) +int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages) { const struct mem_section *ms = __pfn_to_section(pfn); const int order = section_compound_order(ms); @@ -674,7 +673,7 @@ static struct page * __meminit populate_section_memmap(unsigned long pfn, struct page *page = __populate_section_memmap(pfn, nr_pages, nid, altmap, pgmap); - memmap_pages_add(section_nr_vmemmap_pages(pfn, nr_pages, altmap, pgmap)); + memmap_pages_add(section_nr_vmemmap_pages(pfn, nr_pages)); return page; } @@ -685,7 +684,7 @@ static void depopulate_section_memmap(unsigned long pfn, unsigned long nr_pages, unsigned long start = (unsigned long) pfn_to_page(pfn); unsigned long end = start + nr_pages * sizeof(struct page); - memmap_pages_add(-section_nr_vmemmap_pages(pfn, nr_pages, altmap, pgmap)); + memmap_pages_add(-section_nr_vmemmap_pages(pfn, nr_pages)); vmemmap_free(start, end, altmap); } @@ -695,8 +694,7 @@ static void free_map_bootmem(struct page *memmap) unsigned long end = (unsigned long)(memmap + PAGES_PER_SECTION); unsigned long pfn = page_to_pfn(memmap); - memmap_boot_pages_add(-section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION, - NULL, NULL)); + memmap_boot_pages_add(-section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION)); vmemmap_free(start, end, NULL); } diff --git a/mm/sparse.c b/mm/sparse.c index cc28bb41fdb1f1..b75921c622edef 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -250,8 +250,7 @@ static void __init sparse_init_nid(int nid, unsigned long pnum_begin, nid, NULL, NULL); if (!map) panic("Failed to allocate memmap for section %lu\n", pnum); - memmap_boot_pages_add(section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION, - NULL, NULL)); + memmap_boot_pages_add(section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION)); sparse_init_one_section(__nr_to_section(pnum), pnum, map, usage, SECTION_IS_EARLY); usage = (void *)usage + mem_section_usage_size(); diff --git a/mm/sparse.h b/mm/sparse.h index a5111087ee3a3d..530692cdd516fe 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -70,12 +70,10 @@ static inline void sparse_sections_init(void) {} */ #ifdef CONFIG_SPARSEMEM_VMEMMAP void sparse_init_subsection_map(void); -int section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, - struct vmem_altmap *altmap, struct dev_pagemap *pgmap); +int section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages); #else static inline void sparse_init_subsection_map(void) {} -static inline int section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, - struct vmem_altmap *altmap, struct dev_pagemap *pgmap) +static inline int section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages) { return DIV_ROUND_UP(nr_pages * sizeof(struct page), PAGE_SIZE); } From ac4c44fd1ce840b83a3572fa96c82b57fc17862a Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Wed, 30 Sep 2026 22:06:27 +0800 Subject: [PATCH 0908/1352] Documentation/mm: update DAX vmemmap deduplication docs Device DAX now uses the common per-zone shared tail page for vmemmap deduplication. The old documentation still described a DAX-specific layout with a separately populated tail vmemmap page and half the HugeTLB savings. Update the generic and powerpc documentation to describe the shared layout. In the powerpc document, keep the radix and 64K-specific details, drop the duplicated 4K PUD arithmetic, and replace the repeated device-dax diagrams with a single parameterized PMD/PUD diagram. Link: https://lore.kernel.org/20260930140627.57431-13-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Acked-by: David Hildenbrand (Arm) Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Mike Rapoport Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Ritesh Harjani (IBM) Cc: Shrikanth Hegde Cc: Randy Dunlap Cc: Lance Yang --- Documentation/arch/powerpc/vmemmap_dedup.rst | 90 ++++---------------- Documentation/mm/vmemmap_dedup.rst | 32 +------ 2 files changed, 21 insertions(+), 101 deletions(-) diff --git a/Documentation/arch/powerpc/vmemmap_dedup.rst b/Documentation/arch/powerpc/vmemmap_dedup.rst index dc4db59fdf87b4..8286acbca9bc57 100644 --- a/Documentation/arch/powerpc/vmemmap_dedup.rst +++ b/Documentation/arch/powerpc/vmemmap_dedup.rst @@ -19,82 +19,28 @@ With 1G PUD level mapping, we require 16384 struct pages and a single 64K vmemmap page can contain 1024 struct pages (64K/sizeof(struct page)). Hence we require 16 64K pages in vmemmap to map the struct page for 1G PUD level mapping. -Here's how things look like on device-dax after the sections are populated:: - +-----------+ ---virt_to_page---> +-----------+ mapping to +-----------+ - | | | 0 | -------------> | 0 | - | | +-----------+ +-----------+ - | | | 1 | -------------> | 1 | - | | +-----------+ +-----------+ - | | | 2 | ----------------^ ^ ^ ^ ^ ^ - | | +-----------+ | | | | | - | | | 3 | ------------------+ | | | | - | | +-----------+ | | | | - | | | 4 | --------------------+ | | | - | PUD | +-----------+ | | | - | level | | . | ----------------------+ | | - | mapping | +-----------+ | | - | | | . | ------------------------+ | - | | +-----------+ | - | | | 15 | --------------------------+ - | | +-----------+ - | | - | | - | | - +-----------+ - - With 4K page size, 2M PMD level mapping requires 512 struct pages and a single 4K vmemmap page contains 64 struct pages(4K/sizeof(struct page)). Hence we require 8 4K pages in vmemmap to map the struct page for 2M pmd level mapping. -Here's how things look like on device-dax after the sections are populated:: - - +-----------+ ---virt_to_page---> +-----------+ mapping to +-----------+ - | | | 0 | -------------> | 0 | - | | +-----------+ +-----------+ - | | | 1 | -------------> | 1 | - | | +-----------+ +-----------+ - | | | 2 | ----------------^ ^ ^ ^ ^ ^ - | | +-----------+ | | | | | - | | | 3 | ------------------+ | | | | - | | +-----------+ | | | | - | | | 4 | --------------------+ | | | - | PMD | +-----------+ | | | - | level | | 5 | ----------------------+ | | - | mapping | +-----------+ | | - | | | 6 | ------------------------+ | - | | +-----------+ | - | | | 7 | --------------------------+ - | | +-----------+ - | | - | | - | | - +-----------+ - -With 1G PUD level mapping, we require 262144 struct pages and a single 4K -vmemmap page can contain 64 struct pages (4K/sizeof(struct page)). Hence we -require 4096 4K pages in vmemmap to map the struct pages for 1G PUD level -mapping. - -Here's how things look like on device-dax after the sections are populated:: - - +-----------+ ---virt_to_page---> +-----------+ mapping to +-----------+ - | | | 0 | -------------> | 0 | - | | +-----------+ +-----------+ - | | | 1 | -------------> | 1 | - | | +-----------+ +-----------+ - | | | 2 | ----------------^ ^ ^ ^ ^ ^ - | | +-----------+ | | | | | - | | | 3 | ------------------+ | | | | - | | +-----------+ | | | | - | | | 4 | --------------------+ | | | - | PUD | +-----------+ | | | - | level | | . | ----------------------+ | | - | mapping | +-----------+ | | - | | | . | ------------------------+ | - | | +-----------+ | - | | | 4095 | --------------------------+ - | | +-----------+ +Here's how things look on device-dax after vmemmap-optimized sections are +populated. ``N`` is the number of vmemmap pages required by the DAX mapping +above:: + + Device DAX vmemmap pages (N pages) backing page frames + +-----------+ ---virt_to_page---> +-----------+ mapping to +-------------+ + | | | 0 | -------------> | 0 | + | | +-----------+ +-------------+ + | | | 1 | ------+ + | | +-----------+ | + | | | 2 | ------+ + | | +-----------+ | + | | | . | ------+ +-------------+ + | PMD/PUD | +-----------+ | | A single, | + | level | | . | ------+------> | per-zone | + | mapping | +-----------+ | | shared tail | + | | | N - 1 | ------+ | page | + | | +-----------+ +-------------+ | | | | | | diff --git a/Documentation/mm/vmemmap_dedup.rst b/Documentation/mm/vmemmap_dedup.rst index 9fa8642ded483b..8c287ae3f86cc8 100644 --- a/Documentation/mm/vmemmap_dedup.rst +++ b/Documentation/mm/vmemmap_dedup.rst @@ -1,4 +1,3 @@ - .. SPDX-License-Identifier: GPL-2.0 ========================================= @@ -192,32 +191,7 @@ to 4 on HugeTLB pages. There's no remapping of vmemmap given that device-dax memory is not part of System RAM ranges initialized at boot. Thus the tail page deduplication -happens at a later stage when we populate the sections. HugeTLB reuses the -the head vmemmap page representing, whereas device-dax reuses the tail -vmemmap page. This results in only half of the savings compared to HugeTLB. - -Deduplicated tail pages are not mapped read-only. +happens at a later stage when we populate the sections. -Here's how things look like on device-dax after the sections are populated:: - - +-----------+ ---virt_to_page---> +-----------+ mapping to +-----------+ - | | | 0 | -------------> | 0 | - | | +-----------+ +-----------+ - | | | 1 | -------------> | 1 | - | | +-----------+ +-----------+ - | | | 2 | ----------------^ ^ ^ ^ ^ ^ - | | +-----------+ | | | | | - | | | 3 | ------------------+ | | | | - | | +-----------+ | | | | - | | | 4 | --------------------+ | | | - | PMD | +-----------+ | | | - | level | | 5 | ----------------------+ | | - | mapping | +-----------+ | | - | | | 6 | ------------------------+ | - | | +-----------+ | - | | | 7 | --------------------------+ - | | +-----------+ - | | - | | - | | - +-----------+ +Deduplicated tail pages are not mapped read-only. The mapping layout is the same +as HugeTLB. From ae047a8bf7f06d64a4477ac9021e11eddd1fd50c Mon Sep 17 00:00:00 2001 From: Yury Norov Date: Fri, 11 Sep 2026 11:52:43 -0400 Subject: [PATCH 0909/1352] lib: fix lock initialization in region allocation benchmark The Maple Tree benchmark uses MTREE_INIT() for a stack-allocated tree. Its static spinlock initializer leaves lockdep to use the lock address as the class key. Since the address is on the stack, the first allocation triggers "INFO: trying to register non-static key" and disables lockdep. The IDA benchmark has the same problem through IDA_INIT(), but runs after Maple Tree and therefore encounters an already disabled lockdep. Use mt_init_flags() and ida_init() to initialize the locks with persistent lock-class keys. Keep initialization outside the timed allocation paths. Link: https://lore.kernel.org/20260911155244.1406122-1-ynorov@nvidia.com Fixes: f4806cc63cc6 ("lib: test bitmap vs IDA vs Maple Tree performance for region allocations") Signed-off-by: Yury Norov Signed-off-by: Andrew Morton Closes: https://lore.kernel.org/oe-lkp/202609101106.771b567e-lkp@intel.com Acked-by: Liam R. Howlett (Oracle) Cc: Alice Ryhl Cc: Andrew Ballance Cc: Matthew Wilcox (Oracle) Cc: Rasmus Villemoes --- lib/region_alloc_benchmark.c | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/lib/region_alloc_benchmark.c b/lib/region_alloc_benchmark.c index e88b4cf55c629c..a644f3d5431a0c 100644 --- a/lib/region_alloc_benchmark.c +++ b/lib/region_alloc_benchmark.c @@ -78,11 +78,13 @@ static size_t __init ida_size(unsigned long nr_ids) static unsigned long __init benchmark_ida(unsigned long cap) { - struct ida ida = IDA_INIT(ida); + struct ida ida; unsigned long cnt, idx, off, nr_ids = 0; ktime_t alloc_time, free_time; int id = -ENOSPC; + ida_init(&ida); + alloc_time = ktime_get(); for (cnt = 0; cnt <= cap; cnt++) { for (off = 0; off < reg_sz[cnt]; off++) { @@ -125,12 +127,14 @@ static unsigned long __init benchmark_ida(unsigned long cap) static unsigned long __init benchmark_maple_tree(unsigned long cap) { - struct maple_tree mt = MTREE_INIT(mt, MT_FLAGS_ALLOC_RANGE); + struct maple_tree mt; unsigned long cnt, idx; ktime_t alloc_time, free_time; size_t sz; int ret; + mt_init_flags(&mt, MT_FLAGS_ALLOC_RANGE); + alloc_time = ktime_get(); for (cnt = 0; cnt <= cap; cnt++) { ret = mtree_alloc_range(&mt, &idx, xa_mk_value(cnt + 1), From a885889e62808314dc4eeffc94c94c44fe03a175 Mon Sep 17 00:00:00 2001 From: Yeoreum Yun Date: Fri, 11 Sep 2026 15:29:03 +0100 Subject: [PATCH 0910/1352] kselftest: mm: fix potential failure for merged VMA in guard-regions check_vmflag_guard() uses /proc/self/smaps to retrieve the VMA flags, but this can fail if the mapping is merged with an adjacent VMA. To avoid this potential failure, first allocate a temporary region with extra pages at both ends, unmap it, and then map the test region within the temporary address range, leaving an unmapped page on each side to prevent VMA merging. Link: https://lore.kernel.org/20260911142904.1825452-1-yeoreum.yun@arm.com Signed-off-by: Yeoreum Yun Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/mm/guard-regions.c | 14 ++++++++++++-- 1 file changed, 12 insertions(+), 2 deletions(-) diff --git a/tools/testing/selftests/mm/guard-regions.c b/tools/testing/selftests/mm/guard-regions.c index 5c8ec3ca75d7d2..b724d62d2b7555 100644 --- a/tools/testing/selftests/mm/guard-regions.c +++ b/tools/testing/selftests/mm/guard-regions.c @@ -2257,8 +2257,18 @@ TEST_F(guard_regions, smaps) char *ptr, *ptr2; int i; - /* Map a region. */ - ptr = mmap_(self, variant, NULL, 10 * page_size, PROT_READ | PROT_WRITE, 0, 0); + /* Map then unmap placeholder to avoid adjacent merges */ + ptr = mmap_(self, variant, NULL, 12 * page_size, PROT_NONE, 0, 0); + ASSERT_NE(ptr, MAP_FAILED); + ASSERT_EQ(munmap(ptr, 12 * page_size), 0); + + /* + * Map a region for the test. Since the preceding temporary mapping + * succeeded, this mapping should also succeed without merging with + * adjacent VMAs. + */ + ptr = mmap_(self, variant, ptr + page_size, 10 * page_size, + PROT_READ | PROT_WRITE, MAP_FIXED, 0); ASSERT_NE(ptr, MAP_FAILED); /* We shouldn't yet see a guard flag. */ From 43669b5bb035d7d72c6e9d1a584b1e22f8acf275 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Fri, 11 Sep 2026 06:55:03 -0700 Subject: [PATCH 0911/1352] mm/damon/api: introduce DAMOS_FILTER_TYPE_PROBE_HITS_WSUM Patch series "mm/damon: introduce probe_hits_wsum DAMOS core filter", v2. DAMON can do flexible data attributes monitoring. The classical data access monitoring can also be done using the attributes monitoring. The access event is just one of the data attributes that DAMON supports. Users do monitoring to make some actions based on it. DAMOS is a feature for automating that. However, DAMOS cannot utilize the data attributes monitoring results. It is still Data "Access" Monitoring-based Operation Schemes. It requires users to set the target "access" pattern. DAMOS core filter is effectively the same as the target access pattern. It is just a more generalized and flexible way of describing the operation action target region. Introduce a new DAMOS core filter type, probe_hits_wsum. It specifies the filter target based on a range of the probe hits weighted sum. Using this, users can apply DAMOS actions to regions of specific data attributes pattern. Note that the classic target access pattern still works. Hence the target nr_accesses range should still be properly configured. The new filter would be used in only data attributes-only mode. In the mode, classic access monitoring is just turned off, and therefore nr_accesses of regions are always zero. Users could simply set the target nr_accesses range to include the zero nr_Accesses regions. Patches Sequence ================ Patch 1 updates the DAMON kernel API for the new filter type. Patch 2 extends damon_probe_hits_wsum() to do the calculation based on moving sum. Patch 3 implements the filter type in the core layer. Patch 4 refactors DAMON sysfs interface internal data structure for efficient reuse of data structure for the probe hits weighted sum range user inputs. Patch 5 updates DAMON sysfs interface to support the new filter type. Patches 6 and 7 update design and usage documents for the new filter type, respectively. This patch (of 7): Update DAMON kernel API to introduce new DAMOS core filter type, PROBE_HITS_WSUM. It will allow API callers to describe the DAMOS action target regions based on their probe_hits weighted sum. For describing the filtering target weighted sum range, add two type-dependent union fields to damos_filter. Link: https://lore.kernel.org/20260911135510.96914-1-sj@kernel.org Link: https://lore.kernel.org/20260911135510.96914-2-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/damon.h | 20 ++++++++++++++------ 1 file changed, 14 insertions(+), 6 deletions(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index 16800b3dc379b9..9e84bdeb5616df 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -399,14 +399,15 @@ struct damos_stat { * @DAMOS_FILTER_TYPE_UNMAPPED: Unmapped pages. * @DAMOS_FILTER_TYPE_ADDR: Address range. * @DAMOS_FILTER_TYPE_TARGET: Data Access Monitoring target. + * @DAMOS_FILTER_TYPE_PROBE_HITS_WSUM: probe_hits weighted sum range. * @NR_DAMOS_FILTER_TYPES: Number of filter types. * - * All types except &DAMOS_FILTER_TYPE_ADDR and &DAMOS_FILTER_TYPE_TARGET - * are handled by the underlying &struct damon_operations as a part of scheme - * action trying, and therefore accounted as 'tried'. In contrast, - * &DAMOS_FILTER_TYPE_ADDR and &DAMOS_FILTER_TYPE_TARGET filters are handled - * by the core layer before trying of the action, and therefore not accounted - * as 'tried'. + * All types except &DAMOS_FILTER_TYPE_ADDR, &DAMOS_FILTER_TYPE_TARGET and + * &DAMOS_FILTER_TYPE_PROBE_HITS_WSUM are handled by the underlying &struct + * damon_operations as a part of scheme action trying, and therefore accounted + * as 'tried'. In contrast, &DAMOS_FILTER_TYPE_ADDR and + * &DAMOS_FILTER_TYPE_TARGET filters are handled by the core layer before + * trying of the action, and therefore not accounted as 'tried'. * * Support for the operations-handled filters depends on the running * &struct damon_operations. @@ -420,6 +421,7 @@ enum damos_filter_type { DAMOS_FILTER_TYPE_UNMAPPED, DAMOS_FILTER_TYPE_ADDR, DAMOS_FILTER_TYPE_TARGET, + DAMOS_FILTER_TYPE_PROBE_HITS_WSUM, NR_DAMOS_FILTER_TYPES, }; @@ -434,6 +436,8 @@ enum damos_filter_type { * &damon_ctx->adaptive_targets if @type is * DAMOS_FILTER_TYPE_TARGET. * @sz_range: Size range if @type is DAMOS_FILTER_TYPE_HUGEPAGE_SIZE. + * @range_min: Minimum value of range arguments. + * @range_max: Maximum value of range arguments. * * Before applying the &damos->action to a memory region, DAMOS checks if each * byte of the region matches to this given condition and avoid applying the @@ -450,6 +454,10 @@ struct damos_filter { struct damon_addr_range addr_range; int target_idx; struct damon_size_range sz_range; + struct { + unsigned long range_min; + unsigned long range_max; + }; }; /* private: */ /* List head for siblings. */ From 63840b8e117f056f6e1e9c4173d85dff2d9ffbc6 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Fri, 11 Sep 2026 07:20:01 -0700 Subject: [PATCH 0912/1352] mm/damon/api: clarify DAMOS_FILTER_TYPE_PROBE_HITS_WSUM behavior clarify DAMOS_FILTER_TYPE_PROBE_HITS_WSUM behavior Link: https://lore.kernel.org/20260911142522.98013-1-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/damon.h | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index 9e84bdeb5616df..e27ed156e7657e 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -405,9 +405,9 @@ struct damos_stat { * All types except &DAMOS_FILTER_TYPE_ADDR, &DAMOS_FILTER_TYPE_TARGET and * &DAMOS_FILTER_TYPE_PROBE_HITS_WSUM are handled by the underlying &struct * damon_operations as a part of scheme action trying, and therefore accounted - * as 'tried'. In contrast, &DAMOS_FILTER_TYPE_ADDR and - * &DAMOS_FILTER_TYPE_TARGET filters are handled by the core layer before - * trying of the action, and therefore not accounted as 'tried'. + * as 'tried'. In contrast, &DAMOS_FILTER_TYPE_ADDR, &DAMOS_FILTER_TYPE_TARGET + * and &DAMOS_FILTER_TYPE_PROBE_HITS_WSUM filters are handled by the core layer + * before trying of the action, and therefore not accounted as 'tried'. * * Support for the operations-handled filters depends on the running * &struct damon_operations. From 3151ae7241b6905fd648e78ea5d2f8c4ca02090a Mon Sep 17 00:00:00 2001 From: SJ Park Date: Fri, 11 Sep 2026 06:55:04 -0700 Subject: [PATCH 0913/1352] mm/damon/core: extend probe_hits_wsum() for moving sum based calculation damon_probe_hits_wsum() is being called only in aggregation time. In future, it could also be used by DAMOS. In this case, since DAMOS uses its own apply_interval, it could be called in sampling time. Then using the not yet fully aggregated probe_hits could result in suboptimum outcomes. Extend damon_probe_hits_wsum() to get the weighted sum based on moving sum to prepare the DAMOS usage. Link: https://lore.kernel.org/20260911135510.96914-3-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/damon.h | 2 +- mm/damon/core.c | 8 ++++++-- mm/damon/paddr.c | 2 +- mm/damon/vaddr.c | 2 +- 4 files changed, 9 insertions(+), 5 deletions(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index e27ed156e7657e..4be7d1df8e71fa 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -1075,7 +1075,7 @@ unsigned int damon_nr_accesses_mvsum(struct damon_region *r, struct damon_ctx *ctx); unsigned char damon_probe_hits_mvsum(int probe_idx, struct damon_region *r, struct damon_ctx *ctx); -unsigned int damon_probe_hits_wsum(struct damon_region *r, bool last, +unsigned int damon_probe_hits_wsum(struct damon_region *r, bool last, bool mv, struct damon_ctx *ctx); int damon_set_regions(struct damon_target *t, struct damon_addr_range *ranges, diff --git a/mm/damon/core.c b/mm/damon/core.c index ea6df4311ceb7f..2b361ab2407887 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -469,11 +469,12 @@ static bool damon_is_last_region(struct damon_region *r, * damon_probe_hits_wsum() - Returns probe hits weighted sum of a region. * @r: region to get the weighted sum of. * @last: if the request is for last-window aggregated probe hits. + * @mv: use moving sum. * @ctx: context of &r. * * Return: the weighted sum of probe hits of the region. */ -unsigned int damon_probe_hits_wsum(struct damon_region *r, bool last, +unsigned int damon_probe_hits_wsum(struct damon_region *r, bool last, bool mv, struct damon_ctx *ctx) { struct damon_probe *probe; @@ -483,6 +484,9 @@ unsigned int damon_probe_hits_wsum(struct damon_region *r, bool last, damon_for_each_probe(probe, ctx) { if (last) sum += r->last_probe_hits[i++] * probe->weight; + else if (mv) + sum += damon_probe_hits_mvsum(i++, r, ctx) * + probe->weight; else sum += r->probe_hits[i++] * probe->weight; } @@ -3472,7 +3476,7 @@ static unsigned int damon_merge_score(struct damon_region *r, bool last, struct damon_ctx *ctx, bool use_probe_hits) { if (use_probe_hits) - return damon_probe_hits_wsum(r, last, ctx); + return damon_probe_hits_wsum(r, last, false, ctx); if (last) return r->last_nr_accesses; return r->nr_accesses; diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c index 79195026b903a6..5abfabaa339e0e 100644 --- a/mm/damon/paddr.c +++ b/mm/damon/paddr.c @@ -208,7 +208,7 @@ static unsigned int damon_pa_apply_probes(struct damon_ctx *ctx, folio_put(folio); if (return_max_wsum) max_wsum = max(damon_probe_hits_wsum(r, false, - ctx), max_wsum); + false, ctx), max_wsum); } } return max_wsum; diff --git a/mm/damon/vaddr.c b/mm/damon/vaddr.c index 7063356370c34b..d5dde97b3cd0d0 100644 --- a/mm/damon/vaddr.c +++ b/mm/damon/vaddr.c @@ -711,7 +711,7 @@ static unsigned int damon_va_apply_probes(struct damon_ctx *ctx, __damon_va_apply_probes(ctx, mm, r); if (return_max_wsum) max_wsum = max(damon_probe_hits_wsum(r, false, - ctx), max_wsum); + false, ctx), max_wsum); } if (mm) mmput(mm); From fd307b870db2b858ce3dea543c23080f83cdbf6d Mon Sep 17 00:00:00 2001 From: SJ Park Date: Fri, 11 Sep 2026 06:55:05 -0700 Subject: [PATCH 0914/1352] mm/damon/core: support probe_hits_wsum damos core filter Implement probe_hits_wsum DAMOS core filter support in the core layer. Make three small changes for the support. First, update damos_filter_for_ops() to treat probe_hits_wsum filter as core filter. Second, Update destination damos_filter->range_{min,max} for probe_hits_wsum type damos filter commits. Third, extend damos_filter_match() to handle probe_hits_wsum type filter. Calculate the weighted sum of the given region and compare it with the given filter's target weighted sum range. Link: https://lore.kernel.org/20260911135510.96914-4-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/core.c | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 2b361ab2407887..38383ced3dd6c0 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -658,6 +658,7 @@ bool damos_filter_for_ops(enum damos_filter_type type) switch (type) { case DAMOS_FILTER_TYPE_ADDR: case DAMOS_FILTER_TYPE_TARGET: + case DAMOS_FILTER_TYPE_PROBE_HITS_WSUM: return false; default: break; @@ -1332,6 +1333,10 @@ static void damos_commit_filter_arg( case DAMOS_FILTER_TYPE_HUGEPAGE_SIZE: dst->sz_range = src->sz_range; break; + case DAMOS_FILTER_TYPE_PROBE_HITS_WSUM: + dst->range_min = src->range_min; + dst->range_max = src->range_max; + break; default: break; } @@ -2510,7 +2515,7 @@ static bool damos_filter_match(struct damon_ctx *ctx, struct damon_target *t, bool matched = false; struct damon_target *ti; int target_idx = 0; - unsigned long start, end; + unsigned long start, end, wsum; switch (filter->type) { case DAMOS_FILTER_TYPE_TARGET: @@ -2545,6 +2550,11 @@ static bool damos_filter_match(struct damon_ctx *ctx, struct damon_target *t, damon_split_region_at(t, r, end - r->ar.start); matched = true; break; + case DAMOS_FILTER_TYPE_PROBE_HITS_WSUM: + wsum = damon_probe_hits_wsum(r, false, true, ctx); + matched = filter->range_min <= wsum && + wsum <= filter->range_max; + break; default: return false; } From e0eb4d36ba1e924e78b1cdfbbe8171f03314db98 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Fri, 11 Sep 2026 06:55:06 -0700 Subject: [PATCH 0915/1352] mm/damon/sysfs-schemes: rename sysfs_filter->sz_range to range_{min,max} DAMON sysfs interface provides 'min' and 'max' files under the DAMOS filter directory. The purpose is setting the general range arguments for hugepage_size like filters that require range arguments. So far, hugepage_size was the only filter using it. Hence sz_range field of damon_sysfs_scheme_filter struct was connected to the files. In future, we could add a new filter that can reuse the 'min' and 'max' files. And the filter might use a range of a type that is not size. For example, probe_hits weighted sum. In this case, simply reusing the sz_range field would make it a little confusing. Rename sz_range to range_{min,max} to avoid such confusion. Link: https://lore.kernel.org/20260911135510.96914-5-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs-schemes.c | 19 +++++++++++-------- 1 file changed, 11 insertions(+), 8 deletions(-) diff --git a/mm/damon/sysfs-schemes.c b/mm/damon/sysfs-schemes.c index d9b81d7b5910ed..4d9147d2a269ea 100644 --- a/mm/damon/sysfs-schemes.c +++ b/mm/damon/sysfs-schemes.c @@ -534,7 +534,8 @@ struct damon_sysfs_scheme_filter { bool allow; char *memcg_path; struct damon_addr_range addr_range; - struct damon_size_range sz_range; + unsigned long range_min; + unsigned long range_max; int target_idx; }; @@ -588,6 +589,7 @@ damos_sysfs_filter_type_names[] = { .type = DAMOS_FILTER_TYPE_TARGET, .name = "target", }, + }; static ssize_t type_show(struct kobject *kobj, @@ -778,7 +780,7 @@ static ssize_t min_show(struct kobject *kobj, struct damon_sysfs_scheme_filter *filter = container_of(kobj, struct damon_sysfs_scheme_filter, kobj); - return sysfs_emit(buf, "%lu\n", filter->sz_range.min); + return sysfs_emit(buf, "%lu\n", filter->range_min); } static ssize_t min_store(struct kobject *kobj, @@ -786,7 +788,7 @@ static ssize_t min_store(struct kobject *kobj, { struct damon_sysfs_scheme_filter *filter = container_of(kobj, struct damon_sysfs_scheme_filter, kobj); - int err = kstrtoul(buf, 0, &filter->sz_range.min); + int err = kstrtoul(buf, 0, &filter->range_min); return err ? err : count; } @@ -797,7 +799,7 @@ static ssize_t max_show(struct kobject *kobj, struct damon_sysfs_scheme_filter *filter = container_of(kobj, struct damon_sysfs_scheme_filter, kobj); - return sysfs_emit(buf, "%lu\n", filter->sz_range.max); + return sysfs_emit(buf, "%lu\n", filter->range_max); } static ssize_t max_store(struct kobject *kobj, @@ -805,7 +807,7 @@ static ssize_t max_store(struct kobject *kobj, { struct damon_sysfs_scheme_filter *filter = container_of(kobj, struct damon_sysfs_scheme_filter, kobj); - int err = kstrtoul(buf, 0, &filter->sz_range.max); + int err = kstrtoul(buf, 0, &filter->range_max); return err ? err : count; } @@ -2835,12 +2837,13 @@ static int damon_sysfs_add_scheme_filters(struct damos *scheme, } else if (filter->type == DAMOS_FILTER_TYPE_TARGET) { filter->target_idx = sysfs_filter->target_idx; } else if (filter->type == DAMOS_FILTER_TYPE_HUGEPAGE_SIZE) { - if (sysfs_filter->sz_range.min > - sysfs_filter->sz_range.max) { + if (sysfs_filter->range_min > + sysfs_filter->range_max) { damos_destroy_filter(filter); return -EINVAL; } - filter->sz_range = sysfs_filter->sz_range; + filter->sz_range.min = sysfs_filter->range_min; + filter->sz_range.max = sysfs_filter->range_max; } damos_add_filter(scheme, filter); From 9837299c1957c4c21efbe202f274d4d9f68fad59 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Fri, 11 Sep 2026 06:55:07 -0700 Subject: [PATCH 0916/1352] mm/damon/sysfs-schemes: support probe_hits_wsum damos core filter Extend DAMON sysfs interface to support probe_hits_wsum input. Also update sysfs input based scheme build logic to setup the min/max probe hits weighted sum range as user provided. Link: https://lore.kernel.org/20260911135510.96914-6-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs-schemes.c | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/mm/damon/sysfs-schemes.c b/mm/damon/sysfs-schemes.c index 4d9147d2a269ea..3de4d804e049f8 100644 --- a/mm/damon/sysfs-schemes.c +++ b/mm/damon/sysfs-schemes.c @@ -589,7 +589,10 @@ damos_sysfs_filter_type_names[] = { .type = DAMOS_FILTER_TYPE_TARGET, .name = "target", }, - + { + .type = DAMOS_FILTER_TYPE_PROBE_HITS_WSUM, + .name = "probe_hits_wsum", + }, }; static ssize_t type_show(struct kobject *kobj, @@ -2844,6 +2847,13 @@ static int damon_sysfs_add_scheme_filters(struct damos *scheme, } filter->sz_range.min = sysfs_filter->range_min; filter->sz_range.max = sysfs_filter->range_max; + } else if (filter->type == DAMOS_FILTER_TYPE_PROBE_HITS_WSUM) { + filter->range_min = sysfs_filter->range_min; + filter->range_max = sysfs_filter->range_max; + if (filter->range_min > filter->range_max) { + damos_destroy_filter(filter); + return -EINVAL; + } } damos_add_filter(scheme, filter); From 9eaf82c8b2c4ec5280c3cda4d72d2b5634f75cbd Mon Sep 17 00:00:00 2001 From: SJ Park Date: Fri, 11 Sep 2026 06:55:08 -0700 Subject: [PATCH 0917/1352] Docs/mm/damon/design: update for probe_hits_wsum DAMOS core filter Update DAMON design document for the newly added probe hits weighted sum based DAMOS core filter type. Link: https://lore.kernel.org/20260911135510.96914-7-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/mm/damon/design.rst | 3 +++ 1 file changed, 3 insertions(+) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index 22b785cd11dfec..cb116a82ff20b4 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -823,6 +823,9 @@ Below ``type`` of filters are currently supported. - Applied to pages that belonging to a given address range. - target - Applied to pages that belonging to a given DAMON monitoring target. + - probe_hits_wsum + - Matches to monitoring regions having a given range of :ref:`probe + hits weighted sum ` value. - Operations layer handled, supported by only ``paddr`` operations set. - anon - Applied to pages that containing data that not stored in files. From f1cffe232837bf31534addf913af75c362710879 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Fri, 11 Sep 2026 06:55:09 -0700 Subject: [PATCH 0918/1352] Docs/admin-guide/mm/damon/usage: update for probe_hits_wsum DAMOS filter Update DAMON usage document for the newly added probe hits weighted sum based DAMOS core filter type. Link: https://lore.kernel.org/20260911135510.96914-8-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/admin-guide/mm/damon/usage.rst | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/Documentation/admin-guide/mm/damon/usage.rst b/Documentation/admin-guide/mm/damon/usage.rst index 023c6334024f8e..d3e37400367bdd 100644 --- a/Documentation/admin-guide/mm/damon/usage.rst +++ b/Documentation/admin-guide/mm/damon/usage.rst @@ -551,6 +551,10 @@ and ``damon_target_idx``. To ``type`` file, you can write the type of the filter. Refer to :ref:`the design doc ` for available type names, their meaning and on what layer those are handled. +For ``probe_hits_wsum`` type, you can specify the minimum and maximum probe +hits weighted sum value for the filter to ``min`` and ``max`` files, +respectively. + For ``memcg`` type, you can specify the memory cgroup of the interest by writing the path of the memory cgroup from the cgroups mount point to ``memcg_path`` file. For ``addr`` type, you can specify the start and end From 961767ba67392040ff0df3ba33741edc048e9e99 Mon Sep 17 00:00:00 2001 From: Jaeyeon Lee Date: Sat, 12 Sep 2026 22:29:03 +0200 Subject: [PATCH 0919/1352] selftests/mm: skip khugepaged file tests if mkfs.xfs is unavailable The XFS setup for ./khugepaged all:file checks that the kernel supports XFS but not that mkfs.xfs is installed. When CONFIG_XFS_FS=y and xfsprogs is missing, mkfs.xfs and mount both fail, but SPLIT_HUGE_PAGE_TEST_XFS_PATH is assigned from mktemp -d and stays set. The test then runs against plain tmpfs and six subtests fail. The script already has a skip path for this test, but it never runs because the path variable is always set. Assign SPLIT_HUGE_PAGE_TEST_XFS_PATH only once mkfs.xfs and mount have both succeeded, and remove the image and directory otherwise. Link: https://lore.kernel.org/20260912202903.16157-1-jaeyeon.lee.dev@gmail.com Signed-off-by: Jaeyeon Lee Signed-off-by: Andrew Morton Suggested-by: Zi Yan Reviewed-by: Zi Yan Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Assisted-by: LLM Cc: Shuah Khan Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko --- tools/testing/selftests/mm/run_vmtests.sh | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/tools/testing/selftests/mm/run_vmtests.sh b/tools/testing/selftests/mm/run_vmtests.sh index 9bbef9410ccc35..2e5e7975ff4d49 100755 --- a/tools/testing/selftests/mm/run_vmtests.sh +++ b/tools/testing/selftests/mm/run_vmtests.sh @@ -435,11 +435,15 @@ if [ -z "${SPLIT_HUGE_PAGE_TEST_XFS_PATH}" ]; then if test_selected "thp"; then if grep xfs /proc/filesystems &>/dev/null; then XFS_IMG=$(mktemp /tmp/xfs_img_XXXXXX) - SPLIT_HUGE_PAGE_TEST_XFS_PATH=$(mktemp -d /tmp/xfs_dir_XXXXXX) + XFS_DIR=$(mktemp -d /tmp/xfs_dir_XXXXXX) truncate -s 314572800 ${XFS_IMG} - mkfs.xfs -q ${XFS_IMG} - mount -o loop ${XFS_IMG} ${SPLIT_HUGE_PAGE_TEST_XFS_PATH} - MOUNTED_XFS=1 + if mkfs.xfs -q ${XFS_IMG} && mount -t xfs -o loop ${XFS_IMG} ${XFS_DIR}; then + SPLIT_HUGE_PAGE_TEST_XFS_PATH=${XFS_DIR} + MOUNTED_XFS=1 + else + rmdir ${XFS_DIR} + rm -f ${XFS_IMG} + fi fi fi fi From 4f4f21b56c33f9fbd5b09b9a437755174f8458f8 Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Sat, 26 Sep 2026 06:51:09 -0400 Subject: [PATCH 0920/1352] mm/mempolicy: use vm_normal_folio_pmd() in queue_folios_pmd() Patch series "mm: stop calling pmd_folio() on special PMDs", v3. Two page table walkers resolve the folio behind a PMD with pmd_folio(), which is only valid for a PMD mapping a refcounted struct page: madvise_cold_or_pageout_pte_range() mm/madvise.c queue_folios_pmd() mm/mempolicy.c vmf_insert_pfn_pmd() installs special PMDs holding a raw pfn that need not have a memmap entry at all. Both walkers can reach one and fault on the first folio field read. The PTE halves of both already use vm_normal_folio(); these two patches make the PMD halves match. The four callers of vmf_insert_pfn_pmd(), and which walker each reaches: drivers/vfio/pci/vfio_pci_core.c VM_PFNMAP mempolicy drivers/gpu/drm/drm_gem_shmem_helper.c VM_PFNMAP mempolicy drivers/gpu/drm/panthor/panthor_gem.c VM_PFNMAP mempolicy drivers/hv/mshv_vtl_main.c VM_MIXEDMAP both can_madv_lru_vma() rejects VM_PFNMAP, so only mshv_vtl_low reaches the madvise walker, and that needs CAP_SYS_ADMIN. queue_pages_walk_ops supplies its own ->test_walk, so walk_page_test()'s generic VM_PFNMAP skip never runs and vfio-pci is reachable by any process holding the device fd. Hence the different stable tags. One behaviour change: mbind(MPOL_MF_STRICT) over a PMD mapped VM_PFNMAP region now returns 0 rather than -EIO. The PTE loop already returned 0 there. drm_gem_shmem and panthor are where this is observable, since they PMD map pages that do have a memmap entry and so never faulted. Reproducer ========== No hardware needed. An out of tree module stands in for the drivers above: three misc devices, each with a ->huge_fault calling vmf_insert_pfn_pmd(), plus VM_HUGEPAGE so the fault path takes the PMD branch. /dev/pmdspec_mixed VM_MIXEDMAP, pfn at the 1 TiB mark, no memmap /dev/pmdspec_pfnmap VM_PFNMAP, pfn at the 1 TiB mark, no memmap /dev/pmdspec_real VM_PFNMAP, real alloc_pages(PMD_ORDER) on node 0 Userspace maps the device into a PMD aligned window, reads one byte to fault the PMD in, checks a module parameter to confirm it went in, then issues the operation. vng --run --user root --memory 4G --verbose \ --append "numa=fake=2" \ --exec "insmod pmdspec.ko && ./pmdspec_test " numa=fake=2 gives a node 1 to bind to; the module allocates its real page on node 0, which is what makes queue_folio_required() true. subtest operation parent series -------------------------------------------------------------------- madv_cold madvise(MADV_COLD) oops ret=0 madv_pageout madvise(MADV_PAGEOUT) oops ret=0 mbind_mixed mbind(MPOL_BIND, n1, MPOL_MF_MOVE) oops ret=0 mbind_pfnmap mbind(MPOL_BIND, n1, MPOL_MF_STRICT) oops ret=0 mbind_real mbind(MPOL_BIND, n1, MPOL_MF_STRICT) -EIO ret=0 Two things the table shows that are easy to miss in the code: - mbind_mixed passes only MPOL_MF_MOVE. MPOL_MF_STRICT is not needed for a VM_MIXEDMAP vma: walk_page_test() only skips VM_PFNMAP, and vma_migratable() is true for VM_MIXEDMAP. - mbind_real demonstrates the user visible change (-EIO -> 0) This patch (of 2): mmap a VM_PFNMAP region whose ->huge_fault installs a PMD through vmf_insert_pfn_pmd() - a vfio-pci MMIO BAR does this - then mbind(p, len, MPOL_BIND, &mask, maxnode, MPOL_MF_STRICT); With a stand-in module for the driver: BUG: unable to handle page fault for address: fffff96dc0000008 RIP: 0010:queue_folios_pte_range+0xaf/0x440 walk_pgd_range+0x52b/0xaf0 __walk_page_range+0x6a/0x1d0 walk_page_range_mm_unsafe+0x193/0x230 queue_pages_range+0x64/0xa0 do_mbind+0x25e/0x640 queue_folios_pmd(), inlined above, calls pmd_folio() on that PMD. The pfn is raw MMIO with no memmap entry, so the folio lands in unpopulated vmemmap. Neither guard stops the walk: walk_page_test() skips VM_PFNMAP, but queue_pages_walk_ops supplies ->test_walk, so it never runs queue_pages_test_walk() honours vma_migratable(), but only while MPOL_MF_STRICT is clear A VM_MIXEDMAP vma needs neither flag, being vma_migratable(), so plain mbind(MPOL_MF_MOVE) reaches this too - and there the bad folio carries on into migrate_folio_add() and folio_isolate_lru(). mshv_vtl_low is such a mapping. Use vm_normal_folio_pmd() and skip on NULL, as the PTE loop in queue_folios_pte_range() already does with vm_normal_folio(). This also filters the huge zero PMD, so its separate check is no longer needed. mbind(MPOL_MF_STRICT) over a PMD mapped VM_PFNMAP region now returns 0 rather than -EIO. The PTE loop already returned 0 there. Link: https://lore.kernel.org/20260926105110.2156652-1-gourry@gourry.net Link: https://lore.kernel.org/20260926105110.2156652-2-gourry@gourry.net Fixes: 3c8e44c9b369 ("mm: mark special bits for huge pfn mappings when inject") Signed-off-by: Gregory Price (Meta) Signed-off-by: Andrew Morton Reported-by: sashiko-bot Closes: https://sashiko.dev/#/patchset/20260817220810.1175596-1-gourry%40gourry.net Acked-by: David Hildenbrand (Arm) Reviewed-by: Zi Yan Assisted-by: LLM Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Vlastimil Babka Cc: Jann Horn Cc: Joshua Hahn Cc: Rakie Kim Cc: Ying Huang Cc: Peter Xu Cc: Jason Gunthorpe Cc: --- mm/mempolicy.c | 13 +++++-------- 1 file changed, 5 insertions(+), 8 deletions(-) diff --git a/mm/mempolicy.c b/mm/mempolicy.c index 95dba5d919e925..e9860fb9f73f8d 100644 --- a/mm/mempolicy.c +++ b/mm/mempolicy.c @@ -667,7 +667,8 @@ static inline bool queue_folio_required(struct folio *folio, return node_isset(nid, *qp->nmask) == !(flags & MPOL_MF_INVERT); } -static void queue_folios_pmd(pmd_t *pmd, struct mm_walk *walk) +static void queue_folios_pmd(pmd_t *pmd, unsigned long addr, + struct mm_walk *walk) { struct folio *folio; struct queue_pages *qp = walk->private; @@ -678,13 +679,9 @@ static void queue_folios_pmd(pmd_t *pmd, struct mm_walk *walk) qp->nr_failed++; return; } - folio = pmd_folio(pmdval); - if (folio_is_zone_device(folio)) + folio = vm_normal_folio_pmd(walk->vma, addr, pmdval); + if (!folio || folio_is_zone_device(folio)) return; - if (is_huge_zero_folio(folio)) { - walk->action = ACTION_CONTINUE; - return; - } if (!queue_folio_required(folio, qp)) return; if (!(qp->flags & (MPOL_MF_MOVE | MPOL_MF_MOVE_ALL)) || @@ -717,7 +714,7 @@ static int queue_folios_pte_range(pmd_t *pmd, unsigned long addr, ptl = pmd_trans_huge_lock(pmd, vma); if (ptl) { - queue_folios_pmd(pmd, walk); + queue_folios_pmd(pmd, addr, walk); spin_unlock(ptl); goto out; } From dce8a3fe5273d5f0102fa4645df79752c3ccc7d3 Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Sat, 26 Sep 2026 06:51:10 -0400 Subject: [PATCH 0921/1352] mm/madvise: use vm_normal_folio_pmd() in cold/pageout PMD range mmap a VM_MIXEDMAP region whose ->huge_fault installs a PMD through vmf_insert_pfn_pmd() - mshv_vtl_low does this, and needs CAP_SYS_ADMIN to open - then: madvise(p, PMD_SIZE, MADV_PAGEOUT); With a stand-in module for the driver: BUG: unable to handle page fault for address: fffff587c0000008 RIP: 0010:madvise_cold_or_pageout_pte_range+0x410/0x9b0 walk_pgd_range+0x52b/0xaf0 __walk_page_range+0x6a/0x1d0 walk_page_range_vma_unsafe+0x8e/0x120 madvise_pageout+0xb2/0x180 madvise_vma_behavior+0x46b/0xa90 do_madvise+0x108/0x190 __x64_sys_madvise+0x26/0x30 Nothing validates the pfn on the way in: can_madv_lru_vma() rejects VM_PFNMAP, but not VM_MIXEDMAP can_fault() *pfn = vmf->pgoff & ~(mask >> PAGE_SHIFT); vmf_insert_pfn_pmd() no pfn_valid() check pmd_folio() pfn_to_page() -> unpopulated vmemmap Even with a valid pfn the path is wrong. The mapping carries no rmap, so folio_maybe_mapped_shared() sees mapcount 0, and the walker goes on to folio_deactivate(), or folio_isolate_lru() plus reclaim_pages(), against a folio this mapping does not own. Use vm_normal_folio_pmd() and skip on NULL, as the PTE half of this same walker already does with vm_normal_folio(). This also filters the huge zero PMD, so its separate check is no longer needed. Link: https://lore.kernel.org/20260926105110.2156652-3-gourry@gourry.net Fixes: 3c8e44c9b369 ("mm: mark special bits for huge pfn mappings when inject") Signed-off-by: Gregory Price (Meta) Signed-off-by: Andrew Morton Reported-by: sashiko-bot Closes: https://sashiko.dev/#/patchset/20260817220810.1175596-1-gourry%40gourry.net Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) Reviewed-by: Zi Yan Assisted-by: LLM Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Jann Horn Cc: Joshua Hahn Cc: Rakie Kim Cc: Ying Huang Cc: Peter Xu Cc: Jason Gunthorpe Cc: # v6.19+ --- mm/madvise.c | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/mm/madvise.c b/mm/madvise.c index acb5215e2f079f..1cfb0433229310 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -395,16 +395,15 @@ static int madvise_cold_or_pageout_pte_range(pmd_t *pmd, return 0; orig_pmd = *pmd; - if (is_huge_zero_pmd(orig_pmd)) - goto huge_unlock; - if (unlikely(!pmd_present(orig_pmd))) { VM_WARN_ON_ONCE(!pmd_is_migration_entry(orig_pmd) && !pmd_is_device_private_entry(orig_pmd)); goto huge_unlock; } - folio = pmd_folio(orig_pmd); + folio = vm_normal_folio_pmd(vma, addr, orig_pmd); + if (!folio) + goto huge_unlock; if (folio_is_zone_device(folio)) goto huge_unlock; From f084b06e48076765f838ba92801f97da4f9fcf16 Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Tue, 22 Sep 2026 22:29:01 -0400 Subject: [PATCH 0922/1352] mm: refactor find_next_best_node to find_next_best_node_in Patch series "mm: refactor zonelist constructors and iterators", v3. find_next_best_node() picks the next-closest node when building a fallback list, and hardcodes N_MEMORY as the set it picks from. Refactor it into find_next_best_node_in(), which takes the candidate set explicitly. This makes the existing behaviour explicit at both mm/memory-tiers.c call sites - they select demotion targets in fallback order from N_MEMORY - and lets callers narrow that set. Then extract the per-node construction loop out of build_zonelists() into build_node_zonelist(), parameterised on the candidate nodemask and destination zonelist index. Together these allow a zonelist to be built over a candidate set other than N_MEMORY, into a zonelist other than FALLBACK, and iterated in fallback order over a caller-defined subset. These are prerequisites for generating a private node zonelist (nodes unreachable by default), but are otherwise general improvements to the existing interfaces so I'm proposing them separately. No functional change intended - purely refactor commits. This patch (of 2): find_next_best_node() picks the next-closest node for a fallback list from the full N_MEMORY set. Refactor it into find_next_best_node_in(), which takes an explicit candidates nodemask. This enables building fallback lists with non-N_MEMORY candidates. No functional change: every caller still selects from N_MEMORY. Link: https://lore.kernel.org/20260923022902.2433614-1-gourry@gourry.net Link: https://lore.kernel.org/20260923022902.2433614-2-gourry@gourry.net Signed-off-by: Gregory Price Signed-off-by: Andrew Morton Reviewed-by: Vlastimil Babka (SUSE) Acked-by: Balbir Singh Reviewed-by: Zi Yan Reviewed-by: Zenghui Yu (Huawei) Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Joshua Hahn Cc: Rakie Kim Cc: Ying Huang Cc: Brendan Jackman Cc: Johannes Weiner --- mm/internal.h | 6 ++++-- mm/memory-tiers.c | 7 ++++--- mm/page_alloc.c | 13 ++++++++----- 3 files changed, 16 insertions(+), 10 deletions(-) diff --git a/mm/internal.h b/mm/internal.h index 0dca33db068f6b..05179c4b2090ef 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -1124,7 +1124,8 @@ extern int node_reclaim_mode; extern unsigned long node_reclaim(struct pglist_data *pgdat, gfp_t gfp_mask, unsigned int order); -extern int find_next_best_node(int node, nodemask_t *used_node_mask); +int find_next_best_node_in(int node, nodemask_t *used_node_mask, + const nodemask_t *candidates); #else #define node_reclaim_mode 0 @@ -1133,7 +1134,8 @@ static inline unsigned long node_reclaim(struct pglist_data *pgdat, { return 0; } -static inline int find_next_best_node(int node, nodemask_t *used_node_mask) +static inline int find_next_best_node_in(int node, nodemask_t *used_node_mask, + const nodemask_t *candidates) { return NUMA_NO_NODE; } diff --git a/mm/memory-tiers.c b/mm/memory-tiers.c index 54851d8a195b03..25e121851b5862 100644 --- a/mm/memory-tiers.c +++ b/mm/memory-tiers.c @@ -370,7 +370,7 @@ int next_demotion_node(int node, const nodemask_t *allowed_mask) * closest demotion target. */ nodes_complement(mask, *allowed_mask); - return find_next_best_node(node, &mask); + return find_next_best_node_in(node, &mask, &node_states[N_MEMORY]); } static void disable_all_demotion_targets(void) @@ -450,7 +450,7 @@ static void establish_demotion_targets(void) memtier = list_next_entry(memtier, list); tier_nodes = get_memtier_nodemask(memtier); /* - * find_next_best_node, use 'used' nodemask as a skip list. + * find_next_best_node_in, use 'used' nodemask as a skip list. * Add all memory nodes except the selected memory tier * nodelist to skip list so that we find the best node from the * memtier nodelist. @@ -463,7 +463,8 @@ static void establish_demotion_targets(void) * in the preferred mask when allocating pages during demotion. */ do { - target = find_next_best_node(node, &tier_nodes); + target = find_next_best_node_in(node, &tier_nodes, + &node_states[N_MEMORY]); if (target == NUMA_NO_NODE) break; diff --git a/mm/page_alloc.c b/mm/page_alloc.c index 4fb62c6fc3e432..9a80657d70877d 100644 --- a/mm/page_alloc.c +++ b/mm/page_alloc.c @@ -5823,9 +5823,10 @@ static int numa_zonelist_order_handler(const struct ctl_table *table, int write, static int node_load[MAX_NUMNODES]; /** - * find_next_best_node - find the next node that should appear in a given node's fallback list + * find_next_best_node_in - find the next node that should appear in a given node's fallback list * @node: node whose fallback list we're appending * @used_node_mask: nodemask_t of already used nodes + * @candidates: nodemask_t of nodes eligible for selection * * We use a number of factors to determine which is the next node that should * appear on a given node's fallback list. The node should not have appeared @@ -5837,7 +5838,8 @@ static int node_load[MAX_NUMNODES]; * * Return: node id of the found node or %NUMA_NO_NODE if no node is found. */ -int find_next_best_node(int node, nodemask_t *used_node_mask) +int find_next_best_node_in(int node, nodemask_t *used_node_mask, + const nodemask_t *candidates) { int n, val; int min_val = INT_MAX; @@ -5847,12 +5849,12 @@ int find_next_best_node(int node, nodemask_t *used_node_mask) * Use the local node if we haven't already, but for memoryless local * node, we should skip it and fall back to other nodes. */ - if (!node_isset(node, *used_node_mask) && node_state(node, N_MEMORY)) { + if (!node_isset(node, *used_node_mask) && node_isset(node, *candidates)) { node_set(node, *used_node_mask); return node; } - for_each_node_state(n, N_MEMORY) { + for_each_node_mask(n, *candidates) { /* Don't want a node to appear more than once */ if (node_isset(n, *used_node_mask)) @@ -5937,7 +5939,8 @@ static void build_zonelists(pg_data_t *pgdat) prev_node = local_node; memset(node_order, 0, sizeof(node_order)); - while ((node = find_next_best_node(local_node, &used_mask)) >= 0) { + while ((node = find_next_best_node_in(local_node, &used_mask, + &node_states[N_MEMORY])) >= 0) { /* * We don't want to pressure a particular node. * So adding penalty to the first node in same From e80d4f489c28259470fe7926cc41f450e7cdf573 Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Tue, 22 Sep 2026 22:29:02 -0400 Subject: [PATCH 0923/1352] mm/page_alloc: refactor build_node_zonelist() out of build_zonelists() Extract per-node fallback-list construction into build_node_zonelist(). Build each selected node directly into the destination zonelist so no intermediate node_order array or node count is needed. Print the fallback order as each node is added. This lets us build new zonelists from candidate nodemasks instead of just the default N_MEMORY node state list. Add a zlidx argument so callers explicitly select the destination zonelist. The existing caller continues to use ZONELIST_FALLBACK. No functional change: build_zonelists() builds and prints the same FALLBACK list over N_MEMORY with node_load updates as before. Link: https://lore.kernel.org/20260923022902.2433614-3-gourry@gourry.net Signed-off-by: Gregory Price Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Vlastimil Babka (SUSE) Reviewed-by: Zenghui Yu (Huawei) Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Joshua Hahn Cc: Rakie Kim Cc: Ying Huang Cc: Brendan Jackman Cc: Johannes Weiner --- mm/page_alloc.c | 63 ++++++++++++++++++------------------------------- 1 file changed, 23 insertions(+), 40 deletions(-) diff --git a/mm/page_alloc.c b/mm/page_alloc.c index 9a80657d70877d..c9318a97f69460 100644 --- a/mm/page_alloc.c +++ b/mm/page_alloc.c @@ -5887,31 +5887,6 @@ int find_next_best_node_in(int node, nodemask_t *used_node_mask, } -/* - * Build zonelists ordered by node and zones within node. - * This results in maximum locality--normal zone overflows into local - * DMA zone, if any--but risks exhausting DMA zone. - */ -static void build_zonelists_in_node_order(pg_data_t *pgdat, int *node_order, - unsigned nr_nodes) -{ - struct zoneref *zonerefs; - int i; - - zonerefs = pgdat->node_zonelists[ZONELIST_FALLBACK]._zonerefs; - - for (i = 0; i < nr_nodes; i++) { - int nr_zones; - - pg_data_t *node = NODE_DATA(node_order[i]); - - nr_zones = build_zonerefs_node(node, zonerefs); - zonerefs += nr_zones; - } - zonerefs->zone = NULL; - zonerefs->zone_idx = 0; -} - /* * Build __GFP_THISNODE zonelists */ @@ -5927,20 +5902,24 @@ static void build_thisnode_zonelists(pg_data_t *pgdat) zonerefs->zone_idx = 0; } -static void build_zonelists(pg_data_t *pgdat) +/* + * Build one zonelist ordered by node and zones within node. This results in + * maximum locality--normal zone overflows into local DMA zone, if any--but + * risks exhausting DMA zone. + */ +static void build_node_zonelist(pg_data_t *pgdat, const nodemask_t *candidates, + int zlidx) { - static int node_order[MAX_NUMNODES]; - int node, nr_nodes = 0; + struct zoneref *zonerefs = pgdat->node_zonelists[zlidx]._zonerefs; nodemask_t used_mask = NODE_MASK_NONE; - int local_node, prev_node; + int local_node = pgdat->node_id; + int prev_node = local_node; + int node; - /* NUMA-aware ordering of nodes */ - local_node = pgdat->node_id; - prev_node = local_node; + pr_info("Fallback order for Node %d: ", local_node); - memset(node_order, 0, sizeof(node_order)); while ((node = find_next_best_node_in(local_node, &used_mask, - &node_states[N_MEMORY])) >= 0) { + candidates)) >= 0) { /* * We don't want to pressure a particular node. * So adding penalty to the first node in same @@ -5950,18 +5929,22 @@ static void build_zonelists(pg_data_t *pgdat) node_distance(local_node, prev_node)) node_load[node] += 1; - node_order[nr_nodes++] = node; + zonerefs += build_zonerefs_node(NODE_DATA(node), zonerefs); + pr_cont("%d ", node); prev_node = node; } - build_zonelists_in_node_order(pgdat, node_order, nr_nodes); - build_thisnode_zonelists(pgdat); - pr_info("Fallback order for Node %d: ", local_node); - for (node = 0; node < nr_nodes; node++) - pr_cont("%d ", node_order[node]); + zonerefs->zone = NULL; + zonerefs->zone_idx = 0; pr_cont("\n"); } +static void build_zonelists(pg_data_t *pgdat) +{ + build_node_zonelist(pgdat, &node_states[N_MEMORY], ZONELIST_FALLBACK); + build_thisnode_zonelists(pgdat); +} + #ifdef CONFIG_HAVE_MEMORYLESS_NODES /* * Return node id of node used for "local" allocations. From 7397680aac1110d23a6ca9d28d9acaba569ddab5 Mon Sep 17 00:00:00 2001 From: Yury Norov Date: Fri, 11 Sep 2026 18:14:41 -0400 Subject: [PATCH 0924/1352] compiler.h: add ASSERT_STATIC_STORAGE() Patch series "Catch automatic storage in IDA and Maple Tree definitions". A 0day report [1] from the region allocation benchmark exposed a lockdep initialization bug: a stack-local Maple Tree used MTREE_INIT(), whose embedded lock has a static initializer. On the first allocation, lockdep rejected the lock address as a non-static class key and disabled locking validation. The IDA benchmark had the same issue, masked because it ran after Maple Tree had already disabled lockdep. The fix [2] switches the test to using mt_init_flags() and ida_init(). This series adds a compile-time check to the related DEFINE_IDA() and DEFINE_MTREE() declaration macros to catch the same class of mistake earlier. Patch 1 introduces ASSERT_STATIC_STORAGE(). It declares an unused static pointer initialized with the object's address, requiring that address to be a valid static initializer. Patches 2 and 3 apply the helper to IDA and Maple Tree definitions, respectively. The helper is mirrored in the tools compiler header. The existing automatic local IDAs and Maple Trees in the userspace radix-tree tests are converted to runtime initialization. The interval-tree span test keeps its existing mt_init_flags() call and uses a plain Maple Tree declaration. For example, an automatic local definition: void example(void) { DEFINE_IDA(ida); ida_destroy(&ida); } now produces: error: initializer element is not constant note: in expansion of macro 'ASSERT_STATIC_STORAGE' note: in expansion of macro 'DEFINE_IDA' File-scope definitions and static local definitions remain valid. Automatic local objects should use ida_init(), mt_init(), or mt_init_flags(). The check is limited to declaration macros. Direct uses of IDA_INIT(), MTREE_INIT(), and MTREE_INIT_EXT() remain unchanged. The helper cannot be inserted directly into those initializer expressions because it expands to a declaration. Validated by GCC and Clang checks accepting static storage and rejecting automatic storage The userspace IDR/IDA and Maple Tree test are passed as well. This patch (of 3): Static lock initializers rely on a persistent object address when lockdep assigns a lock-class key. Using such an initializer for an automatic local object can compile successfully but disable lockdep on the first lock acquisition. Add ASSERT_STATIC_STORAGE() for declaration macros that require static storage duration. It declares an unused static pointer initialized with the object's address. An automatic local object's address is not a valid static initializer, so the compiler rejects it. Mirror the helper in tools/include/linux/compiler.h because the userspace radix-tree tests include the kernel IDA and Maple Tree headers with the tools compiler definitions. For example: void example(void) { int object; ASSERT_STATIC_STORAGE(object); } GCC reports: error: initializer element is not constant name##_storage_check = &(name) ^ note: in expansion of macro 'ASSERT_STATIC_STORAGE' ASSERT_STATIC_STORAGE(object); File-scope objects and static local objects remain valid. The helper takes an object identifier and must be used as a declaration after that object has been declared. Link: https://lore.kernel.org/all/20260911155244.1406122-1-ynorov@nvidia.com/ Link: https://lore.kernel.org/20260911221444.1523311-2-ynorov@nvidia.com Link: https://download.01.org/0day-ci/archive/20260910/202609101106.771b567e-lkp@intel.com/ [1] Link: https://lore.kernel.org/all/20260911155244.1406122-1-ynorov@nvidia.com/ [2] Signed-off-by: Yury Norov Signed-off-by: Andrew Morton Cc: Alice Ryhl Cc: Andrew Ballance Cc: Christopher Li Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) --- include/linux/compiler.h | 5 +++++ tools/include/linux/compiler.h | 5 +++++ 2 files changed, 10 insertions(+) diff --git a/include/linux/compiler.h b/include/linux/compiler.h index cb2f6050bdf7dc..ef9036fa5413d6 100644 --- a/include/linux/compiler.h +++ b/include/linux/compiler.h @@ -275,6 +275,11 @@ static inline void *offset_to_ptr(const int *off) #define __ADDRESSABLE(sym) \ ___ADDRESSABLE(sym, __section(".discard.addressable")) +/* Enforce static storage duration. */ +#define ASSERT_STATIC_STORAGE(name) \ + static typeof(name) * const __always_unused \ + name##_storage_check = &(name) + /* * This returns a constant expression while determining if an argument is * a constant expression, most importantly without evaluating the argument. diff --git a/tools/include/linux/compiler.h b/tools/include/linux/compiler.h index f2f54b0381680b..03ecf90866436b 100644 --- a/tools/include/linux/compiler.h +++ b/tools/include/linux/compiler.h @@ -73,6 +73,11 @@ # define __same_type(a, b) __builtin_types_compatible_p(typeof(a), typeof(b)) #endif +/* Enforce static storage duration. */ +#define ASSERT_STATIC_STORAGE(name) \ + static typeof(name) * const __always_unused \ + name##_storage_check = &(name) + /* * This returns a constant expression while determining if an argument is * a constant expression, most importantly without evaluating the argument. From fc6100a1ad999b88bd856b7bf5a2d54deeb3051a Mon Sep 17 00:00:00 2001 From: Yury Norov Date: Fri, 11 Sep 2026 18:14:42 -0400 Subject: [PATCH 0925/1352] idr: assert static storage for DEFINE_IDA() DEFINE_IDA() uses IDA_INIT(), which initializes the embedded XArray lock with a static spinlock initializer. For an automatic local IDA, lockdep cannot use the lock address as a persistent class key and reports "INFO: trying to register non-static key" before disabling itself. Apply ASSERT_STATIC_STORAGE() to DEFINE_IDA() so that this misuse is rejected at compile time. For example: void example(void) { DEFINE_IDA(ida); ida_destroy(&ida); } GCC reports: error: initializer element is not constant name##_storage_check = &(name) ^ note: in expansion of macro 'ASSERT_STATIC_STORAGE' ASSERT_STATIC_STORAGE(name) note: in expansion of macro 'DEFINE_IDA' DEFINE_IDA(ida); File-scope definitions and static DEFINE_IDA() within a function remain valid. Automatic local IDAs must instead be initialized with ida_init(). Direct uses of IDA_INIT() are not covered by this declaration check. Convert the five automatic local IDAs in the userspace radix-tree tests to ida_init() so they satisfy the new requirement. Validated file-scope and static local definitions with a kernel object build, and confirmed that an automatic local definition fails to compile. Link: https://lore.kernel.org/20260911221444.1523311-3-ynorov@nvidia.com Signed-off-by: Yury Norov Signed-off-by: Andrew Morton Cc: Alice Ryhl Cc: Andrew Ballance Cc: Christopher Li Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) --- include/linux/idr.h | 5 ++++- tools/testing/radix-tree/idr-test.c | 20 +++++++++++++++----- 2 files changed, 19 insertions(+), 6 deletions(-) diff --git a/include/linux/idr.h b/include/linux/idr.h index 789e23e6744410..e2a4b6298511c7 100644 --- a/include/linux/idr.h +++ b/include/linux/idr.h @@ -16,6 +16,7 @@ #include #include #include +#include struct idr { struct radix_tree_root idr_rt; @@ -269,7 +270,9 @@ struct ida { #define IDA_INIT(name) { \ .xa = XARRAY_INIT(name, IDA_INIT_FLAGS) \ } -#define DEFINE_IDA(name) struct ida name = IDA_INIT(name) +#define DEFINE_IDA(name) \ + struct ida name = IDA_INIT(name); \ + ASSERT_STATIC_STORAGE(name) int ida_alloc_range(struct ida *, unsigned int min, unsigned int max, gfp_t); void ida_free(struct ida *, unsigned int id); diff --git a/tools/testing/radix-tree/idr-test.c b/tools/testing/radix-tree/idr-test.c index 945144e9850724..6fcba5b5870b4e 100644 --- a/tools/testing/radix-tree/idr-test.c +++ b/tools/testing/radix-tree/idr-test.c @@ -460,9 +460,11 @@ void ida_dump(struct ida *); */ void ida_check_nomem(void) { - DEFINE_IDA(ida); + struct ida ida; int id; + ida_init(&ida); + id = ida_alloc_min(&ida, 256, GFP_NOWAIT); IDA_BUG_ON(&ida, id != -ENOMEM); id = ida_alloc_min(&ida, 1UL << 30, GFP_NOWAIT); @@ -475,9 +477,11 @@ void ida_check_nomem(void) */ void ida_check_conv_user(void) { - DEFINE_IDA(ida); + struct ida ida; unsigned long i; + ida_init(&ida); + for (i = 0; i < 1000000; i++) { int id = ida_alloc(&ida, GFP_NOWAIT); if (id == -ENOMEM) { @@ -496,11 +500,13 @@ void ida_check_conv_user(void) void ida_check_random(void) { - DEFINE_IDA(ida); + struct ida ida; DECLARE_BITMAP(bitmap, 2048); unsigned int i; time_t s = time(NULL); + ida_init(&ida); + repeat: memset(bitmap, 0, sizeof(bitmap)); for (i = 0; i < 100000; i++) { @@ -522,9 +528,11 @@ void ida_check_random(void) void ida_alloc_free_test(void) { - DEFINE_IDA(ida); + struct ida ida; unsigned long i; + ida_init(&ida); + for (i = 0; i < 10000; i++) assert(ida_alloc_max(&ida, 20000, GFP_KERNEL) == i); assert(ida_alloc_range(&ida, 5, 30, GFP_KERNEL) < 0); @@ -576,10 +584,12 @@ static void *ida_leak_fn(void *arg) void ida_thread_tests(void) { - DEFINE_IDA(ida); + struct ida ida; pthread_t threads[20]; int i; + ida_init(&ida); + for (i = 0; i < ARRAY_SIZE(threads); i++) if (pthread_create(&threads[i], NULL, ida_random_fn, NULL)) { perror("creating ida thread"); From e0c53df50574e24756435bbcb0d68348a0afbd0a Mon Sep 17 00:00:00 2001 From: Yury Norov Date: Fri, 11 Sep 2026 18:14:43 -0400 Subject: [PATCH 0926/1352] maple_tree: assert static storage for DEFINE_MTREE() DEFINE_MTREE() uses MTREE_INIT(), which initializes the tree's embedded lock with a static spinlock initializer. If the tree is an automatic local object, lockdep rejects its address as a non-static class key and disables locking validation on the first lock acquisition. Apply ASSERT_STATIC_STORAGE() to DEFINE_MTREE() to catch automatic local definitions at compile time. For example: void example(void) { DEFINE_MTREE(mt); mtree_destroy(&mt); } GCC reports: error: initializer element is not constant name##_storage_check = &(name) ^ note: in expansion of macro 'ASSERT_STATIC_STORAGE' ASSERT_STATIC_STORAGE(name) note: in expansion of macro 'DEFINE_MTREE' DEFINE_MTREE(mt); File-scope definitions and static local trees remain valid. Automatic local trees must instead use mt_init() or mt_init_flags(). Direct uses of MTREE_INIT() and MTREE_INIT_EXT() are unchanged. Replace the local DEFINE_MTREE() in the interval-tree span test with a plain declaration; the test already initializes the tree with mt_init_flags() before use. Convert the three local Maple Trees in the userspace radix-tree tests to mt_init(). Validated file-scope and static local definitions with a kernel object build, and confirmed that an automatic local definition fails to compile. The Maple Tree test and region allocation benchmark objects also build with lockdep enabled. Link: https://lore.kernel.org/20260911221444.1523311-4-ynorov@nvidia.com Signed-off-by: Yury Norov Signed-off-by: Andrew Morton Cc: Alice Ryhl Cc: Andrew Ballance Cc: Christopher Li Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) --- include/linux/maple_tree.h | 4 +++- lib/interval_tree_test.c | 2 +- tools/testing/radix-tree/maple.c | 12 +++++++++--- 3 files changed, 13 insertions(+), 5 deletions(-) diff --git a/include/linux/maple_tree.h b/include/linux/maple_tree.h index e595ae5cd0eed4..e30e57f5f3df9f 100644 --- a/include/linux/maple_tree.h +++ b/include/linux/maple_tree.h @@ -9,6 +9,7 @@ */ #include +#include #include #include @@ -297,7 +298,8 @@ struct maple_tree { #endif #define DEFINE_MTREE(name) \ - struct maple_tree name = MTREE_INIT(name, 0) + struct maple_tree name = MTREE_INIT(name, 0); \ + ASSERT_STATIC_STORAGE(name) #define mtree_lock(mt) spin_lock((&(mt)->ma_lock)) #define mtree_lock_nested(mas, subclass) \ diff --git a/lib/interval_tree_test.c b/lib/interval_tree_test.c index b0b07270ce7c3c..06f77fb3179fc8 100644 --- a/lib/interval_tree_test.c +++ b/lib/interval_tree_test.c @@ -244,7 +244,7 @@ static int span_iteration_check(void) unsigned long start, last; struct interval_tree_span_iter span, mas_span; - DEFINE_MTREE(tree); + struct maple_tree tree; MA_STATE(mas, &tree, 0, 0); diff --git a/tools/testing/radix-tree/maple.c b/tools/testing/radix-tree/maple.c index d967e76a3c0650..bfe4b8626c429d 100644 --- a/tools/testing/radix-tree/maple.c +++ b/tools/testing/radix-tree/maple.c @@ -36022,10 +36022,12 @@ static noinline void __init check_erase_rebalance(struct maple_tree *mt) static noinline void __init check_mtree_dup(struct maple_tree *mt) { - DEFINE_MTREE(new); + struct maple_tree new; int i, j, ret, count = 0; unsigned int rand_seed = 17, rand; + mt_init(&new); + /* store a value at [0, 0] */ mt_init_flags(mt, 0); mtree_store_range(mt, 0, 0, xa_mk_value(0), GFP_KERNEL); @@ -36319,7 +36321,9 @@ static inline int check_vma_modification(struct maple_tree *mt) void farmer_tests(void) { struct maple_node *node; - DEFINE_MTREE(tree); + struct maple_tree tree; + + mt_init(&tree); mt_dump(&tree, mt_dump_dec); @@ -36432,9 +36436,11 @@ static unsigned long get_last_index(struct ma_state *mas) static void test_spanning_store_regression(void) { unsigned long from = 0, to = 0; - DEFINE_MTREE(tree); + struct maple_tree tree; MA_STATE(mas, &tree, 0, 0); + mt_init(&tree); + /* * Build a 3-level tree. We require a parent node below the root node * and 2 leaf nodes under it, so we can span the entirety of the right From 158db543c2ed25c4f85ded5022cd528b074b4888 Mon Sep 17 00:00:00 2001 From: Yeoreum Yun Date: Fri, 11 Sep 2026 22:06:11 +0100 Subject: [PATCH 0927/1352] kselftest: mm: remove exclusion of building soft-dirty test in arm64 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Since commit 0389c305ef56 (“selftests/mm: skip soft-dirty tests when CONFIG_MEM_SOFT_DIRTY is disabled”), the soft-dirty test is skipped when soft-dirty is not supported. There is therefore no reason to exclude the test from being built on arm64. Remove the arm64-specific exclusion. Link: https://lore.kernel.org/20260911210611.4001419-1-yeoreum.yun@arm.com Signed-off-by: Yeoreum Yun Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand (Arm) Acked-by: David Hildenbrand (Arm) Acked-by: Lorenzo Stoakes (ARM) Tested-by: Zenghui Yu (Huawei) Cc: Alice Ryhl Cc: Andrew Ballance Cc: Christopher Li Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Yury Norov (NVIDIA) --- tools/testing/selftests/mm/Makefile | 3 --- tools/testing/selftests/mm/run_vmtests.sh | 5 +---- 2 files changed, 1 insertion(+), 7 deletions(-) diff --git a/tools/testing/selftests/mm/Makefile b/tools/testing/selftests/mm/Makefile index 2d5366196e3091..d3e9bd67904aa0 100644 --- a/tools/testing/selftests/mm/Makefile +++ b/tools/testing/selftests/mm/Makefile @@ -104,10 +104,7 @@ TEST_GEN_FILES += guard-regions TEST_GEN_FILES += merge TEST_GEN_FILES += rmap TEST_GEN_FILES += folio_split_race_test - -ifneq ($(ARCH),arm64) TEST_GEN_FILES += soft-dirty -endif ifeq ($(ARCH),x86_64) CAN_BUILD_I386 := $(shell ./../x86/check_cc.sh "$(CC)" ../x86/trivial_32bit_program.c -m32) diff --git a/tools/testing/selftests/mm/run_vmtests.sh b/tools/testing/selftests/mm/run_vmtests.sh index 2e5e7975ff4d49..a1db516557023f 100755 --- a/tools/testing/selftests/mm/run_vmtests.sh +++ b/tools/testing/selftests/mm/run_vmtests.sh @@ -408,10 +408,7 @@ then CATEGORY="pkey" run_test ./protection_keys_64 fi -if [ -x ./soft-dirty ] -then - CATEGORY="soft_dirty" run_test ./soft-dirty -fi +CATEGORY="soft_dirty" run_test ./soft-dirty CATEGORY="pagemap" run_test ./pagemap_ioctl From 13ca27a45abcb7dd2255cedf75ead56eecb50c44 Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Sun, 13 Sep 2026 14:30:31 +0800 Subject: [PATCH 0928/1352] mm: zswap: return -ENOENT when the swap device is gone zswap_writeback_entry() returns -EEXIST when get_swap_device() finds no device. -EEXIST is the shrinker's "page already in swap cache" signal, which makes zswap_shrinker_scan() stop shrinking entirely. A NULL get_swap_device() instead means the device is being swapped off, so the entry is simply stale. Return -ENOENT so the shrinker skips the stale entry and keeps scanning. It affects all swap devices. Link: https://lore.kernel.org/20260913063031.1689420-1-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Acked-by: Nhat Pham Cc: Chengming Zhou Cc: Chris Li Cc: Johannes Weiner Cc: Kairui Song --- mm/zswap.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/zswap.c b/mm/zswap.c index ff7c6742af50f5..584dd306376943 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -1016,7 +1016,7 @@ static int zswap_writeback_entry(struct zswap_entry *entry, /* try to allocate swap cache folio */ si = get_swap_device(swpentry); if (IS_ERR_OR_NULL(si)) - return -EEXIST; + return -ENOENT; mpol = get_task_policy(current); folio = swap_cache_alloc_folio(swpentry, GFP_KERNEL, BIT(0), NULL, mpol, From 88abde9974230245ddd215499ec719289605de09 Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:27:57 -0400 Subject: [PATCH 0929/1352] mm/zsmalloc: replace PG_private with pointer comparison Patch series "Remove PG_private by using page/folio->private checks instead", v5. This patchset removes PG_private to make space for upcoming PG_folio for identifying pages from a folio (more details in Note below). Instead of checking PG_private, all code is changed to check page/folio->private != NULL instead. Overview === Most code uses folio_attach/detach/change_private() functions, so folio refcount is increased and decreased when folio->private is set and reset, respectively. There is no need to change them. Changes are needed for exceptional users: 1. zsmalloc uses PG_private to indicate first component zpdesc page and page->private is used to store zspage in zpdesc. To remove PG_private, is_first_zpdesc() is replaced by pointer comparison. 2. kernel/events/ring_buffer.c stores page order in page->private. Replacing PG_private with page->private != NULL works. 3. drivers/xen/grant-table.c stores xen_page_foreign in page->private, where on 32-bit, a pointer to xen_page_foreign is stored; on 64-bit, page->private is used as xen_page_foreign. PG_private check is replaced by page->private != NULL on 32-bit for xen_page_foreign deallocation. On 64-bit, page->private is cleared unconditionally since {domid=0, gref=0} (xen_page_foreign can be 0) is valid. 4. fs/crypto/crypto.c stores a folio pointer in page->private, PG_private checks are replaced by page->private != NULL. 5. fs/erofs has two different uses: 5a. folio->private is used to form a reversed list of the outputs of readahead_folio(). readahead_folio_last() is added to output folios in reversed order, so that ->private is no longer needed. 5b. folio->private is used as an in-flight I/O counter. Convert the code to use folio_attach/detach/get_private() and add bias==1 to the counter to avoid folio->private being zero. 6. fs/nfs/write.c: folio refcount maintenance is in a bigger scope than folio->private. So folio_attach/detach/get_private() is not used. Nothing to change. 7. fs/f2fs uses attach_page_private() to first reset folio->private then immediately sets PAGE_PRIVATE_NOT_POINTER bit on it. Change it to use attach_page_private() to set PAGE_PRIVATE_NOT_POINTER bit directly to avoid folio->private == NULL gap inside set_page_private_##name(). 8. hugetlb uses folio_change_private(folio, NULL) without folio refcount maintenance. Change it to folio->private = NULL. After the above changes, PG_private ops are converted to page/folio->private ops. folio_has_attached_private() is added to check filesystem-only private data by excluding swapcache and hugetlb folios, because swapcache folios overlap swp_entry_t swap with ->private and hugetlb sets its own flags in ->private. Note === 1. KPF_PRIVATE is removed after PG_private is removed. 2. Documentation/mm/hugetlbfs_reserv.rst is outdated, so I did not remove PG_private related text. It should be rewritten. 3. PG_folio is planned to be set on every page from a folio in page_rmappable_folio(), so folios with any order (currently PG_large_rmappable is used to identify >0 order folios, but not order-0 folios) can be identified. Then vm_insert_*() can correctly reject all folios and rmap code will only see folios. Eventually, page_folio() will return NULL for non-folio pages by checking PG_folio, but before that all existing users that treat compound pages as folios will need to be converted. This patch (of 17): zsmalloc uses PG_private to indicate first zpdesc in a zspage chain. It is equivalent to check zpdesc == zspage->first_zpdesc. Replace is_first_zpdesc() with zpdesc == zspage->first_zpdesc in obj_allocated(). For get_first_zpdesc(), first_zpdesc is from zspage->first_zpdesc, so replace is_first_zpdesc() with first_zpdesc->zspage == zspage, the second requirement of a zspage chain, where all zpdescs point to the same zspage. is_first_zpdesc(), is only used in VM_BUG_ON_PAGE(), so performance impact should be negligible. While at it, change VM_BUG_ON() to VM_WARN_ON_ONCE_PAGE(). It prepares for a future commit that remove PG_private. No functional change intended. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-0-bb68b6a21869@nvidia.com Link: https://lore.kernel.org/20260920-remove-pg_private-v5-1-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Acked-by: Johannes Weiner Reviewed-by: Sergey Senozhatsky Acked-by: David Hildenbrand (Arm) Reviewed-by: Lance Yang Assisted-by: LLM Cc: Minchan Kim --- mm/zpdesc.h | 2 +- mm/zsmalloc.c | 24 ++++++------------------ 2 files changed, 7 insertions(+), 19 deletions(-) diff --git a/mm/zpdesc.h b/mm/zpdesc.h index b8258dc78548d1..4fd81c2e80769f 100644 --- a/mm/zpdesc.h +++ b/mm/zpdesc.h @@ -26,8 +26,8 @@ * with memcg_data. * * Page flags used: - * * PG_private identifies the first component page. * * PG_locked is used by page migration code. + * The first component page has zpdesc->zspage->first_zpdesc == zpdesc */ struct zpdesc { unsigned long flags; diff --git a/mm/zsmalloc.c b/mm/zsmalloc.c index 11be37c4317189..7ef80e0da62676 100644 --- a/mm/zsmalloc.c +++ b/mm/zsmalloc.c @@ -290,11 +290,6 @@ struct zs_pool { atomic_t compaction_in_progress; }; -static inline void zpdesc_set_first(struct zpdesc *zpdesc) -{ - SetPagePrivate(zpdesc_page(zpdesc)); -} - static inline void zpdesc_inc_zone_page_state(struct zpdesc *zpdesc) { inc_zone_page_state(zpdesc_page(zpdesc), NR_ZSPAGES); @@ -476,11 +471,6 @@ static void record_obj(unsigned long handle, unsigned long obj) WRITE_ONCE(*(unsigned long *)handle, obj); } -static inline bool __maybe_unused is_first_zpdesc(struct zpdesc *zpdesc) -{ - return PagePrivate(zpdesc_page(zpdesc)); -} - /* Protected by class->lock */ static inline int get_zspage_inuse(struct zspage *zspage) { @@ -496,7 +486,8 @@ static struct zpdesc *get_first_zpdesc(struct zspage *zspage) { struct zpdesc *first_zpdesc = zspage->first_zpdesc; - VM_BUG_ON_PAGE(!is_first_zpdesc(first_zpdesc), zpdesc_page(first_zpdesc)); + /* the first zpdesc must point back to this zspage */ + VM_WARN_ON_ONCE_PAGE(first_zpdesc->zspage != zspage, zpdesc_page(first_zpdesc)); return first_zpdesc; } @@ -838,7 +829,8 @@ static inline bool obj_allocated(struct zpdesc *zpdesc, void *obj, struct zspage *zspage = get_zspage(zpdesc); if (unlikely(ZsHugePage(zspage))) { - VM_BUG_ON_PAGE(!is_first_zpdesc(zpdesc), zpdesc_page(zpdesc)); + /* only first zpdesc holds the handle */ + VM_WARN_ON_ONCE_PAGE(zspage->first_zpdesc != zpdesc, zpdesc_page(zpdesc)); handle = zpdesc->handle; } else handle = *(unsigned long *)obj; @@ -853,9 +845,6 @@ static inline bool obj_allocated(struct zpdesc *zpdesc, void *obj, static void reset_zpdesc(struct zpdesc *zpdesc) { - struct page *page = zpdesc_page(zpdesc); - - ClearPagePrivate(page); zpdesc->zspage = NULL; zpdesc->next = NULL; /* PageZsmalloc is sticky until the page is freed to the buddy. */ @@ -1006,8 +995,8 @@ static void create_page_chain(struct size_class *class, struct zspage *zspage, * 1. all pages are linked together using zpdesc->next * 2. each sub-page point to zspage using zpdesc->zspage * - * we set PG_private to identify the first zpdesc (i.e. no other zpdesc - * has this flag set). + * The first zpdesc has its zspage->first_zpdesc set to itself, no + * other zpdesc has this set. */ for (i = 0; i < nr_zpdescs; i++) { zpdesc = zpdescs[i]; @@ -1015,7 +1004,6 @@ static void create_page_chain(struct size_class *class, struct zspage *zspage, zpdesc->next = NULL; if (i == 0) { zspage->first_zpdesc = zpdesc; - zpdesc_set_first(zpdesc); if (unlikely(class->objs_per_zspage == 1 && class->pages_per_zspage == 1)) SetZsHugePage(zspage); From d9a14ef047b470c897b0387d810e11d3a5bd5bdd Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:27:58 -0400 Subject: [PATCH 0930/1352] perf/ring_buffer: stop using PG_private as AUX page high-order marker A high-order AUX page sets PG_private on its first page and stores the order in first_page->private. Stop using PG_private and check first_page->private for AUX page order only in the ring buffer and its users. This is fine because page->private is 0 for order-0 AUX pages, matching the page order. It prepares for a future commit that remove PG_private. No functional change intended. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-2-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Acked-by: Usama Arif Acked-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Peter Zijlstra Cc: Ingo Molnar Cc: Arnaldo Carvalho de Melo Cc: Namhyung Kim Cc: Thomas Gleixner Cc: Borislav Petkov Cc: Dave Hansen Cc: Mark Rutland Cc: Alexander Shishkin Cc: Jiri Olsa Cc: Ian Rogers Cc: Adrian Hunter Cc: James Clark Cc: "H. Peter Anvin" --- arch/x86/events/intel/bts.c | 3 --- arch/x86/events/intel/pt.c | 6 ++---- kernel/events/ring_buffer.c | 7 +++---- 3 files changed, 5 insertions(+), 11 deletions(-) diff --git a/arch/x86/events/intel/bts.c b/arch/x86/events/intel/bts.c index cbac54cb3a9ec5..5849392cf26d5b 100644 --- a/arch/x86/events/intel/bts.c +++ b/arch/x86/events/intel/bts.c @@ -66,9 +66,6 @@ static struct pmu bts_pmu; static int buf_nr_pages(struct page *page) { - if (!PagePrivate(page)) - return 1; - return 1 << page_private(page); } diff --git a/arch/x86/events/intel/pt.c b/arch/x86/events/intel/pt.c index 5754cd40556281..49349afee6119f 100644 --- a/arch/x86/events/intel/pt.c +++ b/arch/x86/events/intel/pt.c @@ -781,8 +781,7 @@ static int topa_insert_pages(struct pt_buffer *buf, int cpu, gfp_t gfp) struct page *p; p = virt_to_page(buf->data_pages[buf->nr_pages]); - if (PagePrivate(p)) - order = page_private(p); + order = page_private(p); if (topa_table_full(topa)) { topa = topa_alloc(cpu, gfp); @@ -1296,8 +1295,7 @@ static int pt_buffer_try_single(struct pt_buffer *buf, int nr_pages) if (!intel_pt_validate_hw_cap(PT_CAP_single_range_output)) goto out; - if (PagePrivate(p)) - order = page_private(p); + order = page_private(p); if (1 << order != nr_pages) goto out; diff --git a/kernel/events/ring_buffer.c b/kernel/events/ring_buffer.c index 1b1ffe0533e58a..9d3d324f512720 100644 --- a/kernel/events/ring_buffer.c +++ b/kernel/events/ring_buffer.c @@ -635,11 +635,10 @@ static struct page *rb_alloc_aux_page(int node, int order) /* * Communicate the allocation size to the driver: * if we managed to secure a high-order allocation, - * set its first page's private to this order; - * !PagePrivate(page) means it's just a normal page. + * set its first page's private to this order, otherwise page's + * private remains zero. */ split_page(page, order); - SetPagePrivate(page); set_page_private(page, order); } @@ -650,7 +649,7 @@ static void rb_free_aux_page(struct perf_buffer *rb, int idx) { struct page *page = virt_to_page(rb->aux_pages[idx]); - ClearPagePrivate(page); + set_page_private(page, 0); __free_page(page); } From 4ba13dab3528b7fd3528f21c6e58393e91968585 Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:27:59 -0400 Subject: [PATCH 0931/1352] xen/grant-table: stop setting PG_private on pages for grant mapping gnttab_alloc_pages() stores xen_page_foreign in page->private. On 32-bit, a pointer to an allocated xen_page_foreign is stored; on 64-bit, xen_page_forCcgn is stored inline. Checking page->private != NULL is enough to tell whether a xen_page_foreign needs to be freed on 32-bit and page->private is zeroed unconditionally on 64-bit. It prepares for a future commit that remove PG_private. No functional change intended. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-3-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Juergen Gross Cc: Stefano Stabellini Cc: Oleksandr Tyshchenko --- drivers/xen/grant-table.c | 11 +++++------ 1 file changed, 5 insertions(+), 6 deletions(-) diff --git a/drivers/xen/grant-table.c b/drivers/xen/grant-table.c index 076c1b0ab87fd1..f75b5cc71cfc5a 100644 --- a/drivers/xen/grant-table.c +++ b/drivers/xen/grant-table.c @@ -863,10 +863,10 @@ EXPORT_SYMBOL_GPL(gnttab_free_auto_xlat_frames); int gnttab_pages_set_private(int nr_pages, struct page **pages) { +#if BITS_PER_LONG < 64 int i; for (i = 0; i < nr_pages; i++) { -#if BITS_PER_LONG < 64 struct xen_page_foreign *foreign; foreign = kzalloc_obj(*foreign); @@ -874,9 +874,9 @@ int gnttab_pages_set_private(int nr_pages, struct page **pages) return -ENOMEM; set_page_private(pages[i], (unsigned long)foreign); -#endif - SetPagePrivate(pages[i]); } +#endif + /* Data is stored in page->private on 64-bit */ return 0; } @@ -1031,12 +1031,11 @@ void gnttab_pages_clear_private(int nr_pages, struct page **pages) int i; for (i = 0; i < nr_pages; i++) { - if (PagePrivate(pages[i])) { #if BITS_PER_LONG < 64 + if (page_private(pages[i])) kfree((void *)page_private(pages[i])); #endif - ClearPagePrivate(pages[i]); - } + set_page_private(pages[i], 0); } } EXPORT_SYMBOL_GPL(gnttab_pages_clear_private); From 9bba142cd8cf7bea69b1212cb8b3ef516d40fccf Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:28:00 -0400 Subject: [PATCH 0932/1352] fscrypt: stop setting PG_private on bounce page The pointer to a plain text folio is stored in bound_page->private and cannot be NULL until the bounce_page is freed, making PG_private redundant. It prepares for a future commit that remove PG_private. No functional change intended. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-4-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Acked-by: Eric Biggers Acked-by: Usama Arif Acked-by: David Hildenbrand (Arm) Reviewed-by: Lance Yang Assisted-by: LLM Cc: "Theodore Y. Ts'o" Cc: Jaegeuk Kim --- fs/crypto/crypto.c | 2 -- 1 file changed, 2 deletions(-) diff --git a/fs/crypto/crypto.c b/fs/crypto/crypto.c index 5286a124b0d982..aced5c50a46011 100644 --- a/fs/crypto/crypto.c +++ b/fs/crypto/crypto.c @@ -65,7 +65,6 @@ void fscrypt_free_bounce_page(struct page *bounce_page) if (!bounce_page) return; set_page_private(bounce_page, (unsigned long)NULL); - ClearPagePrivate(bounce_page); mempool_free(bounce_page, fscrypt_bounce_page_pool); } EXPORT_SYMBOL(fscrypt_free_bounce_page); @@ -210,7 +209,6 @@ struct page *fscrypt_encrypt_pagecache_blocks(struct folio *folio, return ERR_PTR(err); } } - SetPagePrivate(ciphertext_page); set_page_private(ciphertext_page, (unsigned long)folio); return ciphertext_page; } From 7c22d8f88514d1421db5df4e45aaea3c5b337bb5 Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:28:01 -0400 Subject: [PATCH 0933/1352] mm/hugetlb: use direct assignment instead of folio_change_private() folio_change_private() should be used along with folio_attach_private() and folio_detach_private(), where adding and remove ->private content requires folio refcount change. add_hugetlb_folio() simply sets folio->private to NULL without refcount manipulation. Change it to direct assignment to avoid semantic confusion. It prepares for a future commit that remove PG_private. No functional change intended. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-5-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Acked-by: Usama Arif Reviewed-by: Gregory Price (Meta) Reviewed-by: Muchun Song Acked-by: David Hildenbrand (Arm) Reviewed-by: Lance Yang Assisted-by: LLM Cc: Oscar Salvador --- mm/hugetlb.c | 7 ++----- 1 file changed, 2 insertions(+), 5 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 2003439ea13c6f..7b27c3c5c3e58e 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -1446,11 +1446,8 @@ void add_hugetlb_folio(struct hstate *h, struct folio *folio, } __folio_set_hugetlb(folio); - folio_change_private(folio, NULL); - /* - * We have to set hugetlb_vmemmap_optimized again as above - * folio_change_private(folio, NULL) cleared it. - */ + /* Clear all folio->private flags except hugetlb_vmemmap_optimized. */ + folio->private = NULL; folio_set_hugetlb_vmemmap_optimized(folio); arch_clear_hugetlb_flags(folio); From 2e57c4820f8997a766f280b74c4dbea87a8583a5 Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:28:02 -0400 Subject: [PATCH 0934/1352] f2fs: stop using PG_private f2fs sets its PAGE_PRIVATE_* flags in page->private and checking page->private != NULL is equivalent to checking PG_private. Change PagePrivate() to page_private(). Meanwhile, in set_page_private_##name(), page->private is first set to 0/NULL before an PAGE_PRIVATE_* flag is set, but it can cause confusion when PG_private is removed and page->private != NULL is used instead. Change it to initialize page->private to PAGE_PRIVATE_NOT_POINTER instead and retain the original semantics. It prepares for a future commit that removes PG_private. No functional change intended. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-6-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Acked-by: Usama Arif Acked-by: Chao Yu Reviewed-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Jaegeuk Kim --- fs/f2fs/f2fs.h | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/fs/f2fs/f2fs.h b/fs/f2fs/f2fs.h index 9940a6cecf1a2c..2f7ab5888b078d 100644 --- a/fs/f2fs/f2fs.h +++ b/fs/f2fs/f2fs.h @@ -2691,7 +2691,7 @@ static inline bool folio_test_f2fs_##name(const struct folio *folio) \ } \ static inline bool page_private_##name(struct page *page) \ { \ - return PagePrivate(page) && \ + return page_private(page) && \ test_bit(PAGE_PRIVATE_NOT_POINTER, &page_private(page)) && \ test_bit(PAGE_PRIVATE_##flagname, &page_private(page)); \ } @@ -2710,9 +2710,9 @@ static inline void folio_set_f2fs_##name(struct folio *folio) \ } \ static inline void set_page_private_##name(struct page *page) \ { \ - if (!PagePrivate(page)) \ - attach_page_private(page, (void *)0); \ - set_bit(PAGE_PRIVATE_NOT_POINTER, &page_private(page)); \ + if (!page_private(page)) \ + attach_page_private(page, \ + (void *)BIT(PAGE_PRIVATE_NOT_POINTER)); \ set_bit(PAGE_PRIVATE_##flagname, &page_private(page)); \ } From 659889b4ba1296c612dc85edbc901fdd8047bd37 Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:28:03 -0400 Subject: [PATCH 0935/1352] f2fs: convert the ->private flag helpers to folio-only page-based ->private flag helpers are used in the compression path, where large folios are not enabled. They can use folio versions with page_folio(). The two remaining users in data.c and segment.c can use fio->folio instead of fio->page (two are in a union). Drop page-based helpers after the conversion and rename PAGE_PRIVATE_{GET,SET,CLEAR}_FUNC() and the PAGE_PRIVATE_* flags to F2FS_FOLIO_PRIVATE_* to match. Convert the folio/page union from f2fs_io_info union to folio only, since no page user is left. The folio helpers do a plain read-modify-write where the page ones used set_bit()/clear_bit(). It is fine because the converted code either holds folio lock or, in f2fs_compress_write_end_io(), matches what the non-compressed code does in f2fs_write_end_bio(). Link: https://lore.kernel.org/20260920-remove-pg_private-v5-7-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Co-developed-by: David Hildenbrand (Arm) Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Suggested-by: Tal Zussman Reviewed-by: Tal Zussman Reviewed-by: Lance Yang Reviewed-by: Chao Yu Assisted-by: LLM Cc: Jaegeuk Kim --- fs/f2fs/compress.c | 35 ++++++++++------ fs/f2fs/data.c | 2 +- fs/f2fs/f2fs.h | 99 ++++++++++++++++++---------------------------- fs/f2fs/segment.c | 2 +- 4 files changed, 63 insertions(+), 75 deletions(-) diff --git a/fs/f2fs/compress.c b/fs/f2fs/compress.c index ce88092d9ce26d..09d9b8d0fdcce0 100644 --- a/fs/f2fs/compress.c +++ b/fs/f2fs/compress.c @@ -1064,13 +1064,15 @@ static void cancel_cluster_writeback(struct compress_ctx *cc, /* Cancel writeback and stay locked. */ for (i = 0; i < cc->cluster_size; i++) { + struct folio *folio = page_folio(cc->rpages[i]); + if (i < submitted) { inode_inc_dirty_pages(cc->inode); - lock_page(cc->rpages[i]); + folio_lock(folio); } - clear_page_private_gcing(cc->rpages[i]); - if (folio_test_writeback(page_folio(cc->rpages[i]))) - end_page_writeback(cc->rpages[i]); + folio_clear_f2fs_gcing(folio); + if (folio_test_writeback(folio)) + folio_end_writeback(folio); } } @@ -1078,11 +1080,15 @@ static void set_cluster_dirty(struct compress_ctx *cc) { int i; - for (i = 0; i < cc->cluster_size; i++) - if (cc->rpages[i]) { - set_page_dirty(cc->rpages[i]); - set_page_private_gcing(cc->rpages[i]); - } + for (i = 0; i < cc->cluster_size; i++) { + struct folio *folio; + + if (!cc->rpages[i]) + continue; + folio = page_folio(cc->rpages[i]); + folio_mark_dirty(folio); + folio_set_f2fs_gcing(folio); + } } static int prepare_compress_overwrite(struct compress_ctx *cc, @@ -1281,7 +1287,7 @@ static int f2fs_write_compressed_pages(struct compress_ctx *cc, .op = REQ_OP_WRITE, .op_flags = wbc_to_write_flags(wbc), .old_blkaddr = NEW_ADDR, - .page = NULL, + .folio = NULL, .encrypted_page = NULL, .compressed_page = NULL, .io_type = io_type, @@ -1370,7 +1376,7 @@ static int f2fs_write_compressed_pages(struct compress_ctx *cc, block_t blkaddr; blkaddr = f2fs_data_blkaddr(&dn); - fio.page = cc->rpages[i]; + fio.folio = page_folio(cc->rpages[i]); fio.old_blkaddr = blkaddr; /* cluster header */ @@ -1476,9 +1482,12 @@ void f2fs_compress_write_end_io(struct bio *bio, struct folio *folio) } for (i = 0; i < cic->nr_rpages; i++) { + struct folio *rfolio; + WARN_ON(!cic->rpages[i]); - clear_page_private_gcing(cic->rpages[i]); - end_page_writeback(cic->rpages[i]); + rfolio = page_folio(cic->rpages[i]); + folio_clear_f2fs_gcing(rfolio); + folio_end_writeback(rfolio); } page_array_free(sbi, cic->rpages, cic->nr_rpages); diff --git a/fs/f2fs/data.c b/fs/f2fs/data.c index 21f396ebe22ca9..ca8232a9095f89 100644 --- a/fs/f2fs/data.c +++ b/fs/f2fs/data.c @@ -2923,7 +2923,7 @@ bool f2fs_should_update_outplace(struct inode *inode, struct f2fs_io_info *fio) return true; if (fio) { - if (page_private_gcing(fio->page)) + if (folio_test_f2fs_gcing(fio->folio)) return true; if (unlikely(is_sbi_flag_set(sbi, SBI_CP_DISABLED) && f2fs_is_checkpointed_data(sbi, fio->old_blkaddr))) diff --git a/fs/f2fs/f2fs.h b/fs/f2fs/f2fs.h index 2f7ab5888b078d..85937de3d7016e 100644 --- a/fs/f2fs/f2fs.h +++ b/fs/f2fs/f2fs.h @@ -1357,10 +1357,7 @@ struct f2fs_io_info { blk_opf_t op_flags; /* req_flag_bits */ block_t new_blkaddr; /* new block address to be written */ block_t old_blkaddr; /* old block address before Cow */ - union { - struct page *page; /* page to be written */ - struct folio *folio; - }; + struct folio *folio; /* folio to be written */ struct page *encrypted_page; /* encrypted page */ struct page *compressed_page; /* compressed page */ struct list_head list; /* serialize IOs */ @@ -1613,27 +1610,27 @@ static inline void f2fs_set_bit(unsigned int nr, char *addr); static inline void f2fs_clear_bit(unsigned int nr, char *addr); /* - * Layout of f2fs page.private: + * Layout of f2fs folio->private: * * Layout A: lowest bit should be 1 * | bit0 = 1 | bit1 | bit2 | ... | bit MAX | private data .... | - * bit 0 PAGE_PRIVATE_NOT_POINTER - * bit 1 PAGE_PRIVATE_ONGOING_MIGRATION - * bit 2 PAGE_PRIVATE_INLINE_INODE - * bit 3 PAGE_PRIVATE_REF_RESOURCE - * bit 4 PAGE_PRIVATE_ATOMIC_WRITE + * bit 0 F2FS_FOLIO_PRIVATE_NOT_POINTER + * bit 1 F2FS_FOLIO_PRIVATE_ONGOING_MIGRATION + * bit 2 F2FS_FOLIO_PRIVATE_INLINE_INODE + * bit 3 F2FS_FOLIO_PRIVATE_REF_RESOURCE + * bit 4 F2FS_FOLIO_PRIVATE_ATOMIC_WRITE * bit 5- f2fs private data * * Layout B: lowest bit should be 0 - * page.private is a wrapped pointer. + * folio->private is a wrapped pointer. */ enum { - PAGE_PRIVATE_NOT_POINTER, /* private contains non-pointer data */ - PAGE_PRIVATE_ONGOING_MIGRATION, /* data page which is on-going migrating */ - PAGE_PRIVATE_INLINE_INODE, /* inode page contains inline data */ - PAGE_PRIVATE_REF_RESOURCE, /* dirty page has referenced resources */ - PAGE_PRIVATE_ATOMIC_WRITE, /* data page from atomic write path */ - PAGE_PRIVATE_MAX + F2FS_FOLIO_PRIVATE_NOT_POINTER, /* private contains non-pointer data */ + F2FS_FOLIO_PRIVATE_ONGOING_MIGRATION, /* data page which is on-going migrating */ + F2FS_FOLIO_PRIVATE_INLINE_INODE, /* inode page contains inline data */ + F2FS_FOLIO_PRIVATE_REF_RESOURCE, /* dirty page has referenced resources */ + F2FS_FOLIO_PRIVATE_ATOMIC_WRITE, /* data page from atomic write path */ + F2FS_FOLIO_PRIVATE_MAX }; /* For compression */ @@ -2681,86 +2678,68 @@ static inline int inc_valid_block_count(struct f2fs_sb_info *sbi, return -ENOSPC; } -#define PAGE_PRIVATE_GET_FUNC(name, flagname) \ +#define F2FS_FOLIO_PRIVATE_GET_FUNC(name, flagname) \ static inline bool folio_test_f2fs_##name(const struct folio *folio) \ { \ unsigned long priv = (unsigned long)folio->private; \ - unsigned long v = (1UL << PAGE_PRIVATE_NOT_POINTER) | \ - (1UL << PAGE_PRIVATE_##flagname); \ + unsigned long v = (1UL << F2FS_FOLIO_PRIVATE_NOT_POINTER) | \ + (1UL << F2FS_FOLIO_PRIVATE_##flagname); \ return (priv & v) == v; \ -} \ -static inline bool page_private_##name(struct page *page) \ -{ \ - return page_private(page) && \ - test_bit(PAGE_PRIVATE_NOT_POINTER, &page_private(page)) && \ - test_bit(PAGE_PRIVATE_##flagname, &page_private(page)); \ } -#define PAGE_PRIVATE_SET_FUNC(name, flagname) \ +#define F2FS_FOLIO_PRIVATE_SET_FUNC(name, flagname) \ static inline void folio_set_f2fs_##name(struct folio *folio) \ { \ - unsigned long v = (1UL << PAGE_PRIVATE_NOT_POINTER) | \ - (1UL << PAGE_PRIVATE_##flagname); \ + unsigned long v = (1UL << F2FS_FOLIO_PRIVATE_NOT_POINTER) | \ + (1UL << F2FS_FOLIO_PRIVATE_##flagname); \ if (!folio->private) \ folio_attach_private(folio, (void *)v); \ else { \ v |= (unsigned long)folio->private; \ folio->private = (void *)v; \ } \ -} \ -static inline void set_page_private_##name(struct page *page) \ -{ \ - if (!page_private(page)) \ - attach_page_private(page, \ - (void *)BIT(PAGE_PRIVATE_NOT_POINTER)); \ - set_bit(PAGE_PRIVATE_##flagname, &page_private(page)); \ } -#define PAGE_PRIVATE_CLEAR_FUNC(name, flagname) \ +#define F2FS_FOLIO_PRIVATE_CLEAR_FUNC(name, flagname) \ static inline void folio_clear_f2fs_##name(struct folio *folio) \ { \ unsigned long v = (unsigned long)folio->private; \ \ - v &= ~(1UL << PAGE_PRIVATE_##flagname); \ - if (v == (1UL << PAGE_PRIVATE_NOT_POINTER)) \ + v &= ~(1UL << F2FS_FOLIO_PRIVATE_##flagname); \ + if (v == (1UL << F2FS_FOLIO_PRIVATE_NOT_POINTER)) \ folio_detach_private(folio); \ else \ folio->private = (void *)v; \ -} \ -static inline void clear_page_private_##name(struct page *page) \ -{ \ - clear_bit(PAGE_PRIVATE_##flagname, &page_private(page)); \ - if (page_private(page) == BIT(PAGE_PRIVATE_NOT_POINTER)) \ - detach_page_private(page); \ } -PAGE_PRIVATE_GET_FUNC(nonpointer, NOT_POINTER); -PAGE_PRIVATE_GET_FUNC(inline, INLINE_INODE); -PAGE_PRIVATE_GET_FUNC(gcing, ONGOING_MIGRATION); -PAGE_PRIVATE_GET_FUNC(atomic, ATOMIC_WRITE); +F2FS_FOLIO_PRIVATE_GET_FUNC(nonpointer, NOT_POINTER); +F2FS_FOLIO_PRIVATE_GET_FUNC(inline, INLINE_INODE); +F2FS_FOLIO_PRIVATE_GET_FUNC(gcing, ONGOING_MIGRATION); +F2FS_FOLIO_PRIVATE_GET_FUNC(atomic, ATOMIC_WRITE); -PAGE_PRIVATE_SET_FUNC(reference, REF_RESOURCE); -PAGE_PRIVATE_SET_FUNC(inline, INLINE_INODE); -PAGE_PRIVATE_SET_FUNC(gcing, ONGOING_MIGRATION); -PAGE_PRIVATE_SET_FUNC(atomic, ATOMIC_WRITE); +F2FS_FOLIO_PRIVATE_SET_FUNC(reference, REF_RESOURCE); +F2FS_FOLIO_PRIVATE_SET_FUNC(inline, INLINE_INODE); +F2FS_FOLIO_PRIVATE_SET_FUNC(gcing, ONGOING_MIGRATION); +F2FS_FOLIO_PRIVATE_SET_FUNC(atomic, ATOMIC_WRITE); -PAGE_PRIVATE_CLEAR_FUNC(reference, REF_RESOURCE); -PAGE_PRIVATE_CLEAR_FUNC(inline, INLINE_INODE); -PAGE_PRIVATE_CLEAR_FUNC(gcing, ONGOING_MIGRATION); -PAGE_PRIVATE_CLEAR_FUNC(atomic, ATOMIC_WRITE); +F2FS_FOLIO_PRIVATE_CLEAR_FUNC(reference, REF_RESOURCE); +F2FS_FOLIO_PRIVATE_CLEAR_FUNC(inline, INLINE_INODE); +F2FS_FOLIO_PRIVATE_CLEAR_FUNC(gcing, ONGOING_MIGRATION); +F2FS_FOLIO_PRIVATE_CLEAR_FUNC(atomic, ATOMIC_WRITE); static inline unsigned long folio_get_f2fs_data(struct folio *folio) { unsigned long data = (unsigned long)folio->private; - if (!test_bit(PAGE_PRIVATE_NOT_POINTER, &data)) + if (!test_bit(F2FS_FOLIO_PRIVATE_NOT_POINTER, &data)) return 0; - return data >> PAGE_PRIVATE_MAX; + return data >> F2FS_FOLIO_PRIVATE_MAX; } static inline void folio_set_f2fs_data(struct folio *folio, unsigned long data) { - data = (1UL << PAGE_PRIVATE_NOT_POINTER) | (data << PAGE_PRIVATE_MAX); + data = (1UL << F2FS_FOLIO_PRIVATE_NOT_POINTER) | + (data << F2FS_FOLIO_PRIVATE_MAX); if (!folio_test_private(folio)) folio_attach_private(folio, (void *)data); diff --git a/fs/f2fs/segment.c b/fs/f2fs/segment.c index 63b712d3d599ed..8c156e1fd37d06 100644 --- a/fs/f2fs/segment.c +++ b/fs/f2fs/segment.c @@ -3803,7 +3803,7 @@ static int __get_segment_type_6(struct f2fs_io_info *fio) if (is_inode_flag_set(inode, FI_ALIGNED_WRITE)) return CURSEG_COLD_DATA_PINNED; - if (page_private_gcing(fio->page)) { + if (folio_test_f2fs_gcing(fio->folio)) { if (fio->sbi->am.atgc_enabled && (fio->io_type == FS_DATA_IO) && (fio->sbi->gc_mode != GC_URGENT_HIGH) && From f77ff3b60ae742722b04665a922945a1667d5412 Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:28:04 -0400 Subject: [PATCH 0936/1352] erofs: mm/pagemap: add readahead_folio_last() to avoid folio->private erofs needs to traverse readahead folios in reverse order to achieve maximum performance by 1. reading all folios from readahead_folio(); 2. storing the prior folio pointer in folio->private; 3. traverse from the last folio to the first one. Add readahead_folio_last() to achieve the same function without using folio->private. __readahead_advance() helper shares readahead_control adjustment code among __readahead_folio(), readahead_folio_last(), and __readahead_batch() by checking new private member, _forward, of readahead_control. It prepares for a future commit that replaces PG_private checks with !folio->private checks. After switching the checks, erofs's use of folio->private without bumping folio refcount can cause unexpected outcomes, e.g., in filemap_release_folio(), try_to_free_buffers() becomes reachable. No functional change intended. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-8-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Lance Yang Assisted-by: LLM Cc: Yue Hu Cc: Jeffle Xu Cc: Sandeep Dhavale Cc: Hongbo Li Cc: Chunhai Guo Cc: Gao Xiang Cc: Chao Yu Cc: "Matthew Wilcox (Oracle)" Cc: Jan Kara --- fs/erofs/zdata.c | 13 +++------ include/linux/pagemap.h | 60 +++++++++++++++++++++++++++++++++++------ 2 files changed, 55 insertions(+), 18 deletions(-) diff --git a/fs/erofs/zdata.c b/fs/erofs/zdata.c index 6b07e73ee2aaa0..e981e371d6c294 100644 --- a/fs/erofs/zdata.c +++ b/fs/erofs/zdata.c @@ -1886,21 +1886,14 @@ static void z_erofs_readahead(struct readahead_control *rac) struct inode *realinode = erofs_real_inode(sharedinode, &need_iput); Z_EROFS_DEFINE_FRONTEND(f, realinode, sharedinode, readahead_pos(rac)); unsigned int nrpages = readahead_count(rac); - struct folio *head = NULL, *folio; + struct folio *folio; int err; trace_erofs_readahead(realinode, readahead_index(rac), nrpages, false); z_erofs_pcluster_readmore(&f, rac, true); - while ((folio = readahead_folio(rac))) { - folio->private = head; - head = folio; - } - - /* traverse in reverse order for best metadata I/O performance */ - while (head) { - folio = head; - head = folio_get_private(folio); + /* traverse from last to first for best metadata I/O performance */ + while ((folio = readahead_folio_last(rac))) { err = z_erofs_scan_folio(&f, folio, true); if (err && err != -EINTR) erofs_err(realinode->i_sb, "readahead error at folio %lu @ nid %llu", diff --git a/include/linux/pagemap.h b/include/linux/pagemap.h index 939f3a5e973f6b..1e3462357aaa45 100644 --- a/include/linux/pagemap.h +++ b/include/linux/pagemap.h @@ -1415,6 +1415,7 @@ struct readahead_control { bool dropbehind; bool _workingset; unsigned long _pflags; + bool _forward; }; #define DEFINE_READAHEAD(ractl, f, r, m, i) \ @@ -1479,18 +1480,29 @@ void page_cache_async_readahead(struct address_space *mapping, page_cache_async_ra(&ractl, folio, req_count); } +/* + * Adjust readahead_control to ensure next folio comes from + * [_index, _index + _nr_pages) afterwards and reset _batch_count. + */ +static inline void __readahead_advance(struct readahead_control *rac) +{ + if (rac->_forward) + rac->_index += rac->_batch_count; + + rac->_nr_pages -= rac->_batch_count; + rac->_batch_count = 0; +} + static inline struct folio *__readahead_folio(struct readahead_control *ractl) { struct folio *folio; BUG_ON(ractl->_batch_count > ractl->_nr_pages); - ractl->_nr_pages -= ractl->_batch_count; - ractl->_index += ractl->_batch_count; + __readahead_advance(ractl); + ractl->_forward = true; - if (!ractl->_nr_pages) { - ractl->_batch_count = 0; + if (!ractl->_nr_pages) return NULL; - } folio = xa_load(&ractl->mapping->i_pages, ractl->_index); VM_BUG_ON_FOLIO(!folio_test_locked(folio), folio); @@ -1516,6 +1528,39 @@ static inline struct folio *readahead_folio(struct readahead_control *ractl) return folio; } +/** + * readahead_folio_last - Get the next folio to read, from the tail. + * @ractl: The current readahead request. + * + * Like readahead_folio(), but walks the range back-to-front. The folio is + * returned locked with its refcount dropped; the caller unlocks it once I/O + * completes. Compound folios are returned once, at their head index. + * + * Context: The folio is locked. + * Return: A pointer to the next folio, or %NULL when done. + */ +static inline struct folio *readahead_folio_last(struct readahead_control *ractl) +{ + struct folio *folio; + + /* Drop the previously returned batch from the remaining range. */ + __readahead_advance(ractl); + ractl->_forward = false; + + if (!ractl->_nr_pages) + return NULL; + + /* xa_load() follows sibling entries, so a tail index returns the head */ + folio = xa_load(&ractl->mapping->i_pages, + ractl->_index + ractl->_nr_pages - 1); + VM_WARN_ON_ONCE_FOLIO(!folio_test_locked(folio), folio); + + ractl->_batch_count = folio_nr_pages(folio); + + folio_put(folio); + return folio; +} + static inline unsigned int __readahead_batch(struct readahead_control *rac, struct page **array, unsigned int array_sz) { @@ -1524,9 +1569,8 @@ static inline unsigned int __readahead_batch(struct readahead_control *rac, struct folio *folio; BUG_ON(rac->_batch_count > rac->_nr_pages); - rac->_nr_pages -= rac->_batch_count; - rac->_index += rac->_batch_count; - rac->_batch_count = 0; + __readahead_advance(rac); + rac->_forward = true; xas_set(&xas, rac->_index); rcu_read_lock(); From f9dd8028b08a1940e878f46facb7e36324b4fcac Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:28:05 -0400 Subject: [PATCH 0937/1352] erofs: use folio_attach/detach_private() instead of direct assignment erofs_onlinefolio_init/split/end() use folio->private without setting PG_private or increasing folio refcount and it works. But after PG_private is replaced by checking folio->private in a future commit, it can break folio_expected_ref_count(), since the folio has private data without elevated refcount. Change them to use folio_attach/detach_private(). Folios during this process are locked as they are in the process of readahead, so no parallel migration/folio split can happen. Furthermore, because folio->private is used to store in-flight I/O counter and the counter reaches 0 when all I/O completes successfully without error or being dirty, ->private=0 causes folio_detach_private() to not drop the elevated folio refcount. Solve this issue by using bias=1 for the counter, so that ->private stays non NULL throughout every attach-to-detach process. Add a macro EROFS_ONLINEFOLIO_BIAS=1. While at it, fix the comment about ->private bit layout and add EROFS_ONLINEFOLIO_COUNT_MASK. It prepares for a future commit that removes PG_private. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-9-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Reviewed-by: Gao Xiang Reviewed-by: David Hildenbrand (Arm) Reviewed-by: Lance Yang Assisted-by: LLM Cc: Chao Yu Cc: Yue Hu Cc: Jeffle Xu Cc: Sandeep Dhavale Cc: Hongbo Li Cc: Chunhai Guo --- fs/erofs/data.c | 16 ++++++++++------ 1 file changed, 10 insertions(+), 6 deletions(-) diff --git a/fs/erofs/data.c b/fs/erofs/data.c index be63b89f086229..8b150aebf0af8c 100644 --- a/fs/erofs/data.c +++ b/fs/erofs/data.c @@ -239,19 +239,23 @@ int erofs_map_dev(struct super_block *sb, struct erofs_map_dev *map) /* * bit 30: I/O error occurred on this folio * bit 29: CPU has dirty data in D-cache (needs aliasing handling); - * bit 0 - 29: remaining parts to complete this folio + * bit 0 - 28: remaining parts to complete this folio, biased by 1 so that + * ->private stays non-NULL while the folio is attached */ #define EROFS_ONLINEFOLIO_EIO 30 #define EROFS_ONLINEFOLIO_DIRTY 29 +#define EROFS_ONLINEFOLIO_COUNT_MASK (BIT(EROFS_ONLINEFOLIO_DIRTY) - 1) +#define EROFS_ONLINEFOLIO_BIAS 1 void erofs_onlinefolio_init(struct folio *folio) { union { atomic_t o; void *v; - } u = { .o = ATOMIC_INIT(1) }; + } u = { .o = ATOMIC_INIT(1 + EROFS_ONLINEFOLIO_BIAS) }; - folio->private = u.v; /* valid only if file-backed folio is locked */ + /* valid only if file-backed folio is locked */ + folio_attach_private(folio, u.v); } void erofs_onlinefolio_split(struct folio *folio) @@ -265,14 +269,14 @@ void erofs_onlinefolio_end(struct folio *folio, int err, bool dirty) do { orig = atomic_read((atomic_t *)&folio->private); - DBG_BUGON(orig <= 0); + DBG_BUGON((orig & EROFS_ONLINEFOLIO_COUNT_MASK) <= EROFS_ONLINEFOLIO_BIAS); v = dirty << EROFS_ONLINEFOLIO_DIRTY; v |= (orig - 1) | (!!err << EROFS_ONLINEFOLIO_EIO); } while (atomic_cmpxchg((atomic_t *)&folio->private, orig, v) != orig); - if (v & (BIT(EROFS_ONLINEFOLIO_DIRTY) - 1)) + if ((v & EROFS_ONLINEFOLIO_COUNT_MASK) != EROFS_ONLINEFOLIO_BIAS) return; - folio->private = 0; + folio_detach_private(folio); if (v & BIT(EROFS_ONLINEFOLIO_DIRTY)) flush_dcache_folio(folio); folio_end_read(folio, !(v & BIT(EROFS_ONLINEFOLIO_EIO))); From 2a52a508c8f0c43fa537ec324f050ef55b1877e0 Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:28:06 -0400 Subject: [PATCH 0938/1352] mm/page-flags: check page/folio->private instead of PG_private After the changes of the prior commits, page/folio->private != NULL is now equivalent to checking PG_private. Stop checking PG_private on pages and folios and use page/folio->private instead, except swapcache and hugetlb folios, because the former uses a field (swp_entry_t swap) overlapping with ->private and the latter sets its flags in ->private. Exclude swapcache and hugetlb when the code is meant to check PG_private only. PG_swapcache and folio->swap.val cannot be set/clear as a whole, so excluding swapcache with folio_test_swapcache() is not reliable. Instead, use folio_test_swapbacked(), since PG_swapbacked is stable when a folio is added to/removed from swapcache. Add a helper, folio_has_attached_private(), for this check. folio_test_private() and PagePrivate() now read folio/page->private plainly instead of an atomic read of PG_private bit, so KCSAN complains about possible data races. Annotate them with data_race(). folio_expected_ref_count() can be called without the folio lock, so annotate folio->mapping with data_race() while at it. folio_set/clear_private() and Set/ClearPagePrivate() become no-ops. PG_private is no longer checked at page free time. They will be removed in an upcoming commit. Remove KPF_PRIVATE since PG_private is no longer used. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-10-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Steven Rostedt Cc: Masami Hiramatsu Cc: Lorenzo Stoakes Cc: "Matthew Wilcox (Oracle)" Cc: Jan Kara Cc: Johannes Weiner Cc: "Liam R. Howlett" Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Mathieu Desnoyers Cc: Baolin Wang Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Matthew Brost Cc: Joshua Hahn Cc: Rakie Kim Cc: Byungchul Park Cc: Gregory Price Cc: Ying Huang Cc: Alistair Popple Cc: Qi Zheng Cc: Shakeel Butt Cc: Kairui Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu --- fs/proc/page.c | 1 - include/linux/kernel-page-flags.h | 1 - include/linux/mm.h | 19 +++++++---- include/linux/page-flags.h | 57 +++++++++++++++++++++++++++---- include/trace/events/pagemap.h | 2 +- mm/huge_memory.c | 2 +- mm/migrate.c | 2 +- mm/page-writeback.c | 2 +- mm/vmscan.c | 2 +- tools/mm/page-types.c | 2 -- 10 files changed, 68 insertions(+), 22 deletions(-) diff --git a/fs/proc/page.c b/fs/proc/page.c index 260772b20bd992..f90e1030825e94 100644 --- a/fs/proc/page.c +++ b/fs/proc/page.c @@ -232,7 +232,6 @@ u64 stable_page_flags(const struct page *page) u |= kpf_copy_bit(k, KPF_RESERVED, PG_reserved); u |= kpf_copy_bit(k, KPF_OWNER_2, PG_owner_2); - u |= kpf_copy_bit(k, KPF_PRIVATE, PG_private); u |= kpf_copy_bit(k, KPF_PRIVATE_2, PG_private_2); u |= kpf_copy_bit(k, KPF_OWNER_PRIVATE, PG_owner_priv_1); u |= kpf_copy_bit(k, KPF_ARCH, PG_arch_1); diff --git a/include/linux/kernel-page-flags.h b/include/linux/kernel-page-flags.h index 196778a087c4df..fe5ab6e50bd70e 100644 --- a/include/linux/kernel-page-flags.h +++ b/include/linux/kernel-page-flags.h @@ -11,7 +11,6 @@ #define KPF_RESERVED 32 #define KPF_MLOCKED 33 #define KPF_OWNER_2 34 -#define KPF_PRIVATE 35 #define KPF_PRIVATE_2 36 #define KPF_OWNER_PRIVATE 37 #define KPF_ARCH 38 diff --git a/include/linux/mm.h b/include/linux/mm.h index 30a3365bca8271..6c8df7715eb261 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -3005,9 +3005,9 @@ static inline bool folio_maybe_mapped_shared(struct folio *folio) * @folio: the folio * * Calculate the expected folio refcount, taking references from the pagecache, - * swapcache, PG_private and page table mappings into account. Useful in - * combination with folio_ref_count() to detect unexpected references (e.g., - * GUP or other temporary references). + * swapcache, private data (folio->private != NULL) and page table mappings into + * account. Useful in combination with folio_ref_count() to detect unexpected + * references (e.g., GUP or other temporary references). * * Does currently not consider references from the LRU cache. If the folio * was isolated from the LRU (which is the case during migration or split), @@ -3045,10 +3045,15 @@ static inline int folio_expected_ref_count(const struct folio *folio) ref_count += folio_test_swapcache(folio) << order; if (!folio_test_anon(folio)) { - /* One reference per page from the pagecache. */ - ref_count += !!folio->mapping << order; - /* One reference from PG_private. */ - ref_count += folio_test_private(folio); + /* + * One reference per page from the pagecache. + * Use data_race() since folio might not be locked. + */ + ref_count += !!data_race(folio->mapping) << order; + /* + * One reference from filesystem private data. + */ + ref_count += folio_has_attached_private(folio); } /* One reference per page table mapping. */ diff --git a/include/linux/page-flags.h b/include/linux/page-flags.h index 7080a6a1a79e72..6d839f50bdcb7e 100644 --- a/include/linux/page-flags.h +++ b/include/linux/page-flags.h @@ -575,9 +575,31 @@ FOLIO_FLAG(swapbacked, FOLIO_HEAD_PAGE) /* * Private page markings that may be used by the filesystem that owns the page * for its own purposes. - * - PG_private and PG_private_2 cause release_folio() and co to be invoked + * - folio->private and PG_private_2 cause release_folio() and co to be invoked */ -PAGEFLAG(Private, private, PF_ANY) + +static __always_inline bool folio_test_private(const struct folio *folio) +{ + /* + * data_race() is added for readers without holding the folio lock. + * Only the NULL/non-NULL answer is used and both are valid while + * private is being attached or detached, so the race is benign. + */ + return data_race(folio->private); +} + +static __always_inline int PagePrivate(const struct page *page) +{ + /* See folio_test_private() for data_race() use */ + return !!data_race(page->private); +} + +/* no-ops during transition */ +static __always_inline void folio_set_private(struct folio *folio) { } +static __always_inline void folio_clear_private(struct folio *folio) { } +static __always_inline void SetPagePrivate(struct page *page) { } +static __always_inline void ClearPagePrivate(struct page *page) { } + FOLIO_FLAG(private_2, FOLIO_HEAD_PAGE) /* owner_2 can be set on tail pages for anon memory */ @@ -1169,7 +1191,7 @@ static __always_inline void __ClearPageAnonExclusive(struct page *page) */ #define PAGE_FLAGS_CHECK_AT_FREE \ (1UL << PG_lru | 1UL << PG_locked | \ - 1UL << PG_private | 1UL << PG_private_2 | \ + 1UL << PG_private_2 | \ 1UL << PG_writeback | 1UL << PG_reserved | \ 1UL << PG_active | \ 1UL << PG_unevictable | __PG_MLOCKED | LRU_GEN_MASK) @@ -1193,8 +1215,31 @@ static __always_inline void __ClearPageAnonExclusive(struct page *page) (0xffUL /* order */ | 1UL << PG_has_hwpoisoned | \ 1UL << PG_large_rmappable | 1UL << PG_partially_mapped) -#define PAGE_FLAGS_PRIVATE \ - (1UL << PG_private | 1UL << PG_private_2) +/** + * folio_has_attached_private - check if the folio has private data attached + * @folio: The folio to check. + * + * Use this in code that may encounter swapcache or hugetlb folios but only + * wants to detect attached private data. + * + * Return: true if the folio has private data attached. + */ +static inline bool folio_has_attached_private(const struct folio *folio) +{ + /* + * Swapcache stores swp_entry_t in folio->swap, a union with + * folio->private, and hugetlb stores its own flags in folio->private; + * both are excluded. + * + * NOTE: For swapcache, folio->swap.val PG_swapcache are not set as + * a whole, so folio_test_swapcache() is not reliable to exclude + * swapcache. Use folio_test_swapbacked() instead, since it remains set + * when a folio is added to/removed from swapcache. + */ + + return folio_test_private(folio) && !folio_test_swapbacked(folio) && + !folio_test_hugetlb(folio); +} /** * folio_has_private - Determine if folio has private stuff * @folio: The folio to be checked @@ -1204,7 +1249,7 @@ static __always_inline void __ClearPageAnonExclusive(struct page *page) */ static inline int folio_has_private(const struct folio *folio) { - return !!(folio->flags.f & PAGE_FLAGS_PRIVATE); + return folio_has_attached_private(folio) || folio_test_private_2(folio); } #undef PF_ANY diff --git a/include/trace/events/pagemap.h b/include/trace/events/pagemap.h index 36c3a90f0accad..5d47b774633a4e 100644 --- a/include/trace/events/pagemap.h +++ b/include/trace/events/pagemap.h @@ -22,7 +22,7 @@ (folio_test_swapcache(folio) ? PAGEMAP_SWAPCACHE : 0) | \ (folio_test_swapbacked(folio) ? PAGEMAP_SWAPBACKED : 0) | \ (folio_test_mappedtodisk(folio) ? PAGEMAP_MAPPEDDISK : 0) | \ - (folio_test_private(folio) ? PAGEMAP_BUFFERS : 0) \ + (folio_has_attached_private(folio) ? PAGEMAP_BUFFERS : 0) \ ) TRACE_EVENT(mm_lru_insertion, diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 194188c292af3d..c822e831554b14 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -4845,7 +4845,7 @@ static int split_huge_pages_pid(int pid, unsigned long vaddr_start, * will try to drop it before split and then check if the folio * can be split or not. So skip the check here. */ - if (!folio_test_private(folio) && + if (!folio_has_attached_private(folio) && folio_expected_ref_count(folio) != folio_ref_count(folio)) goto next; diff --git a/mm/migrate.c b/mm/migrate.c index a369d0c95c3860..b7b92925a28c30 100644 --- a/mm/migrate.c +++ b/mm/migrate.c @@ -1327,7 +1327,7 @@ static int migrate_folio_unmap(new_folio_t get_new_folio, * free the metadata, so the page can be freed. */ if (!src->mapping) { - if (folio_test_private(src)) { + if (folio_has_attached_private(src)) { try_to_free_buffers(src); goto out; } diff --git a/mm/page-writeback.c b/mm/page-writeback.c index eeab25d6ce3647..499a35473e4f31 100644 --- a/mm/page-writeback.c +++ b/mm/page-writeback.c @@ -2705,7 +2705,7 @@ bool filemap_dirty_folio(struct address_space *mapping, struct folio *folio) if (folio_test_set_dirty(folio)) return false; - __folio_mark_dirty(folio, mapping, !folio_test_private(folio)); + __folio_mark_dirty(folio, mapping, !folio_has_attached_private(folio)); if (mapping->host) { /* !PageAnon && !swapper_space */ diff --git a/mm/vmscan.c b/mm/vmscan.c index 80041e2b8049c7..dd6261c862794f 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -1029,7 +1029,7 @@ static void folio_check_dirty_writeback(struct folio *folio, *writeback = folio_test_writeback(folio); /* Verify dirty/writeback state if the filesystem supports it */ - if (!folio_test_private(folio)) + if (!folio_has_attached_private(folio)) return; mapping = folio_mapping(folio); diff --git a/tools/mm/page-types.c b/tools/mm/page-types.c index 7fc5a8be5997fb..47e4781c5fc38a 100644 --- a/tools/mm/page-types.c +++ b/tools/mm/page-types.c @@ -73,7 +73,6 @@ #define KPF_RESERVED 32 #define KPF_MLOCKED 33 #define KPF_OWNER_2 34 -#define KPF_PRIVATE 35 #define KPF_PRIVATE_2 36 #define KPF_OWNER_PRIVATE 37 #define KPF_ARCH 38 @@ -131,7 +130,6 @@ static const char * const page_flag_names[] = { [KPF_RESERVED] = "r:reserved", [KPF_MLOCKED] = "m:mlocked", [KPF_OWNER_2] = "d:owner_2", - [KPF_PRIVATE] = "P:private", [KPF_PRIVATE_2] = "p:private_2", [KPF_OWNER_PRIVATE] = "O:owner_private", [KPF_ARCH] = "h:arch", From e88e7a9c8a7d303f785e9198128283fd9b2234a2 Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:28:07 -0400 Subject: [PATCH 0939/1352] treewide: remove folio_set/clear_private() usage They are no-ops now. Remove them. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-11-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Trond Myklebust Cc: Anna Schumaker Cc: "Matthew Wilcox (Oracle)" Cc: Jan Kara Cc: Matthew Brost Cc: Joshua Hahn Cc: Rakie Kim Cc: Byungchul Park Cc: Gregory Price Cc: Ying Huang Cc: Alistair Popple --- fs/nfs/write.c | 2 -- include/linux/pagemap.h | 4 +--- mm/migrate.c | 1 - 3 files changed, 1 insertion(+), 6 deletions(-) diff --git a/fs/nfs/write.c b/fs/nfs/write.c index 623e7ef1f73d57..b6967b5286691c 100644 --- a/fs/nfs/write.c +++ b/fs/nfs/write.c @@ -717,7 +717,6 @@ static void nfs_inode_add_request(struct nfs_page *req) nfs_lock_request(req); spin_lock(&mapping->i_private_lock); set_bit(PG_MAPPED, &req->wb_flags); - folio_set_private(folio); folio->private = req; spin_unlock(&mapping->i_private_lock); atomic_long_inc(&nfsi->nrequests); @@ -745,7 +744,6 @@ static void nfs_inode_remove_request(struct nfs_page *req) spin_lock(&mapping->i_private_lock); folio->private = NULL; - folio_clear_private(folio); clear_bit(PG_MAPPED, &req->wb_head->wb_flags); spin_unlock(&mapping->i_private_lock); diff --git a/include/linux/pagemap.h b/include/linux/pagemap.h index 1e3462357aaa45..bcbb0afe1a6816 100644 --- a/include/linux/pagemap.h +++ b/include/linux/pagemap.h @@ -594,7 +594,6 @@ static inline void folio_attach_private(struct folio *folio, void *data) { folio_get(folio); folio->private = data; - folio_set_private(folio); } /** @@ -629,9 +628,8 @@ static inline void *folio_detach_private(struct folio *folio) { void *data = folio_get_private(folio); - if (!folio_test_private(folio)) + if (!data) return NULL; - folio_clear_private(folio); folio->private = NULL; folio_put(folio); diff --git a/mm/migrate.c b/mm/migrate.c index b7b92925a28c30..7e3a81f0697442 100644 --- a/mm/migrate.c +++ b/mm/migrate.c @@ -835,7 +835,6 @@ void folio_migrate_flags(struct folio *newfolio, struct folio *folio) */ if (folio_test_swapcache(folio)) folio_clear_swapcache(folio); - folio_clear_private(folio); /* page->private contains hugetlb specific flags */ if (!folio_test_hugetlb(folio)) From 41aa4ba8d043dd3c8eafb3d0a6f31d0383a17b7f Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:28:08 -0400 Subject: [PATCH 0940/1352] ceph: replace PagePrivate() with page_private() PagePrivate() is going to be removed along with PG_private and its implementation is the same as page_private(). Link: https://lore.kernel.org/20260920-remove-pg_private-v5-12-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Ilya Dryomov Cc: Alex Markuze Cc: Viacheslav Dubeyko --- fs/ceph/addr.c | 8 +++----- 1 file changed, 3 insertions(+), 5 deletions(-) diff --git a/fs/ceph/addr.c b/fs/ceph/addr.c index e598b2d424ec16..0c00e9636b516f 100644 --- a/fs/ceph/addr.c +++ b/fs/ceph/addr.c @@ -70,9 +70,7 @@ static int ceph_netfs_check_write_begin(struct file *file, loff_t pos, unsigned static inline struct ceph_snap_context *page_snap_context(struct page *page) { - if (PagePrivate(page)) - return (void *)page->private; - return NULL; + return (void *)page_private(page); } /* @@ -124,8 +122,8 @@ static bool ceph_dirty_folio(struct address_space *mapping, struct folio *folio) spin_unlock(&ci->i_ceph_lock); /* - * Reference snap context in folio->private. Also set - * PagePrivate so that we get invalidate_folio callback. + * Reference snap context in folio->private. Setting folio->private is + * what gets us the invalidate_folio callback. */ VM_WARN_ON_FOLIO(folio->private, folio); folio_attach_private(folio, snapc); From 497ca29fdc5c6220fa1f13b105075428b8bf2b97 Mon Sep 17 00:00:00 2001 From: "Matthew Wilcox (Oracle)" Date: Sun, 20 Sep 2026 22:28:09 -0400 Subject: [PATCH 0941/1352] md: use folio_alloc_buffers() Remove the last user of alloc_page_buffers(). Use folio_alloc_buffers() instead, since alloc_page_buffers() is a wrap over it. Although the pages used in md-bitmap are not folios, as they are not mapped into userspace nor enter the page cache, but they still have buffer heads attached. Cleaning up the code to not use buffer heads is future work. [ziy@nvidia.com: reword commit message] Link: https://lore.kernel.org/20260920-remove-pg_private-v5-13-bb68b6a21869@nvidia.com Signed-off-by: Matthew Wilcox (Oracle) Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) --- drivers/md/md-bitmap.c | 5 +++-- fs/buffer.c | 8 -------- include/linux/buffer_head.h | 1 - 3 files changed, 3 insertions(+), 11 deletions(-) diff --git a/drivers/md/md-bitmap.c b/drivers/md/md-bitmap.c index b8325cb09a371f..5f1637f974c155 100644 --- a/drivers/md/md-bitmap.c +++ b/drivers/md/md-bitmap.c @@ -560,6 +560,7 @@ static int read_file_page(struct file *file, unsigned long index, { int ret = 0; struct inode *inode = file_inode(file); + struct folio *folio = page_folio(page); struct buffer_head *bh; sector_t block, blk_cur; unsigned long blocksize = i_blocksize(inode); @@ -567,12 +568,12 @@ static int read_file_page(struct file *file, unsigned long index, pr_debug("read bitmap file (%dB @ %llu)\n", (int)PAGE_SIZE, (unsigned long long)index << PAGE_SHIFT); - bh = alloc_page_buffers(page, blocksize); + bh = folio_alloc_buffers(folio, blocksize, GFP_NOFS | __GFP_ACCOUNT); if (!bh) { ret = -ENOMEM; goto out; } - attach_page_private(page, bh); + folio_attach_private(folio, bh); blk_cur = index << (PAGE_SHIFT - inode->i_blkbits); while (bh) { block = blk_cur; diff --git a/fs/buffer.c b/fs/buffer.c index ed966fa73b1ba2..020af5dbe2d05f 100644 --- a/fs/buffer.c +++ b/fs/buffer.c @@ -773,14 +773,6 @@ struct buffer_head *folio_alloc_buffers(struct folio *folio, unsigned long size, } EXPORT_SYMBOL_GPL(folio_alloc_buffers); -struct buffer_head *alloc_page_buffers(struct page *page, unsigned long size) -{ - gfp_t gfp = GFP_NOFS | __GFP_ACCOUNT; - - return folio_alloc_buffers(page_folio(page), size, gfp); -} -EXPORT_SYMBOL_GPL(alloc_page_buffers); - static inline void link_dev_buffers(struct folio *folio, struct buffer_head *head) { diff --git a/include/linux/buffer_head.h b/include/linux/buffer_head.h index fd2c7115c05427..6ce2db05c60f33 100644 --- a/include/linux/buffer_head.h +++ b/include/linux/buffer_head.h @@ -197,7 +197,6 @@ void folio_set_bh(struct buffer_head *bh, struct folio *folio, unsigned long offset); struct buffer_head *folio_alloc_buffers(struct folio *folio, unsigned long size, gfp_t gfp); -struct buffer_head *alloc_page_buffers(struct page *page, unsigned long size); struct buffer_head *create_empty_buffers(struct folio *folio, unsigned long blocksize, unsigned long b_state); void end_buffer_read_sync(struct buffer_head *bh, int uptodate); From 329d212f49a697bc4bde183049406ff526eea6f0 Mon Sep 17 00:00:00 2001 From: "Matthew Wilcox (Oracle)" Date: Sun, 20 Sep 2026 22:28:10 -0400 Subject: [PATCH 0942/1352] md: use folio APIs in free_page() Convert the page to a folio. This removes some of the last uses of a few page APIs (detach_page_private(), page_buffers()). Replaces two calls to compound_head() with one. Remove the early return in free_buffers() to avoid memory leak and uninitialized file bitmap pages.[1][2] [ziy@nvidia.com: drop early return] Link: https://lore.kernel.org/20260920-remove-pg_private-v5-14-bb68b6a21869@nvidia.com Link: https://sashiko.dev/#/patchset/20260913-remove-pg_private-v4-0-848550f7574e%40nvidia.com?part=2 [1] Link: https://lore.kernel.org/all/aqgfA0QFi98gXYG2@casper.infradead.org/ [2] Signed-off-by: Matthew Wilcox (Oracle) Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) --- drivers/md/md-bitmap.c | 10 +++------- 1 file changed, 3 insertions(+), 7 deletions(-) diff --git a/drivers/md/md-bitmap.c b/drivers/md/md-bitmap.c index 5f1637f974c155..02a126968ad404 100644 --- a/drivers/md/md-bitmap.c +++ b/drivers/md/md-bitmap.c @@ -533,19 +533,15 @@ static void write_file_page(struct bitmap *bitmap, struct page *page, int wait) static void free_buffers(struct page *page) { - struct buffer_head *bh; - - if (!PagePrivate(page)) - return; + struct folio *folio = page_folio(page); + struct buffer_head *bh = folio_detach_private(folio); - bh = page_buffers(page); while (bh) { struct buffer_head *next = bh->b_this_page; free_buffer_head(bh); bh = next; } - detach_page_private(page); - put_page(page); + folio_put(folio); } /* read a page from a file. From fffd2ab6e6524719f72e40f53f659432209f6657 Mon Sep 17 00:00:00 2001 From: "Matthew Wilcox (Oracle)" Date: Sun, 20 Sep 2026 22:28:11 -0400 Subject: [PATCH 0943/1352] md: remove the last use of page_buffers() Convert the page to a folio and use folio_buffers() instead. This does introduce one extra call to compound_head(), but will simplify a later conversion of md-bitmap to use folios instead of pages. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-15-bb68b6a21869@nvidia.com Signed-off-by: Matthew Wilcox (Oracle) Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) --- drivers/md/md-bitmap.c | 3 ++- include/linux/buffer_head.h | 7 +------ 2 files changed, 3 insertions(+), 7 deletions(-) diff --git a/drivers/md/md-bitmap.c b/drivers/md/md-bitmap.c index 02a126968ad404..8adf6ddca3d605 100644 --- a/drivers/md/md-bitmap.c +++ b/drivers/md/md-bitmap.c @@ -516,7 +516,8 @@ static void end_bitmap_write(struct bio *bio) static void write_file_page(struct bitmap *bitmap, struct page *page, int wait) { - struct buffer_head *bh = page_buffers(page); + struct folio *folio = page_folio(page); + struct buffer_head *bh = folio_buffers(folio); while (bh && bh->b_blocknr) { atomic_inc(&bitmap->pending_writes); diff --git a/include/linux/buffer_head.h b/include/linux/buffer_head.h index 6ce2db05c60f33..4b0b7188472b2d 100644 --- a/include/linux/buffer_head.h +++ b/include/linux/buffer_head.h @@ -175,12 +175,7 @@ static inline unsigned long bh_offset(const struct buffer_head *bh) return (unsigned long)(bh)->b_data & (page_size(bh->b_page) - 1); } -/* If we *know* page->private refers to buffer_heads */ -#define page_buffers(page) \ - ({ \ - BUG_ON(!PagePrivate(page)); \ - ((struct buffer_head *)page_private(page)); \ - }) +/* If we *know* folio->private refers to buffer_heads */ #define folio_buffers(folio) folio_get_private(folio) void buffer_check_dirty_writeback(struct folio *folio, From 913e09349b714c09f3087ebb4a3a2b3619f73b26 Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:28:12 -0400 Subject: [PATCH 0944/1352] treewide: remove PagePrivate() and PG_private from comments and docs PG_private and PagePrivate() are no longer used. Adjust related comments and documentations to refer to page/folio->private instead. hugetlbfs_reserv.rst is outdated and left unchanged. It should be rewritten. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-16-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Ilya Dryomov Cc: Alex Markuze Cc: Viacheslav Dubeyko Cc: Trond Myklebust Cc: Anna Schumaker Cc: Richard Weinberger Cc: Zhihao Cheng Cc: Lorenzo Stoakes Cc: "Liam R. Howlett" Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko --- Documentation/admin-guide/kdump/vmcoreinfo.rst | 2 +- Documentation/filesystems/vfs.rst | 6 +++--- fs/nfs/file.c | 4 ++-- fs/ubifs/file.c | 8 ++++---- include/linux/mm.h | 15 ++++++++------- include/linux/mm_types.h | 4 ++-- 6 files changed, 20 insertions(+), 19 deletions(-) diff --git a/Documentation/admin-guide/kdump/vmcoreinfo.rst b/Documentation/admin-guide/kdump/vmcoreinfo.rst index 7663c610fe9014..5f1df6d080508a 100644 --- a/Documentation/admin-guide/kdump/vmcoreinfo.rst +++ b/Documentation/admin-guide/kdump/vmcoreinfo.rst @@ -325,7 +325,7 @@ NR_FREE_PAGES On linux-2.6.21 or later, the number of free pages is in vm_stat[NR_FREE_PAGES]. Used to get the number of free pages. -PG_lru|PG_private|PG_swapcache|PG_swapbacked|PG_hwpoison|PG_head_mask +PG_lru|PG_swapcache|PG_swapbacked|PG_hwpoison|PG_head_mask -------------------------------------------------------------------------- Page attributes. These flags are used to filter various unnecessary for diff --git a/Documentation/filesystems/vfs.rst b/Documentation/filesystems/vfs.rst index d3a93eec3945f8..dec7816303c6a9 100644 --- a/Documentation/filesystems/vfs.rst +++ b/Documentation/filesystems/vfs.rst @@ -649,8 +649,8 @@ Writeback. The first can be used independently to the others. The VM can try to release clean pages in order to reuse them. To do this it can call -->release_folio on clean folios with the private -flag set. Clean pages without PagePrivate and with no external references +->release_folio on clean folios with folio->private set. Clean pages +without folio->private set and with no external references will be released without notice being given to the address_space. To achieve this functionality, pages need to be placed on an LRU with @@ -674,7 +674,7 @@ filemap_fdatawait_range, to wait for all writeback to complete. An address_space handler may attach extra information to a page, typically using the 'private' field in the 'struct page'. If such -information is attached, the PG_Private flag should be set. This will +information is attached, non-NULL 'private' field will cause various VM routines to make extra calls into the address_space handler to deal with that data. diff --git a/fs/nfs/file.c b/fs/nfs/file.c index e1bdd10b35f10d..38f830a6467c9b 100644 --- a/fs/nfs/file.c +++ b/fs/nfs/file.c @@ -484,7 +484,7 @@ static int nfs_write_end(const struct kiocb *iocb, * Partially or wholly invalidate a page * - Release the private state associated with a page if undergoing complete * page invalidation - * - Called if either PG_private or PG_fscache is set on the page + * - Called if either folio->private or PG_fscache is set on the page * - Caller holds page lock */ static void nfs_invalidate_folio(struct folio *folio, size_t offset, @@ -555,7 +555,7 @@ static void nfs_check_dirty_writeback(struct folio *folio, * Attempt to clear the private state associated with a page when an error * occurs that requires the cached contents of an inode to be written back or * destroyed - * - Called if either PG_private or fscache is set on the page + * - Called if either page->private or fscache is set on the page * - Caller holds page lock * - Return 0 if successful, -error otherwise */ diff --git a/fs/ubifs/file.c b/fs/ubifs/file.c index e73c28b12f97fd..aa0298ce451eff 100644 --- a/fs/ubifs/file.c +++ b/fs/ubifs/file.c @@ -12,14 +12,14 @@ * This file implements VFS file and inode operations for regular files, device * nodes and symlinks as well as address space operations. * - * UBIFS uses 2 page flags: @PG_private and @PG_checked. @PG_private is set if + * UBIFS uses folio->private and page flag @PG_checked. folio->private is set if * the page is dirty and is used for optimization purposes - dirty pages are - * not budgeted so the flag shows that 'ubifs_write_end()' should not release + * not budgeted so it shows that 'ubifs_write_end()' should not release * the budget for this page. The @PG_checked flag is set if full budgeting is * required for the page e.g., when it corresponds to a file hole or it is * beyond the file size. The budgeting is done in 'ubifs_write_begin()', because * it is OK to fail in this function, and the budget is released in - * 'ubifs_write_end()'. So the @PG_private and @PG_checked flags carry + * 'ubifs_write_end()'. So the folio->private and the @PG_checked flag carry * information about how the page was budgeted, to make it possible to release * the budget properly. * @@ -1509,7 +1509,7 @@ static vm_fault_t ubifs_vm_page_mkwrite(struct vm_fault *vmf) * * At the moment we do not know whether the folio is dirty or not, so we * assume that it is not and budget for a new folio. We could look at - * the @PG_private flag and figure this out, but we may race with write + * folio->private and figure this out, but we may race with write * back and the folio state may change by the time we lock it, so this * would need additional care. We do not bother with this at the * moment, although it might be good idea to do. Instead, we allocate diff --git a/include/linux/mm.h b/include/linux/mm.h index 6c8df7715eb261..0a2a7fc4a442fe 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -2049,20 +2049,21 @@ vm_fault_t finish_fault(struct vm_fault *vmf); * * A pagecache page contains an opaque `private' member, which belongs to the * page's address_space. Usually, this is the address of a circular list of - * the page's disk buffers. PG_private must be set to tell the VM to call - * into the filesystem to release these pages. + * the page's disk buffers. It tells the VM to call into the filesystem to + * release these pages. * * A folio may belong to an inode's memory mapping. In this case, * folio->mapping points to the inode, and folio->index is the file * offset of the folio, in units of PAGE_SIZE. * - * If pagecache pages are not associated with an inode, they are said to be - * anonymous pages. These may become associated with the swapcache, and in that - * case PG_swapcache is set, and page->private is an offset into the swapcache. + * If pagecache folios are not associated with an inode, they are said to be + * anonymous folios. These may become associated with the swapcache, and in that + * case PG_swapcache is set, and folio->private is an offset into the swapcache. * * In either case (swapcache or inode backed), the pagecache itself holds one - * reference to the page. Setting PG_private should also increment the - * refcount. The each user mapping also has a reference to the page. + * reference to the folio. Attaching filesystem private data via + * folio_attach_private() also increments the refcount. Each user mapping also + * has a reference to the folio. * * The pagecache pages are stored in a per-mapping radix tree, which is * rooted at mapping->i_pages, and indexed by offset. diff --git a/include/linux/mm_types.h b/include/linux/mm_types.h index 2a3988178adfdc..0720a4e98286b2 100644 --- a/include/linux/mm_types.h +++ b/include/linux/mm_types.h @@ -108,7 +108,7 @@ struct page { }; /** * @private: Mapping-private opaque data. - * Usually used for buffer_heads if PagePrivate. + * Usually used for buffer_heads. * Used for swp_entry_t if swapcache flag set. * Indicates order in the buddy system if PageBuddy * or on pcp_llist. @@ -675,7 +675,7 @@ static inline void ptdesc_pmd_pts_init(struct ptdesc *ptdesc) #define STRUCT_PAGE_MAX_SHIFT (order_base_2(sizeof(struct page))) /* - * page_private can be used on tail pages. However, PagePrivate is only + * page_private can be used on tail pages. However, it is only * checked by the VM on the head page. So page_private on the tail pages * should be used for data that's ancillary to the head page (eg attaching * buffer heads to tail pages after attaching buffer heads to the head page) From 6bd101e349c6c6685eb9e8b017ccca307dc1865b Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:28:13 -0400 Subject: [PATCH 0945/1352] mm/page-flags: remove PG_private folio->private != NULL indicates a folio carries private data, replacing PG_private. All PG_private users are converted. Remove PG_private and reserve the space as PG_folio for future use. Unused PG_private functions are removed too. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-17-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Baoquan He Cc: Mike Rapoport Cc: Pasha Tatashin Cc: Pratyush Yadav Cc: Jonathan Corbet Cc: "Matthew Wilcox (Oracle)" Cc: Jan Kara Cc: Steven Rostedt Cc: Masami Hiramatsu Cc: Dave Young Cc: Shuah Khan Cc: Lorenzo Stoakes Cc: "Liam R. Howlett" Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Mathieu Desnoyers --- include/linux/page-flags.h | 18 +----------------- include/trace/events/mmflags.h | 2 +- kernel/vmcore_info.c | 1 - 3 files changed, 2 insertions(+), 19 deletions(-) diff --git a/include/linux/page-flags.h b/include/linux/page-flags.h index 6d839f50bdcb7e..b0ddc652e76ccb 100644 --- a/include/linux/page-flags.h +++ b/include/linux/page-flags.h @@ -44,10 +44,6 @@ * Consequently, PG_reserved for a page mapped into user space can indicate * the zero page, the vDSO, MMIO pages or device memory. * - * The PG_private bitflag is set on pagecache pages if they contain filesystem - * specific data (which is normally at page->private). It can be used by - * private allocations for its own usage. - * * During initiation of disk I/O, PG_locked is set. This bit is set before I/O * and cleared when writeback _starts_ or when read _completes_. PG_writeback * is set before writeback starts and cleared when it finishes. @@ -105,7 +101,7 @@ enum pageflags { PG_owner_2, /* Owner use. If pagecache, fs may use */ PG_arch_1, PG_reserved, - PG_private, /* If pagecache, has fs-private data */ + PG_folio, /* Do not use: reserved for folio identification */ PG_private_2, /* If pagecache, has fs aux data */ PG_reclaim, /* To be reclaimed asap */ PG_swapbacked, /* Page is backed by RAM/swap */ @@ -588,18 +584,6 @@ static __always_inline bool folio_test_private(const struct folio *folio) return data_race(folio->private); } -static __always_inline int PagePrivate(const struct page *page) -{ - /* See folio_test_private() for data_race() use */ - return !!data_race(page->private); -} - -/* no-ops during transition */ -static __always_inline void folio_set_private(struct folio *folio) { } -static __always_inline void folio_clear_private(struct folio *folio) { } -static __always_inline void SetPagePrivate(struct page *page) { } -static __always_inline void ClearPagePrivate(struct page *page) { } - FOLIO_FLAG(private_2, FOLIO_HEAD_PAGE) /* owner_2 can be set on tail pages for anon memory */ diff --git a/include/trace/events/mmflags.h b/include/trace/events/mmflags.h index ef9aa388b84f7d..3c153b3ad84507 100644 --- a/include/trace/events/mmflags.h +++ b/include/trace/events/mmflags.h @@ -144,7 +144,7 @@ TRACE_DEFINE_ENUM(___GFP_LAST_BIT); DEF_PAGEFLAG_NAME(owner_2), \ DEF_PAGEFLAG_NAME(arch_1), \ DEF_PAGEFLAG_NAME(reserved), \ - DEF_PAGEFLAG_NAME(private), \ + DEF_PAGEFLAG_NAME(folio), \ DEF_PAGEFLAG_NAME(private_2), \ DEF_PAGEFLAG_NAME(writeback), \ DEF_PAGEFLAG_NAME(head), \ diff --git a/kernel/vmcore_info.c b/kernel/vmcore_info.c index 8614430ca212ae..5a417f8a922abd 100644 --- a/kernel/vmcore_info.c +++ b/kernel/vmcore_info.c @@ -216,7 +216,6 @@ static int __init crash_save_vmcoreinfo_init(void) VMCOREINFO_LENGTH(free_area.free_list, MIGRATE_TYPES); VMCOREINFO_NUMBER(NR_FREE_PAGES); VMCOREINFO_NUMBER(PG_lru); - VMCOREINFO_NUMBER(PG_private); VMCOREINFO_NUMBER(PG_swapcache); VMCOREINFO_NUMBER(PG_swapbacked); #define PAGE_SLAB_MAPCOUNT_VALUE (PGTY_slab << 24) From d60f6921ea74b6acbb763d09d0f2116aaf49337c Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:32 +0200 Subject: [PATCH 0946/1352] mm/swap: fix off-by-one in swap cache replace sanity check Patch series "mm/huge_memory: clean up and decouple the anon and file split helpers", v6. The folio split path handles anon, page cache and swap cache folios in one routine. That mixing is what makes the swap cache split restrictions hard to lift and to review. We now support uniform split to order-0 only, and no mappingless swap cache folios. And it has left a fair number of dead or redundant checks behind. This series prepares for lifting those restrictions by cleaning up the code first: split the routine into an anon and a file helper, and keep all swap cache handling in the anon helper. The file helper never sees a swap cache folio, folio_check_splittable() rejects them up front. Apart from two bug fixes (patch 1 and 2) and a slight adjustment of anon splitting (patch 12), this is a pure cleanup. Testing: The in-tree split_huge_page_test selftest (uniform, non-uniform and in-folio-offset splits of anon and pagecache folios) passes 62/62 over 600 runs on the patched kernel. ftrace function_graph tracing filtered on __folio_split() was used to compare per-call durations between the base and the patched kernel on the same x86-64 box (interleaved runs across alternating reboots. 135 split calls per run, 600 test runs): Before: 68.52 us, stddev: 1.58 After: 67.38 us, stddev: 1.33 The patched kernel is slightly faster. The stack usage and object size change as the config and compiler change, but in general the stack usage is reduced and object size is basically unchanged. This patch (of 17): The DEBUG_VM sanity check in __swap_cache_replace_folio() iterates the old folio's range with "while (ci_off++ < ci_end)", so the loop body runs on the already-incremented offset: the first entry is skipped and one entry past the range is read. For a folio split that entry belongs to the first after-split folio and was just repointed by the replacement loop above, so the check would warn spuriously whenever sub-folio orders differ from the head folio's. Currently we don't support non-uniform swapcache split, but this still needs a fix to clean it up and prepare for non-uniform swap cache split. Use the same do-while pattern as the replacement loop. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-0-ba1b4ba72c6f@tencent.com Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-1-ba1b4ba72c6f@tencent.com Fixes: 8578e0c00dcf ("mm, swap: use the swap table for the swap cache and switch API") Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Acked-by: Zi Yan Reviewed-by: Barry Song Acked-by: David Hildenbrand (Arm) Reviewed-by: Yeoreum Yun Acked-by: Kiryl Shutsemau (Meta) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/swap_state.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/mm/swap_state.c b/mm/swap_state.c index 625c185a1ca4d5..cef44aadee6158 100644 --- a/mm/swap_state.c +++ b/mm/swap_state.c @@ -396,8 +396,9 @@ void __swap_cache_replace_folio(struct swap_cluster_info *ci, folio_order(old) != folio_order(new)) { ci_off = swp_cluster_offset(old->swap); ci_end = ci_off + folio_nr_pages(old); - while (ci_off++ < ci_end) + do { WARN_ON_ONCE(swp_tb_to_folio(__swap_table_get(ci, ci_off)) != old); + } while (++ci_off < ci_end); } } From 96b806ad99d3745ca9ebcdc70de2f40875d6d3fe Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:33 +0200 Subject: [PATCH 0947/1352] mm/huge_memory: fix rejection of swap cache folios with a mapping A folio in the swap cache cannot be split if it has a mapping (shmem). The split code does a defensive check for this in __folio_freeze_and_split_unmapped, after the folio ref has been frozen and the NR_SHMEM_THPS/NR_FILE_THPS counters have been decremented. It rejects the split and returns -EINVAL without unfreezing the folio or restoring the counters. That error path is buggy, if it is ever taken. It leaves the folio frozen and stuck, skews the counters, and fires the VM_WARN_ON_ONCE_FOLIO for a state that is actually legitimate. Check for this case up front in folio_check_splittable and return -EBUSY before any state is modified, so the split routine always backs out cleanly. Also fix a bracket style issue that checkpatch.pl keeps complaining about. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-2-ba1b4ba72c6f@tencent.com Fixes: 00527733d0dc ("mm/huge_memory: add two new (not yet used) functions for folio_split()") Fixes: 714b056c8321 ("mm/huge_memory: convert VM_BUG* to VM_WARN* in __folio_split") Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Barry Song Acked-by: David Hildenbrand (Arm) Reviewed-by: Yeoreum Yun Reviewed-by: Kiryl Shutsemau (Meta) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 26 ++++++++++++++++---------- 1 file changed, 16 insertions(+), 10 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index c822e831554b14..7c55bc5e8da840 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -3933,6 +3933,9 @@ static int __split_unmapped_folio(struct folio *folio, int new_order, int folio_check_splittable(struct folio *folio, unsigned int new_order, enum split_type split_type) { + const bool is_anon = folio_test_anon(folio); + const bool is_swapcache = folio_test_swapcache(folio); + VM_WARN_ON_FOLIO(!folio_test_locked(folio), folio); /* * Folios that just got truncated cannot get split. Signal to the @@ -3941,11 +3944,11 @@ int folio_check_splittable(struct folio *folio, unsigned int new_order, * TODO: this will also currently refuse folios without a mapping in the * swapcache (shmem or to-be-anon folios). */ - if (!folio->mapping && !folio_test_anon(folio)) + if (!folio->mapping && !is_anon) return -EBUSY; /* order-1 is not supported for anonymous THP. */ - if (folio_test_anon(folio) && new_order == 1) + if (is_anon && new_order == 1) return -EINVAL; /* @@ -3956,7 +3959,7 @@ int folio_check_splittable(struct folio *folio, unsigned int new_order, * swapcache folio split. Only uniform split to order-0 can be used * here. */ - if ((split_type == SPLIT_TYPE_NON_UNIFORM || new_order) && folio_test_swapcache(folio)) + if ((split_type == SPLIT_TYPE_NON_UNIFORM || new_order) && is_swapcache) return -EINVAL; if (is_huge_zero_folio(folio)) @@ -3965,6 +3968,15 @@ int folio_check_splittable(struct folio *folio, unsigned int new_order, if (folio_test_writeback(folio)) return -EBUSY; + /* + * A non-anon swapcache folio that still has a mapping can only be a + * shmem folio under SWAP IO, it's removed from either swap cache or + * shmem mapping afterward. There is little benefit in splitting them + * hence reject it here up front before touching anything. + */ + if (!is_anon && is_swapcache && folio->mapping) + return -EBUSY; + return 0; } @@ -4037,14 +4049,8 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n } } - if (folio_test_swapcache(folio)) { - if (mapping) { - VM_WARN_ON_ONCE_FOLIO(mapping, folio); - return -EINVAL; - } - + if (folio_test_swapcache(folio)) ci = swap_cluster_get_and_lock(folio); - } /* lock lru list/PageCompound, ref frozen by page_ref_freeze */ if (do_lru) From 1cb57d5e20ce6ee2c23376e39e37b7156285c6c0 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:34 +0200 Subject: [PATCH 0948/1352] mm/huge_memory: invert folio_ref_freeze() check to reduce indentation Invert the folio_ref_freeze() success check in __folio_freeze_and_split_unmapped() to return early on failure, which removes one level of indentation from the entire success path. This is a pure refactoring with no functional change. It prepares the function to be split into separate helpers for anonymous and file-backed folios in a later patch. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-3-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Barry Song Acked-by: David Hildenbrand (Arm) Reviewed-by: Yeoreum Yun Reviewed-by: Kiryl Shutsemau (Meta) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 178 +++++++++++++++++++++++------------------------ 1 file changed, 88 insertions(+), 90 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 7c55bc5e8da840..9b2908271b6989 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -4014,121 +4014,119 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n pgoff_t end, int *nr_shmem_dropped) { struct folio *end_folio = folio_next(folio); + struct swap_cluster_info *ci = NULL; struct folio *new_folio, *next; + struct lruvec *lruvec; int ret = 0; VM_WARN_ON_ONCE(!mapping && end); - if (folio_ref_freeze(folio, folio_cache_ref_count(folio) + 1)) { - struct swap_cluster_info *ci = NULL; - struct lruvec *lruvec; + if (!folio_ref_freeze(folio, folio_cache_ref_count(folio) + 1)) + return -EAGAIN; - /* Take off the deferred split queue while frozen and memcg set */ - folio_unqueue_deferred_split(folio); + /* Take off the deferred split queue while frozen and memcg set */ + folio_unqueue_deferred_split(folio); - /* - * deferred_split_scan() takes the folio off the queue before it - * splits it, so the unqueue above finds an empty list and - * leaves PG_partially_mapped set. - * Clear it here: the flag does not survive the split. - */ - folio_reset_partially_mapped(folio); + /* + * deferred_split_scan() takes the folio off the queue before it + * splits it, so the unqueue above finds an empty list and + * leaves PG_partially_mapped set. + * Clear it here: the flag does not survive the split. + */ + folio_reset_partially_mapped(folio); - if (mapping) { - int nr = folio_nr_pages(folio); - - if (folio_test_pmd_mappable(folio) && - new_order < HPAGE_PMD_ORDER) { - if (folio_test_swapbacked(folio)) { - lruvec_stat_mod_folio(folio, - NR_SHMEM_THPS, -nr); - } else { - lruvec_stat_mod_folio(folio, - NR_FILE_THPS, -nr); - } + if (mapping) { + int nr = folio_nr_pages(folio); + + if (folio_test_pmd_mappable(folio) && + new_order < HPAGE_PMD_ORDER) { + if (folio_test_swapbacked(folio)) { + lruvec_stat_mod_folio(folio, + NR_SHMEM_THPS, -nr); + } else { + lruvec_stat_mod_folio(folio, + NR_FILE_THPS, -nr); } } + } - if (folio_test_swapcache(folio)) - ci = swap_cluster_get_and_lock(folio); - - /* lock lru list/PageCompound, ref frozen by page_ref_freeze */ - if (do_lru) - lruvec = folio_lruvec_lock(folio); + if (folio_test_swapcache(folio)) + ci = swap_cluster_get_and_lock(folio); - ret = __split_unmapped_folio(folio, new_order, split_at, xas, - mapping, split_type); + /* lock lru list/PageCompound, ref frozen by page_ref_freeze */ + if (do_lru) + lruvec = folio_lruvec_lock(folio); - /* - * Unfreeze after-split folios and put them back to the right - * list. @folio should be kept frozon until page cache - * entries are updated with all the other after-split folios - * to prevent others seeing stale page cache entries. - * As a result, new_folio starts from the next folio of - * @folio. - */ - for (new_folio = folio_next(folio); new_folio != end_folio; - new_folio = next) { - unsigned long nr_pages = folio_nr_pages(new_folio); + ret = __split_unmapped_folio(folio, new_order, split_at, xas, + mapping, split_type); - next = folio_next(new_folio); + /* + * Unfreeze after-split folios and put them back to the right + * list. @folio should be kept frozon until page cache + * entries are updated with all the other after-split folios + * to prevent others seeing stale page cache entries. + * As a result, new_folio starts from the next folio of + * @folio. + */ + for (new_folio = folio_next(folio); new_folio != end_folio; + new_folio = next) { + unsigned long nr_pages = folio_nr_pages(new_folio); - zone_device_private_split_cb(folio, new_folio); + next = folio_next(new_folio); - folio_ref_unfreeze(new_folio, - folio_cache_ref_count(new_folio) + 1); + zone_device_private_split_cb(folio, new_folio); - if (do_lru) - lru_add_split_folio(folio, new_folio, lruvec, list); + folio_ref_unfreeze(new_folio, + folio_cache_ref_count(new_folio) + 1); - /* - * Anonymous folio with swap cache. - * NOTE: shmem in swap cache is not supported yet. - */ - if (ci) { - __swap_cache_replace_folio(ci, folio, new_folio); - continue; - } + if (do_lru) + lru_add_split_folio(folio, new_folio, lruvec, list); - /* Anonymous folio without swap cache */ - if (!mapping) - continue; + /* + * Anonymous folio with swap cache. + * NOTE: shmem in swap cache is not supported yet. + */ + if (ci) { + __swap_cache_replace_folio(ci, folio, new_folio); + continue; + } - /* Add the new folio to the page cache. */ - if (new_folio->index < end) { - __xa_store(&mapping->i_pages, new_folio->index, - new_folio, 0); - continue; - } + /* Anonymous folio without swap cache */ + if (!mapping) + continue; - VM_WARN_ON_ONCE(!nr_shmem_dropped); - /* Drop folio beyond EOF: ->index >= end */ - if (shmem_mapping(mapping) && nr_shmem_dropped) - *nr_shmem_dropped += nr_pages; - else if (folio_test_clear_dirty(new_folio)) - folio_account_cleaned( - new_folio, inode_to_wb(mapping->host)); - __filemap_remove_folio(new_folio, NULL); - folio_put_refs(new_folio, nr_pages); + /* Add the new folio to the page cache. */ + if (new_folio->index < end) { + __xa_store(&mapping->i_pages, new_folio->index, + new_folio, 0); + continue; } - zone_device_private_split_cb(folio, NULL); - /* - * Unfreeze @folio only after all page cache entries, which - * used to point to it, have been updated with new folios. - * Otherwise, a parallel folio_try_get() can grab @folio - * and its caller can see stale page cache entries. - */ - folio_ref_unfreeze(folio, folio_cache_ref_count(folio) + 1); + VM_WARN_ON_ONCE(!nr_shmem_dropped); + /* Drop folio beyond EOF: ->index >= end */ + if (shmem_mapping(mapping) && nr_shmem_dropped) + *nr_shmem_dropped += nr_pages; + else if (folio_test_clear_dirty(new_folio)) + folio_account_cleaned( + new_folio, inode_to_wb(mapping->host)); + __filemap_remove_folio(new_folio, NULL); + folio_put_refs(new_folio, nr_pages); + } - if (do_lru) - lruvec_unlock(lruvec); + zone_device_private_split_cb(folio, NULL); + /* + * Unfreeze @folio only after all page cache entries, which + * used to point to it, have been updated with new folios. + * Otherwise, a parallel folio_try_get() can grab @folio + * and its caller can see stale page cache entries. + */ + folio_ref_unfreeze(folio, folio_cache_ref_count(folio) + 1); - if (ci) - swap_cluster_unlock(ci); - } else { - return -EAGAIN; - } + if (do_lru) + lruvec_unlock(lruvec); + + if (ci) + swap_cluster_unlock(ci); return ret; } From d9c450da29025e52449c066d3dba508d274d87ba Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:35 +0200 Subject: [PATCH 0949/1352] mm/huge_memory: split the routine for splitting anon and file folio No functional change intended. Before adding more logic, split __folio_freeze_and_split_unmapped() into an anon and a file variant so each path can evolve independently. The two paths shared little beyond the folio freeze call, the LRU locking, and the unfreeze skeleton, but differed in all other per-folio bookkeeping and routines. While splitting, some cleanups become easy to apply, and helped drop a few now-redundant checks. The zone_device_private_split_cb() calls are only kept in the anon variant, as device private folios can only back anonymous memory, and add a VM_WARN_ON_ONCE_FOLIO() at the entry of the file variant. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-4-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Zi Yan Reviewed-by: Yeoreum Yun Reviewed-by: Kiryl Shutsemau (Meta) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 124 +++++++++++++++++++++++++++++------------------ 1 file changed, 78 insertions(+), 46 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 9b2908271b6989..a9057449d3a458 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -4007,11 +4007,9 @@ static void folio_reset_partially_mapped(struct folio *folio) MTHP_STAT_NR_ANON_PARTIALLY_MAPPED, -1); } -static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int new_order, - struct page *split_at, struct xa_state *xas, - struct address_space *mapping, bool do_lru, - struct list_head *list, enum split_type split_type, - pgoff_t end, int *nr_shmem_dropped) +static int __folio_freeze_split_anon(struct folio *folio, + unsigned int new_order, struct page *split_at, bool do_lru, + struct list_head *list, enum split_type split_type) { struct folio *end_folio = folio_next(folio); struct swap_cluster_info *ci = NULL; @@ -4019,8 +4017,6 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n struct lruvec *lruvec; int ret = 0; - VM_WARN_ON_ONCE(!mapping && end); - if (!folio_ref_freeze(folio, folio_cache_ref_count(folio) + 1)) return -EAGAIN; @@ -4035,24 +4031,75 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n */ folio_reset_partially_mapped(folio); - if (mapping) { + if (folio_test_swapcache(folio)) + ci = swap_cluster_get_and_lock(folio); + + if (do_lru) + lruvec = folio_lruvec_lock(folio); + + ret = __split_unmapped_folio(folio, new_order, split_at, NULL, + NULL, split_type); + + /* + * Unfreeze the after-split folios and put them back to the right + * place. Keep the head @folio frozen until the end: sub entries + * in swap cache must be updated first, so a concurrent + * swap_cache_get_folio() cannot return the head folio for a sub + * entry (folio_try_get() will fail on the head @folio until unfreeze). + */ + for (new_folio = folio_next(folio); new_folio != end_folio; + new_folio = next) { + next = folio_next(new_folio); + zone_device_private_split_cb(folio, new_folio); + folio_ref_unfreeze(new_folio, + folio_cache_ref_count(new_folio) + 1); + if (do_lru) + lru_add_split_folio(folio, new_folio, lruvec, list); + if (ci) + __swap_cache_replace_folio(ci, folio, new_folio); + } + + zone_device_private_split_cb(folio, NULL); + folio_ref_unfreeze(folio, folio_cache_ref_count(folio) + 1); + + if (do_lru) + lruvec_unlock(lruvec); + if (ci) + swap_cluster_unlock(ci); + + return ret; +} + +static int __folio_freeze_split_file(struct folio *folio, + unsigned int new_order, struct page *split_at, + struct xa_state *xas, struct address_space *mapping, + bool do_lru, struct list_head *list, + enum split_type split_type, pgoff_t end, int *nr_shmem_dropped) +{ + struct folio *end_folio = folio_next(folio); + struct folio *new_folio, *next; + struct lruvec *lruvec; + int ret; + + /* Currently device private folios can only back anonymous memory. */ + VM_WARN_ON_ONCE_FOLIO(folio_is_device_private(folio), folio); + + if (!folio_ref_freeze(folio, folio_cache_ref_count(folio) + 1)) + return -EAGAIN; + + if (folio_test_pmd_mappable(folio) && + new_order < HPAGE_PMD_ORDER) { int nr = folio_nr_pages(folio); - if (folio_test_pmd_mappable(folio) && - new_order < HPAGE_PMD_ORDER) { - if (folio_test_swapbacked(folio)) { - lruvec_stat_mod_folio(folio, - NR_SHMEM_THPS, -nr); - } else { - lruvec_stat_mod_folio(folio, - NR_FILE_THPS, -nr); - } + if (folio_test_swapbacked(folio)) { + lruvec_stat_mod_folio(folio, + NR_SHMEM_THPS, -nr); + } else { + lruvec_stat_mod_folio(folio, + NR_FILE_THPS, -nr); } } - if (folio_test_swapcache(folio)) - ci = swap_cluster_get_and_lock(folio); - /* lock lru list/PageCompound, ref frozen by page_ref_freeze */ if (do_lru) lruvec = folio_lruvec_lock(folio); @@ -4062,7 +4109,7 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n /* * Unfreeze after-split folios and put them back to the right - * list. @folio should be kept frozon until page cache + * list. @folio should be kept frozen until page cache * entries are updated with all the other after-split folios * to prevent others seeing stale page cache entries. * As a result, new_folio starts from the next folio of @@ -4072,29 +4119,15 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n new_folio = next) { unsigned long nr_pages = folio_nr_pages(new_folio); + /* compute next before the folio can be freed below */ next = folio_next(new_folio); - zone_device_private_split_cb(folio, new_folio); - folio_ref_unfreeze(new_folio, folio_cache_ref_count(new_folio) + 1); if (do_lru) lru_add_split_folio(folio, new_folio, lruvec, list); - /* - * Anonymous folio with swap cache. - * NOTE: shmem in swap cache is not supported yet. - */ - if (ci) { - __swap_cache_replace_folio(ci, folio, new_folio); - continue; - } - - /* Anonymous folio without swap cache */ - if (!mapping) - continue; - /* Add the new folio to the page cache. */ if (new_folio->index < end) { __xa_store(&mapping->i_pages, new_folio->index, @@ -4113,7 +4146,6 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n folio_put_refs(new_folio, nr_pages); } - zone_device_private_split_cb(folio, NULL); /* * Unfreeze @folio only after all page cache entries, which * used to point to it, have been updated with new folios. @@ -4125,9 +4157,6 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n if (do_lru) lruvec_unlock(lruvec); - if (ci) - swap_cluster_unlock(ci); - return ret; } @@ -4269,7 +4298,10 @@ static int __folio_split(struct folio *folio, unsigned int new_order, /* block interrupt reentry in xa_lock and spinlock */ local_irq_disable(); - if (mapping) { + if (is_anon) { + ret = __folio_freeze_split_anon(folio, new_order, split_at, + true, list, split_type); + } else { /* * Check if the folio is present in page cache. * We assume all tail are present too, if folio is there. @@ -4280,10 +4312,11 @@ static int __folio_split(struct folio *folio, unsigned int new_order, ret = -EAGAIN; goto fail; } + ret = __folio_freeze_split_file(folio, new_order, split_at, &xas, mapping, + true, list, split_type, end, + &nr_shmem_dropped); } - ret = __folio_freeze_and_split_unmapped(folio, new_order, split_at, &xas, mapping, - true, list, split_type, end, &nr_shmem_dropped); fail: if (mapping) xas_unlock(&xas); @@ -4383,9 +4416,8 @@ int folio_split_unmapped(struct folio *folio, unsigned int new_order) return -EAGAIN; local_irq_disable(); - ret = __folio_freeze_and_split_unmapped(folio, new_order, &folio->page, NULL, - NULL, false, NULL, SPLIT_TYPE_UNIFORM, - 0, NULL); + ret = __folio_freeze_split_anon(folio, new_order, &folio->page, + false, NULL, SPLIT_TYPE_UNIFORM); local_irq_enable(); return ret; } From ab685eb979ce05d8f14cf756e20c985dffa3556f Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:36 +0200 Subject: [PATCH 0950/1352] mm/huge_memory: rename __split_unmapped_folio() to __split_frozen_folio() The helper splits a folio whose refcount is frozen: the frozen refcount is the state it relies on, while unmapping is arranged by the caller beforehand. The old name caused confusion and people may try to call the helper on non-frozen folios. Also add a VM_WARN_ON_ONCE_FOLIO(folio_mapped(folio)) to self document that frozen implies unmapped. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-5-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Suggested-by: Zi Yan Reviewed-by: Zi Yan Reviewed-by: Yeoreum Yun Reviewed-by: Kiryl Shutsemau (Meta) Acked-by: David Hildenbrand (Arm) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 23 +++++++++++++---------- 1 file changed, 13 insertions(+), 10 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index a9057449d3a458..b66b855df6f832 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -3810,8 +3810,8 @@ static void __split_folio_to_order(struct folio *folio, int old_order, } /** - * __split_unmapped_folio() - splits an unmapped @folio to lower order folios in - * two ways: uniform split or non-uniform split. + * __split_frozen_folio() - splits a frozen @folio to lower order folios + * in two ways: uniform split or non-uniform split. * @folio: the to-be-split folio * @new_order: the smallest order of the after split folios (since buddy * allocator like split generates folios with orders from @folio's @@ -3850,7 +3850,7 @@ static void __split_folio_to_order(struct folio *folio, int old_order, * Return: 0 - successful, <0 - failed (if -ENOMEM is returned, @folio might be * split but not to @new_order, the caller needs to check) */ -static int __split_unmapped_folio(struct folio *folio, int new_order, +static int __split_frozen_folio(struct folio *folio, int new_order, struct page *split_at, struct xa_state *xas, struct address_space *mapping, enum split_type split_type) { @@ -3860,6 +3860,9 @@ static int __split_unmapped_folio(struct folio *folio, int new_order, struct folio *old_folio = folio; int split_order; + /* Frozen implies unmapped, callers unmap before splitting. */ + VM_WARN_ON_ONCE_FOLIO(folio_mapped(folio), folio); + /* * split to new_order one order at a time. For uniform split, * folio is split to new_order directly. @@ -4037,8 +4040,8 @@ static int __folio_freeze_split_anon(struct folio *folio, if (do_lru) lruvec = folio_lruvec_lock(folio); - ret = __split_unmapped_folio(folio, new_order, split_at, NULL, - NULL, split_type); + ret = __split_frozen_folio(folio, new_order, split_at, NULL, + NULL, split_type); /* * Unfreeze the after-split folios and put them back to the right @@ -4104,8 +4107,8 @@ static int __folio_freeze_split_file(struct folio *folio, if (do_lru) lruvec = folio_lruvec_lock(folio); - ret = __split_unmapped_folio(folio, new_order, split_at, xas, - mapping, split_type); + ret = __split_frozen_folio(folio, new_order, split_at, xas, + mapping, split_type); /* * Unfreeze after-split folios and put them back to the right @@ -4169,9 +4172,9 @@ static int __folio_freeze_split_file(struct folio *folio, * @list: after-split folios will be put on it if non NULL * @split_type: perform uniform split or not (non-uniform split) * - * It calls __split_unmapped_folio() to perform uniform and non-uniform split. + * It calls __split_frozen_folio() to perform uniform and non-uniform split. * It is in charge of checking whether the split is supported or not and - * preparing @folio for __split_unmapped_folio(). + * preparing @folio for __split_frozen_folio(). * * After splitting, the after-split folio containing @lock_at remains locked * and others are unlocked: @@ -4274,7 +4277,7 @@ static int __folio_split(struct folio *folio, unsigned int new_order, i_mmap_lock_read(mapping); /* - *__split_unmapped_folio() may need to trim off pages beyond + * __split_frozen_folio() may need to trim off pages beyond * EOF: but on 32-bit, i_size_read() takes an irq-unsafe * seqlock, which cannot be nested inside the page tree lock. * So note end now: i_size itself may be changed at any moment, From 002df20df840e4ec41012cf3359c833609ec06e9 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:37 +0200 Subject: [PATCH 0951/1352] mm/huge_memory: consolidate irq and locking for folio split Let each split helper handle its own locking instead of relying on the caller, so both helpers manage their own irq and locking state. This lets __folio_split() drop its local irq handling and fail label, preparing for further cleanup. The file path now uses xas_lock_irq() instead of local_irq_disable() with xas_lock(). The two are equivalent on non-RT, and TRANSPARENT_HUGEPAGE cannot be enabled on RT anyway. This conversion also buys consistency: every other place in mm/ that freezes a folio while it is still reachable through the page cache already takes the lock this way. This was actually the last plain xas_lock() on mapping->i_pages left in mm. If we are going to support RT, spinning on frozen folio refs could be a problem, but it already exists in many places and should be fixed generically. The anon helper keeps a single local_irq_disable() as before, because it has to cover several plain spinlocks at once. The dropped xas_reset() was a no-op as the xa_state is not walked before the xas_load() under the lock. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-6-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Kiryl Shutsemau (Meta) Reviewed-by: Yeoreum Yun Acked-by: David Hildenbrand (Arm) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 58 ++++++++++++++++++++++-------------------------- 1 file changed, 27 insertions(+), 31 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index b66b855df6f832..5460cf63a84fbb 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -4020,8 +4020,12 @@ static int __folio_freeze_split_anon(struct folio *folio, struct lruvec *lruvec; int ret = 0; - if (!folio_ref_freeze(folio, folio_cache_ref_count(folio) + 1)) + local_irq_disable(); + + if (!folio_ref_freeze(folio, folio_cache_ref_count(folio) + 1)) { + local_irq_enable(); return -EAGAIN; + } /* Take off the deferred split queue while frozen and memcg set */ folio_unqueue_deferred_split(folio); @@ -4069,6 +4073,7 @@ static int __folio_freeze_split_anon(struct folio *folio, lruvec_unlock(lruvec); if (ci) swap_cluster_unlock(ci); + local_irq_enable(); return ret; } @@ -4087,8 +4092,21 @@ static int __folio_freeze_split_file(struct folio *folio, /* Currently device private folios can only back anonymous memory. */ VM_WARN_ON_ONCE_FOLIO(folio_is_device_private(folio), folio); - if (!folio_ref_freeze(folio, folio_cache_ref_count(folio) + 1)) - return -EAGAIN; + xas_lock_irq(xas); + + /* + * Check if the folio is present in page cache. + * We assume all tail are present too, if folio is there. + */ + if (xas_load(xas) != folio) { + ret = -EAGAIN; + goto fail; + } + + if (!folio_ref_freeze(folio, folio_cache_ref_count(folio) + 1)) { + ret = -EAGAIN; + goto fail; + } if (folio_test_pmd_mappable(folio) && new_order < HPAGE_PMD_ORDER) { @@ -4160,6 +4178,8 @@ static int __folio_freeze_split_file(struct folio *folio, if (do_lru) lruvec_unlock(lruvec); +fail: + xas_unlock_irq(xas); return ret; } @@ -4299,32 +4319,13 @@ static int __folio_split(struct folio *folio, unsigned int new_order, unmap_folio(folio); - /* block interrupt reentry in xa_lock and spinlock */ - local_irq_disable(); - if (is_anon) { + if (is_anon) ret = __folio_freeze_split_anon(folio, new_order, split_at, true, list, split_type); - } else { - /* - * Check if the folio is present in page cache. - * We assume all tail are present too, if folio is there. - */ - xas_lock(&xas); - xas_reset(&xas); - if (xas_load(&xas) != folio) { - ret = -EAGAIN; - goto fail; - } + else ret = __folio_freeze_split_file(folio, new_order, split_at, &xas, mapping, true, list, split_type, end, &nr_shmem_dropped); - } - -fail: - if (mapping) - xas_unlock(&xas); - - local_irq_enable(); if (nr_shmem_dropped) shmem_uncharge(mapping->host, nr_shmem_dropped); @@ -4408,8 +4409,6 @@ static int __folio_split(struct folio *folio, unsigned int new_order, */ int folio_split_unmapped(struct folio *folio, unsigned int new_order) { - int ret = 0; - VM_WARN_ON_ONCE_FOLIO(folio_mapped(folio), folio); VM_WARN_ON_ONCE_FOLIO(!folio_test_locked(folio), folio); VM_WARN_ON_ONCE_FOLIO(!folio_test_large(folio), folio); @@ -4418,11 +4417,8 @@ int folio_split_unmapped(struct folio *folio, unsigned int new_order) if (folio_expected_ref_count(folio) != folio_ref_count(folio) - 1) return -EAGAIN; - local_irq_disable(); - ret = __folio_freeze_split_anon(folio, new_order, &folio->page, - false, NULL, SPLIT_TYPE_UNIFORM); - local_irq_enable(); - return ret; + return __folio_freeze_split_anon(folio, new_order, &folio->page, + false, NULL, SPLIT_TYPE_UNIFORM); } /* From 3b570a3f04a0836dbdcbdf88e6712c3889df7303 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:38 +0200 Subject: [PATCH 0952/1352] mm/huge_memory: move EOF trimming into the file split helper Instead of receiving @end and @nr_shmem_dropped from the caller, the file split helper now computes the EOF boundary and trims pages beyond it itself, as this is only needed for file split. This drops the redundant parameter passing and sanity check. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-7-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Acked-by: David Hildenbrand (Arm) Reviewed-by: Kiryl Shutsemau (Meta) Reviewed-by: Yeoreum Yun Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 41 +++++++++++++++++++---------------------- 1 file changed, 19 insertions(+), 22 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 5460cf63a84fbb..8c621131ee0662 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -4082,16 +4082,29 @@ static int __folio_freeze_split_file(struct folio *folio, unsigned int new_order, struct page *split_at, struct xa_state *xas, struct address_space *mapping, bool do_lru, struct list_head *list, - enum split_type split_type, pgoff_t end, int *nr_shmem_dropped) + enum split_type split_type) { struct folio *end_folio = folio_next(folio); struct folio *new_folio, *next; + int nr_shmem_dropped = 0; struct lruvec *lruvec; + pgoff_t end; int ret; /* Currently device private folios can only back anonymous memory. */ VM_WARN_ON_ONCE_FOLIO(folio_is_device_private(folio), folio); + /* + * The loop below may need to trim off pages beyond + * EOF: but on 32-bit, i_size_read() takes an irq-unsafe + * seqlock, which cannot be nested inside the page tree lock. + * So note end now: i_size itself may be changed at any moment, + * but folio lock is good enough to serialize the trimming. + */ + end = DIV_ROUND_UP(i_size_read(mapping->host), PAGE_SIZE); + if (shmem_mapping(mapping)) + end = shmem_fallocend(mapping->host, end); + xas_lock_irq(xas); /* @@ -4156,10 +4169,9 @@ static int __folio_freeze_split_file(struct folio *folio, continue; } - VM_WARN_ON_ONCE(!nr_shmem_dropped); /* Drop folio beyond EOF: ->index >= end */ - if (shmem_mapping(mapping) && nr_shmem_dropped) - *nr_shmem_dropped += nr_pages; + if (shmem_mapping(mapping)) + nr_shmem_dropped += nr_pages; else if (folio_test_clear_dirty(new_folio)) folio_account_cleaned( new_folio, inode_to_wb(mapping->host)); @@ -4180,6 +4192,8 @@ static int __folio_freeze_split_file(struct folio *folio, fail: xas_unlock_irq(xas); + if (nr_shmem_dropped) + shmem_uncharge(mapping->host, nr_shmem_dropped); return ret; } @@ -4216,9 +4230,7 @@ static int __folio_split(struct folio *folio, unsigned int new_order, struct anon_vma *anon_vma = NULL; int old_order = folio_order(folio); struct folio *new_folio, *next; - int nr_shmem_dropped = 0; enum ttu_flags ttu_flags = 0; - pgoff_t end = 0; int ret; VM_WARN_ON_ONCE_FOLIO(!folio_test_locked(folio), folio); @@ -4295,17 +4307,6 @@ static int __folio_split(struct folio *folio, unsigned int new_order, anon_vma = NULL; i_mmap_lock_read(mapping); - - /* - * __split_frozen_folio() may need to trim off pages beyond - * EOF: but on 32-bit, i_size_read() takes an irq-unsafe - * seqlock, which cannot be nested inside the page tree lock. - * So note end now: i_size itself may be changed at any moment, - * but folio lock is good enough to serialize the trimming. - */ - end = DIV_ROUND_UP(i_size_read(mapping->host), PAGE_SIZE); - if (shmem_mapping(mapping)) - end = shmem_fallocend(mapping->host, end); } /* @@ -4324,11 +4325,7 @@ static int __folio_split(struct folio *folio, unsigned int new_order, true, list, split_type); else ret = __folio_freeze_split_file(folio, new_order, split_at, &xas, mapping, - true, list, split_type, end, - &nr_shmem_dropped); - - if (nr_shmem_dropped) - shmem_uncharge(mapping->host, nr_shmem_dropped); + true, list, split_type); if (!ret && is_anon && !folio_is_device_private(folio)) ttu_flags = TTU_USE_SHARED_ZEROPAGE; From 5f9a5463554c3e4249b17671e36da9c484b13765 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:39 +0200 Subject: [PATCH 0953/1352] mm/huge_memory: move unmap and remap into the split helpers To prepare for further cleanup, move the unmap/remap handling from __folio_split() into the split helpers. Only anon folios need to be remapped, so remap_page() is now only called for anon splits and the anon check in remap_page() is redundant and can be removed. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-8-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Yeoreum Yun Reviewed-by: Kiryl Shutsemau (Meta) Acked-by: David Hildenbrand (Arm) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 32 ++++++++++++++++++-------------- 1 file changed, 18 insertions(+), 14 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 8c621131ee0662..8b041460d977f3 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -3640,9 +3640,6 @@ static void remap_page(struct folio *folio, unsigned long nr, int flags) { int i = 0; - /* If unmap_folio() uses try_to_migrate() on file, remove this check */ - if (!folio_test_anon(folio)) - return; for (;;) { remove_migration_ptes(folio, folio, TTU_RMAP_LOCKED | flags); i += folio_nr_pages(folio); @@ -4016,15 +4013,23 @@ static int __folio_freeze_split_anon(struct folio *folio, { struct folio *end_folio = folio_next(folio); struct swap_cluster_info *ci = NULL; + const int old_order = folio_order(folio); struct folio *new_folio, *next; + enum ttu_flags ttu_flags = 0; struct lruvec *lruvec; + bool need_remap = false; int ret = 0; + if (folio_mapped(folio)) { + need_remap = true; + unmap_folio(folio); + } + local_irq_disable(); if (!folio_ref_freeze(folio, folio_cache_ref_count(folio) + 1)) { - local_irq_enable(); - return -EAGAIN; + ret = -EAGAIN; + goto out_no_split; } /* Take off the deferred split queue while frozen and memcg set */ @@ -4073,7 +4078,13 @@ static int __folio_freeze_split_anon(struct folio *folio, lruvec_unlock(lruvec); if (ci) swap_cluster_unlock(ci); +out_no_split: local_irq_enable(); + if (need_remap) { + if (!ret && !folio_is_device_private(folio)) + ttu_flags = TTU_USE_SHARED_ZEROPAGE; + remap_page(folio, 1 << old_order, ttu_flags); + } return ret; } @@ -4105,6 +4116,8 @@ static int __folio_freeze_split_file(struct folio *folio, if (shmem_mapping(mapping)) end = shmem_fallocend(mapping->host, end); + unmap_folio(folio); + xas_lock_irq(xas); /* @@ -4189,7 +4202,6 @@ static int __folio_freeze_split_file(struct folio *folio, if (do_lru) lruvec_unlock(lruvec); - fail: xas_unlock_irq(xas); if (nr_shmem_dropped) @@ -4230,7 +4242,6 @@ static int __folio_split(struct folio *folio, unsigned int new_order, struct anon_vma *anon_vma = NULL; int old_order = folio_order(folio); struct folio *new_folio, *next; - enum ttu_flags ttu_flags = 0; int ret; VM_WARN_ON_ONCE_FOLIO(!folio_test_locked(folio), folio); @@ -4318,8 +4329,6 @@ static int __folio_split(struct folio *folio, unsigned int new_order, goto out_unlock; } - unmap_folio(folio); - if (is_anon) ret = __folio_freeze_split_anon(folio, new_order, split_at, true, list, split_type); @@ -4327,11 +4336,6 @@ static int __folio_split(struct folio *folio, unsigned int new_order, ret = __folio_freeze_split_file(folio, new_order, split_at, &xas, mapping, true, list, split_type); - if (!ret && is_anon && !folio_is_device_private(folio)) - ttu_flags = TTU_USE_SHARED_ZEROPAGE; - - remap_page(folio, 1 << old_order, ttu_flags); - /* * Drop the mapping while the inode is still pinned. @folio stays * locked and present in the page cache until the loop below, so From e8085a907f788912daedde061b34c949fefbb3e1 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:40 +0200 Subject: [PATCH 0954/1352] mm/huge_memory: rename remap_page() to remap_anon_folio() remap_page() now only has one caller, __folio_freeze_split_anon(), and is only ever called for anon folios: unmap_folio() currently leaves file folios unmapped after the split, so they need no remapping. Rename it to remap_anon_folio() to make that explicit, and add a VM_WARN_ON_FOLIO() documenting it. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-9-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Kiryl Shutsemau (Meta) Reviewed-by: Yeoreum Yun Acked-by: David Hildenbrand (Arm) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 14 ++++++++++---- 1 file changed, 10 insertions(+), 4 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 8b041460d977f3..09de66b6b76e07 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -3551,7 +3551,6 @@ static void unmap_folio(struct folio *folio) /* * Anon pages need migration entries to preserve them, but file * pages can simply be left unmapped, then faulted back on demand. - * If that is ever changed (perhaps for mlock), update remap_page(). */ if (folio_test_anon(folio)) try_to_migrate(folio, ttu_flags); @@ -3636,10 +3635,17 @@ bool unmap_huge_pmd_locked(struct vm_area_struct *vma, unsigned long addr, return __discard_anon_folio_pmd_locked(vma, addr, pmdp, folio); } -static void remap_page(struct folio *folio, unsigned long nr, int flags) +static void remap_anon_folio(struct folio *folio, unsigned long nr, int flags) { int i = 0; + /* + * unmap_folio() installs migration entries only for anon folios, + * so currently only anon folios need to be remapped. File folios + * stay unmapped after the split and are faulted back on demand. + */ + VM_WARN_ON_FOLIO(!folio_test_anon(folio), folio); + for (;;) { remove_migration_ptes(folio, folio, TTU_RMAP_LOCKED | flags); i += folio_nr_pages(folio); @@ -3723,7 +3729,7 @@ static void __split_folio_to_order(struct folio *folio, int old_order, * * Note that for mapped sub-pages of an anonymous THP, * PG_anon_exclusive has been cleared in unmap_folio() and is stored in - * the migration entry instead from where remap_page() will restore it. + * the migration entry instead from where remap_anon_folio() will restore it. * We can still have PG_anon_exclusive set on effectively unmapped and * unreferenced sub-pages of an anonymous THP: we can simply drop * PG_anon_exclusive (-> PG_mappedtodisk) for these here. @@ -4083,7 +4089,7 @@ static int __folio_freeze_split_anon(struct folio *folio, if (need_remap) { if (!ret && !folio_is_device_private(folio)) ttu_flags = TTU_USE_SHARED_ZEROPAGE; - remap_page(folio, 1 << old_order, ttu_flags); + remap_anon_folio(folio, 1 << old_order, ttu_flags); } return ret; From caf99b7451954e53180a92c1c7098943f5de870e Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:41 +0200 Subject: [PATCH 0955/1352] mm/huge_memory: move the racy refcount check into unmap_folio() The check only exists to avoid the expensive PMD-splitting unmap of a folio that cannot be split anyway. Move it from __folio_split() and folio_split_unmapped() into one check in unmap_folio(), right before the PMD split. All split helpers get the same early check without repeating it. unmap_folio() now returns -EAGAIN if the check fails and the split helpers propagate the error. folio_split_unmapped() drops its own copy of the check: it works on already unmapped folios and the definitive folio_ref_freeze() in __folio_freeze_split_anon() still catches unexpected references. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-10-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Kiryl Shutsemau (Meta) Reviewed-by: Yeoreum Yun Acked-by: David Hildenbrand (Arm) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 34 ++++++++++++++++++---------------- 1 file changed, 18 insertions(+), 16 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 09de66b6b76e07..ad92c2b554a044 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -3538,13 +3538,22 @@ void vma_adjust_trans_huge(struct vm_area_struct *vma, split_huge_pmd_if_needed(next, end); } -static void unmap_folio(struct folio *folio) +/* + * A return value of 0 does not mean that unmapping succeeded. It might + * still have failed, but remap_anon_folio() must be called afterwards, + * for anon folios. + */ +static int unmap_folio(struct folio *folio) { enum ttu_flags ttu_flags = TTU_RMAP_LOCKED | TTU_SYNC | TTU_BATCH_FLUSH; VM_BUG_ON_FOLIO(!folio_test_large(folio), folio); + /* Racy check if we can split the page, before we split PMDs */ + if (folio_expected_ref_count(folio) != folio_ref_count(folio) - 1) + return -EAGAIN; + if (folio_test_pmd_mappable(folio)) ttu_flags |= TTU_SPLIT_HUGE_PMD; @@ -3558,6 +3567,8 @@ static void unmap_folio(struct folio *folio) try_to_unmap(folio, ttu_flags | TTU_IGNORE_MLOCK); try_to_unmap_flush(); + + return 0; } static bool __discard_anon_folio_pmd_locked(struct vm_area_struct *vma, @@ -4028,7 +4039,9 @@ static int __folio_freeze_split_anon(struct folio *folio, if (folio_mapped(folio)) { need_remap = true; - unmap_folio(folio); + ret = unmap_folio(folio); + if (ret) + return ret; } local_irq_disable(); @@ -4122,7 +4135,9 @@ static int __folio_freeze_split_file(struct folio *folio, if (shmem_mapping(mapping)) end = shmem_fallocend(mapping->host, end); - unmap_folio(folio); + ret = unmap_folio(folio); + if (ret) + return ret; xas_lock_irq(xas); @@ -4326,15 +4341,6 @@ static int __folio_split(struct folio *folio, unsigned int new_order, i_mmap_lock_read(mapping); } - /* - * Racy check if we can split the page, before unmap_folio() will - * split PMDs - */ - if (folio_expected_ref_count(folio) != folio_ref_count(folio) - 1) { - ret = -EAGAIN; - goto out_unlock; - } - if (is_anon) ret = __folio_freeze_split_anon(folio, new_order, split_at, true, list, split_type); @@ -4373,7 +4379,6 @@ static int __folio_split(struct folio *folio, unsigned int new_order, free_folio_and_swap_cache(new_folio); } -out_unlock: if (anon_vma) { anon_vma_unlock_write(anon_vma); put_anon_vma(anon_vma); @@ -4421,9 +4426,6 @@ int folio_split_unmapped(struct folio *folio, unsigned int new_order) VM_WARN_ON_ONCE_FOLIO(!folio_test_large(folio), folio); VM_WARN_ON_ONCE_FOLIO(!folio_test_anon(folio), folio); - if (folio_expected_ref_count(folio) != folio_ref_count(folio) - 1) - return -EAGAIN; - return __folio_freeze_split_anon(folio, new_order, &folio->page, false, NULL, SPLIT_TYPE_UNIFORM); } From 79262b74850d6e17b6dfe0a80ce03ca0e153f30e Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:42 +0200 Subject: [PATCH 0956/1352] mm/huge_memory: move filemap management into the file split helper Only file split needs the filemap and xarray handling and related variables. Move them out of __folio_split() into the file helper so the helper is self-contained, and simplify the parameters. No functional change. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-11-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Kiryl Shutsemau (Meta) Reviewed-by: Yeoreum Yun Acked-by: David Hildenbrand (Arm) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 102 ++++++++++++++++++++--------------------------- 1 file changed, 44 insertions(+), 58 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index ad92c2b554a044..6457192747aa6e 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -4110,16 +4110,42 @@ static int __folio_freeze_split_anon(struct folio *folio, static int __folio_freeze_split_file(struct folio *folio, unsigned int new_order, struct page *split_at, - struct xa_state *xas, struct address_space *mapping, bool do_lru, struct list_head *list, enum split_type split_type) { + struct address_space *mapping = folio->mapping; + XA_STATE(xas, &mapping->i_pages, folio->index); struct folio *end_folio = folio_next(folio); struct folio *new_folio, *next; int nr_shmem_dropped = 0; + unsigned int min_order; struct lruvec *lruvec; pgoff_t end; - int ret; + gfp_t gfp; + int ret = 0; + + min_order = mapping_min_folio_order(mapping); + if (new_order < min_order) + return -EINVAL; + + gfp = current_gfp_context(mapping_gfp_mask(mapping) & GFP_RECLAIM_MASK); + if (!filemap_release_folio(folio, gfp)) + return -EBUSY; + + mapping_set_update(&xas, mapping); + + if (split_type == SPLIT_TYPE_UNIFORM) { + const int old_order = folio_order(folio); + + xas_set_order(&xas, folio->index, new_order); + xas_split_alloc(&xas, folio, old_order, gfp); + if (xas_error(&xas)) { + ret = xas_error(&xas); + goto fail_free; + } + } + + i_mmap_lock_read(mapping); /* Currently device private folios can only back anonymous memory. */ VM_WARN_ON_ONCE_FOLIO(folio_is_device_private(folio), folio); @@ -4137,15 +4163,15 @@ static int __folio_freeze_split_file(struct folio *folio, ret = unmap_folio(folio); if (ret) - return ret; + goto fail_mmap_unlock; - xas_lock_irq(xas); + xas_lock_irq(&xas); /* * Check if the folio is present in page cache. * We assume all tail are present too, if folio is there. */ - if (xas_load(xas) != folio) { + if (xas_load(&xas) != folio) { ret = -EAGAIN; goto fail; } @@ -4172,7 +4198,7 @@ static int __folio_freeze_split_file(struct folio *folio, if (do_lru) lruvec = folio_lruvec_lock(folio); - ret = __split_frozen_folio(folio, new_order, split_at, xas, + ret = __split_frozen_folio(folio, new_order, split_at, &xas, mapping, split_type); /* @@ -4224,9 +4250,19 @@ static int __folio_freeze_split_file(struct folio *folio, if (do_lru) lruvec_unlock(lruvec); fail: - xas_unlock_irq(xas); + xas_unlock_irq(&xas); +fail_mmap_unlock: if (nr_shmem_dropped) shmem_uncharge(mapping->host, nr_shmem_dropped); + /* + * Drop the mapping while the inode is still pinned. @folio stays + * locked and present in the page cache, so eviction cannot free + * the inode yet, nothing past this point may touch the inode or + * the mapping. + */ + i_mmap_unlock_read(mapping); +fail_free: + xas_destroy(&xas); return ret; } @@ -4255,11 +4291,9 @@ static int __folio_split(struct folio *folio, unsigned int new_order, struct page *split_at, struct page *lock_at, struct list_head *list, enum split_type split_type) { - XA_STATE(xas, &folio->mapping->i_pages, folio->index); struct folio *end_folio = folio_next(folio); bool is_anon = folio_test_anon(folio); struct mem_cgroup *memcg, *old_memcg; - struct address_space *mapping = NULL; struct anon_vma *anon_vma = NULL; int old_order = folio_order(folio); struct folio *new_folio, *next; @@ -4306,60 +4340,15 @@ static int __folio_split(struct folio *folio, unsigned int new_order, goto out; } anon_vma_lock_write(anon_vma); - mapping = NULL; - } else { - unsigned int min_order; - gfp_t gfp; - - mapping = folio->mapping; - min_order = mapping_min_folio_order(mapping); - if (new_order < min_order) { - ret = -EINVAL; - goto out; - } - - gfp = current_gfp_context(mapping_gfp_mask(mapping) & - GFP_RECLAIM_MASK); - - if (!filemap_release_folio(folio, gfp)) { - ret = -EBUSY; - goto out; - } - - mapping_set_update(&xas, mapping); - - if (split_type == SPLIT_TYPE_UNIFORM) { - xas_set_order(&xas, folio->index, new_order); - xas_split_alloc(&xas, folio, old_order, gfp); - if (xas_error(&xas)) { - ret = xas_error(&xas); - goto out; - } - } - - anon_vma = NULL; - i_mmap_lock_read(mapping); } if (is_anon) ret = __folio_freeze_split_anon(folio, new_order, split_at, true, list, split_type); else - ret = __folio_freeze_split_file(folio, new_order, split_at, &xas, mapping, + ret = __folio_freeze_split_file(folio, new_order, split_at, true, list, split_type); - /* - * Drop the mapping while the inode is still pinned. @folio stays - * locked and present in the page cache until the loop below, so - * eviction cannot free the inode yet; @lock_at is not enough, it may - * be a tail beyond EOF that the split already dropped from the page - * cache. Nothing past this point may touch the inode or the mapping. - */ - if (mapping) { - i_mmap_unlock_read(mapping); - mapping = NULL; - } - /* * Unlock all after-split folios except the one containing * @lock_at page. If @folio is not split, it will be kept locked. @@ -4383,14 +4372,11 @@ static int __folio_split(struct folio *folio, unsigned int new_order, anon_vma_unlock_write(anon_vma); put_anon_vma(anon_vma); } - if (mapping) - i_mmap_unlock_read(mapping); out: /* restore to caller's old_memcg */ set_active_memcg(old_memcg); mem_cgroup_put(memcg); out_no_memcg: - xas_destroy(&xas); if (is_pmd_order(old_order)) count_vm_event(!ret ? THP_SPLIT_PAGE : THP_SPLIT_PAGE_FAILED); count_mthp_stat(old_order, !ret ? MTHP_STAT_SPLIT : MTHP_STAT_SPLIT_FAILED); From b1002f27eb9fd1f0db8784e2ee5ea9e0480a9adc Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:43 +0200 Subject: [PATCH 0957/1352] mm/huge_memory: move anon_vma handling into the anon split helper Only anon split needs the anon_vma, and it only needs it to unmap and remap. Move the folio_get_anon_vma()/anon_vma_lock_write() pair out of __folio_split() into the anon helper next to the folio_mapped() check that already gates unmap_folio(). This makes the anon_vma conditional on folio_mapped(), which is a behaviour change but should be fine. folio_get_anon_vma() returns NULL whenever !folio_mapped(), so an anon folio with folio_mapcount() == 0 used to get -EBUSY from split_huge_page() and is now split instead. A realistic case is a THP that has been fully swapped out and is still in the swap cache: swap PTEs do not contribute mapcount, so it is !folio_mapped() but still alive. That should be safe and right to have because: - folio_ref_freeze() below still rejects a folio that picked up any reference, a mapping or a GUP pin, in the meantime. - A parallel split is excluded by the folio lock. The anon_vma write lock was added to serialize split in commit 062f1af2170a ("mm: thp: acquire the anon_vma rwsem for write during split"), when split_huge_page() did not hold the folio lock throughout. commit e9b61f19858a ("thp: reintroduce split_huge_page()") later made the folio lock a caller requirement and added the folio_ref_freeze() scheme, so that has been covered ever since. - Unmapped path is already exercised by folio_split_unmapped(), and the swap cache split already runs well for a partially swapped-out mapped THP. - A !folio_mapped() folio cannot become mapped meanwhile: mapping it requires the folio lock. For mapped folios the anon_vma write lock is now released before __folio_split() unlocks the after-split sub-folios, where previously it was held across that loop; that window is harmless as the sub-folios stay folio-locked and referenced until it. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-12-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Yeoreum Yun Acked-by: David Hildenbrand (Arm) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Kiryl Shutsemau Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 57 +++++++++++++++++++++++++----------------------- 1 file changed, 30 insertions(+), 27 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 6457192747aa6e..ec955488d280f9 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -4032,16 +4032,37 @@ static int __folio_freeze_split_anon(struct folio *folio, struct swap_cluster_info *ci = NULL; const int old_order = folio_order(folio); struct folio *new_folio, *next; + struct anon_vma *anon_vma = NULL; enum ttu_flags ttu_flags = 0; struct lruvec *lruvec; - bool need_remap = false; int ret = 0; + /* + * Unmap/remap needs the anon_vma, so we first take a reference on + * it to prevent it from disappearing, and lock it for write here, + * letting unmap_folio() walk the rmap with TTU_RMAP_LOCKED. + * + * folio_mapped() is not stable here, but it can only change in + * one direction while the folio is locked. The mapcount can drop + * to zero at any time, zap_pte_range() takes no folio lock. It + * cannot go up: swapin, migration and uffd move all lock the folio + * before mapping it, and fork only copies PTEs that already exist. + * + * So if we see the folio mapped, the worst case is an empty rmap + * walk. If we see it unmapped, it stays unmapped and needs neither + * the reference nor the lock. Anything else needs a reference + * first and folio_ref_freeze() below catches it. + * + * Note that entirely swapped-out THPs are unmapped but can be split. + */ if (folio_mapped(folio)) { - need_remap = true; + anon_vma = folio_get_anon_vma(folio); + if (!anon_vma) + return -EBUSY; + anon_vma_lock_write(anon_vma); ret = unmap_folio(folio); if (ret) - return ret; + goto out_unlock; } local_irq_disable(); @@ -4099,11 +4120,16 @@ static int __folio_freeze_split_anon(struct folio *folio, swap_cluster_unlock(ci); out_no_split: local_irq_enable(); - if (need_remap) { + if (anon_vma) { if (!ret && !folio_is_device_private(folio)) ttu_flags = TTU_USE_SHARED_ZEROPAGE; remap_anon_folio(folio, 1 << old_order, ttu_flags); } +out_unlock: + if (anon_vma) { + anon_vma_unlock_write(anon_vma); + put_anon_vma(anon_vma); + } return ret; } @@ -4294,7 +4320,6 @@ static int __folio_split(struct folio *folio, unsigned int new_order, struct folio *end_folio = folio_next(folio); bool is_anon = folio_test_anon(folio); struct mem_cgroup *memcg, *old_memcg; - struct anon_vma *anon_vma = NULL; int old_order = folio_order(folio); struct folio *new_folio, *next; int ret; @@ -4325,23 +4350,6 @@ static int __folio_split(struct folio *folio, unsigned int new_order, memcg = get_mem_cgroup_from_folio(folio); old_memcg = set_active_memcg(memcg); - if (is_anon) { - /* - * The caller does not necessarily hold an mmap_lock that would - * prevent the anon_vma disappearing so we first we take a - * reference to it and then lock the anon_vma for write. This - * is similar to folio_lock_anon_vma_read except the write lock - * is taken to serialise against parallel split or collapse - * operations. - */ - anon_vma = folio_get_anon_vma(folio); - if (!anon_vma) { - ret = -EBUSY; - goto out; - } - anon_vma_lock_write(anon_vma); - } - if (is_anon) ret = __folio_freeze_split_anon(folio, new_order, split_at, true, list, split_type); @@ -4368,11 +4376,6 @@ static int __folio_split(struct folio *folio, unsigned int new_order, free_folio_and_swap_cache(new_folio); } - if (anon_vma) { - anon_vma_unlock_write(anon_vma); - put_anon_vma(anon_vma); - } -out: /* restore to caller's old_memcg */ set_active_memcg(old_memcg); mem_cgroup_put(memcg); From ef1b8825b72a64b96679f50e1de282c5a49f17c7 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:44 +0200 Subject: [PATCH 0958/1352] mm/huge_memory: move memcg switch into the file split helper The xarray node allocations in __folio_freeze_split_file() need to be charged to the folio's memcg, so move the memcg switch from __folio_split() into the helper. The anon split helper and the after-split folio freeing perform no chargeable allocations, so no memcg handling is left in __folio_split(). Rename its out_no_memcg label to out. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-13-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Acked-by: Zi Yan Acked-by: David Hildenbrand (Arm) Reviewed-by: Yeoreum Yun Reviewed-by: Kiryl Shutsemau (Meta) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 36 +++++++++++++++++++----------------- 1 file changed, 19 insertions(+), 17 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index ec955488d280f9..46d00bf6e43796 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -4142,6 +4142,7 @@ static int __folio_freeze_split_file(struct folio *folio, struct address_space *mapping = folio->mapping; XA_STATE(xas, &mapping->i_pages, folio->index); struct folio *end_folio = folio_next(folio); + struct mem_cgroup *memcg, *old_memcg; struct folio *new_folio, *next; int nr_shmem_dropped = 0; unsigned int min_order; @@ -4154,9 +4155,18 @@ static int __folio_freeze_split_file(struct folio *folio, if (new_order < min_order) return -EINVAL; + /* + * Switch to folio's memcg as xarray node allocation can happen and + * needs to charge to it. + */ + memcg = get_mem_cgroup_from_folio(folio); + old_memcg = set_active_memcg(memcg); + gfp = current_gfp_context(mapping_gfp_mask(mapping) & GFP_RECLAIM_MASK); - if (!filemap_release_folio(folio, gfp)) - return -EBUSY; + if (!filemap_release_folio(folio, gfp)) { + ret = -EBUSY; + goto fail_free; + } mapping_set_update(&xas, mapping); @@ -4288,6 +4298,9 @@ static int __folio_freeze_split_file(struct folio *folio, */ i_mmap_unlock_read(mapping); fail_free: + /* Restore the previously active memcg */ + set_active_memcg(old_memcg); + mem_cgroup_put(memcg); xas_destroy(&xas); return ret; } @@ -4319,7 +4332,6 @@ static int __folio_split(struct folio *folio, unsigned int new_order, { struct folio *end_folio = folio_next(folio); bool is_anon = folio_test_anon(folio); - struct mem_cgroup *memcg, *old_memcg; int old_order = folio_order(folio); struct folio *new_folio, *next; int ret; @@ -4329,27 +4341,20 @@ static int __folio_split(struct folio *folio, unsigned int new_order, if (folio != page_folio(split_at) || folio != page_folio(lock_at)) { ret = -EINVAL; - goto out_no_memcg; + goto out; } if (new_order >= old_order) { ret = -EINVAL; - goto out_no_memcg; + goto out; } ret = folio_check_splittable(folio, new_order, split_type); if (ret) { VM_WARN_ONCE(ret == -EINVAL, "Tried to split an unsplittable folio"); - goto out_no_memcg; + goto out; } - /* - * switch to folio's memcg as xarray node allocation can happen and - * needs to charge to it. - */ - memcg = get_mem_cgroup_from_folio(folio); - old_memcg = set_active_memcg(memcg); - if (is_anon) ret = __folio_freeze_split_anon(folio, new_order, split_at, true, list, split_type); @@ -4376,10 +4381,7 @@ static int __folio_split(struct folio *folio, unsigned int new_order, free_folio_and_swap_cache(new_folio); } - /* restore to caller's old_memcg */ - set_active_memcg(old_memcg); - mem_cgroup_put(memcg); -out_no_memcg: +out: if (is_pmd_order(old_order)) count_vm_event(!ret ? THP_SPLIT_PAGE : THP_SPLIT_PAGE_FAILED); count_mthp_stat(old_order, !ret ? MTHP_STAT_SPLIT : MTHP_STAT_SPLIT_FAILED); From e142994b36372f597628b7602fb867d532c9f5c5 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:45 +0200 Subject: [PATCH 0959/1352] mm/huge_memory: drop the unused do_lru argument of the file split helper The only caller of __folio_freeze_split_file() always passes do_lru as true, so the argument and the branches gated on it are dead code. Drop it. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-14-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Acked-by: David Hildenbrand (Arm) Reviewed-by: Yeoreum Yun Reviewed-by: Kiryl Shutsemau (Meta) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 16 +++++----------- 1 file changed, 5 insertions(+), 11 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 46d00bf6e43796..4bcd57540eea13 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -4136,8 +4136,7 @@ static int __folio_freeze_split_anon(struct folio *folio, static int __folio_freeze_split_file(struct folio *folio, unsigned int new_order, struct page *split_at, - bool do_lru, struct list_head *list, - enum split_type split_type) + struct list_head *list, enum split_type split_type) { struct address_space *mapping = folio->mapping; XA_STATE(xas, &mapping->i_pages, folio->index); @@ -4231,9 +4230,7 @@ static int __folio_freeze_split_file(struct folio *folio, } /* lock lru list/PageCompound, ref frozen by page_ref_freeze */ - if (do_lru) - lruvec = folio_lruvec_lock(folio); - + lruvec = folio_lruvec_lock(folio); ret = __split_frozen_folio(folio, new_order, split_at, &xas, mapping, split_type); @@ -4255,8 +4252,7 @@ static int __folio_freeze_split_file(struct folio *folio, folio_ref_unfreeze(new_folio, folio_cache_ref_count(new_folio) + 1); - if (do_lru) - lru_add_split_folio(folio, new_folio, lruvec, list); + lru_add_split_folio(folio, new_folio, lruvec, list); /* Add the new folio to the page cache. */ if (new_folio->index < end) { @@ -4282,9 +4278,7 @@ static int __folio_freeze_split_file(struct folio *folio, * and its caller can see stale page cache entries. */ folio_ref_unfreeze(folio, folio_cache_ref_count(folio) + 1); - - if (do_lru) - lruvec_unlock(lruvec); + lruvec_unlock(lruvec); fail: xas_unlock_irq(&xas); fail_mmap_unlock: @@ -4360,7 +4354,7 @@ static int __folio_split(struct folio *folio, unsigned int new_order, true, list, split_type); else ret = __folio_freeze_split_file(folio, new_order, split_at, - true, list, split_type); + list, split_type); /* * Unlock all after-split folios except the one containing From 7728a730256557b8382152b96e07fca42e8bb51d Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:46 +0200 Subject: [PATCH 0960/1352] mm/huge_memory: clean up after-split folio freeing in __folio_split Replace free_folio_and_swap_cache() with an explicit folio_free_swap() and folio_put() in the after-split loop. free_folio_and_swap_cache() must trylock it again and re-check folio_mapped() before freeing the swap cache entries. If the trylock loses a race, the entries are left behind even though the folio reference is dropped. The sub folios are still locked here, so just directly call folio_free_swap() under the lock if it's unmapped, then unlock and drop the reference. This makes the swap cache freeing deterministic and the reference drop explicit. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-15-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Yeoreum Yun Reviewed-by: Kiryl Shutsemau (Meta) Acked-by: David Hildenbrand (Arm) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 13 ++++++++----- 1 file changed, 8 insertions(+), 5 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 4bcd57540eea13..2614a4676d7ba6 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -4325,7 +4325,8 @@ static int __folio_split(struct folio *folio, unsigned int new_order, struct list_head *list, enum split_type split_type) { struct folio *end_folio = folio_next(folio); - bool is_anon = folio_test_anon(folio); + const bool is_anon = folio_test_anon(folio); + const bool is_swapcache = folio_test_swapcache(folio); int old_order = folio_order(folio); struct folio *new_folio, *next; int ret; @@ -4365,14 +4366,16 @@ static int __folio_split(struct folio *folio, unsigned int new_order, if (new_folio == page_folio(lock_at)) continue; - folio_unlock(new_folio); /* * Subpages whose mapping has been zapped may be freed * earlier, but freeing them requires taking the - * lru_lock, so we defer put_page() on tail pages until + * lru_lock, so we defer folio_put() on tail pages until * after the split completes. */ - free_folio_and_swap_cache(new_folio); + if (is_swapcache && !folio_mapped(new_folio)) + folio_free_swap(new_folio); + folio_unlock(new_folio); + folio_put(new_folio); } out: @@ -4399,7 +4402,7 @@ static int __folio_split(struct folio *folio, unsigned int new_order, * isolated from LRU (if applicable) * * Upon return, the folio is not remapped, split folios are not added to LRU, - * free_folio_and_swap_cache() is not called, and new folios remain locked. + * folio_free_swap() is not called, and new folios remain locked. * * Return: 0 on success, -EAGAIN if the folio cannot be split (e.g., due to * insufficient reference count or extra pins). From 966217b4676e37a84d6480128f06a8942e72ff71 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:47 +0200 Subject: [PATCH 0961/1352] mm/huge_memory: count only swap cache refs in anon folio split Only __folio_freeze_split_anon() sees anon folios and swap cache folios now. The file split helper only handles page cache folios, which hold exactly folio_nr_pages() references. Rename folio_cache_ref_count() to folio_swapcache_ref_count() and drop the anon check so the helper counts what its name says. The file split helper now uses folio_nr_pages() directly. No feature change. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-16-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Yeoreum Yun Reviewed-by: Kiryl Shutsemau (Meta) Acked-by: David Hildenbrand (Arm) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 35 +++++++++++++++-------------------- 1 file changed, 15 insertions(+), 20 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 2614a4676d7ba6..349c75b832cd9e 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -3997,10 +3997,10 @@ int folio_check_splittable(struct folio *folio, unsigned int new_order, return 0; } -/* Number of folio references from the pagecache or the swapcache. */ -static unsigned int folio_cache_ref_count(const struct folio *folio) +/* Number of folio references from the swapcache. */ +static unsigned int folio_swapcache_ref_count(const struct folio *folio) { - if (folio_test_anon(folio) && !folio_test_swapcache(folio)) + if (!folio_test_swapcache(folio)) return 0; return folio_nr_pages(folio); } @@ -4067,7 +4067,7 @@ static int __folio_freeze_split_anon(struct folio *folio, local_irq_disable(); - if (!folio_ref_freeze(folio, folio_cache_ref_count(folio) + 1)) { + if (!folio_ref_freeze(folio, folio_swapcache_ref_count(folio) + 1)) { ret = -EAGAIN; goto out_no_split; } @@ -4104,7 +4104,7 @@ static int __folio_freeze_split_anon(struct folio *folio, next = folio_next(new_folio); zone_device_private_split_cb(folio, new_folio); folio_ref_unfreeze(new_folio, - folio_cache_ref_count(new_folio) + 1); + folio_swapcache_ref_count(new_folio) + 1); if (do_lru) lru_add_split_folio(folio, new_folio, lruvec, list); if (ci) @@ -4112,7 +4112,7 @@ static int __folio_freeze_split_anon(struct folio *folio, } zone_device_private_split_cb(folio, NULL); - folio_ref_unfreeze(folio, folio_cache_ref_count(folio) + 1); + folio_ref_unfreeze(folio, folio_swapcache_ref_count(folio) + 1); if (do_lru) lruvec_unlock(lruvec); @@ -4138,6 +4138,7 @@ static int __folio_freeze_split_file(struct folio *folio, unsigned int new_order, struct page *split_at, struct list_head *list, enum split_type split_type) { + const long old_nr_pages = folio_nr_pages(folio); struct address_space *mapping = folio->mapping; XA_STATE(xas, &mapping->i_pages, folio->index); struct folio *end_folio = folio_next(folio); @@ -4211,22 +4212,16 @@ static int __folio_freeze_split_file(struct folio *folio, goto fail; } - if (!folio_ref_freeze(folio, folio_cache_ref_count(folio) + 1)) { + if (!folio_ref_freeze(folio, old_nr_pages + 1)) { ret = -EAGAIN; goto fail; } - if (folio_test_pmd_mappable(folio) && - new_order < HPAGE_PMD_ORDER) { - int nr = folio_nr_pages(folio); - - if (folio_test_swapbacked(folio)) { - lruvec_stat_mod_folio(folio, - NR_SHMEM_THPS, -nr); - } else { - lruvec_stat_mod_folio(folio, - NR_FILE_THPS, -nr); - } + if (folio_test_pmd_mappable(folio) && new_order < HPAGE_PMD_ORDER) { + if (folio_test_swapbacked(folio)) + lruvec_stat_mod_folio(folio, NR_SHMEM_THPS, -old_nr_pages); + else + lruvec_stat_mod_folio(folio, NR_FILE_THPS, -old_nr_pages); } /* lock lru list/PageCompound, ref frozen by page_ref_freeze */ @@ -4250,7 +4245,7 @@ static int __folio_freeze_split_file(struct folio *folio, next = folio_next(new_folio); folio_ref_unfreeze(new_folio, - folio_cache_ref_count(new_folio) + 1); + folio_nr_pages(new_folio) + 1); lru_add_split_folio(folio, new_folio, lruvec, list); @@ -4277,7 +4272,7 @@ static int __folio_freeze_split_file(struct folio *folio, * Otherwise, a parallel folio_try_get() can grab @folio * and its caller can see stale page cache entries. */ - folio_ref_unfreeze(folio, folio_cache_ref_count(folio) + 1); + folio_ref_unfreeze(folio, folio_nr_pages(folio) + 1); lruvec_unlock(lruvec); fail: xas_unlock_irq(&xas); From de1dab75d21cc8c695e87b43479330d0c65a69df Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:48 +0200 Subject: [PATCH 0962/1352] mm/huge_memory: drop the redundant mapping argument of __split_frozen_folio The mapping parameter only served as a non-NULL check to detect whether page cache entries need updating. The xa_state pointer conveys exactly the same information: the anon split helper passes NULL and the file split helper passes &xas, which is non-NULL iff the folio is in the page cache. Use the xas pointer instead and drop the parameter, along with its kerneldoc entry. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-17-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Yeoreum Yun Reviewed-by: Kiryl Shutsemau (Meta) Acked-by: David Hildenbrand (Arm) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 11 ++++------- 1 file changed, 4 insertions(+), 7 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 349c75b832cd9e..8f8bf60a22649a 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -3833,7 +3833,6 @@ static void __split_folio_to_order(struct folio *folio, int old_order, * @split_at: in buddy allocator like split, the folio containing @split_at * will be split until its order becomes @new_order. * @xas: xa_state pointing to folio->mapping->i_pages and locked by caller - * @mapping: @folio->mapping * @split_type: if the split is uniform or not (buddy allocator like split) * * @@ -3866,7 +3865,7 @@ static void __split_folio_to_order(struct folio *folio, int old_order, */ static int __split_frozen_folio(struct folio *folio, int new_order, struct page *split_at, struct xa_state *xas, - struct address_space *mapping, enum split_type split_type) + enum split_type split_type) { const bool is_anon = folio_test_anon(folio); int old_order = folio_order(folio); @@ -3890,7 +3889,7 @@ static int __split_frozen_folio(struct folio *folio, int new_order, if (is_anon && split_order == 1) continue; - if (mapping) { + if (xas) { /* * uniform split has xas_split_alloc() called before * irq is disabled to allocate enough memory, whereas @@ -4089,8 +4088,7 @@ static int __folio_freeze_split_anon(struct folio *folio, if (do_lru) lruvec = folio_lruvec_lock(folio); - ret = __split_frozen_folio(folio, new_order, split_at, NULL, - NULL, split_type); + ret = __split_frozen_folio(folio, new_order, split_at, NULL, split_type); /* * Unfreeze the after-split folios and put them back to the right @@ -4226,8 +4224,7 @@ static int __folio_freeze_split_file(struct folio *folio, /* lock lru list/PageCompound, ref frozen by page_ref_freeze */ lruvec = folio_lruvec_lock(folio); - ret = __split_frozen_folio(folio, new_order, split_at, &xas, - mapping, split_type); + ret = __split_frozen_folio(folio, new_order, split_at, &xas, split_type); /* * Unfreeze after-split folios and put them back to the right From 75dd53fdf0728268c7a6d5d9200bc29bd98959c0 Mon Sep 17 00:00:00 2001 From: Guillaume Morin Date: Mon, 14 Sep 2026 17:36:21 +0200 Subject: [PATCH 0963/1352] selftests/mm: hugetlb_madv_vs_map: add underflow test Add a test that checks for underflows when a parent unmaps the page first. Also check that when the child exits the reserve count is correct. Link: https://lore.kernel.org/all/alEJkwn5VlTTH_ZX@bender.morinfr.org/ Link: https://lore.kernel.org/aqgUdbtumaO8RiIb@bender.morinfr.org Signed-off-by: Guillaume Morin Signed-off-by: Andrew Morton Reviewed-by: Breno Leitao Reviewed-by: Mike Rapoport --- .../testing/selftests/mm/hugepage_settings.c | 9 ++ .../testing/selftests/mm/hugepage_settings.h | 1 + .../selftests/mm/hugetlb_madv_vs_map.c | 141 +++++++++++++++--- 3 files changed, 130 insertions(+), 21 deletions(-) diff --git a/tools/testing/selftests/mm/hugepage_settings.c b/tools/testing/selftests/mm/hugepage_settings.c index d7917dce3abac7..584054736ce99f 100644 --- a/tools/testing/selftests/mm/hugepage_settings.c +++ b/tools/testing/selftests/mm/hugepage_settings.c @@ -449,6 +449,15 @@ unsigned long hugetlb_free_pages(unsigned long size) return read_num(path); } +unsigned long hugetlb_nr_resv_pages(unsigned long size) +{ + char path[PATH_MAX]; + + hugetlb_sysfs_path(path, sizeof(path), size, "resv_hugepages"); + + return read_num(path); +} + static bool __hugetlb_setup(unsigned long size, unsigned long nr) { unsigned long free = hugetlb_free_pages(size); diff --git a/tools/testing/selftests/mm/hugepage_settings.h b/tools/testing/selftests/mm/hugepage_settings.h index 726c73c43c05ba..548e9d288d1d16 100644 --- a/tools/testing/selftests/mm/hugepage_settings.h +++ b/tools/testing/selftests/mm/hugepage_settings.h @@ -98,6 +98,7 @@ unsigned long default_huge_page_size(void); unsigned long hugetlb_nr_pages(unsigned long size); void hugetlb_set_nr_pages(unsigned long size, unsigned long nr); unsigned long hugetlb_free_pages(unsigned long size); +unsigned long hugetlb_nr_resv_pages(unsigned long size); static inline void hugetlb_save_settings(void) { diff --git a/tools/testing/selftests/mm/hugetlb_madv_vs_map.c b/tools/testing/selftests/mm/hugetlb_madv_vs_map.c index f94549efcc6ff3..0f15eff1da0403 100644 --- a/tools/testing/selftests/mm/hugetlb_madv_vs_map.c +++ b/tools/testing/selftests/mm/hugetlb_madv_vs_map.c @@ -3,18 +3,6 @@ * A test case that must run on a system with one and only one huge page available. * # echo 1 > /sys/kernel/mm/hugepages/hugepages-2048kB/nr_hugepages * - * During setup, the test allocates the only available page, and starts three threads: - * - thread1: - * * madvise(MADV_DONTNEED) on the allocated huge page - * - thread 2: - * * Write to the allocated huge page - * - thread 3: - * * Try to allocated an extra huge page (which must not available) - * - * The test fails if thread3 is able to allocate a page. - * - * Touching the first page after thread3's allocation will raise a SIGBUS - * * Author: Breno Leitao */ #include @@ -22,6 +10,7 @@ #include #include #include +#include #include #include "vm_util.h" @@ -74,7 +63,21 @@ void *map_extra(void *unused) return NULL; } -int main(void) +/* During setup in main, the only available page was allocated. This test then + * starts three threads: + * + * - thread1: + * * madvise(MADV_DONTNEED) on the allocated huge page + * - thread 2: + * * Write to the allocated huge page + * - thread 3: + * * Try to allocated an extra huge page (which must not available) + * + * The test fails if thread3 is able to allocate a page. + * + * Touching the first page after thread3's allocation will raise a SIGBUS + */ +void test_madv_vs_map(void) { pthread_t thread1, thread2, thread3; void *ret; @@ -85,13 +88,6 @@ int main(void) */ int max = 10; - ksft_print_header(); - ksft_set_plan(1); - - if (!hugetlb_setup_default_exact(1)) - ksft_exit_skip("This test needs one and only one page to execute. Got %lu\n", - hugetlb_free_default_pages()); - mmap_size = default_huge_page_size(); while (max--) { @@ -100,7 +96,7 @@ int main(void) -1, 0); if ((unsigned long)huge_ptr == -1) - ksft_exit_fail_msg("Failed to allocate huge page\n"); + ksft_exit_fail_perror("Failed to allocate huge page"); pthread_create(&thread1, NULL, madv, NULL); pthread_create(&thread2, NULL, touch, NULL); @@ -120,5 +116,108 @@ int main(void) } ksft_test_result_pass("No unexpected huge page allocations\n"); +} + +/* We create a child process, then unmap the page in the parent while the child + * waits and verify that there is no underflow of the reserved count. We also + * verify that after the child exits, the reserved count is properly restored. + */ +void test_underflow(void) +{ + pid_t pid; + int pipe_fds[2]; + unsigned long nr_reserved = 0; + + huge_ptr = mmap(NULL, mmap_size, PROT_READ | PROT_WRITE, + MAP_PRIVATE | MAP_ANONYMOUS | MAP_HUGETLB, -1, 0); + + if ((unsigned long)huge_ptr == -1) + ksft_exit_fail_perror("Failed to allocate huge page"); + + nr_reserved = hugetlb_nr_resv_pages(default_huge_page_size()); + if (nr_reserved != 1) + ksft_exit_fail_msg("Unexpected number of reserved pages: %lu, expected 1\n", + nr_reserved); + + /* Force the fault to ensure the reservation is consumed */ + *huge_ptr = 0; + nr_reserved = hugetlb_nr_resv_pages(default_huge_page_size()); + if (nr_reserved != 0) + ksft_exit_fail_msg("Unexpected number of reserved pages: %lu, expected 0\n", + nr_reserved); + + if (pipe(pipe_fds) != 0) + ksft_exit_fail_perror("pipe failed"); + + pid = fork(); + if (pid < 0) + ksft_exit_fail_perror("fork failed"); + + if (pid == 0) { + /* Child: Simply wait for the parent */ + char b; + + close(pipe_fds[1]); + if (read(pipe_fds[0], &b, 1) < 0) + ksft_perror("child read failed"); + /* Let the parent do the cleanup */ + _exit(0); + } + + /* Parent */ + close(pipe_fds[0]); + + /* First unmap, this will close the vma */ + if (munmap(huge_ptr, mmap_size) != 0) { + ksft_perror("munmap failed"); + goto err_cleanup; + } + + nr_reserved = hugetlb_nr_resv_pages(default_huge_page_size()); + if (nr_reserved == 0) { + ksft_test_result_pass("Underflow not present!\n"); + } else { + ksft_test_result_fail("Unexpected HugePages_Rsvd=%ld after munmap, should be 0\n", + nr_reserved); + goto err_cleanup; + } + /* Make the child exit, this should restore HugePages_Rsvd to 0 */ + if (write(pipe_fds[1], &nr_reserved, 1) < 0) { + /* If write failed, the child is likely already gone */ + ksft_exit_fail_perror("write failed"); + } + close(pipe_fds[1]); + if (waitpid(pid, NULL, 0) <= 0) + ksft_exit_fail_msg("waitpid failed\n"); + + nr_reserved = hugetlb_nr_resv_pages(default_huge_page_size()); + if (nr_reserved == 0) + ksft_test_result_pass("After the child dies, HugePages_Rsvd is properly set to 0\n"); + else + ksft_exit_fail_msg("Unexpected HugePages_Rsvd=%ld after the child termination munmap, should be 0\n", nr_reserved); + + return; + +err_cleanup: + if (write(pipe_fds[1], &nr_reserved, 1) < 0) + ksft_exit_fail_perror("write failed"); + if (waitpid(pid, NULL, 0) <= 0) + ksft_exit_fail_perror("waitpid failed"); + + ksft_exit_fail(); +} + +int main(void) +{ + ksft_print_header(); + ksft_set_plan(3); + + if (!hugetlb_setup_default_exact(1)) + ksft_exit_skip("This test needs one and only one page to execute. Got %lu\n", + hugetlb_free_default_pages()); + + test_madv_vs_map(); + test_underflow(); + ksft_finished(); } From 6d3fec7bf7c57827a3b5f4203c8cb184a24ca695 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:10 +0100 Subject: [PATCH 0964/1352] mm/vma: fix mmap_prepare file handling, remove file_doesnt_need_get Patch series "mm: make VMA flag semantics explicit, eliminate VM_SPECIAL", v3. The VM_SPECIAL / VMA_SPECIAL_FLAGS mask conflates several unrelated properties: * Is this kernel-owned, whether MMIO, kernel-allocated pages, or ordinary pages a driver maps itself? * Can it be expanded or merged? * Is this a 'weird' case like mlock where migration might race and we 'have' to set invalid flags to notify? * Is it another 'weird' case where we just want to stop GUP from touching it? Driver writers have often been confused about this, and who can blame them? It also interacts badly with the eternal edgecase known as hugetlb - which sets VMA_DONTEXPAND_BIT but doesn't also want to be treated like a 'special' flag. Another issue is that we cannot make sensible assumptions about flag use. It's not possible to assume VMA_IO_BIT means iommu because drivers abuse it and mlock abuses it. Special is also an overloaded term in mm. VDSO and VVAR mappings are also called 'special' but they're special in a... special way. Sometimes things are called special that are a subset of VMA_SPECIAL_FLAGS (VMA_PFNMAP_BIT and VMA_MIXEDMAP_BIT for instance when it comes to zapping or vm_normal_folio()). There's a specific kind of special for THP too, which considers PFN map, mixed map 'special' but DAX not. It's all rather a mess. This series brings some order to things by both limiting what drivers can do with VMA flags and switching to using predicates that describe behaviour, not arbitrary flags. It establishes the invariant that only kernel-owned mappings may set VMA_IO_BIT or clear VMA_MAYWRITE_BIT in an mmap hook, enforcing this by validating VMA state after every mmap and mmap_prepare hook. It updates usbmon and sg to mmap_prepare in order to do so, adding a new mmap action for mapping discontiguous kernel pages, and has hfi1 and the ALSA PCM status page map their pages eagerly instead. It also establishes the invariant that VMA_MIXEDMAP_BIT be set when mapping kernel memory, something that is usually the case but happens not to be for some users - specifically defio, cmt_speech, uprobes and the bpf arena, all of which are updated to do the right thing. It replaces VM_SPECIAL and arbitrary flag tests with predicates that say what is actually being tested: vma_is_kernel_owned() Does a driver or kernel code manage a VMA's life cycle? vma_is_fixed_mapping() Is the VMA not permitted to be expanded or merged? vma_is_persistent() Do bytes written to the VMA stay written, and bytes read stay the same unless userland changes them? vma_can_merge() Can the VMA be merged with a compatible neighbour? vma_can_gup() Can GUP obtain pages from the VMA, i.e. is it neither a PFN map nor memory-mapped I/O? Remaining raw VMA_IO_BIT, VMA_PFNMAP_BIT and VMA_MIXEDMAP_BIT tests scattered across mm are also converted to predicates where it makes sense to do so. And also the opportunity is taken to eliminate THP's vma_is_special_huge() which was an existing source of confusion. This patch (of 39): The map->file_doesnt_need_get flag is confusing and the existing implementation has holes. Drivers are permitted to change the owning file of a mapping. If they do so, they are required to take a reference on that file. The mmap() operation which ultimately invokes __mmap_region() is guaranteed to drop the refcount for the original file the mapping was made under, but this is not true for the replaced file. This has been addressed so far by tracking map->file_doesnt_need_get, which is rather poorly named and unfortunately fails to correctly track whether or not an additional put were needed in a number of cases. Make life easier by removing this flag, and instead drop the reference for both mmap_prepare and the deprecated mmap callback in a new function put_map(). Track whether this needs to be done by aligning mmap_state with vm_area_desc and store the original file in the map->file field, keeping the updated file in map->vm_file. In order to have the same behaviour for both types of hooks, only drop the reference __mmap_new_file_vma() itself took in its error path, deferring the replaced file's reference to put_map(). To make this work correctly, map->vm_file has to be updated before any error handling, so update __mmap_new_file_vma() and call_mmap_prepare() to set this field first. Also when mmap_prepare() changes the file and is then merged, the reference count also must be decremented, so update the logic to call put_map() in this case too. Also update __compat_vma_mmap() to manually perform this step for stacked file systems using the compatibility layer, and update compat_set_vma_from_desc() to replace vma_set_file() with a correct refcount/file update. No in-tree driver is impacted by the incorrect implementation of this currently (no driver that does this is mergeable for one), so this does not need to be a fix. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-0-4583d8a23bca@kernel.org Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-1-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis (Meta) Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- mm/internal.h | 1 + mm/util.c | 5 +++- mm/vma.c | 81 +++++++++++++++++++++++++++++++-------------------- mm/vma.h | 6 ++-- 4 files changed, 58 insertions(+), 35 deletions(-) diff --git a/mm/internal.h b/mm/internal.h index 05179c4b2090ef..323fa1514b92a3 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -7,6 +7,7 @@ #ifndef __MM_INTERNAL_H #define __MM_INTERNAL_H +#include #include #include #include diff --git a/mm/util.c b/mm/util.c index bf0513d1d3d086..016932780925e1 100644 --- a/mm/util.c +++ b/mm/util.c @@ -1228,8 +1228,11 @@ int __compat_vma_mmap(struct vm_area_desc *desc, /* Perform any preparatory tasks for mmap action. */ err = mmap_action_prepare(desc); - if (err) + if (err) { + if (desc->vm_file != vma->vm_file) + fput(desc->vm_file); return err; + } /* Update the VMA from the descriptor. */ compat_set_vma_from_desc(vma, desc); /* Complete any specified mmap actions. */ diff --git a/mm/vma.c b/mm/vma.c index f56317ee824717..9e35fe723a4d22 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -24,7 +24,8 @@ struct mmap_state { vm_flags_t vm_flags; vma_flags_t vma_flags; }; - struct file *file; + struct file *file; /* mmap()-specified file. */ + struct file *vm_file; /* May be updated by mmap_prepare. */ pgprot_t page_prot; /* User-defined fields, perhaps updated by .mmap_prepare(). */ @@ -43,8 +44,6 @@ struct mmap_state { /* Determine if we can check KSM flags early in mmap() logic. */ bool check_ksm_early :1; - /* If .mmap_prepare changed the file, we don't need to pin. */ - bool file_doesnt_need_get :1; }; #define MMAP_STATE(name, mm_, vmi_, addr_, len_, pgoff_, anon_pgoff_, vma_flags_, file_) \ @@ -58,6 +57,7 @@ struct mmap_state { .pglen = PHYS_PFN(len_), \ .vma_flags = vma_flags_, \ .file = file_, \ + .vm_file = file_, \ .page_prot = vma_flags_to_page_prot(vma_flags_), \ } @@ -70,7 +70,7 @@ struct mmap_state { .vma_flags = (map_)->vma_flags, \ .pgoff = (map_)->pgoff, \ .anon_pgoff = (map_)->anon_pgoff, \ - .file = (map_)->file, \ + .file = (map_)->vm_file, \ .prev = (map_)->prev, \ .middle = vma_, \ .next = (vma_) ? NULL : (map_)->next, \ @@ -2462,7 +2462,7 @@ void mm_drop_all_locks(struct mm_struct *mm) */ static bool accountable_mapping(struct mmap_state *map) { - const struct file *file = map->file; + const struct file *file = map->vm_file; /* * hugetlb has its own accounting separate from the core VM @@ -2511,7 +2511,7 @@ static void vms_abort_munmap_vmas(struct vma_munmap_struct *vms, static void update_ksm_flags(struct mmap_state *map) { - map->vma_flags = ksm_vma_flags(map->mm, map->file, map->vma_flags); + map->vma_flags = ksm_vma_flags(map->mm, map->vm_file, map->vma_flags); } static void set_desc_from_map(struct vm_area_desc *desc, @@ -2521,7 +2521,7 @@ static void set_desc_from_map(struct vm_area_desc *desc, desc->end = map->end; desc->pgoff = map->pgoff; - desc->vm_file = map->file; + desc->vm_file = map->vm_file; desc->vma_flags = map->vma_flags; desc->page_prot = map->page_prot; } @@ -2601,6 +2601,10 @@ static int __mmap_setup(struct mmap_state *map, struct vm_area_desc *desc, return 0; } +static bool map_same_file(struct mmap_state *map) +{ + return map->vm_file == map->file; +} static int __mmap_new_file_vma(struct mmap_state *map, struct vm_area_struct *vma) @@ -2608,20 +2612,23 @@ static int __mmap_new_file_vma(struct mmap_state *map, struct vma_iterator *vmi = map->vmi; int error; - vma->vm_file = map->file; - if (!map->file_doesnt_need_get) - get_file(map->file); + vma->vm_file = map->vm_file; + if (map_same_file(map)) + get_file(map->vm_file); - if (!map->file->f_op->mmap) + if (!map->vm_file->f_op->mmap) return 0; error = mmap_file(vma->vm_file, vma); + map->vm_file = vma->vm_file; + if (error) { UNMAP_STATE(unmap, vmi, vma, vma->vm_start, vma->vm_end, map->prev, map->next); - fput(vma->vm_file); - vma->vm_file = NULL; + if (map_same_file(map)) + fput(map->vm_file); + vma->vm_file = NULL; vma_iter_set(vmi, vma->vm_end); /* Undo any partial mapping done by a device driver. */ unmap_region(&unmap); @@ -2638,7 +2645,6 @@ static int __mmap_new_file_vma(struct mmap_state *map, !vma_flags_test(&map->vma_flags, VMA_MAYWRITE_BIT) && vma_test(vma, VMA_MAYWRITE_BIT)); - map->file = vma->vm_file; map->vma_flags = vma->flags; return 0; @@ -2646,7 +2652,7 @@ static int __mmap_new_file_vma(struct mmap_state *map, static void map_set_anon(struct mmap_state *map) { - map->file = NULL; + map->vm_file = NULL; map->vm_ops = NULL; map->pgoff = map->addr >> PAGE_SHIFT; } @@ -2658,7 +2664,7 @@ static bool map_is_private(const struct mmap_state *map) static bool map_is_anon(const struct mmap_state *map) { - return map_is_private(map) && !map->file; + return map_is_private(map) && !map->vm_file; } /* @@ -2703,7 +2709,7 @@ static int __mmap_new_vma(struct mmap_state *map, struct vm_area_struct **vmap, } /* Invoke callbacks. */ - if (map->file) + if (map->vm_file) error = __mmap_new_file_vma(map, vma); else if (!is_anon) error = shmem_zero_setup(vma); @@ -2812,10 +2818,14 @@ static int call_mmap_prepare(struct mmap_state *map, int err; /* Invoke the hook. */ - err = vfs_mmap_prepare(map->file, desc); + err = vfs_mmap_prepare(map->vm_file, desc); if (err) return err; + /* Update first so file refcount tracked correctly. */ + if (desc->vm_file != map->vm_file) + map->vm_file = desc->vm_file; + /* It's invalid for mmap_prepare hooks to clear vm_ops. */ if (!desc->vm_ops) return -EINVAL; @@ -2826,10 +2836,6 @@ static int call_mmap_prepare(struct mmap_state *map, /* Update fields permitted to be changed. */ map->pgoff = desc->pgoff; - if (desc->vm_file != map->file) { - map->file_doesnt_need_get = true; - map->file = desc->vm_file; - } map->vma_flags = desc->vma_flags; map->page_prot = desc->page_prot; /* User-defined fields. */ @@ -2841,7 +2847,7 @@ static int call_mmap_prepare(struct mmap_state *map, * anonymous mappings. Rather than allowing these mappings to be odd * outliers, simply make them truly anonymous. */ - if (map_is_private(map) && file_is_dev_zero(map->file)) + if (map_is_private(map) && file_is_dev_zero(map->vm_file)) map_set_anon(map); return 0; @@ -2860,7 +2866,7 @@ static void set_vma_user_defined_fields(struct vm_area_struct *vma, */ static bool can_set_ksm_flags_early(struct mmap_state *map) { - struct file *file = map->file; + struct file *file = map->vm_file; /* Anonymous mappings have no driver which can change them. */ if (!file) @@ -2883,6 +2889,20 @@ static bool can_set_ksm_flags_early(struct mmap_state *map) return false; } +static void put_map(struct mmap_state *map) +{ + /* + * An error occurred or the VMA was merged. + * + * If the file was changed by the driver (which is required to increment + * the replacement file's reference count), drop its reference count. + * + * On error, the caller always drops the original file regardless. + */ + if (map->vm_file && !map_same_file(map)) + fput(map->vm_file); +} + static unsigned long __mmap_region(struct file *file, unsigned long addr, unsigned long len, vma_flags_t vma_flags, unsigned long pgoff, struct list_head *uf) @@ -2937,7 +2957,10 @@ static unsigned long __mmap_region(struct file *file, unsigned long addr, __mmap_complete(&map, vma); - if (have_mmap_prepare && allocated_new) { + if (!allocated_new) { + /* Merged, so need to drop refcount. */ + put_map(&map); + } else if (have_mmap_prepare) { error = mmap_action_complete(vma, &desc.action, /*is_compat=*/false); if (error) @@ -2951,13 +2974,7 @@ static unsigned long __mmap_region(struct file *file, unsigned long addr, if (map.charged) vm_unacct_memory(map.charged); abort_munmap: - /* - * This indicates that .mmap_prepare has set a new file, differing from - * desc->vm_file. But since we're aborting the operation, only the - * original file will be cleaned up. Ensure we clean up both. - */ - if (map.file_doesnt_need_get) - fput(map.file); + put_map(&map); vms_abort_munmap_vmas(&map.vms, &map.mas_detach); return error; } diff --git a/mm/vma.h b/mm/vma.h index f856d9ace3a6a6..4665b40163fa67 100644 --- a/mm/vma.h +++ b/mm/vma.h @@ -394,8 +394,10 @@ static inline void compat_set_vma_from_desc(struct vm_area_struct *vma, /* Mutable fields. Populated with initial state. */ vma_set_pgoff(vma, desc->pgoff); - if (desc->vm_file != vma->vm_file) - vma_set_file(vma, desc->vm_file); + if (desc->vm_file != vma->vm_file) { + fput(vma->vm_file); + vma->vm_file = desc->vm_file; + } vma->flags = desc->vma_flags; vma->vm_page_prot = desc->page_prot; From 39aebf443d3974767c4e590c9ee2d59d72060abb Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:12 +0100 Subject: [PATCH 0965/1352] mm/vma: introduce and use vma_[flags_]can_merge() Replace the open-coded VMA_SPECIAL_FLAGS check in the VMA merge logic with two new functions vma_flags_can_merge() and vma_can_merge() and update the merge logic to use the former. This abstracts the check and expresses it in terms of the desired behaviour rather than an arbitrary and confusing VMA flag. This also lays the groundwork for making further improvements in VMA flag usage. Also update the userland VMA tests to reflect the change. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-3-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Suren Baghdasaryan Reviewed-by: Zi Yan Reviewed-by: Gregory Price (Meta) Acked-by: David Hildenbrand (Arm) Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- include/linux/mm.h | 21 +++++++++++++++++++++ mm/vma.c | 19 +++++++++++-------- tools/testing/vma/include/dup.h | 5 +++++ 3 files changed, 37 insertions(+), 8 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index 0a2a7fc4a442fe..b3fdc5e100c48a 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -1613,6 +1613,27 @@ static inline bool vma_is_shared_maywrite(const struct vm_area_struct *vma) return is_shared_maywrite(&vma->flags); } +/** + * vma_flags_can_merge() - Do the specified VMA flags permit the VMA to be + * merged with another? + * @flags: The VMA flags to test. + * Returns: true if the flags permit merging, false otherwise. + */ +static inline bool vma_flags_can_merge(const vma_flags_t *flags) +{ + return !vma_flags_test_any_mask(flags, VMA_SPECIAL_FLAGS); +} + +/** + * vma_can_merge() - Do @vma's flags permit it to be merged with another VMA? + * @vma: The VMA to test. + * Returns: true if the flags permit merging, otherwise false. + */ +static inline bool vma_can_merge(const struct vm_area_struct *vma) +{ + return vma_flags_can_merge(&vma->flags); +} + /** * vma_kernel_pagesize - Default page size granularity for this VMA. * @vma: The user mapping. diff --git a/mm/vma.c b/mm/vma.c index 9e35fe723a4d22..ead13520e90e97 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -924,13 +924,14 @@ static __must_check struct vm_area_struct *vma_merge_existing_range( vmg->state = VMA_MERGE_NOMERGE; + if (!vma_flags_can_merge(&vmg->vma_flags)) + return NULL; /* - * If a special mapping or if the range being modified is neither at the - * furthermost left or right side of the VMA, then we have no chance of - * merging and should abort. + * If the range being modified is neither at the furthermost left or + * right side of the VMA, then we have no chance of merging and should + * abort. */ - if (vma_flags_test_any_mask(&vmg->vma_flags, VMA_SPECIAL_FLAGS) || - (!left_side && !right_side)) + if (!left_side && !right_side) return NULL; if (left_side) @@ -1152,9 +1153,11 @@ struct vm_area_struct *vma_merge_new_range(struct vma_merge_struct *vmg) vmg->state = VMA_MERGE_NOMERGE; - /* Special VMAs are unmergeable, also if no prev/next. */ - if (vma_flags_test_any_mask(&vmg->vma_flags, VMA_SPECIAL_FLAGS) || - (!prev && !next)) + if (!vma_flags_can_merge(&vmg->vma_flags)) + return NULL; + + /* VMAs with no prev/next are unmergeable. */ + if (!prev && !next) return NULL; can_merge_left = can_vma_merge_left(vmg); diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index 16c09dac59d9b4..2fd422789717fe 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -1647,3 +1647,8 @@ static inline bool file_is_dev_zero(const struct file *file) { return file && file->f_op == &zero_fops; } + +static inline bool vma_flags_can_merge(const vma_flags_t *flags) +{ + return !vma_flags_test_any_mask(flags, VMA_SPECIAL_FLAGS); +} From 6b86058164cb29ae5a3a082f41411833cc4fe82a Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:13 +0100 Subject: [PATCH 0966/1352] mm: consistently validate VMA state after mmap[_prepare] hooks When the f_op->mmap_prepare or deprecated f_op->mmap hooks are invoked, the driver might have done something crazy that is not permitted by the kernel. Currently we check for three such cases in __mmap_new_file_vma(), but only if the legacy f_op->mmap hook is used: * Did sparc ADI result in invalid flags? * Did the driver alter vma->vm_start? * Did the driver make a file-backed mapping on a read-only file writable? Generalise these checks for both mmap_prepare and mmap and apply to all invocations of mmap_file(), the f_op->mmap and f_op->mmap_prepare handling in the core VMA code and the mmap_prepare compatibility layer. Also extend the vm_start check to vm_end also - drivers must not change the VMA range at all. We also WARN_ON_ONCE() on these conditions as they are things that should simply not occur in the kernel and it's important to call it out when it does. We invoke mmap_prepare_validate() after mmap_action_prepare(), as mmap actions often manipulate state in the descriptor thus providing the final state the VMA will be derived from. Also call mmap_validate_vma_flags() in insert_vm_struct() to ensure that special regions which are inserted (such as a VDSO or VVAR) also satisfy the sanity checks. This way every VMA established through an mmap hook, whether via mmap() or the compatibility layer, or inserted via insert_vm_struct(), has been validated. brk() VMAs never pass through a driver hook and so need no such check. While we're here, also fixup a couple disjoint blocks of #ifdef CONFIG_MMU. Finally, update the VMA userland tests to reflect the change. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-4-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Suren Baghdasaryan Reviewed-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/internal.h | 51 ++++++++++------ mm/util.c | 19 ++++-- mm/vma.c | 100 +++++++++++++++++++++++++++----- mm/vma.h | 25 +++++++- tools/testing/vma/include/dup.h | 10 ++++ 5 files changed, 163 insertions(+), 42 deletions(-) diff --git a/mm/internal.h b/mm/internal.h index 323fa1514b92a3..88f310cc215078 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -213,6 +213,24 @@ static inline void *folio_raw_mapping(const struct folio *folio) return (void *)(mapping & ~FOLIO_MAPPING_FLAGS); } +/* + * If the VMA has a close hook then close it, and since closing it might leave + * it in an inconsistent state which makes the use of any hooks suspect, clear + * them down by installing dummy empty hooks. + */ +static inline void vma_close(struct vm_area_struct *vma) +{ + if (vma->vm_ops && vma->vm_ops->close) { + vma->vm_ops->close(vma); + + /* + * The mapping is in an inconsistent state, and no further hooks + * may be invoked upon it. + */ + vma->vm_ops = &vma_dummy_vm_ops; + } +} + /* * This is a file-backed mapping, and is about to be memory mapped - invoke its * mmap hook and safely handle error conditions. On error, VMA hooks will be @@ -225,8 +243,12 @@ static inline void *folio_raw_mapping(const struct folio *folio) */ static inline int mmap_file(struct file *file, struct vm_area_struct *vma) { - int err = vfs_mmap(file, vma); + const unsigned long prev_start = vma->vm_start; + const unsigned long prev_end = vma->vm_end; + const vma_flags_t prev_flags = vma->flags; + int err; + err = vfs_mmap(file, vma); /* * Either we tried to call the file hook for mmap() and an error arose * or a driver set vma->vm_ops = NULL intending there to be no VMA @@ -239,26 +261,17 @@ static inline int mmap_file(struct file *file, struct vm_area_struct *vma) */ if (unlikely(err || !vma->vm_ops)) vma->vm_ops = &vma_dummy_vm_ops; + if (unlikely(err)) + return err; - return err; -} - -/* - * If the VMA has a close hook then close it, and since closing it might leave - * it in an inconsistent state which makes the use of any hooks suspect, clear - * them down by installing dummy empty hooks. - */ -static inline void vma_close(struct vm_area_struct *vma) -{ - if (vma->vm_ops && vma->vm_ops->close) { - vma->vm_ops->close(vma); - - /* - * The mapping is in an inconsistent state, and no further hooks - * may be invoked upon it. - */ - vma->vm_ops = &vma_dummy_vm_ops; + err = mmap_hook_validate(prev_start, prev_end, &prev_flags, vma); + if (unlikely(err)) { + vma->vm_start = prev_start; + vma->vm_end = prev_end; + vma_close(vma); } + + return err; } /* unmap_vmas is in mm/memory.c */ diff --git a/mm/util.c b/mm/util.c index 016932780925e1..bdd5923eebc7be 100644 --- a/mm/util.c +++ b/mm/util.c @@ -1224,19 +1224,28 @@ EXPORT_SYMBOL(compat_set_desc_from_vma); int __compat_vma_mmap(struct vm_area_desc *desc, struct vm_area_struct *vma) { + struct vm_area_desc prev_desc; int err; + /* Derive state prior to mmap_prepare hook. */ + compat_set_desc_from_vma(&prev_desc, desc->file, vma); /* Perform any preparatory tasks for mmap action. */ err = mmap_action_prepare(desc); - if (err) { - if (desc->vm_file != vma->vm_file) - fput(desc->vm_file); - return err; - } + if (err) + goto err_put; + /* Check the caller did nothing crazy. */ + err = mmap_prepare_validate(&prev_desc, desc); + if (err) + goto err_put; /* Update the VMA from the descriptor. */ compat_set_vma_from_desc(vma, desc); /* Complete any specified mmap actions. */ return mmap_action_complete(vma, &desc->action, /*is_compat=*/true); + +err_put: + if (desc->vm_file != vma->vm_file) + fput(desc->vm_file); + return err; } EXPORT_SYMBOL(__compat_vma_mmap); diff --git a/mm/vma.c b/mm/vma.c index ead13520e90e97..aa8b30b6734dc5 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -2638,16 +2638,6 @@ static int __mmap_new_file_vma(struct mmap_state *map, return error; } - /* Drivers cannot alter the address of the VMA. */ - WARN_ON_ONCE(map->addr != vma->vm_start); - /* - * Drivers should not permit writability when previously it was - * disallowed. - */ - VM_WARN_ON_ONCE(!vma_flags_same_pair(&map->vma_flags, &vma->flags) && - !vma_flags_test(&map->vma_flags, VMA_MAYWRITE_BIT) && - vma_test(vma, VMA_MAYWRITE_BIT)); - map->vma_flags = vma->flags; return 0; @@ -2725,11 +2715,6 @@ static int __mmap_new_vma(struct mmap_state *map, struct vm_area_struct **vmap, vma->flags = map->vma_flags; } -#ifdef CONFIG_SPARC64 - /* TODO: Fix SPARC ADI! */ - WARN_ON_ONCE(!arch_validate_flags(map->vm_flags)); -#endif - /* Lock the VMA since it is modified after insertion into VMA tree */ vma_start_write(vma); vma_iter_store_new(vmi, vma); @@ -2792,6 +2777,80 @@ static void __mmap_complete(struct mmap_state *map, struct vm_area_struct *vma) vma_set_page_prot(vma); } +/* Check to ensure that the VMA flags of a newly mapped VMA are sane. */ +static int mmap_validate_vma_flags(const vma_flags_t *flags) +{ +#ifdef CONFIG_SPARC64 + const vm_flags_t legacy_flags = vma_flags_to_legacy(*flags); + + /* TODO: Fix SPARC ADI! */ + if (WARN_ON_ONCE(!arch_validate_flags(legacy_flags))) + return -EINVAL; +#endif + + return 0; +} + +/* Check to ensure a driver hasn't done something crazy. */ +static int mmap_validate(unsigned long prev_start, unsigned long prev_end, + unsigned long curr_start, unsigned long curr_end, + const vma_flags_t *prev_flags, + const vma_flags_t *curr_flags) +{ + bool was_maywrite, is_maywrite; + + /* Drivers cannot alter the range of the VMA. */ + if (WARN_ON_ONCE(prev_start != curr_start || prev_end != curr_end)) + return -EINVAL; + + was_maywrite = vma_flags_test(prev_flags, VMA_MAYWRITE_BIT); + is_maywrite = vma_flags_test(curr_flags, VMA_MAYWRITE_BIT); + + /* A driver may not make a previously unwritable mapping writable. */ + if (WARN_ON_ONCE(!was_maywrite && is_maywrite)) + return -EINVAL; + + return mmap_validate_vma_flags(curr_flags); +} + +/** + * mmap_prepare_validate() - Ensure the driver hasn't violated invariants in its + * f_op->mmap_prepare hook. + * @prev_desc: The VMA descriptor prior to the mmap_prepare hook being called. + * @desc: The VMA descriptor after the mmap_prepare hook has been called. + * + * Returns: 0 on success, otherwise an error. + */ +int mmap_prepare_validate(const struct vm_area_desc *prev_desc, + const struct vm_area_desc *desc) +{ + return mmap_validate(prev_desc->start, prev_desc->end, + desc->start, desc->end, + &prev_desc->vma_flags, &desc->vma_flags); +} + +/** + * mmap_hook_validate() - Ensure the driver hasn't violated invariants in + * its f_op->mmap hook. + * @prev_start: The start of the mapping prior to the mmap hook. + * @prev_end: The end of the mapping prior to the mmap hook. + * @prev_flags: The VMA flags set for the VMA prior to the mmap hook. + * @vma: The VMA after the hook has been applied. + * + * Returns: 0 on success, otherwise an error. + */ +int mmap_hook_validate(unsigned long prev_start, unsigned long prev_end, + const vma_flags_t *prev_flags, + const struct vm_area_struct *vma) +{ + const unsigned long start = vma->vm_start; + const unsigned long end = vma->vm_end; + const vma_flags_t *flags = &vma->flags; + + return mmap_validate(prev_start, prev_end, start, end, prev_flags, + flags); +} + static int call_action_prepare(struct mmap_state *map, struct vm_area_desc *desc) { @@ -2818,6 +2877,7 @@ static int call_action_prepare(struct mmap_state *map, static int call_mmap_prepare(struct mmap_state *map, struct vm_area_desc *desc) { + const struct vm_area_desc prev_desc = *desc; int err; /* Invoke the hook. */ @@ -2837,6 +2897,11 @@ static int call_mmap_prepare(struct mmap_state *map, if (err) return err; + /* Check the caller did nothing crazy. */ + err = mmap_prepare_validate(&prev_desc, desc); + if (err) + return err; + /* Update fields permitted to be changed. */ map->pgoff = desc->pgoff; map->vma_flags = desc->vma_flags; @@ -3472,10 +3537,15 @@ int __vm_munmap(unsigned long start, size_t len, bool unlock) int insert_vm_struct(struct mm_struct *mm, struct vm_area_struct *vma) { unsigned long charged = vma_pages(vma); + int err; if (find_vma_intersection(mm, vma->vm_start, vma->vm_end)) return -ENOMEM; + err = mmap_validate_vma_flags(&vma->flags); + if (err) + return err; + if (vma_test(vma, VMA_ACCOUNT_BIT) && security_vm_enough_memory_mm(mm, charged)) return -ENOMEM; diff --git a/mm/vma.h b/mm/vma.h index 4665b40163fa67..b9b99fa02a861d 100644 --- a/mm/vma.h +++ b/mm/vma.h @@ -787,14 +787,19 @@ struct vm_area_struct *vm_area_alloc(struct mm_struct *mm); struct vm_area_struct *vm_area_dup(struct vm_area_struct *orig); void vm_area_free(struct vm_area_struct *vma); -/* vma_exec.c */ #ifdef CONFIG_MMU +int mmap_prepare_validate(const struct vm_area_desc *prev_desc, + const struct vm_area_desc *desc); + +int mmap_hook_validate(unsigned long prev_start, unsigned long prev_end, + const vma_flags_t *prev_flags, + const struct vm_area_struct *vma); + +/* vma_exec.c */ int create_init_stack_vma(struct mm_struct *mm, struct vm_area_struct **vmap, unsigned long *top_mem_p); int relocate_vma_down(struct vm_area_struct *vma, unsigned long shift); -#endif -#ifdef CONFIG_MMU /* * Denies creating a writable executable mapping or gaining executable permissions. * @@ -843,6 +848,20 @@ static inline bool map_deny_write_exec(const vma_flags_t *old, return false; } +#else +static inline int mmap_prepare_validate(const struct vm_area_desc *prev_desc, + const struct vm_area_desc *desc) +{ + return 0; +} + +static inline int mmap_hook_validate(unsigned long prev_start, + unsigned long prev_end, + const vma_flags_t *prev_flags, + const struct vm_area_struct *vma) +{ + return 0; +} #endif struct vm_area_struct *__install_special_mapping(struct mm_struct *mm, diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index 2fd422789717fe..2986ae6ca1e5ed 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -1359,13 +1359,23 @@ static inline int vfs_mmap_prepare(struct file *file, struct vm_area_desc *desc) return file->f_op->mmap_prepare(desc); } +int mmap_prepare_validate(const struct vm_area_desc *prev_desc, + const struct vm_area_desc *desc); + static inline int __compat_vma_mmap(struct vm_area_desc *desc, struct vm_area_struct *vma) { + struct vm_area_desc prev_desc; int err; + /* Derive state prior to mmap_prepare hook. */ + compat_set_desc_from_vma(&prev_desc, desc->file, vma); /* Perform any preparatory tasks for mmap action. */ err = mmap_action_prepare(desc); + if (err) + return err; + /* Check the caller did nothing crazy. */ + err = mmap_prepare_validate(&prev_desc, desc); if (err) return err; /* Update the VMA from the descriptor. */ From 34f43ddb6c9fd6f017aad71ba8aaf42f4bddb651 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:14 +0100 Subject: [PATCH 0967/1352] mm/vma: ensure mmap_prepare doesn't set actions on a mergeable vma When a user requests an mmap_action be performed in mmap_prepare, this involves populating the VMA range with data. However, if the VMA is mergeable, it might then mistakenly be merged with another VMA without having populated the range. Every mmap action currently available sets VMA flags such that the VMA cannot be merged. However, to ensure that no future mmap action falls foul of this, assert that this is the case upon mmap_prepare validation. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-5-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Suren Baghdasaryan Acked-by: David Hildenbrand (Arm) Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/vma.c | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/mm/vma.c b/mm/vma.c index aa8b30b6734dc5..cfaf217d9c1c3f 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -2824,6 +2824,15 @@ static int mmap_validate(unsigned long prev_start, unsigned long prev_end, int mmap_prepare_validate(const struct vm_area_desc *prev_desc, const struct vm_area_desc *desc) { + /* + * It is not valid to execute mmap actions for VMAs which can be merged, + * as any such merge would leave portions of the mapping incorrectly + * unmapped. + */ + if (vma_flags_can_merge(&desc->vma_flags) && + WARN_ON_ONCE(desc->action.type != MMAP_NOTHING)) + return -EINVAL; + return mmap_validate(prev_desc->start, prev_desc->end, desc->start, desc->end, &prev_desc->vma_flags, &desc->vma_flags); From 4983b0482e0c392b0f7026222500bc025488a277 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:15 +0100 Subject: [PATCH 0968/1352] mm: make map_kernel_pages_[prepare,complete] internal and unexported There's no reason to export the symbols for these functions which are only called from internal mm logic, additionally there's no reason for them to be declared in mm.h. This patch therefore removes the exports and moves the declarations to mm/internal.h. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-6-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Suren Baghdasaryan Reviewed-by: Gregory Price (Meta) Acked-by: David Hildenbrand (Arm) Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- include/linux/mm.h | 3 --- mm/internal.h | 3 +++ mm/memory.c | 2 -- 3 files changed, 3 insertions(+), 5 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index b3fdc5e100c48a..ac99e77f0a2230 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -4775,9 +4775,6 @@ int remap_pfn_range(struct vm_area_struct *vma, unsigned long addr, int vm_insert_page(struct vm_area_struct *, unsigned long addr, struct page *); int vm_insert_pages(struct vm_area_struct *vma, unsigned long addr, struct page **pages, unsigned long *num); -int map_kernel_pages_prepare(struct vm_area_desc *desc); -int map_kernel_pages_complete(struct vm_area_struct *vma, - struct mmap_action *action); int vm_map_pages(struct vm_area_struct *vma, struct page **pages, unsigned long num); int vm_map_pages_zero(struct vm_area_struct *vma, struct page **pages, diff --git a/mm/internal.h b/mm/internal.h index 88f310cc215078..e75b3fc8ce0041 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -1515,6 +1515,9 @@ int remap_pfn_range_prepare(struct vm_area_desc *desc); int remap_pfn_range_complete(struct vm_area_struct *vma, struct mmap_action *action); int simple_ioremap_prepare(struct vm_area_desc *desc); +int map_kernel_pages_prepare(struct vm_area_desc *desc); +int map_kernel_pages_complete(struct vm_area_struct *vma, + struct mmap_action *action); static inline int io_remap_pfn_range_prepare(struct vm_area_desc *desc) { diff --git a/mm/memory.c b/mm/memory.c index 926276d4192026..448342883e9daa 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -2628,7 +2628,6 @@ int map_kernel_pages_prepare(struct vm_area_desc *desc) return 0; } -EXPORT_SYMBOL(map_kernel_pages_prepare); int map_kernel_pages_complete(struct vm_area_struct *vma, struct mmap_action *action) @@ -2640,7 +2639,6 @@ int map_kernel_pages_complete(struct vm_area_struct *vma, action->map_kernel.pages, &nr_pages, vma->vm_page_prot); } -EXPORT_SYMBOL(map_kernel_pages_complete); /** * vm_insert_page - insert single page into user vma From 8b25f518e3587b525de53073af754731f22d00cc Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:16 +0100 Subject: [PATCH 0969/1352] mm/vma: tidy up map kernel pages enum values MMAP_MAP_KERNEL_PAGES is a mouthful, discard the MAP_ as that's implied by MMAP. Also while we're here delete useless comments for mmap actions whose names clearly indicate what they are for. Also update the userland VMA tests to reflect this change. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-7-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Suren Baghdasaryan Reviewed-by: Gregory Price (Meta) Acked-by: David Hildenbrand (Arm) Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- include/linux/mm.h | 2 +- include/linux/mm_types.h | 8 ++++---- mm/util.c | 8 ++++---- tools/testing/vma/include/dup.h | 8 ++++---- 4 files changed, 13 insertions(+), 13 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index ac99e77f0a2230..f8c715557ec930 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -4630,7 +4630,7 @@ static inline void mmap_action_map_kernel_pages(struct vm_area_desc *desc, { struct mmap_action *action = &desc->action; - action->type = MMAP_MAP_KERNEL_PAGES; + action->type = MMAP_KERNEL_PAGES; action->map_kernel.start = start; action->map_kernel.pages = pages; action->map_kernel.nr_pages = nr_pages; diff --git a/include/linux/mm_types.h b/include/linux/mm_types.h index 0720a4e98286b2..9b6bbdc8e53f39 100644 --- a/include/linux/mm_types.h +++ b/include/linux/mm_types.h @@ -815,11 +815,11 @@ struct pfnmap_track_ctx { /* What action should be taken after an .mmap_prepare call is complete? */ enum mmap_action_type { - MMAP_NOTHING, /* Mapping is complete, no further action. */ - MMAP_REMAP_PFN, /* Remap PFN range. */ - MMAP_IO_REMAP_PFN, /* I/O remap PFN range. */ + MMAP_NOTHING, + MMAP_REMAP_PFN, + MMAP_IO_REMAP_PFN, MMAP_SIMPLE_IO_REMAP, /* I/O remap with guardrails. */ - MMAP_MAP_KERNEL_PAGES, /* Map kernel page range from array. */ + MMAP_KERNEL_PAGES, /* Map kernel page range from array. */ }; /* diff --git a/mm/util.c b/mm/util.c index bdd5923eebc7be..438170490e7fd6 100644 --- a/mm/util.c +++ b/mm/util.c @@ -1467,7 +1467,7 @@ int mmap_action_prepare(struct vm_area_desc *desc) return io_remap_pfn_range_prepare(desc); case MMAP_SIMPLE_IO_REMAP: return simple_ioremap_prepare(desc); - case MMAP_MAP_KERNEL_PAGES: + case MMAP_KERNEL_PAGES: return map_kernel_pages_prepare(desc); } @@ -1498,7 +1498,7 @@ int mmap_action_complete(struct vm_area_struct *vma, case MMAP_REMAP_PFN: err = remap_pfn_range_complete(vma, action); break; - case MMAP_MAP_KERNEL_PAGES: + case MMAP_KERNEL_PAGES: err = map_kernel_pages_complete(vma, action); break; case MMAP_IO_REMAP_PFN: @@ -1521,7 +1521,7 @@ int mmap_action_prepare(struct vm_area_desc *desc) case MMAP_REMAP_PFN: case MMAP_IO_REMAP_PFN: case MMAP_SIMPLE_IO_REMAP: - case MMAP_MAP_KERNEL_PAGES: + case MMAP_KERNEL_PAGES: WARN_ON_ONCE(1); /* nommu cannot handle these. */ break; } @@ -1542,7 +1542,7 @@ int mmap_action_complete(struct vm_area_struct *vma, case MMAP_REMAP_PFN: case MMAP_IO_REMAP_PFN: case MMAP_SIMPLE_IO_REMAP: - case MMAP_MAP_KERNEL_PAGES: + case MMAP_KERNEL_PAGES: WARN_ON_ONCE(1); /* nommu cannot handle this. */ err = -EINVAL; diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index 2986ae6ca1e5ed..1098655a5f4a3d 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -454,11 +454,11 @@ static __always_inline bool vma_flags_empty(const vma_flags_t *flags) /* What action should be taken after an .mmap_prepare call is complete? */ enum mmap_action_type { - MMAP_NOTHING, /* Mapping is complete, no further action. */ - MMAP_REMAP_PFN, /* Remap PFN range. */ - MMAP_IO_REMAP_PFN, /* I/O remap PFN range. */ + MMAP_NOTHING, + MMAP_REMAP_PFN, + MMAP_IO_REMAP_PFN, MMAP_SIMPLE_IO_REMAP, /* I/O remap with guardrails. */ - MMAP_MAP_KERNEL_PAGES, /* Map kernel page range from an array. */ + MMAP_KERNEL_PAGES, /* Map kernel page range from array. */ }; /* From 6bcfd3888aca2437877528a5c534046c2bad879c Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:17 +0100 Subject: [PATCH 0970/1352] mm: add mmap action for discontiguous kernel page mapping The existing kernel page mapping mmap actions allow for partial and full mapping of an array of struct page pointers. However some drivers require the mapping of discontiguous ranges. Permit this by providing discontig_kernel_page_ops which allows a driver to specify how the operation should begin and how batches of pages should be retrieved. It uses the minimum exposed interface to do so, providing address, page offset and both vm_private_data state and a local private state object. ops->init can establish state for the operation, and ops->get outputs the pages to map and their count. Should an error arise the core unmaps the VMA, invoking vm_ops->close, which is therefore where any state established by ops->init is released. Batches may not exceed the VMA, but may map less than its full range in case the driver wishes to allow the user to map an area larger than the available data. To use it, users invoke mmap_action_map_discontig_kernel_pages() with initial local private state and a set of operations. Users can then use one of the provided helper functions to perform an action: * discontig_kernel_map_abort() - Abort and leave the mapping as it has been accumulated so far. * discontig_kernel_map_page() - Map a single page, or a compound page given its head page. * discontig_kernel_map_page_range() - Maps a struct page ** array of a specified count. The userland VMA tests are updated accordingly. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-8-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- include/linux/mm.h | 45 +++++++++++++ include/linux/mm_types.h | 44 ++++++++++++- mm/internal.h | 3 + mm/memory.c | 108 ++++++++++++++++++++++++++++++-- mm/util.c | 7 +++ tools/testing/vma/include/dup.h | 11 +++- 6 files changed, 209 insertions(+), 9 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index f8c715557ec930..5602a89156775c 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -4654,10 +4654,55 @@ static inline void mmap_action_map_kernel_pages_full(struct vm_area_desc *desc, vma_desc_pages(desc)); } +static inline +void mmap_action_map_discontig_kernel_pages(struct vm_area_desc *desc, + void *init_private, const struct discontig_kernel_page_ops *ops) +{ + struct mmap_action *action = &desc->action; + + action->type = MMAP_DISCONTIG_KERNEL_PAGES; + action->map_kernel_discontig.init_private = init_private; + action->map_kernel_discontig.ops = ops; +} + int mmap_action_prepare(struct vm_area_desc *desc); int mmap_action_complete(struct vm_area_struct *vma, struct mmap_action *action, bool is_compat); +static inline void +discontig_kernel_map_abort(struct discontig_kernel_page_state *state) +{ + state->action = DISCONTIG_KERNEL_PAGE_ABORT; +} + +static inline void +discontig_kernel_map_page(struct discontig_kernel_page_state *state, + struct page *page) +{ + struct folio *folio = page_folio(page); + + if (folio_test_large(folio)) { + VM_WARN_ON_ONCE(page != folio_page(folio, 0)); + state->action = DISCONTIG_KERNEL_PAGE_MAP_COMPOUND_PAGE; + state->__folio = folio; + state->__nr_pages = min(state->nr_pages_remain, + folio_nr_pages(folio)); + } else { + state->action = DISCONTIG_KERNEL_PAGE_MAP_PAGE; + state->__page = page; + state->__nr_pages = 1; + } +} + +static inline void +discontig_kernel_map_page_range(struct discontig_kernel_page_state *state, + struct page **page_arr, unsigned long nr_pages) +{ + state->action = DISCONTIG_KERNEL_PAGE_MAP_PAGE_RANGE; + state->__page_arr = page_arr; + state->__nr_pages = nr_pages; +} + /* Look up the first VMA which exactly match the interval vm_start ... vm_end */ static inline struct vm_area_struct *find_exact_vma(struct mm_struct *mm, unsigned long vm_start, unsigned long vm_end) diff --git a/include/linux/mm_types.h b/include/linux/mm_types.h index 9b6bbdc8e53f39..95a768fac987a5 100644 --- a/include/linux/mm_types.h +++ b/include/linux/mm_types.h @@ -818,8 +818,44 @@ enum mmap_action_type { MMAP_NOTHING, MMAP_REMAP_PFN, MMAP_IO_REMAP_PFN, - MMAP_SIMPLE_IO_REMAP, /* I/O remap with guardrails. */ - MMAP_KERNEL_PAGES, /* Map kernel page range from array. */ + MMAP_SIMPLE_IO_REMAP, /* I/O remap with guardrails. */ + MMAP_KERNEL_PAGES, /* Map kernel page range from array. */ + MMAP_DISCONTIG_KERNEL_PAGES, /* Map kernel discontig page range. */ +}; + +enum discontig_kernel_page_action { + DISCONTIG_KERNEL_PAGE_ABORT, + DISCONTIG_KERNEL_PAGE_MAP_PAGE, + DISCONTIG_KERNEL_PAGE_MAP_COMPOUND_PAGE, + DISCONTIG_KERNEL_PAGE_MAP_PAGE_RANGE, +}; + +struct discontig_kernel_page_state { + /* Map state. */ + const unsigned long start; /* Start address of VMA. */ + const unsigned long end; /* End address of VMA. */ + unsigned long addr; /* The current address to be mapped. */ + pgoff_t pgoff; /* The current pgoff to be mapped. */ + unsigned long nr_pages_mapped; /* The number of pages mapped. */ + unsigned long nr_pages_remain; /* The number of pages remaining. */ + + /* User-defined state. */ + void *vm_private_data; /* VMA private data. */ + void *private; /* Mapping private data. */ + + /* Users should not touch these, use discontig_kernel_map_*() helpers. */ + enum discontig_kernel_page_action action; + union { + struct page *__page; + struct folio *__folio; + struct page **__page_arr; + }; + unsigned long __nr_pages; +}; + +struct discontig_kernel_page_ops { + int (*init)(void *vm_private_data, void **private); + int (*get)(struct discontig_kernel_page_state *state); }; /* @@ -844,6 +880,10 @@ struct mmap_action { unsigned long nr_pages; pgoff_t pgoff; } map_kernel; + struct { + void *init_private; + const struct discontig_kernel_page_ops *ops; + } map_kernel_discontig; }; enum mmap_action_type type; diff --git a/mm/internal.h b/mm/internal.h index e75b3fc8ce0041..a1970ff52ad7b9 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -1518,6 +1518,9 @@ int simple_ioremap_prepare(struct vm_area_desc *desc); int map_kernel_pages_prepare(struct vm_area_desc *desc); int map_kernel_pages_complete(struct vm_area_struct *vma, struct mmap_action *action); +int map_discontig_kernel_pages_prepare(struct vm_area_desc *desc); +int map_discontig_kernel_pages_complete(struct vm_area_struct *vma, + struct mmap_action *action); static inline int io_remap_pfn_range_prepare(struct vm_area_desc *desc) { diff --git a/mm/memory.c b/mm/memory.c index 448342883e9daa..45b21bb04a18b9 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -2609,17 +2609,23 @@ int vm_insert_pages(struct vm_area_struct *vma, unsigned long addr, } EXPORT_SYMBOL(vm_insert_pages); +static void __map_kernel_pages_prepare(struct vm_area_desc *desc) +{ + if (vma_desc_test(desc, VMA_MIXEDMAP_BIT)) + return; + + VM_WARN_ON_ONCE(mmap_read_trylock(desc->mm)); + VM_WARN_ON_ONCE(vma_desc_test(desc, VMA_PFNMAP_BIT)); + vma_desc_set_flags(desc, VMA_MIXEDMAP_BIT); +} + int map_kernel_pages_prepare(struct vm_area_desc *desc) { const struct mmap_action *action = &desc->action; const unsigned long addr = action->map_kernel.start; unsigned long nr_pages, end; - if (!vma_desc_test(desc, VMA_MIXEDMAP_BIT)) { - VM_WARN_ON_ONCE(mmap_read_trylock(desc->mm)); - VM_WARN_ON_ONCE(vma_desc_test(desc, VMA_PFNMAP_BIT)); - vma_desc_set_flags(desc, VMA_MIXEDMAP_BIT); - } + __map_kernel_pages_prepare(desc); nr_pages = action->map_kernel.nr_pages; end = addr + PAGE_SIZE * nr_pages; @@ -2640,6 +2646,98 @@ int map_kernel_pages_complete(struct vm_area_struct *vma, &nr_pages, vma->vm_page_prot); } +int map_discontig_kernel_pages_prepare(struct vm_area_desc *desc) +{ + const struct mmap_action *action = &desc->action; + const struct discontig_kernel_page_ops *ops = + action->map_kernel_discontig.ops; + + /* At minimum need to be able to get pages. */ + if (WARN_ON_ONCE(!ops || !ops->get)) + return -EINVAL; + + __map_kernel_pages_prepare(desc); + return 0; +} + +static int apply_discontig_action(struct vm_area_struct *vma, + struct discontig_kernel_page_state *state) +{ + unsigned long nr_pages = state->__nr_pages; + unsigned long addr = state->addr; + unsigned long i; + + if (state->action == DISCONTIG_KERNEL_PAGE_MAP_PAGE) + return insert_page(vma, addr, state->__page, + vma->vm_page_prot, /*mkwrite=*/false); + if (state->action == DISCONTIG_KERNEL_PAGE_MAP_PAGE_RANGE) + return insert_pages(vma, addr, state->__page_arr, + &nr_pages, vma->vm_page_prot); + + /* Compound folio - have to iterate through each page. */ + for (i = 0; i < nr_pages; i++, addr += PAGE_SIZE) { + struct page *page = folio_page(state->__folio, i); + int err; + + err = insert_page(vma, addr, page, vma->vm_page_prot, + /*mkwrite=*/false); + if (err) + return err; + } + return 0; +} + +int map_discontig_kernel_pages_complete(struct vm_area_struct *vma, + struct mmap_action *action) +{ + const struct discontig_kernel_page_ops *ops = + action->map_kernel_discontig.ops; + struct discontig_kernel_page_state state = { + .start = vma->vm_start, + .end = vma->vm_end, + .addr = vma->vm_start, + .pgoff = vma->vm_pgoff, + .nr_pages_mapped = 0, + .nr_pages_remain = vma_pages(vma), + .vm_private_data = vma->vm_private_data, + .private = action->map_kernel_discontig.init_private, + }; + int err = 0; + + if (ops->init) + err = ops->init(vma->vm_private_data, &state.private); + if (err) + return err; + + do { + unsigned long end, pgoff_end; + unsigned long nr_pages; + + /* Default to abort. */ + state.action = DISCONTIG_KERNEL_PAGE_ABORT; + err = ops->get(&state); + if (err || state.action == DISCONTIG_KERNEL_PAGE_ABORT) + return err; + nr_pages = state.__nr_pages; + + if (!nr_pages || nr_pages > state.nr_pages_remain) + return -EINVAL; + end = state.addr + PAGE_SIZE * nr_pages; + pgoff_end = state.pgoff + nr_pages; + + err = apply_discontig_action(vma, &state); + if (err) + return err; + + state.addr = end; + state.pgoff = pgoff_end; + state.nr_pages_mapped += nr_pages; + state.nr_pages_remain -= nr_pages; + } while (state.addr < vma->vm_end); + + return 0; +} + /** * vm_insert_page - insert single page into user vma * @vma: user vma to map to diff --git a/mm/util.c b/mm/util.c index 438170490e7fd6..c5ee52aede1e41 100644 --- a/mm/util.c +++ b/mm/util.c @@ -1469,6 +1469,8 @@ int mmap_action_prepare(struct vm_area_desc *desc) return simple_ioremap_prepare(desc); case MMAP_KERNEL_PAGES: return map_kernel_pages_prepare(desc); + case MMAP_DISCONTIG_KERNEL_PAGES: + return map_discontig_kernel_pages_prepare(desc); } WARN_ON_ONCE(1); @@ -1501,6 +1503,9 @@ int mmap_action_complete(struct vm_area_struct *vma, case MMAP_KERNEL_PAGES: err = map_kernel_pages_complete(vma, action); break; + case MMAP_DISCONTIG_KERNEL_PAGES: + err = map_discontig_kernel_pages_complete(vma, action); + break; case MMAP_IO_REMAP_PFN: case MMAP_SIMPLE_IO_REMAP: /* Should have been delegated. */ @@ -1522,6 +1527,7 @@ int mmap_action_prepare(struct vm_area_desc *desc) case MMAP_IO_REMAP_PFN: case MMAP_SIMPLE_IO_REMAP: case MMAP_KERNEL_PAGES: + case MMAP_DISCONTIG_KERNEL_PAGES: WARN_ON_ONCE(1); /* nommu cannot handle these. */ break; } @@ -1543,6 +1549,7 @@ int mmap_action_complete(struct vm_area_struct *vma, case MMAP_IO_REMAP_PFN: case MMAP_SIMPLE_IO_REMAP: case MMAP_KERNEL_PAGES: + case MMAP_DISCONTIG_KERNEL_PAGES: WARN_ON_ONCE(1); /* nommu cannot handle this. */ err = -EINVAL; diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index 1098655a5f4a3d..1d5f6b3cbd21e8 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -457,14 +457,17 @@ enum mmap_action_type { MMAP_NOTHING, MMAP_REMAP_PFN, MMAP_IO_REMAP_PFN, - MMAP_SIMPLE_IO_REMAP, /* I/O remap with guardrails. */ - MMAP_KERNEL_PAGES, /* Map kernel page range from array. */ + MMAP_SIMPLE_IO_REMAP, /* I/O remap with guardrails. */ + MMAP_KERNEL_PAGES, /* Map kernel page range from array. */ + MMAP_DISCONTIG_KERNEL_PAGES, /* Map kernel discontig page range. */ }; /* * Describes an action an mmap_prepare hook can instruct to be taken to complete * the mapping of a VMA. Specified in vm_area_desc. */ +struct discontig_kernel_page_ops; + struct mmap_action { union { struct { @@ -483,6 +486,10 @@ struct mmap_action { unsigned long nr_pages; pgoff_t pgoff; } map_kernel; + struct { + void *init_private; + const struct discontig_kernel_page_ops *ops; + } map_kernel_discontig; }; enum mmap_action_type type; From 78e9ad3eb1d53c95be3cb23546916d1b866bc847 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:18 +0100 Subject: [PATCH 0971/1352] docs: filesystems: update mmap_prepare docs for discontig kernel pgs Describe the newly introduced discontiguous kernel page mapping mechanism, detailing how to use it sensibly and how the API looks. Explicitly detail the various discontiguous actions available and how to use them. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-9-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- Documentation/filesystems/mmap_prepare.rst | 81 ++++++++++++++++++++++ 1 file changed, 81 insertions(+) diff --git a/Documentation/filesystems/mmap_prepare.rst b/Documentation/filesystems/mmap_prepare.rst index 82c99c95ad854e..a476e1006bf126 100644 --- a/Documentation/filesystems/mmap_prepare.rst +++ b/Documentation/filesystems/mmap_prepare.rst @@ -164,5 +164,86 @@ pointer. These are: sufficient entries in the page array to cover the entire range of the described VMA. +* mmap_action_map_discontig_kernel_pages() - Maps a discontiguous range of + `struct page` pointers over the VMA. They must span from the start of the VMA, + but may terminate prior to the end (leaving the remainder unmapped). + **NOTE:** The ``action`` field should never normally be manipulated directly, rather you ought to use one of these helpers. + +Discontiguous Actions +===================== + +Some actions can be performed across discontiguous ranges. + +Map kernel pages +---------------- + +To map kernel pages discontiguously, you must provide hooks using ``struct +discontig_kernel_page_ops``: + +.. code-block:: C + + struct discontig_kernel_page_ops { + int (*init)(void *vm_private_data, void **private); + int (*get)(struct discontig_kernel_page_state *state); + }; + +The ``init`` hook is optional and allows state to be established before the +operation starts, for instance taking a reference count. Nothing is invoked +after the operation, so ``init`` must not leave locks held, and state that must +be released once the mapping goes away should be released in +``vm_ops->close``. + +The ``init`` hook, if provided, is invoked prior to the operation starting. It +may update what is pointed to by ``vm_private_data`` and/or ``private``. If an +error is returned, then the operation is aborted. The ``private`` field can be +reassigned. + +**NOTE:** The operation may sleep between invocations of ``get``, so locks +needed to stabilise state must be taken and released within each hook. + +The ``get`` handler is the key means through which the operation is +executed. The current state of the operation is provided through ``struct +discontig_kernel_page_state``: + +.. code-block:: C + + struct discontig_kernel_page_state { + /* Map state. */ + unsigned long start; /* Start address of VMA. */ + unsigned long end; /* End address of VMA. */ + unsigned long addr; /* The current address to be mapped. */ + pgoff_t pgoff; /* The current pgoff to be mapped. */ + unsigned long nr_pages_mapped; /* The number of pages mapped. */ + unsigned long nr_pages_remain; /* The number of pages remaining. */ + + /* User-defined state. */ + void *vm_private_data; /* VMA private data. */ + void *private; /* Mapping private data. */ + + /* Users should not touch these, use discontig_kernel_map_*() helpers. */ + ... internal fields ... + }; + +With ``private`` being an additional user-controllable state variable, +initialised via ``mmap_action_map_discontig_kernel_pages()``, and +``vm_private_data`` being equal to the ``desc->private_data`` field set in +the ``mmap_prepare()`` hook. + +In the ``get`` hook, the user must choose how to map kernel pages: + +* ``discontig_kernel_map_abort()`` - Call this to abort the operation, whatever + has been mapped so far will be retained, the rest of the mapping will SIGBUS + if accessed. +* ``discontig_kernel_map_page()`` - Maps a single page, correctly handling + compound pages (if the compound page is bigger than the remaining pages in the + VMA, then only those pages that fit will be mapped). For a compound page, the + head page must be passed. +* ``discontig_kernel_map_page_range()`` - Map an array of pages of a specified + size. Note that if the number of pages specified exceeds the VMA size then an + error will arise. + +If an error arises after ``init`` succeeded, the core unmaps the VMA, invoking +``vm_ops->close`` if set, which is therefore the place to release any state +that ``init`` established. From 0a8c67dfd003295a6b03984341ee2fb0591d3371 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:19 +0100 Subject: [PATCH 0972/1352] drivers/usb/mon: update to use mmap_prepare + map kernel pages Replace the deprecated .mmap hook with its replacement .mmap_prepare. As part of this change, additionally take the approach of mapping pages upon mmap rather than providing a fault handler. The page span cannot be mutated when an mmap mapping is in place, so this is safe to do in advance (the MON_IOCT_RING_SIZE ioctl operation exits -EBUSY if it's attempted, gated by the rp->mmap_active reference count). Utilise the newly introduced mmap_action_map_discontig_kernel_pages() to do this, which allows for iteration over pages in mon_bin_discontig_get(). mon_bin_discontig_init() increments the rp->mmap_active reference count to stabilise page spans. Should an error arise the core unmaps the VMA and mon_bin_vma_close() drops the reference again. The vm_ops->close hook implemented in mon_bin_vma_close() will ensure correct reference count arithmetic upon unmap (with mon_bin_vma_open() accounting for splitting). The existing semantics are all retained, including not mapping past the range of available pages, with a SIGBUS being raised in a userland process that attempts to access past this point. Ultimately insert_page() is invoked to insert each page, which increments the reference count on each mapped page. This mimics what was being done previously, only we pre-map the entire range rather than doing so on demand. The existing fault handler did nothing that required demand paging, and was presumably implemented this way for historical reasons. One behavioural difference: pages are no longer faulted in on demand, so a page discarded with MADV_DONTNEED is not repopulated and a subsequent access raises SIGBUS, as with other pre-populated kernel mappings. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-10-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Greg Kroah-Hartman Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- drivers/usb/mon/mon_bin.c | 82 +++++++++++++++++++++++++-------------- 1 file changed, 53 insertions(+), 29 deletions(-) diff --git a/drivers/usb/mon/mon_bin.c b/drivers/usb/mon/mon_bin.c index 687f6a8981f34f..9d00b21a8153be 100644 --- a/drivers/usb/mon/mon_bin.c +++ b/drivers/usb/mon/mon_bin.c @@ -1219,6 +1219,15 @@ mon_bin_poll(struct file *file, struct poll_table_struct *wait) return mask; } +static void __mon_bin_vma_open(struct mon_reader_bin *rp) +{ + unsigned long flags; + + spin_lock_irqsave(&rp->b_lock, flags); + rp->mmap_active++; + spin_unlock_irqrestore(&rp->b_lock, flags); +} + /* * open and close: just keep track of how many times the device is * mapped, to use the proper memory allocation function. @@ -1226,64 +1235,79 @@ mon_bin_poll(struct file *file, struct poll_table_struct *wait) static void mon_bin_vma_open(struct vm_area_struct *vma) { struct mon_reader_bin *rp = vma->vm_private_data; - unsigned long flags; - spin_lock_irqsave(&rp->b_lock, flags); - rp->mmap_active++; - spin_unlock_irqrestore(&rp->b_lock, flags); + __mon_bin_vma_open(rp); } -static void mon_bin_vma_close(struct vm_area_struct *vma) +static void __mon_bin_vma_close(struct mon_reader_bin *rp) { unsigned long flags; - struct mon_reader_bin *rp = vma->vm_private_data; spin_lock_irqsave(&rp->b_lock, flags); rp->mmap_active--; spin_unlock_irqrestore(&rp->b_lock, flags); } -/* - * Map ring pages to user space. - */ -static vm_fault_t mon_bin_vma_fault(struct vm_fault *vmf) +static void mon_bin_vma_close(struct vm_area_struct *vma) { - struct mon_reader_bin *rp = vmf->vma->vm_private_data; + struct mon_reader_bin *rp = vma->vm_private_data; + + __mon_bin_vma_close(rp); +} + +static const struct vm_operations_struct mon_bin_vm_ops = { + .open = mon_bin_vma_open, + .close = mon_bin_vma_close, +}; + +static int mon_bin_discontig_init(void *vm_private_data, void **private) +{ + struct mon_reader_bin *rp = vm_private_data; + + /* Dropped by mon_bin_vma_close() on unmap, including on error. */ + __mon_bin_vma_open(rp); + return 0; +} + +static int mon_bin_discontig_get(struct discontig_kernel_page_state *state) +{ + struct mon_reader_bin *rp = state->vm_private_data; unsigned long offset, chunk_idx; - struct page *pageptr; unsigned long flags; spin_lock_irqsave(&rp->b_lock, flags); - offset = vmf->pgoff << PAGE_SHIFT; + + offset = state->pgoff << PAGE_SHIFT; if (offset >= rp->b_size) { spin_unlock_irqrestore(&rp->b_lock, flags); - return VM_FAULT_SIGBUS; + discontig_kernel_map_abort(state); + return 0; } chunk_idx = offset / CHUNK_SIZE; - pageptr = rp->b_vec[chunk_idx].pg; - get_page(pageptr); - vmf->page = pageptr; + discontig_kernel_map_page(state, rp->b_vec[chunk_idx].pg); + spin_unlock_irqrestore(&rp->b_lock, flags); return 0; } -static const struct vm_operations_struct mon_bin_vm_ops = { - .open = mon_bin_vma_open, - .close = mon_bin_vma_close, - .fault = mon_bin_vma_fault, +static const struct discontig_kernel_page_ops mon_discontig_ops = { + .init = mon_bin_discontig_init, + .get = mon_bin_discontig_get, }; -static int mon_bin_mmap(struct file *filp, struct vm_area_struct *vma) +static int mon_bin_mmap_prepare(struct vm_area_desc *desc) { - /* don't do anything here: "fault" will set up page table entries */ - vma->vm_ops = &mon_bin_vm_ops; + const struct file *filp = desc->file; - if (vma->vm_flags & VM_WRITE) + if (vma_desc_test(desc, VMA_WRITE_BIT)) return -EPERM; - vm_flags_mod(vma, VM_DONTEXPAND | VM_DONTDUMP, VM_MAYWRITE); - vma->vm_private_data = filp->private_data; - mon_bin_vma_open(vma); + desc->vm_ops = &mon_bin_vm_ops; + vma_desc_clear_flags(desc, VMA_MAYWRITE_BIT); + vma_desc_set_flags(desc, VMA_DONTEXPAND_BIT, VMA_DONTDUMP_BIT); + desc->private_data = filp->private_data; + + mmap_action_map_discontig_kernel_pages(desc, NULL, &mon_discontig_ops); return 0; } @@ -1298,7 +1322,7 @@ static const struct file_operations mon_fops_binary = { .compat_ioctl = mon_bin_compat_ioctl, #endif .release = mon_bin_release, - .mmap = mon_bin_mmap, + .mmap_prepare = mon_bin_mmap_prepare, }; static int mon_bin_wait_event(struct file *file, struct mon_reader_bin *rp) From 6e9eeb6ae0f7000b64388ded04ec0c17806df46c Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:20 +0100 Subject: [PATCH 0973/1352] infiniband: update hfi1 to use remap_vmalloc_range() In cases which map chip memory from vmalloc()'d ranges, the hfi1 infiniband drivers currently installs a fault handler, and then smuggles the kernel virtual address of this range in vma->vm_pgoff. This is exposing KASLR-sensitive internal kernel state in the VMA, and is entirely unnecessary. Instead, use remap_vmalloc_range() to remap the VMA to the span, and eliminate the fault handler altogether. remap_vmalloc_range() checks that the VMA does not extend beyond the vmalloc area, and the driver already requires the VMA to exactly match the span of the memory being mapped, so this has no impact. The memory is all preallocated so not having a fault handler has no impact either, other than pre-mapping the ranges which is beneficial. We also remove the VM_IO flag as it's not appropriate here, and the VM_DONTEXPAND flag as remap_vmalloc_range() will set it (and also mark the range correctly as a mixed map). We also update the vmalloc paths to place the virtual kernel address in memvirt, rather than overloading the physical address memaddr. We predicate the vmalloc handling on the vmalloc flag before we check memvirt for the virtual address-derived PFN remap path, so this works fine. remap_vmalloc_range() requires that the vmalloc()'d areas were all allocated using vmalloc_user() - each of cq->comps, uctxt->subctxt_rcvegrbuf, uctxt->subctxt_rcvhdr_base, uctxt->subctxt_uregbase and dd->events were allocated this way, so that requirement is satisfied. We also remove VM_IO and VM_DONTEXPAND from the STATUS command, as these are both set on remap. Finally, we remove VM_DONTEXPAND from the PIO_BUFS, PIO_BUFS_SOP and UREGS commands, as these are also all set on remap. PIO_CRED retains it, as dma_mmap_coherent() may map via vm_insert_page() on the IOMMU-DMA path, which sets only VM_MIXEDMAP. The RCV_HDRQ, RCV_EGRBUF and RTAIL commands also map via dma_mmap_coherent() and never set VM_DONTEXPAND, so set it for them for the same reason. Note that we retain expected behaviour throughout - the vmalloc remapped ranges set VM_MIXEDMAP | VM_DONTDUMP | VM_DONTEXPAND for each range. VM_IO was never appropriate as the ranges are explicitly not MMIO, and the reference to the v3.7 VM_RESERVED semantics map on to VM_MIXEDMAP | VM_DONTDUMP | VM_DONTEXPAND correctly - no core dump, unmergeable, no normal vm page for purposes of reclaim/migration/etc. There is a change in behaviour in that pages mapped using remap_vmalloc_range() will now have normal GUP-able pages, however this should have no impact as there is no reason not to allow this. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-11-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- drivers/infiniband/hw/hfi1/file_ops.c | 84 +++++++++------------------ 1 file changed, 26 insertions(+), 58 deletions(-) diff --git a/drivers/infiniband/hw/hfi1/file_ops.c b/drivers/infiniband/hw/hfi1/file_ops.c index 1a36f995c4f6db..0fc2ba5f1f9113 100644 --- a/drivers/infiniband/hw/hfi1/file_ops.c +++ b/drivers/infiniband/hw/hfi1/file_ops.c @@ -70,7 +70,6 @@ static int set_ctxt_pkey(struct hfi1_ctxtdata *uctxt, unsigned long arg); static int ctxt_reset(struct hfi1_ctxtdata *uctxt); static int manage_rcvq(struct hfi1_ctxtdata *uctxt, u16 subctxt, unsigned long arg); -static vm_fault_t vma_fault(struct vm_fault *vmf); static long hfi1_file_ioctl(struct file *fp, unsigned int cmd, unsigned long arg); @@ -85,10 +84,6 @@ static const struct file_operations hfi1_file_ops = { .llseek = noop_llseek, }; -static const struct vm_operations_struct vm_ops = { - .fault = vma_fault, -}; - /* * Types of memories mapped into user processes' space */ @@ -304,13 +299,13 @@ static ssize_t hfi1_write_iter(struct kiocb *kiocb, struct iov_iter *from) return reqs; } -static inline void mmap_cdbg(u16 ctxt, u8 subctxt, u8 type, u8 mapio, u8 vmf, +static inline void mmap_cdbg(u16 ctxt, u8 subctxt, u8 type, u8 mapio, u8 is_vmalloc, u64 memaddr, void *memvirt, dma_addr_t memdma, ssize_t memlen, struct vm_area_struct *vma) { hfi1_cdbg(PROC, - "%u:%u type:%u io/vf/dma:%d/%d/%d, addr:0x%llx, len:%lu(%lu), flags:0x%lx", - ctxt, subctxt, type, mapio, vmf, !!memdma, + "%u:%u type:%u io/vmalloc/dma:%d/%d/%d, addr:0x%llx, len:%lu(%lu), flags:0x%lx", + ctxt, subctxt, type, mapio, is_vmalloc, !!memdma, memaddr ?: (u64)memvirt, memlen, vma->vm_end - vma->vm_start, vma->vm_flags); } @@ -325,7 +320,8 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) memaddr = 0; void *memvirt = NULL; dma_addr_t memdma = 0; - u8 subctxt, mapio = 0, vmf = 0, type; + u8 subctxt, mapio = 0, type; + u8 is_vmalloc = 0; size_t memdmalen = 0; ssize_t memlen = 0; int ret = 0; @@ -348,7 +344,7 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) /* * vm_pgoff is used as a buffer selector cookie. Always mmap from * the beginning. - */ + */ vma->vm_pgoff = 0; flags = vma->vm_flags; @@ -367,7 +363,7 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) */ memlen = PAGE_ALIGN(uctxt->sc->credits * PIO_BLOCK_SIZE); flags &= ~VM_MAYREAD; - flags |= VM_DONTCOPY | VM_DONTEXPAND; + flags |= VM_DONTCOPY; vma->vm_page_prot = pgprot_writecombine(vma->vm_page_prot); mapio = 1; break; @@ -411,6 +407,7 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) memlen = rcvhdrq_size(uctxt); memvirt = uctxt->rcvhdrq; memdma = uctxt->rcvhdrq_dma; + flags |= VM_DONTEXPAND; break; case RCV_EGRBUF: { unsigned long vm_start_save; @@ -432,7 +429,7 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) ret = -EPERM; goto done; } - vm_flags_clear(vma, VM_MAYWRITE); + vm_flags_mod(vma, VM_DONTEXPAND, VM_MAYWRITE); /* * Mmap multiple separate allocations into a single vma. From * here, dma_mmap_coherent() calls dma_direct_mmap(), which @@ -448,7 +445,7 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) memvirt = uctxt->egrbufs.buffers[i].addr; memdma = uctxt->egrbufs.buffers[i].dma; vma->vm_end += memlen; - mmap_cdbg(ctxt, subctxt, type, mapio, vmf, memaddr, + mmap_cdbg(ctxt, subctxt, type, mapio, is_vmalloc, memaddr, memvirt, memdma, memlen, vma); ret = dma_mmap_coherent(&dd->pcidev->dev, vma, memvirt, memdma, memlen); @@ -477,7 +474,7 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) * user registers. */ memlen = PAGE_SIZE; - flags |= VM_DONTCOPY | VM_DONTEXPAND; + flags |= VM_DONTCOPY; vma->vm_page_prot = pgprot_noncached(vma->vm_page_prot); mapio = 1; break; @@ -486,15 +483,10 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) * Use the page where this context's flags are. User level * knows where it's own bitmap is within the page. */ - memaddr = (unsigned long) - (dd->events + uctxt_offset(uctxt)) & PAGE_MASK; + memvirt = dd->events + uctxt_offset(uctxt); + memvirt = (void *)(((uintptr_t)memvirt) & PAGE_MASK); memlen = PAGE_SIZE; - /* - * v3.7 removes VM_RESERVED but the effect is kept by - * using VM_IO. - */ - flags |= VM_IO | VM_DONTEXPAND; - vmf = 1; + is_vmalloc = 1; break; case STATUS: if (flags & VM_WRITE) { @@ -503,7 +495,6 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) } memaddr = kvirt_to_phys((void *)dd->status); memlen = PAGE_SIZE; - flags |= VM_IO | VM_DONTEXPAND; break; case RTAIL: if (!HFI1_CAP_IS_USET(DMA_RTAIL)) { @@ -522,25 +513,23 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) memvirt = (void *)hfi1_rcvhdrtail_kvaddr(uctxt); memdma = uctxt->rcvhdrqtailaddr_dma; flags &= ~VM_MAYWRITE; + flags |= VM_DONTEXPAND; break; case SUBCTXT_UREGS: - memaddr = (u64)uctxt->subctxt_uregbase; + memvirt = uctxt->subctxt_uregbase; memlen = PAGE_SIZE; - flags |= VM_IO | VM_DONTEXPAND; - vmf = 1; + is_vmalloc = 1; break; case SUBCTXT_RCV_HDRQ: - memaddr = (u64)uctxt->subctxt_rcvhdr_base; + memvirt = uctxt->subctxt_rcvhdr_base; memlen = rcvhdrq_size(uctxt) * uctxt->subctxt_cnt; - flags |= VM_IO | VM_DONTEXPAND; - vmf = 1; + is_vmalloc = 1; break; case SUBCTXT_EGRBUF: - memaddr = (u64)uctxt->subctxt_rcvegrbuf; + memvirt = uctxt->subctxt_rcvegrbuf; memlen = uctxt->egrbufs.size * uctxt->subctxt_cnt; - flags |= VM_IO | VM_DONTEXPAND; flags &= ~VM_MAYWRITE; - vmf = 1; + is_vmalloc = 1; break; case SDMA_COMP: { struct hfi1_user_sdma_comp_q *cq = fd->cq; @@ -549,10 +538,9 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) ret = -EFAULT; goto done; } - memaddr = (u64)cq->comps; + memvirt = cq->comps; memlen = PAGE_ALIGN(sizeof(*cq->comps) * cq->nentries); - flags |= VM_IO | VM_DONTEXPAND; - vmf = 1; + is_vmalloc = 1; break; } default: @@ -569,12 +557,10 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) } vm_flags_reset(vma, flags); - mmap_cdbg(ctxt, subctxt, type, mapio, vmf, memaddr, memvirt, memdma, + mmap_cdbg(ctxt, subctxt, type, mapio, is_vmalloc, memaddr, memvirt, memdma, memlen, vma); - if (vmf) { - vma->vm_pgoff = PFN_DOWN(memaddr); - vma->vm_ops = &vm_ops; - ret = 0; + if (is_vmalloc) { + ret = remap_vmalloc_range(vma, memvirt, 0); } else if (memdma) { ret = dma_mmap_coherent(&dd->pcidev->dev, vma, memvirt, memdma, @@ -599,24 +585,6 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) return ret; } -/* - * Local (non-chip) user memory is not mapped right away but as it is - * accessed by the user-level code. - */ -static vm_fault_t vma_fault(struct vm_fault *vmf) -{ - struct page *page; - - page = vmalloc_to_page((void *)(vmf->pgoff << PAGE_SHIFT)); - if (!page) - return VM_FAULT_SIGBUS; - - get_page(page); - vmf->page = page; - - return 0; -} - static __poll_t hfi1_poll(struct file *fp, struct poll_table_struct *pt) { struct hfi1_ctxtdata *uctxt; From b21bbd8e745b3dde0628bb7e0f24204c5813dac3 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:21 +0100 Subject: [PATCH 0974/1352] selinux: reject writable opens of policy file, drop mmap shared/write check The policy file has no write method and is exposed read-only (S_IRUGO in selinux_files[]), yet sel_open_policy() performs no open mode check, so a CAP_DAC_OVERRIDE caller can open it O_RDWR. Reject FMODE_WRITE at open, as kernfs does. The file can then never be mapped with FMODE_WRITE, so do_mmap() always clears VM_MAYWRITE and VM_SHARED for MAP_SHARED mappings and the VM_SHARED check in sel_mmap_policy() cannot be reached. Remove it. This also stops sel_mmap_policy() clearing VM_MAYWRITE on a mapping that is neither a PFN map nor a mixed map, ahead of the core enforcing that only such mappings may do so. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-12-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Stephen Smalley Reviewed-by: Jann Horn Acked-by: Paul Moore Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- security/selinux/selinuxfs.c | 11 +++-------- 1 file changed, 3 insertions(+), 8 deletions(-) diff --git a/security/selinux/selinuxfs.c b/security/selinux/selinuxfs.c index c7d91476971cb5..545a6f89f9e763 100644 --- a/security/selinux/selinuxfs.c +++ b/security/selinux/selinuxfs.c @@ -340,6 +340,9 @@ static int sel_open_policy(struct inode *inode, struct file *filp) struct policy_load_memory *plm = NULL; int rc; + if (filp->f_mode & FMODE_WRITE) + return -EACCES; + rc = avc_has_perm(current_sid(), SECINITSID_SECURITY, SECCLASS_SECURITY, SECURITY__READ_POLICY, NULL); if (rc) @@ -424,14 +427,6 @@ static const struct vm_operations_struct sel_mmap_policy_ops = { static int sel_mmap_policy(struct file *filp, struct vm_area_struct *vma) { - if (vma->vm_flags & VM_SHARED) { - /* do not allow mprotect to make mapping writable */ - vm_flags_clear(vma, VM_MAYWRITE); - - if (vma->vm_flags & VM_WRITE) - return -EACCES; - } - vm_flags_set(vma, VM_DONTEXPAND | VM_DONTDUMP); vma->vm_ops = &sel_mmap_policy_ops; From 7e6136319fd7172e6368f61e6a56fe8c5a9bd807 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:22 +0100 Subject: [PATCH 0975/1352] ALSA: pcm: use vm_insert_page() to map PCM status page There's no need to keep a fault handler around for this, instead map on mmap. While we're here, rename area to vma to be consistent. This correctly makes the mapping a mixed map mapping. This works towards establishing the invariant that only PFN mapped or mixed map mappings may clear the VM_MAYWRITE flag. The status page mapping clears VM_MAYWRITE, so it must be kernel-owned; the control page mapping remains writable and is left fault-based. The assumption is made that the struct pcm_mmap_status structure is at most a page in size, which is asserted as a build bug. This is safe to assume, as the size of the structure is 56 bytes at most. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-13-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Takashi Iwai Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- sound/core/pcm_native.c | 38 +++++++++++++------------------------- 1 file changed, 13 insertions(+), 25 deletions(-) diff --git a/sound/core/pcm_native.c b/sound/core/pcm_native.c index 6d32c12fb79bdf..af80488baaf427 100644 --- a/sound/core/pcm_native.c +++ b/sound/core/pcm_native.c @@ -3760,39 +3760,27 @@ static __poll_t snd_pcm_poll(struct file *file, poll_table *wait) /* * mmap status record */ -static vm_fault_t snd_pcm_mmap_status_fault(struct vm_fault *vmf) +static int snd_pcm_mmap_status(struct snd_pcm_substream *substream, struct file *file, + struct vm_area_struct *vma) { - struct snd_pcm_substream *substream = vmf->vma->vm_private_data; + const unsigned long size = vma->vm_end - vma->vm_start; struct snd_pcm_runtime *runtime; - - if (substream == NULL) - return VM_FAULT_SIGBUS; - runtime = substream->runtime; - vmf->page = virt_to_page(runtime->status); - get_page(vmf->page); - return 0; -} + struct page *page; -static const struct vm_operations_struct snd_pcm_vm_ops_status = -{ - .fault = snd_pcm_mmap_status_fault, -}; + BUILD_BUG_ON(sizeof(struct snd_pcm_mmap_status) > PAGE_SIZE); -static int snd_pcm_mmap_status(struct snd_pcm_substream *substream, struct file *file, - struct vm_area_struct *area) -{ - long size; - if (!(area->vm_flags & VM_READ)) + if (!(vma->vm_flags & VM_READ)) return -EINVAL; - size = area->vm_end - area->vm_start; - if (size != PAGE_ALIGN(sizeof(struct snd_pcm_mmap_status))) + if (size != PAGE_SIZE) return -EINVAL; - area->vm_ops = &snd_pcm_vm_ops_status; - area->vm_private_data = substream; - vm_flags_mod(area, VM_DONTEXPAND | VM_DONTDUMP, + + vm_flags_mod(vma, VM_DONTEXPAND | VM_DONTDUMP, VM_WRITE | VM_MAYWRITE); + vma->vm_page_prot = vm_get_page_prot(vma->vm_flags); - return 0; + runtime = substream->runtime; + page = virt_to_page(runtime->status); + return vm_insert_page(vma, vma->vm_start, page); } /* From 2ea6655b9c850a1bd4737edb0982b4dffc5e692c Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:23 +0100 Subject: [PATCH 0976/1352] bpf: arena: mark arena_map_mmap() mappings VM_MIXEDMAP The bpf_map->ops->map_mmap callback invoked by bpf_map_mmap() can be set to one of ringbuf_map_mmap_kern(), ringbuf_map_mmap_user(), array_map_mmap() or arena_map_mmap(). It is convention in mm to mark mappings whose pages the kernel manages itself with VM_MIXEDMAP, so the core mm knows not to treat them as ordinary page cache or anonymous memory. The map_mmap callbacks ringbuf_map_mmap_kern() and ringbuf_map_mmap_user() use remap_vmalloc_range(), which ultimately invokes vm_insert_page() and so marks the ranges VM_MIXEDMAP, and array_map_mmap() sets VM_MIXEDMAP explicitly. However, the exception to this is arena_map_mmap(), which doesn't set the flag. This patch corrects this and updates the comment to reflect it. The pages are refcounted and vm_normal_page() finds them regardless of the flag, and VM_DONTEXPAND remains set (marking the memory as VM_SPECIAL and thus unmergeable). The one effect is that NUMA balancing now skips these VMAs, as it already does for the other bpf map mappings, which is the reason array_map_mmap() gives for setting the flag. The intent of this patch is to be able to establish the invariant that only PFN-mapped or mixed map ranges may clear the VM_MAYWRITE flag, as is done in bpf_map_mmap(). Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-14-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Emil Tsalapatis Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- kernel/bpf/arena.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/kernel/bpf/arena.c b/kernel/bpf/arena.c index 7b6847200b4312..b69fe5e343393d 100644 --- a/kernel/bpf/arena.c +++ b/kernel/bpf/arena.c @@ -620,8 +620,9 @@ static int arena_map_mmap(struct bpf_map *map, struct vm_area_struct *vma) * clears VM_MAYEXEC. Set VM_DONTEXPAND to avoid potential change * of user_vm_start. Set VM_DONTCOPY to prevent arena VMA from * being copied into the child process on fork. + * This is a kernel page so set VM_MIXEDMAP. */ - vm_flags_set(vma, VM_DONTEXPAND | VM_DONTCOPY); + vm_flags_set(vma, VM_MIXEDMAP | VM_DONTEXPAND | VM_DONTCOPY); vma->vm_ops = &arena_vm_ops; return 0; } From 4358cd6f469f7d99660a0268296d9697fc6ef6c9 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:24 +0100 Subject: [PATCH 0977/1352] mm/vma: add vma[_flags]_is_kernel_owned() predicates Rather than referring to VMA flags with uncertain meaning, add a new predicate that explicitly describes what possession of the VMA_PFNMAP_BIT or VMA_MIXEDMAP_BIT flags mean, and then refer to that function for determining VMA mergeability. Either flag means the contents of the mapping are owned by the kernel, usually a driver, rather than by the core mm: the memory may be MMIO, kernel-allocated pages or even ordinary pages the driver maps itself, but the core must not populate, reclaim, migrate, copy-on-write or merge the range on its own initiative. We initially also include VMA_IO_BIT here, as by implication, these must be kernel-owned. (mlock() also sets VMA_IO_BIT transiently on ordinary VMAs while locking them, which is addressed later in this series.) However the intent is to in future remove this, as no mapping should be marked as an I/O mapping without also being marked with VMA_PFNMAP_BIT. This forms the basis of further work intended to improve how we express VMA properties such as this. Also update the VMA userland tests to reflect the change. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-15-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- include/linux/mm.h | 56 ++++++++++++++++++++++++++++++++- tools/testing/vma/include/dup.h | 29 ++++++++++++++++- 2 files changed, 83 insertions(+), 2 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index 5602a89156775c..4f235e0ec385fb 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -1613,6 +1613,44 @@ static inline bool vma_is_shared_maywrite(const struct vm_area_struct *vma) return is_shared_maywrite(&vma->flags); } +/** + * vma_flags_is_kernel_owned() - Do the specified VMA flags indicate that the + * contents of the VMA are owned by the kernel rather than the core mm? + * @flags: The VMA flags to test. + * + * A kernel-owned mapping is one whose contents are established and controlled + * by the kernel, typically a driver, rather than by the core mm's fault and + * rmap machinery. + * + * The mapping may be memory-mapped I/O, kernel-allocated pages or ordinary + * pages the owner has chosen to map itself (shmem via a PFN map, for instance). + * + * In all cases the core mm must not populate, reclaim, migrate, copy-on-write + * or merge it of its own accord. + * + * Pages mapped this way are not necessarily reference counted or map counted. + * + * Returns: true if the flags indicate a kernel-owned mapping. + */ +static inline bool vma_flags_is_kernel_owned(const vma_flags_t *flags) +{ + return vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT, + VMA_IO_BIT); +} + +/** + * vma_is_kernel_owned() - Are the contents of @vma owned by the kernel? + * @vma: The VMA to test. + * + * See vma_flags_is_kernel_owned() for a description of this property. + * + * Returns: true if the VMA is kernel-owned. + */ +static inline bool vma_is_kernel_owned(const struct vm_area_struct *vma) +{ + return vma_flags_is_kernel_owned(&vma->flags); +} + /** * vma_flags_can_merge() - Do the specified VMA flags permit the VMA to be * merged with another? @@ -1621,7 +1659,23 @@ static inline bool vma_is_shared_maywrite(const struct vm_area_struct *vma) */ static inline bool vma_flags_can_merge(const vma_flags_t *flags) { - return !vma_flags_test_any_mask(flags, VMA_SPECIAL_FLAGS); + /* + * VMA merging assumes that a VMA's flags and fields completely describe + * its state. + * + * However, kernel-owned mappings may have established state upon mapping + * not embodied in any attribute of the VMA. + * + * Additionally, private (CoW) PFN maps encode the source PFN of the + * range in vma->vm_pgoff, which may otherwise cause spurious merges. + */ + if (vma_flags_is_kernel_owned(flags)) + return false; + /* VMA explicitly marked as being unmergeable. */ + if (vma_flags_test(flags, VMA_DONTEXPAND_BIT)) + return false; + + return true; } /** diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index 1d5f6b3cbd21e8..d09148ce23054d 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -1665,7 +1665,34 @@ static inline bool file_is_dev_zero(const struct file *file) return file && file->f_op == &zero_fops; } +static inline bool vma_flags_is_kernel_owned(const vma_flags_t *flags) +{ + return vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT, + VMA_IO_BIT); +} + +static inline bool vma_is_kernel_owned(const struct vm_area_struct *vma) +{ + return vma_flags_is_kernel_owned(&vma->flags); +} + static inline bool vma_flags_can_merge(const vma_flags_t *flags) { - return !vma_flags_test_any_mask(flags, VMA_SPECIAL_FLAGS); + /* + * VMA merging assumes that a VMA's flags and fields completely describe + * its state. + * + * However, kernel-owned mappings may have established state upon mapping + * not embodied in any attribute of the VMA. + * + * Additionally, private (CoW) PFN maps encode the source PFN of the + * range in vma->vm_pgoff, which may otherwise cause spurious merges. + */ + if (vma_flags_is_kernel_owned(flags)) + return false; + /* VMA explicitly marked as being unmergeable. */ + if (vma_flags_test(flags, VMA_DONTEXPAND_BIT)) + return false; + + return true; } From 8e30f7ed8eb20624032499ea1f1b4bcd9cf78a62 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:25 +0100 Subject: [PATCH 0978/1352] mm/vma: only allow mmap to clear VMA_MAYWRITE_BIT if kernel-owned For ordinary files the only way the VMA_MAYWRITE_BIT flag is cleared is if the underlying file is itself read-only. This means that mprotect() cannot mark a shared mapping of a read-only file as read/write, as doing so would violate the read only attribute, and permit writes. In general, we do not want file systems to be able to do this for read/write files. Doing so would violate fundamental user expectation of file attributes and likely break userspace. However, drivers pose a tricky problem here - the /dev/xxx file may be read/write but provide access to a resource which is fundamentally read-only. Therefore we must allow drivers to be able to clear VMA_MAYWRITE_BIT. To achieve both of these things, restrict this ability to kernel-owned mappings as identified by vma_flags_is_kernel_owned(). This constrains this ability to drivers which own the mapping's contents, whether memory-mapped I/O, kernel-allocated pages, or ordinary pages they map themselves, and so define its semantics. Every in-tree mmap hook which clears VMA_MAYWRITE_BIT, some twenty sites across drivers, filesystems and bpf, establishes a kernel-owned mapping, with usbmon and the ALSA PCM status page converted earlier in this series to do so. Note that drivers may, if they do not gate on VMA_SHARED_BIT, be able to disable MAP_PRIVATE-file-backed mapping CoW semantics. This is perhaps not always intended, but we retain this capacity to maintain existing behaviour. As all drivers which clear VMA_MAYWRITE_BIT establish kernel-owned mappings, no functional change is intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-16-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/vma.c | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/mm/vma.c b/mm/vma.c index cfaf217d9c1c3f..81d7061f3e9ecf 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -2810,6 +2810,11 @@ static int mmap_validate(unsigned long prev_start, unsigned long prev_end, if (WARN_ON_ONCE(!was_maywrite && is_maywrite)) return -EINVAL; + /* Only kernel-owned mappings may clear VMA_MAYWRITE_BIT. */ + if (!vma_flags_is_kernel_owned(curr_flags) && + WARN_ON_ONCE(was_maywrite && !is_maywrite)) + return -EINVAL; + return mmap_validate_vma_flags(curr_flags); } From 630efd07c105cf7058148d1f62ef6ff5d59f6623 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:26 +0100 Subject: [PATCH 0979/1352] mm/vma: add and use vma_[flags]_is_fixed_mapping This determines whether a VMA cannot be expanded or merged because what they mapped was determined to be a set size at mmap time. This typically refers to kernel-owned mappings, however VMA_DONTEXPAND_BIT is not reliably set alongside VMA_PFNMAP_BIT or VMA_MIXEDMAP_BIT, so we must explicitly test for this for now. We also explicitly test for VMA_PFNMAP_BIT as VMA_DONTEXPAND_BIT may not be set for VMA_PFNMAP_BIT's despite the one implying the other. Use this predicate in vma_flags_can_merge() and in check_prep_vma() in the mremap logic testing to see if mremap() can expand the VMA. The criteria for khugepaged and MADV_COLLAPSE eligibility in __thp_vma_allowable_orders() are precisely those for mergeability, so use vma_can_merge() there (with an expanded comment). This obviates the need for the VM_NO_KHUGEPAGED mask, so remove it. Hugetlb VMAs remain excluded from khugepaged as hugetlbfs always sets VMA_DONTEXPAND_BIT. Also update the userland VMA tests to reflect the change. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-17-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- include/linux/mm.h | 39 +++++++++++++++++++++++++++++---- mm/huge_memory.c | 11 ++++++---- mm/mremap.c | 5 ++--- tools/testing/vma/include/dup.h | 16 +++++++++++++- 4 files changed, 59 insertions(+), 12 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index 4f235e0ec385fb..a7fa4df6fd4747 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -601,9 +601,6 @@ enum { #define VMA_REMAP_FLAGS mk_vma_flags(VMA_IO_BIT, VMA_PFNMAP_BIT, \ VMA_DONTEXPAND_BIT, VMA_DONTDUMP_BIT) -/* This mask prevents VMA from being scanned with khugepaged */ -#define VM_NO_KHUGEPAGED (VM_SPECIAL | VM_HUGETLB) - /* This mask defines which mm->def_flags a process can inherit its parent */ #define VM_INIT_DEF_MASK VM_NOHUGEPAGE @@ -1651,6 +1648,40 @@ static inline bool vma_is_kernel_owned(const struct vm_area_struct *vma) return vma_flags_is_kernel_owned(&vma->flags); } +/** + * vma_flags_is_fixed_mapping() - Do the specified VMA flags indicate that this + * is a fixed mapping that cannot be expanded or merged? + * @flags: The VMA flags to test. + * + * Fixed mappings are those whose size is set at the point of mmap (for + * instance, a kernel-owned mapping of a fixed range of memory), and thus + * cannot be expanded or merged. + * + * Returns: true if the flags indicate a fixed mapping. + */ +static inline bool vma_flags_is_fixed_mapping(const vma_flags_t *flags) +{ + /* + * VMA_PFNMAP_BIT should imply VMA_DONTEXPAND_BIT, but some callers set + * only the former. + */ + return vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_DONTEXPAND_BIT); +} + +/** + * vma_is_fixed_mapping() - Is this VMA a fixed mapping that cannot be + * expanded or merged? + * @vma: The VMA to test. + * + * See vma_flags_is_fixed_mapping() for a description of this property. + * + * Returns: true if the VMA maps a fixed mapping. + */ +static inline bool vma_is_fixed_mapping(const struct vm_area_struct *vma) +{ + return vma_flags_is_fixed_mapping(&vma->flags); +} + /** * vma_flags_can_merge() - Do the specified VMA flags permit the VMA to be * merged with another? @@ -1672,7 +1703,7 @@ static inline bool vma_flags_can_merge(const vma_flags_t *flags) if (vma_flags_is_kernel_owned(flags)) return false; /* VMA explicitly marked as being unmergeable. */ - if (vma_flags_test(flags, VMA_DONTEXPAND_BIT)) + if (vma_flags_is_fixed_mapping(flags)) return false; return true; diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 8f8bf60a22649a..6ee21854b6dfa9 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -212,11 +212,14 @@ unsigned long __thp_vma_allowable_orders(struct vm_area_struct *vma, return in_pf ? orders : 0; /* - * khugepaged special VMA and hugetlb VMA. - * Must be checked after dax since some dax mappings may have - * VM_MIXEDMAP set. + * khugepaged moves data from VMAs once collapsed, after they have been + * faulted in, relying on refaulting for file-backed memory. + * + * Kernel-owned mappings cannot be reliably reconstructed from page + * faults, and fixed mappings (including hugetlb) may not be marked as + * kernel-owned - precisely the mappings which cannot be merged. */ - if (!in_pf && !smaps && (vm_flags & VM_NO_KHUGEPAGED)) + if (!in_pf && !smaps && !vma_can_merge(vma)) return 0; /* diff --git a/mm/mremap.c b/mm/mremap.c index 444f37d7cd28ec..982460ef6bdbb1 100644 --- a/mm/mremap.c +++ b/mm/mremap.c @@ -1822,8 +1822,7 @@ static int check_prep_vma(struct vma_remap_struct *vrm) return -EINVAL; } - if ((vrm->flags & MREMAP_DONTUNMAP) && - vma_test_any(vma, VMA_DONTEXPAND_BIT, VMA_PFNMAP_BIT)) + if ((vrm->flags & MREMAP_DONTUNMAP) && vma_is_fixed_mapping(vma)) return -EINVAL; /* @@ -1861,7 +1860,7 @@ static int check_prep_vma(struct vma_remap_struct *vrm) if (pgoff + (new_len >> PAGE_SHIFT) < pgoff) return -EINVAL; - if (vma_test_any(vma, VMA_DONTEXPAND_BIT, VMA_PFNMAP_BIT)) + if (vma_is_fixed_mapping(vma)) return -EFAULT; if (!mlock_future_ok(mm, vma_test(vma, VMA_LOCKED_BIT), vrm->delta)) diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index d09148ce23054d..b8b1462ca71052 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -1676,6 +1676,20 @@ static inline bool vma_is_kernel_owned(const struct vm_area_struct *vma) return vma_flags_is_kernel_owned(&vma->flags); } +static inline bool vma_flags_is_fixed_mapping(const vma_flags_t *flags) +{ + /* + * VMA_PFNMAP_BIT should imply VMA_DONTEXPAND_BIT, but some callers set + * only the former. + */ + return vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_DONTEXPAND_BIT); +} + +static inline bool vma_is_fixed_mapping(const struct vm_area_struct *vma) +{ + return vma_flags_is_fixed_mapping(&vma->flags); +} + static inline bool vma_flags_can_merge(const vma_flags_t *flags) { /* @@ -1691,7 +1705,7 @@ static inline bool vma_flags_can_merge(const vma_flags_t *flags) if (vma_flags_is_kernel_owned(flags)) return false; /* VMA explicitly marked as being unmergeable. */ - if (vma_flags_test(flags, VMA_DONTEXPAND_BIT)) + if (vma_flags_is_fixed_mapping(flags)) return false; return true; From 1abc4eb4c2ee8f58ddf393b833c353f1dfb720c1 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:27 +0100 Subject: [PATCH 0980/1352] scsi: sg: convert mmap hook to mmap_prepare and rework Move from the deprecated mmap hook to the new mmap_prepare hook. We are mapping kernel pages here, so use the discontiguous kernel mapping mmap action to do so. Unwind the rather confusing loop and instead map as many pages as we can at one time. Note that we do not need to pay attention to rsv_schp->k_use_sg here, as the pages are populated for the length of the buffer at rsv_schp->page_order granularity as compound pages. The discontiguous kernel page mapping logic handles the compound pages for us. sfp->mmap_called keeps the buffer stable for us. As before it is never cleared, so a failed mmap also leaves it set. We also remove some useless vma, vma->vm_file NULL checks - these will always be non-NULL if you reached the mmap hook logic. We retain log output for consistency, but change what's output on page mapping to indicate that sg_discontig_get() does the work now. Note that we drop the VMA_IO_BIT flag for the VMA here. It was never necessary as we invoke alloc_pages() which gives us refcounted folios that are fine for GUP to access (VMA_IO_BIT would prevent that). Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-18-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- drivers/scsi/sg.c | 115 ++++++++++++++++++++-------------------------- 1 file changed, 51 insertions(+), 64 deletions(-) diff --git a/drivers/scsi/sg.c b/drivers/scsi/sg.c index 5408f002e6c01f..3f9e08725602ca 100644 --- a/drivers/scsi/sg.c +++ b/drivers/scsi/sg.c @@ -1212,85 +1212,72 @@ sg_fasync(int fd, struct file *filp, int mode) return fasync_helper(fd, filp, mode, &sfp->async_qp); } -static vm_fault_t -sg_vma_fault(struct vm_fault *vmf) +static int sg_discontig_init(void *vm_private_data, void **private) { - struct vm_area_struct *vma = vmf->vma; - Sg_fd *sfp; - unsigned long offset, len, sa; - Sg_scatter_hold *rsv_schp; - int k, length; - - if ((NULL == vma) || (!(sfp = (Sg_fd *) vma->vm_private_data))) - return VM_FAULT_SIGBUS; - rsv_schp = &sfp->reserve; - offset = vmf->pgoff << PAGE_SHIFT; - if (offset >= rsv_schp->bufflen) - return VM_FAULT_SIGBUS; - SCSI_LOG_TIMEOUT(3, sg_printk(KERN_INFO, sfp->parentdp, - "sg_vma_fault: offset=%lu, scatg=%d\n", - offset, rsv_schp->k_use_sg)); - sa = vma->vm_start; - length = 1 << (PAGE_SHIFT + rsv_schp->page_order); - for (k = 0; k < rsv_schp->k_use_sg && sa < vma->vm_end; k++) { - len = vma->vm_end - sa; - len = (len < length) ? len : length; - if (offset < len) { - struct page *page = rsv_schp->pages[k] + (offset >> PAGE_SHIFT); - get_page(page); /* increment page count */ - vmf->page = page; - return 0; /* success */ - } - sa += len; - offset -= len; + const unsigned long req_sz = (unsigned long)*private; + Sg_fd *sfp = vm_private_data; + Sg_scatter_hold *rsv_schp = &sfp->reserve; + int err = 0; + + mutex_lock(&sfp->f_mutex); + if (req_sz > rsv_schp->bufflen) { + err = -ENOMEM; /* cannot map more than reserved buffer */ + goto out; + } + sfp->mmap_called = 1; /* Prevents changes to buffer size. */ +out: + mutex_unlock(&sfp->f_mutex); + return err; +} + +static int +sg_discontig_get(struct discontig_kernel_page_state *state) +{ + Sg_fd *sfp = state->vm_private_data; + Sg_scatter_hold *rsv_schp = &sfp->reserve; + const unsigned int order = rsv_schp->page_order; + const pgoff_t nr_pages = state->nr_pages_mapped; + + if (nr_pages >= (rsv_schp->bufflen >> PAGE_SHIFT)) { + discontig_kernel_map_abort(state); + return 0; } - return VM_FAULT_SIGBUS; + SCSI_LOG_TIMEOUT(3, sg_printk(KERN_INFO, sfp->parentdp, + "%s: offset=%lu, scatg=%d\n", __func__, + nr_pages << PAGE_SHIFT, rsv_schp->k_use_sg)); + + discontig_kernel_map_page(state, rsv_schp->pages[nr_pages >> order]); + return 0; } -static const struct vm_operations_struct sg_mmap_vm_ops = { - .fault = sg_vma_fault, +static const struct discontig_kernel_page_ops sg_discontig_ops = { + .init = sg_discontig_init, + .get = sg_discontig_get, }; static int -sg_mmap(struct file *filp, struct vm_area_struct *vma) +sg_mmap_prepare(struct vm_area_desc *desc) { - Sg_fd *sfp; - unsigned long req_sz, len, sa; - Sg_scatter_hold *rsv_schp; - int k, length; - int ret = 0; + Sg_fd *sfp = desc->file->private_data; + const unsigned long req_sz = vma_desc_size(desc); - if ((!filp) || (!vma) || (!(sfp = (Sg_fd *) filp->private_data))) + if (!sfp) return -ENXIO; - req_sz = vma->vm_end - vma->vm_start; + SCSI_LOG_TIMEOUT(3, sg_printk(KERN_INFO, sfp->parentdp, "sg_mmap starting, vm_start=%p, len=%d\n", - (void *) vma->vm_start, (int) req_sz)); - if (vma->vm_pgoff) + (void *) desc->start, (int) req_sz)); + + if (desc->pgoff) return -EINVAL; /* want no offset */ - rsv_schp = &sfp->reserve; - mutex_lock(&sfp->f_mutex); - if (req_sz > rsv_schp->bufflen) { - ret = -ENOMEM; /* cannot map more than reserved buffer */ - goto out; - } - sa = vma->vm_start; - length = 1 << (PAGE_SHIFT + rsv_schp->page_order); - for (k = 0; k < rsv_schp->k_use_sg && sa < vma->vm_end; k++) { - len = vma->vm_end - sa; - len = (len < length) ? len : length; - sa += len; - } + vma_desc_set_flags(desc, VMA_DONTEXPAND_BIT, VMA_DONTDUMP_BIT); + desc->private_data = sfp; - sfp->mmap_called = 1; - vm_flags_set(vma, VM_IO | VM_DONTEXPAND | VM_DONTDUMP); - vma->vm_private_data = sfp; - vma->vm_ops = &sg_mmap_vm_ops; -out: - mutex_unlock(&sfp->f_mutex); - return ret; + mmap_action_map_discontig_kernel_pages(desc, (void *)req_sz, + &sg_discontig_ops); + return 0; } static void @@ -1415,7 +1402,7 @@ static const struct file_operations sg_fops = { .unlocked_ioctl = sg_ioctl, .compat_ioctl = compat_ptr_ioctl, .open = sg_open, - .mmap = sg_mmap, + .mmap_prepare = sg_mmap_prepare, .release = sg_release, .fasync = sg_fasync, }; From 15f895f7a9081591cdee5c2b7d1419f31b7b8224 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:28 +0100 Subject: [PATCH 0981/1352] fbdev: defio: assert FBINFO_VIRTFB, drop VM_IO, add VM_MIXEDMAP Currently all drivers which use defio allocate system memory. All of them also set FBINFO_VIRTFB, other than ssd1307fb, however this driver allocates system RAM, so simply failed to set this flag when it ought to. This patch sets FBINFO_VIRTFB on ssd1307fb probe, then drops setting VM_IO in fb_deferred_io_mmap() and instead requires FBINFO_VIRTFB to be set, erroring out with a kernel warning if not. The logic requires a page from the driver and since commit 1ecbc7dd2902 ("fbdev/deferred-io: Always call get_page() for framebuffer pages") has always required it to be refcounted, so this was implicitly already the case. Finally this patch sets VM_MIXEDMAP, as the logic is mapping kernel-allocated memory so this is appropriate. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-19-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Thomas Zimmermann Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- drivers/video/fbdev/core/fb_defio.c | 6 +++--- drivers/video/fbdev/ssd1307fb.c | 2 ++ 2 files changed, 5 insertions(+), 3 deletions(-) diff --git a/drivers/video/fbdev/core/fb_defio.c b/drivers/video/fbdev/core/fb_defio.c index fd00b86e1ae602..fb359ecc396619 100644 --- a/drivers/video/fbdev/core/fb_defio.c +++ b/drivers/video/fbdev/core/fb_defio.c @@ -366,13 +366,13 @@ int fb_deferred_io_mmap(struct fb_info *info, struct vm_area_struct *vma) { vma->vm_page_prot = pgprot_decrypted(vma->vm_page_prot); + if (WARN_ON_ONCE(!(info->flags & FBINFO_VIRTFB))) + return -EINVAL; if (!try_module_get(THIS_MODULE)) return -EINVAL; vma->vm_ops = &fb_deferred_io_vm_ops; - vm_flags_set(vma, VM_DONTEXPAND | VM_DONTDUMP); - if (!(info->flags & FBINFO_VIRTFB)) - vm_flags_set(vma, VM_IO); + vm_flags_set(vma, VM_MIXEDMAP | VM_DONTEXPAND | VM_DONTDUMP); vma->vm_private_data = info->fbdefio_state; fb_deferred_io_state_get(info->fbdefio_state); /* released in vma->vm_ops->close() */ diff --git a/drivers/video/fbdev/ssd1307fb.c b/drivers/video/fbdev/ssd1307fb.c index 4d185c75428438..db61d710fccb01 100644 --- a/drivers/video/fbdev/ssd1307fb.c +++ b/drivers/video/fbdev/ssd1307fb.c @@ -767,6 +767,8 @@ static int ssd1307fb_probe(struct i2c_client *client) info->fix.smem_start = __pa(vmem); info->fix.smem_len = vmem_size; + info->flags = FBINFO_VIRTFB; + fb_deferred_io_init(info); i2c_set_clientdata(client, info); From aa47f9d9a42e01219c0ace3b3199bc9115c4db23 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:29 +0100 Subject: [PATCH 0982/1352] HSI: cmt_speech: convert mmap hook to mmap_prepare, refactor Use the mmap_prepare in favour of the deprecated mmap hook as part of the work to convert one to another. Since this is simply a refcounted kernel page that has been allocated, it should not be marked VM_IO and should be inserted using the kernel page insertion mechanism, so convert it to do this instead. Use the VMA descriptor's private data field as a scratch buffer to store the page in - this stays valid throughout the kernel page mapping operation. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-20-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- drivers/hsi/clients/cmt_speech.c | 33 +++++++++----------------------- 1 file changed, 9 insertions(+), 24 deletions(-) diff --git a/drivers/hsi/clients/cmt_speech.c b/drivers/hsi/clients/cmt_speech.c index 7226677ebde7ac..801697b74d4f8b 100644 --- a/drivers/hsi/clients/cmt_speech.c +++ b/drivers/hsi/clients/cmt_speech.c @@ -1084,22 +1084,6 @@ static void cs_hsi_stop(struct cs_hsi_iface *hi) kfree(hi); } -static vm_fault_t cs_char_vma_fault(struct vm_fault *vmf) -{ - struct cs_char *csdata = vmf->vma->vm_private_data; - struct page *page; - - page = virt_to_page((void *)csdata->mmap_base); - get_page(page); - vmf->page = page; - - return 0; -} - -static const struct vm_operations_struct cs_char_vm_ops = { - .fault = cs_char_vma_fault, -}; - static int cs_char_fasync(int fd, struct file *file, int on) { struct cs_char *csdata = file->private_data; @@ -1256,18 +1240,19 @@ static long cs_char_ioctl(struct file *file, unsigned int cmd, return r; } -static int cs_char_mmap(struct file *file, struct vm_area_struct *vma) +static int cs_char_mmap_prepare(struct vm_area_desc *desc) { - if (vma->vm_end < vma->vm_start) - return -EINVAL; + struct file *file = desc->file; + struct cs_char *csdata = file->private_data; + struct page **pages = (struct page **)&desc->private_data; - if (vma_pages(vma) != 1) + if (vma_desc_pages(desc) != 1) return -EINVAL; - vm_flags_set(vma, VM_IO | VM_DONTDUMP | VM_DONTEXPAND); - vma->vm_ops = &cs_char_vm_ops; - vma->vm_private_data = file->private_data; + vma_desc_set_flags(desc, VMA_DONTDUMP_BIT, VMA_DONTEXPAND_BIT); + *pages = virt_to_page((void *)csdata->mmap_base); + mmap_action_map_kernel_pages_full(desc, pages); return 0; } @@ -1353,7 +1338,7 @@ static const struct file_operations cs_char_fops = { .write = cs_char_write, .poll = cs_char_poll, .unlocked_ioctl = cs_char_ioctl, - .mmap = cs_char_mmap, + .mmap_prepare = cs_char_mmap_prepare, .open = cs_char_open, .release = cs_char_release, .fasync = cs_char_fasync, From ea1e73e012d09806841dee91992396ddc06ed6e9 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:30 +0100 Subject: [PATCH 0983/1352] mm/gup: error out early on !VMA_MAYREAD_BIT VMAs When populating a VMA range via the aptly named populate_vma_page_range() an unreadable VMA will always eventually fail with -EFAULT. That a VMA is accessible is always checked, however VMA_MAYREAD_BIT is not. All user mappings always have VMA_MAYREAD_BIT set, so this check only impacts kernel mappings. It is implemented specifically to disallow population of uprobes XOL mappings which are exec-only. A nasty interaction with these mappings may occur if they are mlocked, so actively disallow this early. This allows a subsequent commit to remove the VM_IO check in __mm_populate() which otherwise requires non-MMIO mappings to be wrongly flagged simply as a workaround. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-21-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/gup.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/mm/gup.c b/mm/gup.c index c2dfcb4744bc3f..e6310a7cc05b2b 100644 --- a/mm/gup.c +++ b/mm/gup.c @@ -1836,6 +1836,10 @@ long populate_vma_page_range(struct vm_area_struct *vma, if (!vma_is_accessible(vma)) return -EFAULT; + /* Unreadable VMAs also cannot be faulted in. */ + if (!vma_test(vma, VMA_MAYREAD_BIT)) + return -EFAULT; + gup_flags = FOLL_TOUCH; /* * We want to touch writable mappings with a write fault in order From f85086d3ad29f603ce30e87f2cd112adbcc5c869 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:31 +0100 Subject: [PATCH 0984/1352] uprobes: remove VM_IO, set VM_MIXEDMAP for mapped kernel pages These are not MMIO pages so VMA_IO_BIT is an inappropriate flag to set. Instead, set them VMA_MIXEDMAP_BIT as they are kernel mappings and this is the appropriate flag to set for those. This provides the semantics required - no VMA merging is permitted, but does not prevent GUP. However this has no meaningful impact as these are refcounted and thus can be GUPed. A previous commit already prevented __mm_populate() from being invoked on XOL areas, which VMA_IO_BIT was previously relied upon to do, so that is no longer required. Both VMAs set a VMA name, so always_dump_vma() returns true before vma_dump_size() reaches its VMA_IO_BIT check, and thus there is no change in core dump behaviour. Change this for both the core xol_add_vma() function and the x86-specific get_uprobe_trampoline() function. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-22-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- arch/x86/kernel/uprobes.c | 2 +- kernel/events/uprobes.c | 4 ++-- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/arch/x86/kernel/uprobes.c b/arch/x86/kernel/uprobes.c index 65a2de82ecd292..0f60c0d076b620 100644 --- a/arch/x86/kernel/uprobes.c +++ b/arch/x86/kernel/uprobes.c @@ -715,7 +715,7 @@ static struct vm_area_struct *get_uprobe_trampoline(struct mm_struct *mm, unsign *new_mapping = true; return _install_special_mapping(mm, vaddr, PAGE_SIZE, - VM_READ|VM_EXEC|VM_MAYEXEC|VM_MAYREAD|VM_IO, + VM_READ|VM_EXEC|VM_MAYEXEC|VM_MAYREAD|VM_MIXEDMAP, &tramp_mapping); } diff --git a/kernel/events/uprobes.c b/kernel/events/uprobes.c index 7709ea88247785..b89cc5cee00274 100644 --- a/kernel/events/uprobes.c +++ b/kernel/events/uprobes.c @@ -1726,8 +1726,8 @@ static int xol_add_vma(struct mm_struct *mm, struct xol_area *area) } vma = _install_special_mapping(mm, area->vaddr, PAGE_SIZE, - VM_EXEC|VM_MAYEXEC|VM_DONTCOPY|VM_IO| - VM_SEALED_SYSMAP, + VM_EXEC|VM_MAYEXEC|VM_DONTCOPY| + VM_MIXEDMAP|VM_SEALED_SYSMAP, &xol_mapping); if (IS_ERR(vma)) { ret = PTR_ERR(vma); From 97ae92f30d3ebc62edd4ffb5e8f10a530814efe5 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:32 +0100 Subject: [PATCH 0985/1352] mm/mlock: clear VMA_LOCKED_MASK over mmap callback Currently there's a confusing mess around VMA_LOCKED_BIT and VMA_LOCKONFAULT_BIT. It is permitted for drivers to set any flags they like, with the VMA already possessing lock flags. This results in the absurd situation of a VMA possessing both VMA_SPECIAL_FLAGS and VMA_LOCKED_MASK flags, which is not permitted. This has resulted in mlock_vma_folio() having a very silly check for this scenario to work around it. There is no need for this - just clear the flags before invoking the hook and reinstate them afterwards if they are required. Nothing relies upon this being set during the mmap operation. mmap_prepare is unaffected by this so requires no fix. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-23-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/internal.h | 9 +-------- mm/vma.c | 14 ++++++++++++++ 2 files changed, 15 insertions(+), 8 deletions(-) diff --git a/mm/internal.h b/mm/internal.h index a1970ff52ad7b9..1794b10ffeb4f3 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -971,14 +971,7 @@ void mlock_folio(struct folio *folio); static inline void mlock_vma_folio(struct folio *folio, struct vm_area_struct *vma) { - /* - * The VM_SPECIAL check here serves two purposes. - * 1) VM_IO check prevents migration from double-counting during mlock. - * 2) Although mmap_region() and mlock_fixup() take care that VM_LOCKED - * is never left set on a VM_SPECIAL vma, there is an interval while - * file->f_op->mmap() is using vm_insert_page(s), when VM_LOCKED may - * still be set while VM_SPECIAL bits are added: so ignore it then. - */ + /* The VM_IO check prevents migration from double-counting during mlock. */ if (unlikely((vma->vm_flags & (VM_LOCKED|VM_SPECIAL)) == VM_LOCKED)) mlock_folio(folio); } diff --git a/mm/vma.c b/mm/vma.c index 81d7061f3e9ecf..58264a717b9eb1 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -2622,6 +2622,11 @@ static int __mmap_new_file_vma(struct mmap_state *map, if (!map->vm_file->f_op->mmap) return 0; + /* + * Driver-specified flags may make the lock flags invalid, so clear + * VMA_LOCKED_MASK and reinstate it afterwards if appropriate. + */ + vma_clear_flags_mask(vma, VMA_LOCKED_MASK); error = mmap_file(vma->vm_file, vma); map->vm_file = vma->vm_file; @@ -2638,6 +2643,15 @@ static int __mmap_new_file_vma(struct mmap_state *map, return error; } + /* If VMA flags still valid for locked mask, reinstate. */ + if (vma_supports_mlock(vma)) { + const vma_flags_t mask = + vma_flags_and_mask(&map->vma_flags, + VMA_LOCKED_MASK); + + vma_set_flags_mask(vma, mask); + } + map->vma_flags = vma->flags; return 0; From 65f9df9402222b7388a6858f7e924718b0c3af76 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:33 +0100 Subject: [PATCH 0986/1352] mm/mlock: eliminate weird VMA_IO_BIT abuse and simplify When performing mlock() or munlock() otherwise normal VMAs have VMA_IO_BIT solely to fix a race with migration which might otherwise double-count mlock VMAs. This is unnecessary - at the point of applying folio mlock state, whether setting or clearing PG_mlocked, we know whether or not we are locking. Solve this in two ways - thread a boolean through the page table walk indicating whether a lock or unlock is being performed, and run a locking walk with VMA_LOCKONFAULT_BIT set and VMA_LOCKED_BIT cleared. This state never occurs otherwise, as VMA_LOCKONFAULT_BIT always implies VMA_LOCKED_BIT. These are also always cleared together. Then, update folio_add_lru_vma() and mlock_folio() to check only for VMA_LOCKED_BIT, and update try_to_unmap_one() to check for VMA_LOCKED_MASK instead. Also remove the useless invocation of allow_mlock_munlock() which simply returns true if unlocking and instead rename it to allow_mlock() and only call it when locking. Finally, with the other mlock abuse of VMA_IO_BIT addressed, update mlock_vma_folio() and folio_add_lru_vma() to simply test for VMA_LOCKED_BIT. munlock_vma_folio() tests VMA_LOCKED_MASK instead, as an unmap racing with the locking walk must still munlock folios the walk has already counted. While here, also replace some deprecated VMA flag predicates. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-24-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/folio.c | 2 +- mm/internal.h | 10 +++++++--- mm/mlock.c | 51 +++++++++++++++++++-------------------------------- mm/rmap.c | 4 +++- 4 files changed, 30 insertions(+), 37 deletions(-) diff --git a/mm/folio.c b/mm/folio.c index 47a437e0f7fde6..35e242b48870b0 100644 --- a/mm/folio.c +++ b/mm/folio.c @@ -505,7 +505,7 @@ void folio_add_lru_vma(struct folio *folio, struct vm_area_struct *vma) { VM_BUG_ON_FOLIO(folio_test_lru(folio), folio); - if (unlikely((vma->vm_flags & (VM_LOCKED | VM_SPECIAL)) == VM_LOCKED)) + if (vma_test(vma, VMA_LOCKED_BIT)) mlock_new_folio(folio); else folio_add_lru(folio); diff --git a/mm/internal.h b/mm/internal.h index 1794b10ffeb4f3..1bf6517cf38932 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -971,8 +971,7 @@ void mlock_folio(struct folio *folio); static inline void mlock_vma_folio(struct folio *folio, struct vm_area_struct *vma) { - /* The VM_IO check prevents migration from double-counting during mlock. */ - if (unlikely((vma->vm_flags & (VM_LOCKED|VM_SPECIAL)) == VM_LOCKED)) + if (vma_test(vma, VMA_LOCKED_BIT)) mlock_folio(folio); } @@ -989,7 +988,12 @@ static inline void munlock_vma_folio(struct folio *folio, * always munlock the folio and page reclaim will correct it * if it's wrong. */ - if (unlikely(vma->vm_flags & VM_LOCKED)) + /* + * VMA_LOCKONFAULT_BIT alone marks an mlock walk in progress, see + * mlock_vma_pages_range(). An unmap racing with the walk must still + * munlock folios the walk has already counted. + */ + if (unlikely(vma_test_any_mask(vma, VMA_LOCKED_MASK))) munlock_folio(folio); } diff --git a/mm/mlock.c b/mm/mlock.c index 39215a3eab1fbf..4235a1518fc9e4 100644 --- a/mm/mlock.c +++ b/mm/mlock.c @@ -316,22 +316,10 @@ static inline unsigned int folio_mlock_step(struct folio *folio, return folio_pte_batch(folio, pte, ptent, count); } -static inline bool allow_mlock_munlock(struct folio *folio, +static inline bool allow_mlock(struct folio *folio, struct vm_area_struct *vma, unsigned long start, unsigned long end, unsigned int step) { - /* - * For unlock, allow munlock large folio which is partially - * mapped to VMA. As it's possible that large folio is - * mlocked and VMA is split later. - * - * During memory pressure, such kind of large folio can - * be split. And the pages are not in VM_LOCKed VMA - * can be reclaimed. - */ - if (!vma_test(vma, VMA_LOCKED_BIT)) - return true; - /* folio_within_range() cannot take KSM, but any small folio is OK */ if (!folio_test_large(folio)) return true; @@ -352,6 +340,7 @@ static int mlock_pte_range(pmd_t *pmd, unsigned long addr, { struct vm_area_struct *vma = walk->vma; + const bool lock = walk->private; spinlock_t *ptl; pte_t *start_pte, *pte; pte_t ptent; @@ -368,7 +357,7 @@ static int mlock_pte_range(pmd_t *pmd, unsigned long addr, folio = pmd_folio(*pmd); if (folio_is_zone_device(folio)) goto out; - if (vma_test(vma, VMA_LOCKED_BIT)) + if (lock) mlock_folio(folio); else munlock_folio(folio); @@ -390,10 +379,10 @@ static int mlock_pte_range(pmd_t *pmd, unsigned long addr, continue; step = folio_mlock_step(folio, pte, addr, end); - if (!allow_mlock_munlock(folio, vma, start, end, step)) + if (lock && !allow_mlock(folio, vma, start, end, step)) goto next_entry; - if (vma_test(vma, VMA_LOCKED_BIT)) + if (lock) mlock_folio(folio); else munlock_folio(folio); @@ -428,31 +417,29 @@ static void mlock_vma_pages_range(struct vm_area_struct *vma, .pmd_entry = mlock_pte_range, .walk_lock = PGWALK_WRLOCK_VERIFY, }; + const bool lock = vma_flags_test(new_vma_flags, VMA_LOCKED_BIT); + vma_flags_t walk_flags = *new_vma_flags; /* - * There is a slight chance that concurrent page migration, - * or page reclaim finding a page of this now-VMA_LOCKED_BIT vma, - * will call mlock_vma_folio() and raise page's mlock_count: - * double counting, leaving the page unevictable indefinitely. - * Communicate this danger to mlock_vma_folio() with VMA_IO_BIT, - * which is a VMA_SPECIAL_FLAGS flag not allowed on VMA_LOCKED_BIT vmas. - * mmap_lock is held in write mode here, so this weird - * combination should not be visible to other mmap_lock users; - * but WRITE_ONCE so rmap walkers must see VMA_IO_BIT if VMA_LOCKED_BIT. + * LOCKONFAULT without LOCKED never otherwise occurs: it marks a walk in + * progress so that rmap-side callers, which test VMA_LOCKED_BIT, do not + * count folios, while try_to_unmap_one(), which tests VMA_LOCKED_MASK, + * still refuses to unmap them. */ - if (vma_flags_test(new_vma_flags, VMA_LOCKED_BIT)) - vma_flags_set(new_vma_flags, VMA_IO_BIT); + if (lock) { + vma_flags_clear(&walk_flags, VMA_LOCKED_BIT); + vma_flags_set(&walk_flags, VMA_LOCKONFAULT_BIT); + } + vma_start_write(vma); - vma_flags_reset_once(vma, new_vma_flags); + vma_flags_reset_once(vma, &walk_flags); lru_add_drain(); - walk_page_range_vma(vma, start, end, &mlock_walk_ops, NULL); + walk_page_range_vma(vma, start, end, &mlock_walk_ops, (void *)lock); lru_add_drain(); - if (vma_flags_test(new_vma_flags, VMA_IO_BIT)) { - vma_flags_clear(new_vma_flags, VMA_IO_BIT); + if (lock) vma_flags_reset_once(vma, new_vma_flags); - } } /* diff --git a/mm/rmap.c b/mm/rmap.c index 5332c52909be18..6661bc11ce658b 100644 --- a/mm/rmap.c +++ b/mm/rmap.c @@ -2239,9 +2239,11 @@ static bool try_to_unmap_one(struct folio *folio, struct vm_area_struct *vma, /* * If the folio is in an mlock()d vma, we must not swap it out. + * VMA_LOCKONFAULT_BIT alone marks an mlock walk in progress, see + * mlock_vma_pages_range(). */ if (!(flags & TTU_IGNORE_MLOCK) && - (vma->vm_flags & VM_LOCKED)) { + vma_test_any_mask(vma, VMA_LOCKED_MASK)) { ptes++; /* From 5cebfd961237ae5c3d1afdca75ebd90466b494db Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:34 +0100 Subject: [PATCH 0987/1352] mm/vma: enforce that only kernel-owned mappings may set VMA_IO_BIT It makes no sense for a mapping whose contents the kernel does not own to specify that the range is MMIO. Prior to this patch, all in-tree drivers which did so have been updated such that they are marked as kernel-owned. The check WARNs and fails the mmap for any out-of-tree driver that still sets VMA_IO_BIT without a kernel mapping. No functional change intended for in-tree code. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-25-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/vma.c | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/mm/vma.c b/mm/vma.c index 58264a717b9eb1..11e664fc3f200f 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -2802,6 +2802,12 @@ static int mmap_validate_vma_flags(const vma_flags_t *flags) return -EINVAL; #endif + if (!vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT)) { + /* Only kernel-owned mappings may set VMA_IO_BIT. */ + if (WARN_ON_ONCE(vma_flags_test(flags, VMA_IO_BIT))) + return -EINVAL; + } + return 0; } From 5c14179edb4f618f76442146127eed330a15cc55 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:35 +0100 Subject: [PATCH 0988/1352] mm: remove VMA_IO_BIT check in vma[_flags]_is_kernel_owned() We have now made it such that every driver which sets VMA_IO_BIT marks it as kernel-owned. However, vma_flags_is_kernel_owned() currently checks for VMA_IO_BIT. This was a product of drivers previously marking a range as kernel-owned by setting VMA_IO_BIT alone. Fix this by removing the VMA_IO_BIT check in vma_flags_is_kernel_owned(), and update mmap_validate_vma_flags() to use vma_flags_is_kernel_owned() rather than open-coding the VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT check. This change means that vma[_flags]_can_merge() doesn't check VMA_IO_BIT any longer (which is now redundant) as it calls vma_flags_is_kernel_owned(). Now that the predicate means precisely VMA_PFNMAP_BIT or VMA_MIXEDMAP_BIT, also use it at the other sites which open-code that pair, so the intent is stated rather than the flags, with no functional change: zap_special_vma_range() only zaps kernel-owned mappings, as drivers use it to tear down ranges they established themselves. The mprotect() arch PFN modification check applies to kernel-owned mappings, which may map PFNs without struct pages. NUMA balancing skips VM_MIXEDMAP mappings having already excluded VM_IO and VM_PFNMAP mappings via vma_migratable(), so it skips exactly the kernel-owned mappings - say so. Finally, update the VMA userland merge 'special' flag tests to no longer assert that VMA_IO_BIT prevents merge as VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT now suffices. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-26-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- include/linux/mm.h | 3 +-- kernel/sched/fair.c | 2 +- mm/memory.c | 6 +++--- mm/mprotect.c | 3 +-- mm/vma.c | 2 +- tools/testing/vma/include/dup.h | 3 +-- tools/testing/vma/tests/merge.c | 10 ++-------- 7 files changed, 10 insertions(+), 19 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index a7fa4df6fd4747..e0fe10e0375993 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -1631,8 +1631,7 @@ static inline bool vma_is_shared_maywrite(const struct vm_area_struct *vma) */ static inline bool vma_flags_is_kernel_owned(const vma_flags_t *flags) { - return vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT, - VMA_IO_BIT); + return vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT); } /** diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index 57360f5cdde4fa..3a8730a627c897 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -4417,7 +4417,7 @@ static void task_numa_work(struct callback_head *work) for (; vma; vma = vma_next(&vmi)) { if (!vma_migratable(vma) || !vma_policy_mof(vma) || - is_vm_hugetlb_page(vma) || (vma->vm_flags & VM_MIXEDMAP)) { + is_vm_hugetlb_page(vma) || vma_is_kernel_owned(vma)) { trace_sched_skip_vma_numa(mm, vma, NUMAB_SKIP_UNSUITABLE); continue; } diff --git a/mm/memory.c b/mm/memory.c index 45b21bb04a18b9..1e6cd2e504089a 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -2343,19 +2343,19 @@ void zap_vma_range(struct vm_area_struct *vma, unsigned long address, } /** - * zap_special_vma_range - zap all page table entries in a special vma range + * zap_special_vma_range - zap all page table entries in a kernel-owned VMA * @vma: the vma covering the range to zap * @address: starting address of the range to zap * @size: number of bytes to zap * * This function does nothing when the provided address range is not fully - * contained in @vma, or when the @vma is not VM_PFNMAP or VM_MIXEDMAP. + * contained in @vma, or when @vma is not kernel-owned. */ void zap_special_vma_range(struct vm_area_struct *vma, unsigned long address, unsigned long size) { if (!range_in_vma(vma, address, address + size) || - !(vma->vm_flags & (VM_PFNMAP | VM_MIXEDMAP))) + !vma_is_kernel_owned(vma)) return; zap_vma_range(vma, address, size); diff --git a/mm/mprotect.c b/mm/mprotect.c index 2888ee638d872a..fe32fd87cf5cd7 100644 --- a/mm/mprotect.c +++ b/mm/mprotect.c @@ -783,8 +783,7 @@ mprotect_fixup(struct vma_iterator *vmi, struct mmu_gather *tlb, * uncommon case, so doesn't need to be very optimized. */ if (arch_has_pfn_modify_check() && - vma_flags_test_any(&old_vma_flags, VMA_PFNMAP_BIT, - VMA_MIXEDMAP_BIT) && + vma_flags_is_kernel_owned(&old_vma_flags) && !vma_flags_test_any_mask(&new_vma_flags, VMA_ACCESS_FLAGS)) { pgprot_t new_pgprot = vm_get_page_prot(newflags); diff --git a/mm/vma.c b/mm/vma.c index 11e664fc3f200f..6e4c337a8e1a2b 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -2802,7 +2802,7 @@ static int mmap_validate_vma_flags(const vma_flags_t *flags) return -EINVAL; #endif - if (!vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT)) { + if (!vma_flags_is_kernel_owned(flags)) { /* Only kernel-owned mappings may set VMA_IO_BIT. */ if (WARN_ON_ONCE(vma_flags_test(flags, VMA_IO_BIT))) return -EINVAL; diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index b8b1462ca71052..97d3bf6cd5b712 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -1667,8 +1667,7 @@ static inline bool file_is_dev_zero(const struct file *file) static inline bool vma_flags_is_kernel_owned(const vma_flags_t *flags) { - return vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT, - VMA_IO_BIT); + return vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT); } static inline bool vma_is_kernel_owned(const struct vm_area_struct *vma) diff --git a/tools/testing/vma/tests/merge.c b/tools/testing/vma/tests/merge.c index acaab282939c0b..b26f1a66a17074 100644 --- a/tools/testing/vma/tests/merge.c +++ b/tools/testing/vma/tests/merge.c @@ -496,17 +496,11 @@ static bool test_vma_merge_special_flags(void) .mm = &mm, .vmi = &vmi, }; - vma_flag_t special_flags[] = { VMA_IO_BIT, VMA_DONTEXPAND_BIT, + vma_flag_t special_flags[] = { VMA_DONTEXPAND_BIT, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT }; - vma_flags_t all_special_flags = EMPTY_VMA_FLAGS; int i; struct vm_area_struct *vma_left, *vma; - /* Make sure there aren't new VM_SPECIAL flags. */ - for (i = 0; i < ARRAY_SIZE(special_flags); i++) - vma_flags_set(&all_special_flags, special_flags[i]); - ASSERT_FLAGS_SAME_MASK(&all_special_flags, VMA_SPECIAL_FLAGS); - /* * 01234 * AAA @@ -520,7 +514,7 @@ static bool test_vma_merge_special_flags(void) * 01234 * AAA* * - * This should merge if not for the VM_SPECIAL flag. + * This should merge if not for the 'special' flag. */ vmg_set_range(&vmg, 0x3000, 0x4000, 3, vma_flags); for (i = 0; i < ARRAY_SIZE(special_flags); i++) { From a5796f5ad5aa84395a4e088312832cf2162a4ad0 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:36 +0100 Subject: [PATCH 0989/1352] mm: remove hugetlb_inline.h This header really makes little sense - every place it is included mm.h is also included, and the header itself includes mm.h, so it does nothing to reduce header size. It also oddly does an #ifdef around checking VMA_HUGETLB_BIT, however VMA_HUGETLB_BIT is unconditionally available, and will never be set if hugetlb is not enabled. Simply remove the header, eliminate the odd ifdeffery and place the predicates in mm.h. The naming of these predicates is odd, but to keep changes separate, we will address this in a separate patch. The file was never put into MAINTAINERS so there's no change required there. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-27-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- drivers/gpu/drm/drm_gpusvm.c | 2 +- include/asm-generic/tlb.h | 2 +- include/linux/hugetlb.h | 1 - include/linux/hugetlb_inline.h | 28 ---------------------------- include/linux/mm.h | 11 +++++++++++ include/linux/pagemap.h | 1 - include/linux/userfaultfd_k.h | 1 - kernel/sched/fair.c | 1 - mm/vma_internal.h | 1 - 9 files changed, 13 insertions(+), 35 deletions(-) delete mode 100644 include/linux/hugetlb_inline.h diff --git a/drivers/gpu/drm/drm_gpusvm.c b/drivers/gpu/drm/drm_gpusvm.c index a93eee7ddb9e95..793dacec210034 100644 --- a/drivers/gpu/drm/drm_gpusvm.c +++ b/drivers/gpu/drm/drm_gpusvm.c @@ -9,9 +9,9 @@ #include #include #include -#include #include #include +#include #include #include diff --git a/include/asm-generic/tlb.h b/include/asm-generic/tlb.h index 9d827076db1969..bb7b05e1009963 100644 --- a/include/asm-generic/tlb.h +++ b/include/asm-generic/tlb.h @@ -11,9 +11,9 @@ #ifndef _ASM_GENERIC__TLB_H #define _ASM_GENERIC__TLB_H +#include #include #include -#include #include #include diff --git a/include/linux/hugetlb.h b/include/linux/hugetlb.h index 80a5a03e9cee72..d7e6563cef753c 100644 --- a/include/linux/hugetlb.h +++ b/include/linux/hugetlb.h @@ -7,7 +7,6 @@ #include #include #include -#include #include #include #include diff --git a/include/linux/hugetlb_inline.h b/include/linux/hugetlb_inline.h deleted file mode 100644 index 5c29cd3223a1e4..00000000000000 --- a/include/linux/hugetlb_inline.h +++ /dev/null @@ -1,28 +0,0 @@ -/* SPDX-License-Identifier: GPL-2.0 */ -#ifndef _LINUX_HUGETLB_INLINE_H -#define _LINUX_HUGETLB_INLINE_H - -#include - -#ifdef CONFIG_HUGETLB_PAGE - -static inline bool is_vma_hugetlb_flags(const vma_flags_t *flags) -{ - return vma_flags_test(flags, VMA_HUGETLB_BIT); -} - -#else - -static inline bool is_vma_hugetlb_flags(const vma_flags_t *flags) -{ - return false; -} - -#endif - -static inline bool is_vm_hugetlb_page(const struct vm_area_struct *vma) -{ - return is_vma_hugetlb_flags(&vma->flags); -} - -#endif diff --git a/include/linux/mm.h b/include/linux/mm.h index e0fe10e0375993..04eee802912faf 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -1610,6 +1610,17 @@ static inline bool vma_is_shared_maywrite(const struct vm_area_struct *vma) return is_shared_maywrite(&vma->flags); } +static inline bool is_vma_hugetlb_flags(const vma_flags_t *flags) +{ + return IS_ENABLED(CONFIG_HUGETLB_PAGE) && + vma_flags_test(flags, VMA_HUGETLB_BIT); +} + +static inline bool is_vm_hugetlb_page(const struct vm_area_struct *vma) +{ + return is_vma_hugetlb_flags(&vma->flags); +} + /** * vma_flags_is_kernel_owned() - Do the specified VMA flags indicate that the * contents of the VMA are owned by the kernel rather than the core mm? diff --git a/include/linux/pagemap.h b/include/linux/pagemap.h index bcbb0afe1a6816..73af18a3736706 100644 --- a/include/linux/pagemap.h +++ b/include/linux/pagemap.h @@ -14,7 +14,6 @@ #include #include #include /* for in_interrupt() */ -#include struct folio_batch; diff --git a/include/linux/userfaultfd_k.h b/include/linux/userfaultfd_k.h index a4351cffc60ce3..a14b8a9ffb7b1c 100644 --- a/include/linux/userfaultfd_k.h +++ b/include/linux/userfaultfd_k.h @@ -18,7 +18,6 @@ #include #include #include -#include /* The set of all possible UFFD-related VM flags. */ #define __VM_UFFD_FLAGS (VM_UFFD_MISSING | VM_UFFD_MINOR | \ diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index 3a8730a627c897..e2b00e56d76e16 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -22,7 +22,6 @@ */ #include #include -#include #include #include #include diff --git a/mm/vma_internal.h b/mm/vma_internal.h index 4d300e7bbaf4c2..4f73f0a4db796b 100644 --- a/mm/vma_internal.h +++ b/mm/vma_internal.h @@ -18,7 +18,6 @@ #include #include #include -#include #include #include #include From f09773c2c4d0028035d6b19617b54d3b4f6064cd Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:37 +0100 Subject: [PATCH 0990/1352] mm: rename is_vm_hugetlb_page() to vma_is_hugetlb() The is_vm_hugetlb_page() predicate is badly named - the mapping can span more than a page and it is inconsistent with other VMA predicates that typically are prefixed by vma_. Rename to vma_is_hugetlb() for consistency, and while we're here update some VM_BUG_ON_VMA() to VM_WARN_ON_ONCE_VMA() as to avoid unnecessary oopses. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-28-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Marc Zyngier Acked-by: Claudio Imbrenda Acked-by: Anup Patel Acked-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- arch/arm64/kvm/mmu.c | 4 ++-- arch/powerpc/mm/book3s64/radix_tlb.c | 6 +++--- arch/powerpc/mm/nohash/e500_hugetlbpage.c | 2 +- arch/powerpc/mm/nohash/tlb.c | 2 +- arch/riscv/kvm/mmu.c | 2 +- arch/riscv/mm/tlbflush.c | 2 +- arch/s390/mm/gmap_helpers.c | 6 +++--- arch/sparc/mm/init_64.c | 2 +- drivers/gpu/drm/drm_gpusvm.c | 2 +- fs/coredump.c | 2 +- fs/hugetlbfs/inode.c | 2 +- fs/proc/task_mmu.c | 8 +++---- include/asm-generic/tlb.h | 2 +- include/linux/hugetlb.h | 4 ++-- include/linux/mm.h | 19 ++++++++++++++--- include/linux/rmap.h | 2 +- kernel/events/core.c | 2 +- kernel/sched/fair.c | 2 +- mm/gup.c | 4 ++-- mm/huge_memory.c | 2 +- mm/hugetlb.c | 14 ++++++------ mm/internal.h | 2 +- mm/madvise.c | 4 ++-- mm/memory.c | 12 +++++------ mm/mempolicy.c | 2 +- mm/migrate_device.c | 2 +- mm/mmap.c | 2 +- mm/mmu_gather.c | 2 +- mm/mprotect.c | 2 +- mm/mremap.c | 6 +++--- mm/page_vma_mapped.c | 4 ++-- mm/pagewalk.c | 2 +- mm/swapfile.c | 2 +- mm/userfaultfd.c | 26 +++++++++++------------ mm/vma.c | 8 +++---- mm/vmscan.c | 2 +- tools/testing/vma/include/stubs.h | 2 +- 37 files changed, 92 insertions(+), 79 deletions(-) diff --git a/arch/arm64/kvm/mmu.c b/arch/arm64/kvm/mmu.c index 2d44cd6a5aed90..b8e28d1b5461d5 100644 --- a/arch/arm64/kvm/mmu.c +++ b/arch/arm64/kvm/mmu.c @@ -1472,13 +1472,13 @@ static int get_vma_page_shift(struct vm_area_struct *vma, unsigned long hva) { unsigned long pa; - if (is_vm_hugetlb_page(vma) && !(vma->vm_flags & VM_PFNMAP)) + if (vma_is_hugetlb(vma) && !(vma->vm_flags & VM_PFNMAP)) return huge_page_shift(hstate_vma(vma)); if (!(vma->vm_flags & VM_PFNMAP)) return PAGE_SHIFT; - VM_BUG_ON(is_vm_hugetlb_page(vma)); + VM_BUG_ON(vma_is_hugetlb(vma)); pa = (vma->vm_pgoff << PAGE_SHIFT) + (hva - vma->vm_start); diff --git a/arch/powerpc/mm/book3s64/radix_tlb.c b/arch/powerpc/mm/book3s64/radix_tlb.c index 7de5760164a90f..b4603a98224b32 100644 --- a/arch/powerpc/mm/book3s64/radix_tlb.c +++ b/arch/powerpc/mm/book3s64/radix_tlb.c @@ -627,7 +627,7 @@ void radix__local_flush_tlb_page(struct vm_area_struct *vma, unsigned long vmadd { #ifdef CONFIG_HUGETLB_PAGE /* need the return fix for nohash.c */ - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) return radix__local_flush_hugetlb_page(vma, vmaddr); #endif radix__local_flush_tlb_page_psize(vma->vm_mm, vmaddr, mmu_virtual_psize); @@ -945,7 +945,7 @@ void radix__flush_tlb_page_psize(struct mm_struct *mm, unsigned long vmaddr, void radix__flush_tlb_page(struct vm_area_struct *vma, unsigned long vmaddr) { #ifdef CONFIG_HUGETLB_PAGE - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) return radix__flush_hugetlb_page(vma, vmaddr); #endif radix__flush_tlb_page_psize(vma->vm_mm, vmaddr, mmu_virtual_psize); @@ -1113,7 +1113,7 @@ void radix__flush_tlb_range(struct vm_area_struct *vma, unsigned long start, { #ifdef CONFIG_HUGETLB_PAGE - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) return radix__flush_hugetlb_tlb_range(vma, start, end); #endif diff --git a/arch/powerpc/mm/nohash/e500_hugetlbpage.c b/arch/powerpc/mm/nohash/e500_hugetlbpage.c index a134d28a0e4d39..b87623f04be53c 100644 --- a/arch/powerpc/mm/nohash/e500_hugetlbpage.c +++ b/arch/powerpc/mm/nohash/e500_hugetlbpage.c @@ -180,7 +180,7 @@ book3e_hugetlb_preload(struct vm_area_struct *vma, unsigned long ea, pte_t pte) */ void __update_mmu_cache(struct vm_area_struct *vma, unsigned long address, pte_t *ptep) { - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) book3e_hugetlb_preload(vma, address, *ptep); } diff --git a/arch/powerpc/mm/nohash/tlb.c b/arch/powerpc/mm/nohash/tlb.c index 0a650742f3a008..07a2db16c2b153 100644 --- a/arch/powerpc/mm/nohash/tlb.c +++ b/arch/powerpc/mm/nohash/tlb.c @@ -278,7 +278,7 @@ void __flush_tlb_page(struct mm_struct *mm, unsigned long vmaddr, void flush_tlb_page(struct vm_area_struct *vma, unsigned long vmaddr) { #ifdef CONFIG_HUGETLB_PAGE - if (vma && is_vm_hugetlb_page(vma)) + if (vma && vma_is_hugetlb(vma)) flush_hugetlb_page(vma, vmaddr); #endif diff --git a/arch/riscv/kvm/mmu.c b/arch/riscv/kvm/mmu.c index 3e955d808743b4..371fccaf9df08a 100644 --- a/arch/riscv/kvm/mmu.c +++ b/arch/riscv/kvm/mmu.c @@ -665,7 +665,7 @@ int kvm_riscv_mmu_map(struct kvm_vcpu *vcpu, struct kvm_memory_slot *memslot, return -EFAULT; } - is_hugetlb = is_vm_hugetlb_page(vma); + is_hugetlb = vma_is_hugetlb(vma); if (is_hugetlb) vma_pageshift = huge_page_shift(hstate_vma(vma)); else diff --git a/arch/riscv/mm/tlbflush.c b/arch/riscv/mm/tlbflush.c index 962db300a16659..a74a7d5258aa1d 100644 --- a/arch/riscv/mm/tlbflush.c +++ b/arch/riscv/mm/tlbflush.c @@ -149,7 +149,7 @@ void flush_tlb_range(struct vm_area_struct *vma, unsigned long start, { unsigned long stride_size; - if (!is_vm_hugetlb_page(vma)) { + if (!vma_is_hugetlb(vma)) { stride_size = PAGE_SIZE; } else { stride_size = huge_page_size(hstate_vma(vma)); diff --git a/arch/s390/mm/gmap_helpers.c b/arch/s390/mm/gmap_helpers.c index ff63ffb1dbd29c..3f6783b93e679f 100644 --- a/arch/s390/mm/gmap_helpers.c +++ b/arch/s390/mm/gmap_helpers.c @@ -102,7 +102,7 @@ __context_unsafe(/* pte_unmap_unlock() not instrumented */) /* Find the vm address for the guest address */ vma = vma_lookup(mm, vmaddr); - if (!vma || is_vm_hugetlb_page(vma)) + if (!vma || vma_is_hugetlb(vma)) return; /* Get pointer to the page table entry */ @@ -139,7 +139,7 @@ void gmap_helper_discard(struct mm_struct *mm, unsigned long vmaddr, unsigned lo vma = find_vma_intersection(mm, vmaddr, end); if (!vma) return; - if (!is_vm_hugetlb_page(vma)) + if (!vma_is_hugetlb(vma)) zap_vma_range(vma, vmaddr, min(end, vma->vm_end) - vmaddr); vmaddr = vma->vm_end; } @@ -247,7 +247,7 @@ static int __gmap_helper_unshare_zeropages(struct mm_struct *mm) * proof to catch unexpected zeropages in other mappings and * fail. */ - if ((vma->vm_flags & VM_PFNMAP) || is_vm_hugetlb_page(vma)) + if ((vma->vm_flags & VM_PFNMAP) || vma_is_hugetlb(vma)) continue; addr = vma->vm_start; diff --git a/arch/sparc/mm/init_64.c b/arch/sparc/mm/init_64.c index 103db4683b165e..9bbccb5d23a8f1 100644 --- a/arch/sparc/mm/init_64.c +++ b/arch/sparc/mm/init_64.c @@ -413,7 +413,7 @@ void update_mmu_cache_range(struct vm_fault *vmf, struct vm_area_struct *vma, if (mm->context.hugetlb_pte_count || mm->context.thp_pte_count) { unsigned long hugepage_size = PAGE_SIZE; - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) hugepage_size = huge_page_size(hstate_vma(vma)); if (hugepage_size >= PUD_SIZE) { diff --git a/drivers/gpu/drm/drm_gpusvm.c b/drivers/gpu/drm/drm_gpusvm.c index 793dacec210034..a1d4989b0b616f 100644 --- a/drivers/gpu/drm/drm_gpusvm.c +++ b/drivers/gpu/drm/drm_gpusvm.c @@ -1142,7 +1142,7 @@ drm_gpusvm_range_find_or_insert(struct drm_gpusvm *gpusvm, * have to change. */ migrate_devmem = ctx->devmem_possible && - vma_is_anonymous(vas) && !is_vm_hugetlb_page(vas); + vma_is_anonymous(vas) && !vma_is_hugetlb(vas); chunk_size = drm_gpusvm_range_chunk_size(gpusvm, notifier, vas, fault_addr, gpuva_start, diff --git a/fs/coredump.c b/fs/coredump.c index 6114839f5178b0..5820cb8ec88e70 100644 --- a/fs/coredump.c +++ b/fs/coredump.c @@ -1608,7 +1608,7 @@ static unsigned long vma_dump_size(struct vm_area_struct *vma, } /* Hugetlb memory check */ - if (is_vm_hugetlb_page(vma)) { + if (vma_is_hugetlb(vma)) { if ((vma->vm_flags & VM_SHARED) && FILTER(HUGETLB_SHARED)) goto whole; if (!(vma->vm_flags & VM_SHARED) && FILTER(HUGETLB_PRIVATE)) diff --git a/fs/hugetlbfs/inode.c b/fs/hugetlbfs/inode.c index 7611a8470ea265..ba7097d5720c07 100644 --- a/fs/hugetlbfs/inode.c +++ b/fs/hugetlbfs/inode.c @@ -108,7 +108,7 @@ static int hugetlbfs_file_mmap(struct file *file, struct vm_area_struct *vma) * vma address alignment (but not the pgoff alignment) has * already been checked by prepare_hugepage_range. If you add * any error returns here, do so after setting VM_HUGETLB, so - * is_vm_hugetlb_page tests below unmap_region go the right + * vma_is_hugetlb tests below unmap_region go the right * way when do_mmap unwinds (may be important on powerpc * and ia64). */ diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c index e671b4fd8dedd9..565e6446bd3127 100644 --- a/fs/proc/task_mmu.c +++ b/fs/proc/task_mmu.c @@ -3015,7 +3015,7 @@ static int pagemap_scan_pte_hole(unsigned long addr, unsigned long end, * hugetlb differs, see pagemap_hugetlb_category(). */ categories = p->cur_vma_category; - if (userfaultfd_wp(vma) && !is_vm_hugetlb_page(vma)) + if (userfaultfd_wp(vma) && !vma_is_hugetlb(vma)) categories |= PAGE_IS_WRITTEN; if (!pagemap_scan_is_interesting_page(categories, p)) @@ -3028,7 +3028,7 @@ static int pagemap_scan_pte_hole(unsigned long addr, unsigned long end, if (~p->arg.flags & PM_SCAN_WP_MATCHING) return ret; - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) err = pagemap_scan_hugetlb_hole_wp(vma, addr, end); else err = uffd_wp_range(vma, addr, end - addr, true); @@ -3470,7 +3470,7 @@ static int show_numa_map(struct seq_file *m, void *v) seq_puts(m, " stack"); } - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) seq_puts(m, " huge"); /* Skip walking pages if gate VMA */ @@ -3499,7 +3499,7 @@ static int show_numa_map(struct seq_file *m, void *v) if (md->swapcache) seq_printf(m, " swapcache=%lu", md->swapcache); - if (md->active < md->pages && !is_vm_hugetlb_page(vma)) + if (md->active < md->pages && !vma_is_hugetlb(vma)) seq_printf(m, " active=%lu", md->active); if (md->writeback) diff --git a/include/asm-generic/tlb.h b/include/asm-generic/tlb.h index bb7b05e1009963..fbe1114e9a443d 100644 --- a/include/asm-generic/tlb.h +++ b/include/asm-generic/tlb.h @@ -438,7 +438,7 @@ tlb_update_vma_flags(struct mmu_gather *tlb, struct vm_area_struct *vma) * We rely on tlb_end_vma() to issue a flush, such that when we reset * these values the batch is empty. */ - tlb->vma_huge = is_vm_hugetlb_page(vma); + tlb->vma_huge = vma_is_hugetlb(vma); tlb->vma_exec = !!(vma->vm_flags & VM_EXEC); /* diff --git a/include/linux/hugetlb.h b/include/linux/hugetlb.h index d7e6563cef753c..24727ece20fe52 100644 --- a/include/linux/hugetlb.h +++ b/include/linux/hugetlb.h @@ -251,14 +251,14 @@ extern void __hugetlb_zap_end(struct vm_area_struct *vma, static inline void hugetlb_zap_begin(struct vm_area_struct *vma, unsigned long *start, unsigned long *end) { - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) __hugetlb_zap_begin(vma, start, end); } static inline void hugetlb_zap_end(struct vm_area_struct *vma, struct zap_details *details) { - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) __hugetlb_zap_end(vma, details); } diff --git a/include/linux/mm.h b/include/linux/mm.h index 04eee802912faf..dbd4c1a70a2359 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -1610,15 +1610,28 @@ static inline bool vma_is_shared_maywrite(const struct vm_area_struct *vma) return is_shared_maywrite(&vma->flags); } -static inline bool is_vma_hugetlb_flags(const vma_flags_t *flags) +/** + * vma_flags_is_hugetlb() - Do the specified VMA flags indicate that the + * VMA is a hugetlb mapping? + * @flags: The VMA flags to test. + * + * Returns: true if the flags indicate a hugetlb mapping, false otherwise. + */ +static inline bool vma_flags_is_hugetlb(const vma_flags_t *flags) { return IS_ENABLED(CONFIG_HUGETLB_PAGE) && vma_flags_test(flags, VMA_HUGETLB_BIT); } -static inline bool is_vm_hugetlb_page(const struct vm_area_struct *vma) +/** + * vma_is_hugetlb() - Is @vma a hugetlb mapping? + * @vma: The VMA to test. + * + * Returns: true if @vma is a hugetlb mapping, false otherwise. + */ +static inline bool vma_is_hugetlb(const struct vm_area_struct *vma) { - return is_vma_hugetlb_flags(&vma->flags); + return vma_flags_is_hugetlb(&vma->flags); } /** diff --git a/include/linux/rmap.h b/include/linux/rmap.h index 0b332770abeed5..74cca0e3c72641 100644 --- a/include/linux/rmap.h +++ b/include/linux/rmap.h @@ -888,7 +888,7 @@ struct page_vma_mapped_walk { static inline void page_vma_mapped_walk_done(struct page_vma_mapped_walk *pvmw) { /* HugeTLB pte is set to the relevant page table entry without pte_mapped. */ - if (pvmw->pte && !is_vm_hugetlb_page(pvmw->vma)) + if (pvmw->pte && !vma_is_hugetlb(pvmw->vma)) pte_unmap(pvmw->pte); if (pvmw->ptl) spin_unlock(pvmw->ptl); diff --git a/kernel/events/core.c b/kernel/events/core.c index 634d2ccbab82d8..601e8d944c240c 100644 --- a/kernel/events/core.c +++ b/kernel/events/core.c @@ -9818,7 +9818,7 @@ static void perf_event_mmap_event(struct perf_mmap_event *mmap_event) if (vma->vm_flags & VM_LOCKED) flags |= MAP_LOCKED; - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) flags |= MAP_HUGETLB; if (file) { diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index e2b00e56d76e16..8d38c3b7d7920b 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -4416,7 +4416,7 @@ static void task_numa_work(struct callback_head *work) for (; vma; vma = vma_next(&vmi)) { if (!vma_migratable(vma) || !vma_policy_mof(vma) || - is_vm_hugetlb_page(vma) || vma_is_kernel_owned(vma)) { + vma_is_hugetlb(vma) || vma_is_kernel_owned(vma)) { trace_sched_skip_vma_numa(mm, vma, NUMAB_SKIP_UNSUITABLE); continue; } diff --git a/mm/gup.c b/mm/gup.c index e6310a7cc05b2b..f166acf794e308 100644 --- a/mm/gup.c +++ b/mm/gup.c @@ -621,7 +621,7 @@ static struct page *no_page_table(struct vm_area_struct *vma, * But we can only make this optimization where a hole would surely * be zero-filled if handle_mm_fault() actually did handle it. */ - if (is_vm_hugetlb_page(vma)) { + if (vma_is_hugetlb(vma)) { struct hstate *h = hstate_vma(vma); if (!hugetlbfs_pagecache_present(h, vma, address)) @@ -1213,7 +1213,7 @@ static int check_vma_flags(struct vm_area_struct *vma, unsigned long gup_flags) if ((gup_flags & FOLL_LONGTERM) && vma_is_fsdax(vma)) return -EOPNOTSUPP; - if ((gup_flags & FOLL_SPLIT_PMD) && is_vm_hugetlb_page(vma)) + if ((gup_flags & FOLL_SPLIT_PMD) && vma_is_hugetlb(vma)) return -EOPNOTSUPP; if (vma_is_secretmem(vma)) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 6ee21854b6dfa9..05c17a01551df1 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -4792,7 +4792,7 @@ static inline bool vma_not_suitable_for_thp_split(struct vm_area_struct *vma) return true; if (vma_test(vma, VMA_IO_BIT)) return true; - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) return true; return false; diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 7b27c3c5c3e58e..1b53ba991d360a 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -1147,7 +1147,7 @@ static inline struct resv_map *inode_resv_map(struct inode *inode) static struct resv_map *vma_resv_map(struct vm_area_struct *vma) { - VM_BUG_ON_VMA(!is_vm_hugetlb_page(vma), vma); + VM_WARN_ON_ONCE_VMA(!vma_is_hugetlb(vma), vma); if (vma->vm_flags & VM_MAYSHARE) { struct address_space *mapping = vma->vm_file->f_mapping; struct inode *inode = mapping->host; @@ -1162,7 +1162,7 @@ static struct resv_map *vma_resv_map(struct vm_area_struct *vma) static void set_vma_resv_map(struct vm_area_struct *vma, struct resv_map *map) { - VM_WARN_ON_ONCE_VMA(!is_vm_hugetlb_page(vma), vma); + VM_WARN_ON_ONCE_VMA(!vma_is_hugetlb(vma), vma); VM_WARN_ON_ONCE_VMA(vma_test(vma, VMA_MAYSHARE_BIT), vma); set_vma_private_data(vma, (unsigned long)map); @@ -1170,7 +1170,7 @@ static void set_vma_resv_map(struct vm_area_struct *vma, struct resv_map *map) static void set_vma_resv_flags(struct vm_area_struct *vma, unsigned long flags) { - VM_WARN_ON_ONCE_VMA(!is_vm_hugetlb_page(vma), vma); + VM_WARN_ON_ONCE_VMA(!vma_is_hugetlb(vma), vma); VM_WARN_ON_ONCE_VMA(vma_test(vma, VMA_MAYSHARE_BIT), vma); set_vma_private_data(vma, get_vma_private_data(vma) | flags); @@ -1178,7 +1178,7 @@ static void set_vma_resv_flags(struct vm_area_struct *vma, unsigned long flags) static int is_vma_resv_set(struct vm_area_struct *vma, unsigned long flag) { - VM_BUG_ON_VMA(!is_vm_hugetlb_page(vma), vma); + VM_WARN_ON_ONCE_VMA(!vma_is_hugetlb(vma), vma); return (get_vma_private_data(vma) & flag) != 0; } @@ -1192,7 +1192,7 @@ bool __vma_private_lock(struct vm_area_struct *vma) void hugetlb_dup_vma_private(struct vm_area_struct *vma) { - VM_BUG_ON_VMA(!is_vm_hugetlb_page(vma), vma); + VM_WARN_ON_ONCE_VMA(!vma_is_hugetlb(vma), vma); /* * Clear vm_private_data * - For shared mappings this is a per-vma semaphore that may be @@ -5276,7 +5276,7 @@ void __unmap_hugepage_range(struct mmu_gather *tlb, struct vm_area_struct *vma, unsigned long last_addr_mask; i_mmap_assert_write_locked(vma->vm_file->f_mapping); - WARN_ON(!is_vm_hugetlb_page(vma)); + WARN_ON(!vma_is_hugetlb(vma)); BUG_ON(start & ~huge_page_mask(h)); BUG_ON(end & ~huge_page_mask(h)); @@ -7502,6 +7502,6 @@ void hugetlb_unshare_all_pmds(struct vm_area_struct *vma) */ void fixup_hugetlb_reservations(struct vm_area_struct *vma) { - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) clear_vma_resv_huge_pages(vma); } diff --git a/mm/internal.h b/mm/internal.h index 1bf6517cf38932..c63df7b7d77272 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -1117,7 +1117,7 @@ static inline bool vma_supports_mlock(const struct vm_area_struct *vma) return false; if (vma_test_single_mask(vma, VMA_DROPPABLE)) return false; - if (vma_is_dax(vma) || is_vm_hugetlb_page(vma)) + if (vma_is_dax(vma) || vma_is_hugetlb(vma)) return false; return vma != get_gate_vma(current->mm); } diff --git a/mm/madvise.c b/mm/madvise.c index 1cfb0433229310..f7d03e8988e214 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -881,7 +881,7 @@ bool madvise_dontneed_free_valid_vma(struct madvise_behavior *madv_behavior) int behavior = madv_behavior->behavior; struct madvise_behavior_range *range = &madv_behavior->range; - if (!is_vm_hugetlb_page(vma)) { + if (!vma_is_hugetlb(vma)) { unsigned int forbidden = VM_PFNMAP; if (behavior != MADV_DONTNEED_LOCKED) @@ -1579,7 +1579,7 @@ static int madvise_vma_behavior(struct madvise_behavior *madv_behavior) new_flags |= VM_DONTDUMP; break; case MADV_DODUMP: - if ((!is_vm_hugetlb_page(vma) && (new_flags & VM_SPECIAL)) || + if ((!vma_is_hugetlb(vma) && (new_flags & VM_SPECIAL)) || (new_flags & VM_DROPPABLE)) return -EINVAL; new_flags &= ~VM_DONTDUMP; diff --git a/mm/memory.c b/mm/memory.c index 1e6cd2e504089a..6c011979401aab 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -1564,7 +1564,7 @@ copy_page_range(struct vm_area_struct *dst_vma, struct vm_area_struct *src_vma) if (!vma_needs_copy(dst_vma, src_vma)) return 0; - if (is_vm_hugetlb_page(src_vma)) + if (vma_is_hugetlb(src_vma)) return copy_hugetlb_page_range(dst_mm, src_mm, dst_vma, src_vma); /* @@ -2178,7 +2178,7 @@ static void __zap_vma_range(struct mmu_gather *tlb, struct vm_area_struct *vma, if (vma->vm_file && !reaping) uprobe_munmap(vma, start, end); - if (unlikely(is_vm_hugetlb_page(vma))) { + if (unlikely(vma_is_hugetlb(vma))) { zap_flags_t zap_flags = details ? details->zap_flags : 0; VM_WARN_ON_ONCE(reaping); @@ -2313,7 +2313,7 @@ void zap_vma_range_batched(struct mmu_gather *tlb, */ __zap_vma_range(tlb, vma, address, end, details); mmu_notifier_invalidate_range_end(&range); - if (is_vm_hugetlb_page(vma)) { + if (vma_is_hugetlb(vma)) { /* * flush tlb and free resources before hugetlb_zap_end(), to * avoid concurrent page faults' allocation failure. @@ -6933,7 +6933,7 @@ vm_fault_t handle_mm_fault(struct vm_area_struct *vma, unsigned long address, lru_gen_enter_fault(vma); - if (unlikely(is_vm_hugetlb_page(vma))) + if (unlikely(vma_is_hugetlb(vma))) ret = hugetlb_fault(vma->vm_mm, vma, address, flags); else ret = __handle_mm_fault(vma, address, flags); @@ -7803,12 +7803,12 @@ void ptlock_free(struct ptdesc *ptdesc) void vma_pgtable_walk_begin(struct vm_area_struct *vma) { - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) hugetlb_vma_lock_read(vma); } void vma_pgtable_walk_end(struct vm_area_struct *vma) { - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) hugetlb_vma_unlock_read(vma); } diff --git a/mm/mempolicy.c b/mm/mempolicy.c index e9860fb9f73f8d..70298fded1b4a8 100644 --- a/mm/mempolicy.c +++ b/mm/mempolicy.c @@ -2018,7 +2018,7 @@ bool vma_migratable(struct vm_area_struct *vma) if (vma_is_dax(vma)) return false; - if (is_vm_hugetlb_page(vma) && + if (vma_is_hugetlb(vma) && !hugepage_migration_supported(hstate_vma(vma))) return false; diff --git a/mm/migrate_device.c b/mm/migrate_device.c index 0c437004329d9c..c38cbaaef5a492 100644 --- a/mm/migrate_device.c +++ b/mm/migrate_device.c @@ -743,7 +743,7 @@ int migrate_vma_setup(struct migrate_vma *args) args->start &= PAGE_MASK; args->end &= PAGE_MASK; - if (!args->vma || is_vm_hugetlb_page(args->vma) || + if (!args->vma || vma_is_hugetlb(args->vma) || (args->vma->vm_flags & VM_SPECIAL) || vma_is_dax(args->vma)) return -EINVAL; if (nr_pages <= 0) diff --git a/mm/mmap.c b/mm/mmap.c index 4bf26b0f1e6e36..98449f364af1c4 100644 --- a/mm/mmap.c +++ b/mm/mmap.c @@ -1786,7 +1786,7 @@ __latent_entropy int dup_mmap(struct mm_struct *mm, struct mm_struct *oldmm) /* * Copy/update hugetlb private vma information. */ - if (is_vm_hugetlb_page(tmp)) + if (vma_is_hugetlb(tmp)) hugetlb_dup_vma_private(tmp); /* diff --git a/mm/mmu_gather.c b/mm/mmu_gather.c index 2a72a9686773a3..9f353f0e2ef4d2 100644 --- a/mm/mmu_gather.c +++ b/mm/mmu_gather.c @@ -480,7 +480,7 @@ void tlb_gather_mmu_vma(struct mmu_gather *tlb, struct vm_area_struct *vma) { tlb_gather_mmu(tlb, vma->vm_mm); tlb_update_vma_flags(tlb, vma); - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) /* All entries have the same size. */ tlb_change_page_size(tlb, huge_page_size(hstate_vma(vma))); } diff --git a/mm/mprotect.c b/mm/mprotect.c index fe32fd87cf5cd7..a1b6d29bf03908 100644 --- a/mm/mprotect.c +++ b/mm/mprotect.c @@ -717,7 +717,7 @@ long change_protection(struct mmu_gather *tlb, (cp_flags & MM_CP_UFFD_RWP)) newprot = PAGE_NONE; - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) pages = hugetlb_change_protection(vma, start, end, newprot, cp_flags); else diff --git a/mm/mremap.c b/mm/mremap.c index 982460ef6bdbb1..5c72545db1752c 100644 --- a/mm/mremap.c +++ b/mm/mremap.c @@ -812,7 +812,7 @@ unsigned long move_page_tables(struct pagetable_move_control *pmc) if (!pmc->len_in) return 0; - if (is_vm_hugetlb_page(pmc->old)) + if (vma_is_hugetlb(pmc->old)) return move_hugetlb_page_tables(pmc->old, pmc->new, pmc->old_addr, pmc->new_addr, pmc->len_in); @@ -1769,7 +1769,7 @@ static bool vma_multi_allowed(struct vm_area_struct *vma) /* Known good. */ if (vma_is_shmem(vma)) return true; - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) return true; if (file->f_op->get_unmapped_area == thp_get_unmapped_area) return true; @@ -1792,7 +1792,7 @@ static int check_prep_vma(struct vma_remap_struct *vrm) return -EPERM; /* Align to hugetlb page size, if required. */ - if (is_vm_hugetlb_page(vma) && !align_hugetlb(vrm)) + if (vma_is_hugetlb(vma) && !align_hugetlb(vrm)) return -EINVAL; vrm_set_delta(vrm); diff --git a/mm/page_vma_mapped.c b/mm/page_vma_mapped.c index 28e306fdb3a5b8..8408aee7571b56 100644 --- a/mm/page_vma_mapped.c +++ b/mm/page_vma_mapped.c @@ -109,7 +109,7 @@ static bool check_pte(struct page_vma_mapped_walk *pvmw, unsigned long pte_nr) unsigned long pfn; pte_t ptent; - if (is_vm_hugetlb_page(pvmw->vma)) + if (vma_is_hugetlb(pvmw->vma)) ptent = huge_ptep_get(pvmw->vma->vm_mm, pvmw->address, pvmw->pte); else @@ -206,7 +206,7 @@ bool page_vma_mapped_walk(struct page_vma_mapped_walk *pvmw) if (pvmw->pmd && !pvmw->pte) return not_found(pvmw); - if (unlikely(is_vm_hugetlb_page(vma))) { + if (unlikely(vma_is_hugetlb(vma))) { struct hstate *hstate = hstate_vma(vma); unsigned long size = huge_page_size(hstate); /* The only possible mapping was handled on last iteration */ diff --git a/mm/pagewalk.c b/mm/pagewalk.c index 7411702a37f58d..e6493bbe6919e5 100644 --- a/mm/pagewalk.c +++ b/mm/pagewalk.c @@ -408,7 +408,7 @@ static int __walk_page_range(unsigned long start, unsigned long end, int err = 0; struct vm_area_struct *vma = walk->vma; const struct mm_walk_ops *ops = walk->ops; - bool is_hugetlb = is_vm_hugetlb_page(vma); + bool is_hugetlb = vma_is_hugetlb(vma); /* We do not support hugetlb PTE installation. */ if (ops->install_pte && is_hugetlb) diff --git a/mm/swapfile.c b/mm/swapfile.c index 2cd0d0ba966c38..c1c5fbb3c909d3 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -2707,7 +2707,7 @@ static int unuse_mm(struct mm_struct *mm, unsigned int type) if (check_stable_address_space(mm)) goto unlock; for_each_vma(vmi, vma) { - if (vma->anon_vma && !is_vm_hugetlb_page(vma)) { + if (vma->anon_vma && !vma_is_hugetlb(vma)) { ret = unuse_vma(vma, type); if (ret) break; diff --git a/mm/userfaultfd.c b/mm/userfaultfd.c index bf50bff3838aa4..215b993d6df4fd 100644 --- a/mm/userfaultfd.c +++ b/mm/userfaultfd.c @@ -237,7 +237,7 @@ static int mfill_get_vma(struct mfill_state *state) if ((flags & MFILL_ATOMIC_WP) && !(dst_vma->vm_flags & VM_UFFD_WP)) goto out_unlock; - if (is_vm_hugetlb_page(dst_vma)) + if (vma_is_hugetlb(dst_vma)) return 0; ops = vma_uffd_ops(dst_vma); @@ -804,7 +804,7 @@ static __always_inline ssize_t mfill_atomic_hugetlb( } err = -ENOENT; - if (!is_vm_hugetlb_page(dst_vma)) + if (!vma_is_hugetlb(dst_vma)) goto out_unlock_vma; err = -EINVAL; @@ -967,7 +967,7 @@ static __always_inline ssize_t mfill_atomic(struct userfaultfd_ctx *ctx, /* * If this is a HUGETLB vma, pass off to appropriate routine */ - if (is_vm_hugetlb_page(state.vma)) + if (vma_is_hugetlb(state.vma)) return mfill_atomic_hugetlb(ctx, state.vma, dst_start, src_start, len, flags); @@ -1114,7 +1114,7 @@ static int mwriteprotect_range(struct userfaultfd_ctx *ctx, unsigned long start, break; } - if (is_vm_hugetlb_page(dst_vma)) { + if (vma_is_hugetlb(dst_vma)) { err = -EINVAL; page_mask = vma_kernel_pagesize(dst_vma) - 1; if ((start & page_mask) || (len & page_mask)) @@ -1172,7 +1172,7 @@ int mrwprotect_range(struct userfaultfd_ctx *ctx, unsigned long start, if (!userfaultfd_rwp(dst_vma)) return -ENOENT; - if (is_vm_hugetlb_page(dst_vma)) { + if (vma_is_hugetlb(dst_vma)) { unsigned long page_mask; page_mask = vma_kernel_pagesize(dst_vma) - 1; @@ -2150,7 +2150,7 @@ static bool vma_can_userfault(struct vm_area_struct *vma, vm_flags_t vm_flags, if (vma->vm_flags & (VM_DROPPABLE | VM_SHADOW_STACK)) return false; - if (!is_vm_hugetlb_page(vma) && (vma->vm_flags & VM_SPECIAL)) + if (!vma_is_hugetlb(vma) && (vma->vm_flags & VM_SPECIAL)) return false; vm_flags &= __VM_UFFD_FLAGS; @@ -2320,7 +2320,7 @@ static int userfaultfd_register_range(struct userfaultfd_ctx *ctx, */ userfaultfd_set_ctx(vma, ctx, vm_flags); - if (is_vm_hugetlb_page(vma) && uffd_disable_huge_pmd_share(vma)) + if (vma_is_hugetlb(vma) && uffd_disable_huge_pmd_share(vma)) hugetlb_unshare_all_pmds(vma); skip: @@ -2896,7 +2896,7 @@ vm_fault_t handle_userfault(struct vm_fault *vmf, unsigned long reason) * (sleepable) vma lock can modify the current task state, that * must be before explicitly calling set_current_state(). */ - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) hugetlb_vma_lock_read(vma); spin_lock_irq(&ctx->fault_pending_wqh.lock); @@ -2913,7 +2913,7 @@ vm_fault_t handle_userfault(struct vm_fault *vmf, unsigned long reason) set_current_state(blocking_state); spin_unlock_irq(&ctx->fault_pending_wqh.lock); - if (is_vm_hugetlb_page(vma)) { + if (vma_is_hugetlb(vma)) { must_wait = userfaultfd_huge_must_wait(ctx, vmf, reason); hugetlb_vma_unlock_read(vma); } else { @@ -3745,7 +3745,7 @@ static int userfaultfd_register(struct userfaultfd_ctx *ctx, * If the first vma contains huge pages, make sure start address * is aligned to huge page size. */ - if (is_vm_hugetlb_page(vma)) { + if (vma_is_hugetlb(vma)) { unsigned long vma_hpagesize = vma_kernel_pagesize(vma); if (start & (vma_hpagesize - 1)) @@ -3796,7 +3796,7 @@ static int userfaultfd_register(struct userfaultfd_ctx *ctx, * If this vma contains ending address, and huge pages * check alignment. */ - if (is_vm_hugetlb_page(cur) && end <= cur->vm_end && + if (vma_is_hugetlb(cur) && end <= cur->vm_end && end > cur->vm_start) { unsigned long vma_hpagesize = vma_kernel_pagesize(cur); @@ -3832,7 +3832,7 @@ static int userfaultfd_register(struct userfaultfd_ctx *ctx, /* * Note vmas containing huge pages */ - if (is_vm_hugetlb_page(cur)) + if (vma_is_hugetlb(cur)) basic_ioctls = true; found = true; @@ -3918,7 +3918,7 @@ static int userfaultfd_unregister(struct userfaultfd_ctx *ctx, * If the first vma contains huge pages, make sure start address * is aligned to huge page size. */ - if (is_vm_hugetlb_page(vma)) { + if (vma_is_hugetlb(vma)) { unsigned long vma_hpagesize = vma_kernel_pagesize(vma); if (start & (vma_hpagesize - 1)) diff --git a/mm/vma.c b/mm/vma.c index 6e4c337a8e1a2b..ac3908d6967ab8 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -599,7 +599,7 @@ __split_vma(struct vma_iterator *vmi, struct vm_area_struct *vma, * boundary. */ vma_adjust_trans_huge(vma, vma->vm_start, addr, NULL); - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) hugetlb_split(vma, addr); if (new_below) { @@ -2251,7 +2251,7 @@ bool vma_wants_writenotify(struct vm_area_struct *vma, pgprot_t vm_page_prot) * Do we need to track softdirty? hugetlb does not support softdirty * tracking yet. */ - if (vma_soft_dirty_enabled(vma) && !is_vm_hugetlb_page(vma)) + if (vma_soft_dirty_enabled(vma) && !vma_is_hugetlb(vma)) return true; /* Do we need write faults for uffd-wp tracking? */ @@ -2370,7 +2370,7 @@ int mm_take_all_locks(struct mm_struct *mm) if (signal_pending(current)) goto out_unlock; if (vma->vm_file && vma->vm_file->f_mapping && - is_vm_hugetlb_page(vma)) + vma_is_hugetlb(vma)) vm_lock_mapping(mm, vma->vm_file->f_mapping); } @@ -2379,7 +2379,7 @@ int mm_take_all_locks(struct mm_struct *mm) if (signal_pending(current)) goto out_unlock; if (vma->vm_file && vma->vm_file->f_mapping && - !is_vm_hugetlb_page(vma)) + !vma_is_hugetlb(vma)) vm_lock_mapping(mm, vma->vm_file->f_mapping); } diff --git a/mm/vmscan.c b/mm/vmscan.c index dd6261c862794f..9fc4282da0aab5 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -3474,7 +3474,7 @@ static int should_skip_vma(unsigned long start, unsigned long end, struct mm_wal if (!vma_is_accessible(vma)) return true; - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) return true; if (!vma_has_recency(vma)) diff --git a/tools/testing/vma/include/stubs.h b/tools/testing/vma/include/stubs.h index d6136e19a8af3d..48d1dc53df42cb 100644 --- a/tools/testing/vma/include/stubs.h +++ b/tools/testing/vma/include/stubs.h @@ -193,7 +193,7 @@ static inline bool mapping_can_writeback(struct address_space *mapping) return true; } -static inline bool is_vm_hugetlb_page(struct vm_area_struct *vma) +static inline bool vma_is_hugetlb(struct vm_area_struct *vma) { return false; } From 7a41ed87f53fea408f29bee3f4551515b23daa2a Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:38 +0100 Subject: [PATCH 0991/1352] mm: drop some redundant checks around hugetlb VMAs Adjust code which inadvertently perform redundant checks on hugetlb VMAs and clean them up: * hugetlb VMAs have VMA_DONTEXPAND_BIT set so a VMA_SPECIAL_FLAGS check suffices. (migrate_vma_setup() regains an explicit hugetlb test later in the series, once VMA_SPECIAL_FLAGS is removed.) * hugetlb VMAs unconditionally set vma->vm_ops, so they are never anonymous. * hugetlb VMAs do not set VMA_PFNMAP_BIT so checking for this is redundant. While we're here also drop a VM_BUG_ON() which the simplified check above makes unreachable, and use the new VMA flag API. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-29-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Marc Zyngier Reviewed-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- arch/arm64/kvm/mmu.c | 4 +--- drivers/gpu/drm/drm_gpusvm.c | 3 +-- mm/migrate_device.c | 4 ++-- 3 files changed, 4 insertions(+), 7 deletions(-) diff --git a/arch/arm64/kvm/mmu.c b/arch/arm64/kvm/mmu.c index b8e28d1b5461d5..73c8492e85c4e0 100644 --- a/arch/arm64/kvm/mmu.c +++ b/arch/arm64/kvm/mmu.c @@ -1472,14 +1472,12 @@ static int get_vma_page_shift(struct vm_area_struct *vma, unsigned long hva) { unsigned long pa; - if (vma_is_hugetlb(vma) && !(vma->vm_flags & VM_PFNMAP)) + if (vma_is_hugetlb(vma)) return huge_page_shift(hstate_vma(vma)); if (!(vma->vm_flags & VM_PFNMAP)) return PAGE_SHIFT; - VM_BUG_ON(vma_is_hugetlb(vma)); - pa = (vma->vm_pgoff << PAGE_SHIFT) + (hva - vma->vm_start); #ifndef __PAGETABLE_PMD_FOLDED diff --git a/drivers/gpu/drm/drm_gpusvm.c b/drivers/gpu/drm/drm_gpusvm.c index a1d4989b0b616f..fab34fea99c2fe 100644 --- a/drivers/gpu/drm/drm_gpusvm.c +++ b/drivers/gpu/drm/drm_gpusvm.c @@ -1141,8 +1141,7 @@ drm_gpusvm_range_find_or_insert(struct drm_gpusvm *gpusvm, * limitations. If/when migrate_vma_* add more support, this logic will * have to change. */ - migrate_devmem = ctx->devmem_possible && - vma_is_anonymous(vas) && !vma_is_hugetlb(vas); + migrate_devmem = ctx->devmem_possible && vma_is_anonymous(vas); chunk_size = drm_gpusvm_range_chunk_size(gpusvm, notifier, vas, fault_addr, gpuva_start, diff --git a/mm/migrate_device.c b/mm/migrate_device.c index c38cbaaef5a492..b9c453c28795f3 100644 --- a/mm/migrate_device.c +++ b/mm/migrate_device.c @@ -743,8 +743,8 @@ int migrate_vma_setup(struct migrate_vma *args) args->start &= PAGE_MASK; args->end &= PAGE_MASK; - if (!args->vma || vma_is_hugetlb(args->vma) || - (args->vma->vm_flags & VM_SPECIAL) || vma_is_dax(args->vma)) + if (!args->vma || vma_test_any_mask(args->vma, VMA_SPECIAL_FLAGS) || + vma_is_dax(args->vma)) return -EINVAL; if (nr_pages <= 0) return -EINVAL; From acf8d49d55501b788a4fd0ddb1e030811d5e16df Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:39 +0100 Subject: [PATCH 0992/1352] mm/madvise: update is_valid_guard_vma() to use vma_can_merge() We currently disallow the installation of lightweight guard regions in VMAs whose flags intersect VMA_SPECIAL_FLAGS or VMA_HUGETLB_BIT, or VMA_LOCKED_BIT unless allow_locked is set. hugetlb VMAs set VMA_DONTEXPAND_BIT so this was already redundant, VMA_SPECIAL_FLAGS already sufficed. However, now that VMA_IO_BIT is only set if VMA_PFNMAP or VMA_MIXEDMAP_BIT is set, this check collapses to being the equivalent of !vma_can_merge(). Update is_valid_guard_vma() to reflect this. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-30-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/madvise.c | 20 +++++++++++++------- 1 file changed, 13 insertions(+), 7 deletions(-) diff --git a/mm/madvise.c b/mm/madvise.c index f7d03e8988e214..7a038837b2a8b7 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -1222,19 +1222,25 @@ static long madvise_remove(struct madvise_behavior *madv_behavior) return error; } -static bool is_valid_guard_vma(struct vm_area_struct *vma, bool allow_locked) +static bool is_valid_guard_vma(const struct vm_area_struct *vma, + bool allow_locked) { - vm_flags_t disallowed = VM_SPECIAL | VM_HUGETLB; - /* - * A user could lock after setting a guard range but that's fine, as + * A user could lock after setting a guard range but that's fine as * they'd not be able to fault in. The issue arises when we try to zap * existing locked VMAs. We don't want to do that. */ - if (!allow_locked) - disallowed |= VM_LOCKED; + if (!allow_locked && vma_test(vma, VMA_LOCKED_BIT)) + return false; + /* + * Guard regions require a VMA whose page tables are managed solely by + * the core, which is also what merging requires, so disallow any flags + * that would prevent a merge. + */ + if (!vma_can_merge(vma)) + return false; - return !(vma->vm_flags & disallowed); + return true; } static bool is_guard_pte_marker(pte_t ptent) From 85011e2a06b56f814e366aed8556a993ef84320b Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:40 +0100 Subject: [PATCH 0993/1352] mm/vma: introduce vma[_flags]_is_persistent() Introduce vma[_flags]_is_persistent() for the purposes of identifying mappings that are persistent in the sense that bytes to the mapping stay there, and bytes read from the mapping are the same unless changed by actions taken by userland. Kernel-owned mappings do not fall into this category, as their owner may change the contents without the user having initiated it, and nor of course does memory-mapped I/O. We exclude fixed mappings as these are singled out as being unmergeable and so cannot be guaranteed to persist user data. hugetlb mappings are fixed mappings, but their contents are entirely the user's, so they are explicitly carved out as persistent, as the MADV_DODUMP check already does. It excludes droppable mappings, which by their nature are ephemeral. Use this functionality to update the madvise MADV_DODUMP check to test for persistence rather than open-coding this. This replaces the VM_SPECIAL check which means it no longer checks for VMA_IO_BIT, however this is safe as we have established the invariant that only kernel-owned mappings may set VMA_IO_BIT, so we implicitly include these. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-31-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- include/linux/mm.h | 44 ++++++++++++++++++++++++++++++++++++++++++++ mm/madvise.c | 4 ++-- 2 files changed, 46 insertions(+), 2 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index dbd4c1a70a2359..eaf3a4110f91ed 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -1742,6 +1742,50 @@ static inline bool vma_can_merge(const struct vm_area_struct *vma) return vma_flags_can_merge(&vma->flags); } +/** + * vma_flags_is_persistent() - Do the specified VMA flags imply that the VMA + * contains persistent data? + * @flags: The VMA flags to test. + * + * Persistent in the sense that - if you write bytes to the mapping - do they + * stay written? + * + * If the kernel or a device could write to the memory independently of + * userland, or the kernel could arbitrarily discard it, then it is not + * persistent. + * + * Returns: true if the flags imply this VMA is persistent, otherwise false. + */ +static inline bool vma_flags_is_persistent(const vma_flags_t *flags) +{ + /* hugetlb is a fixed mapping, but its contents are the user's own. */ + if (vma_flags_is_hugetlb(flags)) + return true; + /* + * MMIO mappings may not store what is written and may be changed by the + * device. Kernel-owned and fixed mappings may be changed by their owner + * without the user having initiated it. + */ + if (vma_flags_is_kernel_owned(flags) || + vma_flags_is_fixed_mapping(flags)) + return false; + /* Droppable memory is discardable by definition. */ + return !vma_flags_test_single_mask(flags, VMA_DROPPABLE); +} + +/** + * vma_is_persistent() - Does the VMA contain persistent data? + * @vma: The VMA to test. + * + * See vma_flags_is_persistent() for details. + * + * Returns: true if the VMA is persistent, otherwise false. + */ +static inline bool vma_is_persistent(const struct vm_area_struct *vma) +{ + return vma_flags_is_persistent(&vma->flags); +} + /** * vma_kernel_pagesize - Default page size granularity for this VMA. * @vma: The user mapping. diff --git a/mm/madvise.c b/mm/madvise.c index 7a038837b2a8b7..f5307b3c2191f6 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -1585,8 +1585,8 @@ static int madvise_vma_behavior(struct madvise_behavior *madv_behavior) new_flags |= VM_DONTDUMP; break; case MADV_DODUMP: - if ((!vma_is_hugetlb(vma) && (new_flags & VM_SPECIAL)) || - (new_flags & VM_DROPPABLE)) + /* Non-persistent memory cannot be dumped. */ + if (!vma_is_persistent(vma)) return -EINVAL; new_flags &= ~VM_DONTDUMP; break; From 590656071d679211dd16f7a0ca46137129cb1e8b Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:41 +0100 Subject: [PATCH 0994/1352] mm/uffd: use predicates for userfaultfd checks Rather than directly checking VMA flags, use the newly introduced vma_is_kernel_owned() and vma_is_persistent() helpers in userfaultfd when assessing VMA suitability for userfaultfd and UFFDIO_MOVE. Update vma_move_compatible() so it's expressed in terms of VMA characteristics rather than arbitrary flags. Additionally, update the use of the deprecated VMA flag API when checking VMA_SHADOW_STACK_BIT. A VMA_IO_BIT check is no longer required but that is fine as a hard invariant has been established that only kernel-owned mappings may set VMA_IO_BIT so the check is now redundant. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-32-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/userfaultfd.c | 21 +++++++++++++++------ 1 file changed, 15 insertions(+), 6 deletions(-) diff --git a/mm/userfaultfd.c b/mm/userfaultfd.c index 215b993d6df4fd..17ecbb0ceddf11 100644 --- a/mm/userfaultfd.c +++ b/mm/userfaultfd.c @@ -1755,10 +1755,18 @@ static inline bool move_splits_huge_pmd(unsigned long dst_addr, } #endif -static inline bool vma_move_compatible(struct vm_area_struct *vma) +static inline bool vma_move_compatible(const struct vm_area_struct *vma) { - return !(vma->vm_flags & (VM_PFNMAP | VM_IO | VM_HUGETLB | - VM_MIXEDMAP | VM_SHADOW_STACK)); + /* uffd is generally incompatible with kernel-owned mappings. */ + if (vma_is_kernel_owned(vma)) + return false; + /* The shadow stack should not be written to by userspace. */ + if (vma_test_single_mask(vma, VMA_SHADOW_STACK)) + return false; + /* hugetlb mappings cannot be safely moved. */ + if (vma_is_hugetlb(vma)) + return false; + return true; } static int validate_move_areas(struct userfaultfd_ctx *ctx, @@ -2147,10 +2155,11 @@ static bool vma_can_userfault(struct vm_area_struct *vma, vm_flags_t vm_flags, { const struct vm_uffd_ops *ops = vma_uffd_ops(vma); - if (vma->vm_flags & (VM_DROPPABLE | VM_SHADOW_STACK)) + /* Non-persistent memory is inherently not controllable by userspace. */ + if (!vma_is_persistent(vma)) return false; - - if (!vma_is_hugetlb(vma) && (vma->vm_flags & VM_SPECIAL)) + /* The shadow stack should not be written to by userspace. */ + if (vma_test_single_mask(vma, VMA_SHADOW_STACK)) return false; vm_flags &= __VM_UFFD_FLAGS; From 9c9a4e5137184ef48c1b4512584ea4a4a975c3e4 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:42 +0100 Subject: [PATCH 0995/1352] mm/madvise: use predicates for madvise(..., MADV_DOFORK) Make it clear what we're blocking in MADV_DOFORK. Previously we simply disallowed VM_SPECIAL i.e. kernel-owned mappings, fixed mappings and VMA_IO_BIT. Now the invariant is established that only kernel-owned mappings can set VMA_IO_BIT, the VMA_IO_BIT check is redundant. The rest is equivalent to testing for a kernel-owned or fixed mapping, i.e. exactly the same check as whether the VMA is permitted to be merged. This was established by commit 0b2758f48f22 ("Require (reasonably) normal mappings for MADV_DOFORK") containing my hands-down favourite call out of all time. Express the same thing differently - if we wouldn't be allowed to merge it, then we aren't allowed to manipulate CoW behaviour on fork. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-33-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/madvise.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/madvise.c b/mm/madvise.c index f5307b3c2191f6..80ea991ce350a3 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -1566,7 +1566,7 @@ static int madvise_vma_behavior(struct madvise_behavior *madv_behavior) new_flags |= VM_DONTCOPY; break; case MADV_DOFORK: - if (new_flags & VM_SPECIAL) + if (!vma_can_merge(vma)) return -EINVAL; new_flags &= ~VM_DONTCOPY; break; From bd7ccf5cc1d3bb78d0b535f6ef021e54301f3b26 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:43 +0100 Subject: [PATCH 0996/1352] mm: eliminate VMA_SPECIAL_FLAGS usage when hugetlb explicitly tested It is now an invariant that VMA_IO_BIT is not set except by kernel-owned mappings, so each existing VMA_SPECIAL_FLAGS test need only test for VMA_DONTEXPAND_BIT, VMA_PFNMAP_BIT and VMA_MIXEDMAP_BIT. This is precisely a test for a kernel-owned or fixed mapping. Update a number of callsites which already explicitly handle hugetlb mappings. vma_supports_mlock() and ksm_compatible() also explicitly bail on droppable mappings - detecting kernel-owned, fixed or droppable mappings is handled by vma_is_persistent(), so in these cases use this predicate. should_skip_vma() tests for locked, kernel-owned or fixed memory (having already excluded hugetlb mappings) so simply test for those there. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-34-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/internal.h | 4 +--- mm/ksm.c | 4 +--- mm/vmscan.c | 3 ++- 3 files changed, 4 insertions(+), 7 deletions(-) diff --git a/mm/internal.h b/mm/internal.h index c63df7b7d77272..3b9fdb826162df 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -1113,9 +1113,7 @@ static inline struct file *maybe_unlock_mmap_for_io(struct vm_fault *vmf, static inline bool vma_supports_mlock(const struct vm_area_struct *vma) { - if (vma_test_any_mask(vma, VMA_SPECIAL_FLAGS)) - return false; - if (vma_test_single_mask(vma, VMA_DROPPABLE)) + if (!vma_is_persistent(vma)) return false; if (vma_is_dax(vma) || vma_is_hugetlb(vma)) return false; diff --git a/mm/ksm.c b/mm/ksm.c index 624f37975e1295..f80372bfd4b2fc 100644 --- a/mm/ksm.c +++ b/mm/ksm.c @@ -747,9 +747,7 @@ static bool ksm_compatible(const struct file *file, vma_flags_t vma_flags) if (vma_flags_test_any(&vma_flags, VMA_SHARED_BIT, VMA_MAYSHARE_BIT, VMA_HUGETLB_BIT)) return false; - if (vma_flags_test_single_mask(&vma_flags, VMA_DROPPABLE)) - return false; - if (vma_flags_test_any_mask(&vma_flags, VMA_SPECIAL_FLAGS)) + if (!vma_flags_is_persistent(&vma_flags)) return false; if (file_is_dax(file)) return false; diff --git a/mm/vmscan.c b/mm/vmscan.c index 9fc4282da0aab5..76f6ece5f20134 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -3480,7 +3480,8 @@ static int should_skip_vma(unsigned long start, unsigned long end, struct mm_wal if (!vma_has_recency(vma)) return true; - if (vma->vm_flags & (VM_LOCKED | VM_SPECIAL)) + if (vma_test(vma, VMA_LOCKED_BIT) || vma_is_kernel_owned(vma) || + vma_is_fixed_mapping(vma)) return true; if (vma == get_gate_vma(vma->vm_mm)) From 395d5e30db0340ca99d787ceb775c8b076493644 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:44 +0100 Subject: [PATCH 0997/1352] mm: eliminate VMA_SPECIAL_FLAGS check in lru_gen_look_around() A kernel-owned or fixed mapping is one which sets VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT or VMA_DONTEXPAND_BIT, which is precisely what VMA_SPECIAL_FLAGS tests for other than VMA_IO_BIT, which is safe to drop as only kernel-owned mappings may set it. Using these predicates rather than VMA_SPECIAL_FLAGS makes the check self-documenting and helps eliminate the confusion around 'special' flags. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-35-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/vmscan.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 76f6ece5f20134..836f50814ffae5 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -4421,8 +4421,8 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr) if (spin_is_contended(pvmw->ptl)) return true; - /* exclude special VMAs containing anon pages from COW */ - if (vma->vm_flags & VM_SPECIAL) + /* exclude kernel-owned and fixed VMAs containing anon pages from COW */ + if (vma_is_kernel_owned(vma) || vma_is_fixed_mapping(vma)) return true; /* avoid taking the LRU lock under the PTL when possible */ From e97c45587bdf78a96d328ca5df7141b80c2be00a Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:45 +0100 Subject: [PATCH 0998/1352] mm: avoid use of VMA_SPECIAL_FLAGS in migrate_vma_setup() Now we have the expressive vma_is_kernel_owned() and vma_is_fixed_mapping() predicates, use them to determine whether to proceed with migration. This drops the VMA_IO_BIT test, which is safe as only kernel-owned mappings may set it. hugetlb mappings remain excluded, as they are fixed mappings. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-36-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/migrate_device.c | 12 +++++++----- 1 file changed, 7 insertions(+), 5 deletions(-) diff --git a/mm/migrate_device.c b/mm/migrate_device.c index b9c453c28795f3..b74c0ae4276826 100644 --- a/mm/migrate_device.c +++ b/mm/migrate_device.c @@ -739,19 +739,21 @@ static void migrate_vma_unmap(struct migrate_vma *migrate) */ int migrate_vma_setup(struct migrate_vma *args) { + const struct vm_area_struct *vma = args->vma; long nr_pages = (args->end - args->start) >> PAGE_SHIFT; args->start &= PAGE_MASK; args->end &= PAGE_MASK; - if (!args->vma || vma_test_any_mask(args->vma, VMA_SPECIAL_FLAGS) || - vma_is_dax(args->vma)) + if (!vma) + return -EINVAL; + if (vma_is_kernel_owned(vma) || vma_is_fixed_mapping(vma) || + vma_is_dax(vma)) return -EINVAL; if (nr_pages <= 0) return -EINVAL; - if (args->start < args->vma->vm_start || - args->start >= args->vma->vm_end) + if (args->start < vma->vm_start || args->start >= vma->vm_end) return -EINVAL; - if (args->end <= args->vma->vm_start || args->end > args->vma->vm_end) + if (args->end <= vma->vm_start || args->end > vma->vm_end) return -EINVAL; if (!args->src || !args->dst) return -EINVAL; From 64b7f8d4a45d3285951dae8d2d53625b0acd1a70 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:46 +0100 Subject: [PATCH 0999/1352] mm: eliminate VM_SPECIAL, VMA_SPECIAL_FLAGS Every user of the VM_SPECIAL or VMA_SPECIAL_FLAGS has now been converted to predicates which explicitly express what is actually being checked for rather than the nebulous concept of possessing 'special' VMA flags. In any case 'special' is not so special a term of art in mm - it includes VDSO/VVAR mappings, special in the sense of vm_normal_folio() and probably other cases too. Therefore make things less special by eliminating these now unused flags. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-37-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- include/linux/mm.h | 8 -------- tools/testing/vma/include/dup.h | 8 -------- 2 files changed, 16 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index eaf3a4110f91ed..448384fdb594d2 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -577,14 +577,6 @@ enum { #define VM_ACCESS_FLAGS (VM_READ | VM_WRITE | VM_EXEC) #define VMA_ACCESS_FLAGS mk_vma_flags(VMA_READ_BIT, VMA_WRITE_BIT, VMA_EXEC_BIT) -/* - * Special vmas that are non-mergable, non-mlock()able. - */ - -#define VMA_SPECIAL_FLAGS mk_vma_flags(VMA_IO_BIT, VMA_DONTEXPAND_BIT, \ - VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT) -#define VM_SPECIAL vma_flags_to_legacy(VMA_SPECIAL_FLAGS) - /* * Physically remapped pages are special. Tell the * rest of the world about it: diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index 97d3bf6cd5b712..c21f67decab58d 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -352,14 +352,6 @@ enum { #define VM_ACCESS_FLAGS (VM_READ | VM_WRITE | VM_EXEC) #define VMA_ACCESS_FLAGS mk_vma_flags(VMA_READ_BIT, VMA_WRITE_BIT, VMA_EXEC_BIT) -/* - * Special vmas that are non-mergable, non-mlock()able. - */ -#define VM_SPECIAL (VM_IO | VM_DONTEXPAND | VM_PFNMAP | VM_MIXEDMAP) - -#define VMA_SPECIAL_FLAGS mk_vma_flags(VMA_IO_BIT, VMA_DONTEXPAND_BIT, \ - VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT) - #define VMA_REMAP_FLAGS mk_vma_flags(VMA_IO_BIT, VMA_PFNMAP_BIT, \ VMA_DONTEXPAND_BIT, VMA_DONTDUMP_BIT) From 24e97a31abe69f158452b853502fcddb3b2460e0 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:47 +0100 Subject: [PATCH 1000/1352] fuse: dax: do not set VM_MIXEDMAP Commit e1fb4a086495 ("dax: remove VM_MIXEDMAP for fsdax and device dax") prevented fsdax and device-dax from setting VM_MIXEDMAP, as DAX no longer relies on it to direct core mm paths. The fuse DAX implementation, added later, copied the old pattern and still sets it. Fuse DAX maps pages the same way fsdax does, via dax_iomap_fault() and ultimately vmf_insert_page_mkwrite() and vmf_insert_folio_pmd(), which insert ordinary refcounted pages and so do not require VM_MIXEDMAP. Setting it only serves to mark the mapping as kernel-owned, making fuse DAX the sole DAX implementation whose mappings are unmergeable, cannot be mlock()'d, eagerly copy page tables on fork and reject MADV_DOFORK and MADV_DODUMP. It also requires vma_is_special_huge() in mm/huge_memory.c to carve DAX out of its kernel-owned check explicitly. There is no reason for fuse DAX to keep on using this flag so drop it. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-38-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- fs/fuse/dax.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/fuse/dax.c b/fs/fuse/dax.c index 85cdf0199bc0b8..a5994f1c637d92 100644 --- a/fs/fuse/dax.c +++ b/fs/fuse/dax.c @@ -826,7 +826,7 @@ int fuse_dax_mmap(struct file *file, struct vm_area_struct *vma) { file_accessed(file); vma->vm_ops = &fuse_dax_vm_ops; - vm_flags_set(vma, VM_MIXEDMAP | VM_HUGEPAGE); + vma_set_flags(vma, VMA_HUGEPAGE_BIT); return 0; } From 43f67f3fae19c40c889a1053cdaadd141d6e36a0 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:48 +0100 Subject: [PATCH 1001/1352] mm/huge_memory: remove vma_is_special_huge() vma_is_special_huge() tests whether either the VMA_PFNMAP_BIT or VMA_MIXEDMAP_BIT is set (i.e. whether the VMA is a kernel-owned mapping), but with a DAX carve-out. DAX however no longer sets VMA_MIXEDMAP_BIT, so this carve-out is no longer required. Therefore test for vma_is_kernel_owned() instead and also drop the VMA_IO_BIT check, as it is now redundant since it is enforced that only kernel-owned mappings can set this flag. This also eliminates another overloaded use of 'special' within mm. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-39-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/huge_memory.c | 18 ++++-------------- 1 file changed, 4 insertions(+), 14 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 05c17a01551df1..ec37a63b8a2ec7 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -110,14 +110,6 @@ static inline bool file_thp_enabled(const struct vm_area_struct *vma) return S_ISREG(inode->i_mode); } -/* If returns true, we are unable to access the VMA's folios. */ -static bool vma_is_special_huge(const struct vm_area_struct *vma) -{ - if (vma_is_dax(vma)) - return false; - return vma_test_any(vma, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT); -} - static bool vma_file_bypass_thp_tuneables(const struct vm_area_struct *vma, enum tva_type type) { @@ -192,7 +184,7 @@ unsigned long __thp_vma_allowable_orders(struct vm_area_struct *vma, /* Check the intersection of requested and supported orders. */ if (vma_is_anonymous(vma)) supported_orders = THP_ORDERS_ALL_ANON; - else if (vma_is_dax(vma) || vma_is_special_huge(vma)) + else if (vma_is_dax(vma) || vma_is_kernel_owned(vma)) supported_orders = THP_ORDERS_ALL_SPECIAL_DAX; else supported_orders = THP_ORDERS_ALL_FILE_DEFAULT; @@ -3065,7 +3057,7 @@ int zap_huge_pud(struct mmu_gather *tlb, struct vm_area_struct *vma, orig_pud = pudp_huge_get_and_clear_full(vma, addr, pud, tlb->fullmm); arch_check_zapped_pud(vma, orig_pud); tlb_remove_pud_tlb_entry(tlb, pud, addr); - if (vma_is_special_huge(vma)) { + if (vma_is_kernel_owned(vma)) { spin_unlock(ptl); /* No zero page support yet */ } else { @@ -3221,7 +3213,7 @@ static void __split_huge_pmd_locked(struct vm_area_struct *vma, pmd_t *pmd, */ if (arch_needs_pgtable_deposit()) zap_deposited_table(mm, pmd); - if (vma_is_special_huge(vma)) + if (vma_is_kernel_owned(vma)) return; if (unlikely(pmd_is_migration_entry(old_pmd))) { const softleaf_t old_entry = softleaf_from_pmd(old_pmd); @@ -4788,9 +4780,7 @@ static inline bool vma_not_suitable_for_thp_split(struct vm_area_struct *vma) { if (vma_is_dax(vma)) return true; - if (vma_is_special_huge(vma)) - return true; - if (vma_test(vma, VMA_IO_BIT)) + if (vma_is_kernel_owned(vma)) return true; if (vma_is_hugetlb(vma)) return true; From c60d5ae3aa3820d2dafaf45f8bfb032983f8a024 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:49 +0100 Subject: [PATCH 1002/1352] mm/vma: introduce and use vma[_flags]_can_gup() GUP cannot be used for VMAs which set VMA_IO_BIT - because memory-mapped I/O must not be accessed on the user's behalf - or VMA_PFNMAP_BIT - because PFN maps have no folios which the kernel is permitted to access. Rather than keeping these checks open-coded, abstract them to vma_flags_can_gup() and its VMA wrapper vma_can_gup(). A number of other places make the same check to decide whether a mapping can be populated or accessed as GUP would, so update those too. While here, drop a reference to 'special' and replace a use of the deprecated VMA flags API in vma_dump_size(). No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-40-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- fs/coredump.c | 4 ++-- include/linux/mm.h | 29 +++++++++++++++++++++++++++++ mm/gup.c | 7 +++---- mm/hmm.c | 3 +-- mm/memory.c | 14 ++++++++------ mm/mempolicy.c | 3 ++- 6 files changed, 45 insertions(+), 15 deletions(-) diff --git a/fs/coredump.c b/fs/coredump.c index 5820cb8ec88e70..4d03826dd24981 100644 --- a/fs/coredump.c +++ b/fs/coredump.c @@ -1616,8 +1616,8 @@ static unsigned long vma_dump_size(struct vm_area_struct *vma, return 0; } - /* Do not dump I/O mapped devices or special mappings */ - if (vma->vm_flags & VM_IO) + /* Do not dump memory-mapped I/O, which may have side effects on read. */ + if (vma_test(vma, VMA_IO_BIT)) return 0; /* By default, dump shared memory if mapped from an anonymous file. */ diff --git a/include/linux/mm.h b/include/linux/mm.h index 448384fdb594d2..4fd47cc796a6c1 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -1778,6 +1778,35 @@ static inline bool vma_is_persistent(const struct vm_area_struct *vma) return vma_flags_is_persistent(&vma->flags); } +/** + * vma_flags_can_gup() - Do the specified VMA flags permit GUP to access the + * mapping's pages? + * @flags: The VMA flags to test. + * + * GUP cannot obtain pages from a PFN map (VMA_PFNMAP_BIT), which may have no + * struct pages behind it, and must not provide access to memory-mapped I/O + * (VMA_IO_BIT). + * + * Returns: true if GUP may access pages from the mapping, otherwise false. + */ +static inline bool vma_flags_can_gup(const vma_flags_t *flags) +{ + return !vma_flags_test_any(flags, VMA_IO_BIT, VMA_PFNMAP_BIT); +} + +/** + * vma_can_gup() - May GUP obtain pages from @vma? + * @vma: The VMA to test. + * + * See vma_flags_can_gup() for details. + * + * Returns: true if GUP may access pages from the mapping, otherwise false. + */ +static inline bool vma_can_gup(const struct vm_area_struct *vma) +{ + return vma_flags_can_gup(&vma->flags); +} + /** * vma_kernel_pagesize - Default page size granularity for this VMA. * @vma: The user mapping. diff --git a/mm/gup.c b/mm/gup.c index f166acf794e308..8e9ef5ee7498c6 100644 --- a/mm/gup.c +++ b/mm/gup.c @@ -1204,7 +1204,7 @@ static int check_vma_flags(struct vm_area_struct *vma, unsigned long gup_flags) int foreign = (gup_flags & FOLL_REMOTE); bool vma_anon = vma_is_anonymous(vma); - if (vm_flags & (VM_IO | VM_PFNMAP)) + if (!vma_can_gup(vma)) return -EFAULT; if ((gup_flags & FOLL_ANON) && !vma_anon) @@ -1955,7 +1955,7 @@ int __mm_populate(unsigned long start, unsigned long len, int ignore_errors) * range with the first VMA. Also, skip undesirable VMA types. */ nend = min(end, vma->vm_end); - if (vma->vm_flags & (VM_IO | VM_PFNMAP)) + if (!vma_can_gup(vma)) continue; if (nstart < vma->vm_start) nstart = vma->vm_start; @@ -2017,8 +2017,7 @@ static long __get_user_pages_locked(struct mm_struct *mm, unsigned long start, break; /* protect what we can, including chardevs */ - if ((vma->vm_flags & (VM_IO | VM_PFNMAP)) || - !(vm_flags & vma->vm_flags)) + if (!vma_can_gup(vma) || !(vm_flags & vma->vm_flags)) break; if (pages) { diff --git a/mm/hmm.c b/mm/hmm.c index 2f1e98c6b6440b..e9569b82a1f0cc 100644 --- a/mm/hmm.c +++ b/mm/hmm.c @@ -595,8 +595,7 @@ static int hmm_vma_walk_test(unsigned long start, unsigned long end, struct hmm_range *range = hmm_vma_walk->range; struct vm_area_struct *vma = walk->vma; - if (!(vma->vm_flags & (VM_IO | VM_PFNMAP)) && - vma->vm_flags & VM_READ) + if (vma_can_gup(vma) && vma_test(vma, VMA_READ_BIT)) return 0; /* diff --git a/mm/memory.c b/mm/memory.c index 6c011979401aab..338fce99e71197 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -2417,11 +2417,11 @@ static bool vm_mixed_zeropage_allowed(struct vm_area_struct *vma) * be problematic as soon as the zeropage gets replaced by a different * page due to vma->vm_ops->pfn_mkwrite, because what's mapped would * now differ to what GUP looked up. FSDAX is incompatible to - * FOLL_LONGTERM and VM_IO is incompatible to GUP completely (see - * check_vma_flags). + * FOLL_LONGTERM and memory-mapped I/O is incompatible to GUP completely + * (see vma_can_gup()). */ return vma->vm_ops && vma->vm_ops->pfn_mkwrite && - (vma_is_fsdax(vma) || vma->vm_flags & VM_IO); + (vma_is_fsdax(vma) || vma_test(vma, VMA_IO_BIT)); } static int validate_page_before_insert(struct vm_area_struct *vma, @@ -7116,7 +7116,8 @@ int follow_pfnmap_start(struct follow_pfnmap_args *args) if (unlikely(address < vma->vm_start || address >= vma->vm_end)) goto out; - if (!(vma->vm_flags & (VM_IO | VM_PFNMAP))) + /* Only mappings GUP cannot handle are followed here. */ + if (vma_can_gup(vma)) goto out; retry: pgdp = pgd_offset(mm, address); @@ -7316,8 +7317,9 @@ static int __access_remote_vm(struct mm_struct *mm, unsigned long addr, } /* - * Check if this is a VM_IO | VM_PFNMAP VMA, which - * we can access using slightly different code. + * GUP failed, perhaps because this is a mapping it + * cannot handle (see vma_can_gup()) - such mappings may + * provide access via vm_ops->access() instead. */ bytes = 0; #ifdef CONFIG_HAVE_IOREMAP_PROT diff --git a/mm/mempolicy.c b/mm/mempolicy.c index 70298fded1b4a8..40744658483b27 100644 --- a/mm/mempolicy.c +++ b/mm/mempolicy.c @@ -2008,7 +2008,8 @@ SYSCALL_DEFINE5(get_mempolicy, int __user *, policy, bool vma_migratable(struct vm_area_struct *vma) { - if (vma->vm_flags & (VM_IO | VM_PFNMAP)) + /* Pages which GUP cannot obtain cannot be migrated either. */ + if (!vma_can_gup(vma)) return false; /* From 9433d8d71d7a8bdf243172550f61f7bdc627cd18 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Mon, 14 Sep 2026 07:23:18 -0700 Subject: [PATCH 1003/1352] mm/damon/sysfs-schemes: read sysfs_filter->addr_range only once Patch series "mm/damon: move damos filter range arguments validation to core". DAMOS filter range arguments are being validated in the DAMON sysfs interface. Some of those have minor time-of-check to time-of-use (TOCTOU) bugs. In future, other callers might have duplicated validations with similar bugs. Fix the bugs and further refactor the code to move the validation to the core layer. Also add kunit test cases for the validations. Patches 1 and 2 fix the existing TOCTOU bugs. Patches 3 and 4 adds the validation in the core layer. Patch 5 drops the replicated validation in DAMON sysfs interface. Patch 6 further cleanup the code. Patches 7 and 8 extends kunit tests to test the validation. This patch (of 8): DAMON sysfs interface reads the user-provided address range arguments for addr type DAMOS filter twice. Once for validation, and once again for assignments to the variable that will be passed to the core layer. If the user updates the argument in parallel, an invalid address range could be passed to the core layer. Avoid it by doing the assignments first, and then validating the assigned variables before passing those to the core layer. User impact of the bug should be trivial. From the core layer's perspective, the invalid address range is not really invalid. It just works as having a weird address range. No critical issues such as a crash or a leak could happen. And sane users ain't do such parallel arguments update anyway. If they do, such racy behavior is arguably somewhat expected and deserved. That said, there is no reason to keep such races. Link: https://lore.kernel.org/20260914142327.92510-1-sj@kernel.org Link: https://lore.kernel.org/20260914142327.92510-2-sj@kernel.org Fixes: 2f1abcfccd86 ("mm/damon/sysfs-schemes: support address range type DAMOS filter") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Usama Arif Cc: --- mm/damon/sysfs-schemes.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/mm/damon/sysfs-schemes.c b/mm/damon/sysfs-schemes.c index 3de4d804e049f8..3c1c1cb387fec2 100644 --- a/mm/damon/sysfs-schemes.c +++ b/mm/damon/sysfs-schemes.c @@ -2831,12 +2831,12 @@ static int damon_sysfs_add_scheme_filters(struct damos *scheme, return err; } } else if (filter->type == DAMOS_FILTER_TYPE_ADDR) { - if (sysfs_filter->addr_range.end < - sysfs_filter->addr_range.start) { + filter->addr_range = sysfs_filter->addr_range; + if (filter->addr_range.end < + filter->addr_range.start) { damos_destroy_filter(filter); return -EINVAL; } - filter->addr_range = sysfs_filter->addr_range; } else if (filter->type == DAMOS_FILTER_TYPE_TARGET) { filter->target_idx = sysfs_filter->target_idx; } else if (filter->type == DAMOS_FILTER_TYPE_HUGEPAGE_SIZE) { From 935c6cc324d4245eb837b03aa94013d8857172b6 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Mon, 14 Sep 2026 07:23:19 -0700 Subject: [PATCH 1004/1352] mm/damon/sysfs-schemes: read sysfs_filter->sz_range only once DAMON sysfs interface reads the user-provided size range arguments for hugepage_size type DAMOS filter twice. Once for validation, and once again for assignments to the variable that will be passed to the core layer. If the user updates the arguments in parallel, an invalid size range could be passed to the core layer. Avoid it by doing the assignments first, and then validating the assigned variables before passing those to the core layer. User impact of the bug should be trivial. From the core layer's perspective, the invalid size range is not really invalid. It just works as having a weird size range. No critical issues such as a crash or a leak could happen. And sane users ain't do such parallel arguments update anyway. If they do, such racy behavior is arguably somewhat expected and deserved. That said, there is no reason to keep such races. Link: https://lore.kernel.org/20260914142327.92510-3-sj@kernel.org Fixes: ea1f204ba29a ("mm/damon/sysfs-schemes: add files for setting damos_filter->sz_range") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Usama Arif Cc: --- mm/damon/sysfs-schemes.c | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/mm/damon/sysfs-schemes.c b/mm/damon/sysfs-schemes.c index 3c1c1cb387fec2..8c8ab82c8facc9 100644 --- a/mm/damon/sysfs-schemes.c +++ b/mm/damon/sysfs-schemes.c @@ -2840,13 +2840,12 @@ static int damon_sysfs_add_scheme_filters(struct damos *scheme, } else if (filter->type == DAMOS_FILTER_TYPE_TARGET) { filter->target_idx = sysfs_filter->target_idx; } else if (filter->type == DAMOS_FILTER_TYPE_HUGEPAGE_SIZE) { - if (sysfs_filter->range_min > - sysfs_filter->range_max) { + filter->sz_range.min = sysfs_filter->range_min; + filter->sz_range.max = sysfs_filter->range_max; + if (filter->range_min > filter->range_max) { damos_destroy_filter(filter); return -EINVAL; } - filter->sz_range.min = sysfs_filter->range_min; - filter->sz_range.max = sysfs_filter->range_max; } else if (filter->type == DAMOS_FILTER_TYPE_PROBE_HITS_WSUM) { filter->range_min = sysfs_filter->range_min; filter->range_max = sysfs_filter->range_max; From e8fb9d82cf8494de53ba7e96a232ac6460892bb0 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Mon, 14 Sep 2026 07:23:20 -0700 Subject: [PATCH 1005/1352] mm/damon/core: return an error from damos_commit_filter_arg() damos_commit_filter_arg() is supposed to always succeed. It may not in future, for example, if the given filter is invalid. Prepare the case by modifying its signature to return an error when it failed. Also pipe the return value to its callers and let them handle the error. Link: https://lore.kernel.org/20260914142327.92510-4-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: Usama Arif Cc: --- mm/damon/core.c | 41 ++++++++++++++++++++++++++++------------- 1 file changed, 28 insertions(+), 13 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 38383ced3dd6c0..1932a50252079b 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -1317,7 +1317,7 @@ static struct damos_filter *damos_nth_ops_filter(int n, struct damos *s) return NULL; } -static void damos_commit_filter_arg( +static int damos_commit_filter_arg( struct damos_filter *dst, struct damos_filter *src) { switch (dst->type) { @@ -1340,28 +1340,32 @@ static void damos_commit_filter_arg( default: break; } + return 0; } -static void damos_commit_filter( +static int damos_commit_filter( struct damos_filter *dst, struct damos_filter *src) { dst->type = src->type; dst->matching = src->matching; dst->allow = src->allow; - damos_commit_filter_arg(dst, src); + return damos_commit_filter_arg(dst, src); } static int damos_commit_core_filters(struct damos *dst, struct damos *src) { struct damos_filter *dst_filter, *next, *src_filter, *new_filter; - int i = 0, j = 0; + int i = 0, j = 0, err; damos_for_each_core_filter_safe(dst_filter, next, dst) { src_filter = damos_nth_core_filter(i++, src); - if (src_filter) - damos_commit_filter(dst_filter, src_filter); - else + if (src_filter) { + err = damos_commit_filter(dst_filter, src_filter); + if (err) + return err; + } else { damos_destroy_filter(dst_filter); + } } damos_for_each_core_filter_safe(src_filter, next, src) { @@ -1373,7 +1377,11 @@ static int damos_commit_core_filters(struct damos *dst, struct damos *src) src_filter->allow); if (!new_filter) return -ENOMEM; - damos_commit_filter_arg(new_filter, src_filter); + err = damos_commit_filter_arg(new_filter, src_filter); + if (err) { + damos_destroy_filter(new_filter); + return err; + } damos_add_filter(dst, new_filter); } return 0; @@ -1382,14 +1390,17 @@ static int damos_commit_core_filters(struct damos *dst, struct damos *src) static int damos_commit_ops_filters(struct damos *dst, struct damos *src) { struct damos_filter *dst_filter, *next, *src_filter, *new_filter; - int i = 0, j = 0; + int i = 0, j = 0, err; damos_for_each_ops_filter_safe(dst_filter, next, dst) { src_filter = damos_nth_ops_filter(i++, src); - if (src_filter) - damos_commit_filter(dst_filter, src_filter); - else + if (src_filter) { + err = damos_commit_filter(dst_filter, src_filter); + if (err) + return err; + } else { damos_destroy_filter(dst_filter); + } } damos_for_each_ops_filter_safe(src_filter, next, src) { @@ -1401,7 +1412,11 @@ static int damos_commit_ops_filters(struct damos *dst, struct damos *src) src_filter->allow); if (!new_filter) return -ENOMEM; - damos_commit_filter_arg(new_filter, src_filter); + err = damos_commit_filter_arg(new_filter, src_filter); + if (err) { + damos_destroy_filter(new_filter); + return err; + } damos_add_filter(dst, new_filter); } return 0; From 5142730a1522339df5d1a0ae45b0304dd64dfc26 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Mon, 14 Sep 2026 07:23:21 -0700 Subject: [PATCH 1006/1352] mm/damon/core: disallow max < min damos filter range arguments commit damos_commit_filter_arg() receives range arguments for a few types of DAMOS filters. It allows any range including max < min range. It is fine for the logic, but makes no sense to support it. Actually DAMON sysfs interface is doing the validation on its own. To avoid duplicated validations in multiple DAMON API callers, it would be better to do the validation in the core layer. Add a validation of the given range. Link: https://lore.kernel.org/20260914142327.92510-5-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Usama Arif Cc: --- mm/damon/core.c | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/mm/damon/core.c b/mm/damon/core.c index 1932a50252079b..5fdacb9dcee5f6 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -1325,15 +1325,21 @@ static int damos_commit_filter_arg( dst->memcg_id = src->memcg_id; break; case DAMOS_FILTER_TYPE_ADDR: + if (src->addr_range.end < src->addr_range.start) + return -EINVAL; dst->addr_range = src->addr_range; break; case DAMOS_FILTER_TYPE_TARGET: dst->target_idx = src->target_idx; break; case DAMOS_FILTER_TYPE_HUGEPAGE_SIZE: + if (src->sz_range.max < src->sz_range.min) + return -EINVAL; dst->sz_range = src->sz_range; break; case DAMOS_FILTER_TYPE_PROBE_HITS_WSUM: + if (src->range_max < src->range_min) + return -EINVAL; dst->range_min = src->range_min; dst->range_max = src->range_max; break; From 5862ad6c459bcced76117880df40a56665addcb6 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Mon, 14 Sep 2026 07:23:22 -0700 Subject: [PATCH 1007/1352] mm/damon/sysfs-schemes: drop centralized filter range arg validations DAMON sysfs interface is validating wrong range arguments for DAMOS filters. Now the core layer is doing the same validation. Drop the duplicated validation in DAMON sysfs interface. Link: https://lore.kernel.org/20260914142327.92510-6-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Usama Arif Cc: --- mm/damon/sysfs-schemes.c | 13 ------------- 1 file changed, 13 deletions(-) diff --git a/mm/damon/sysfs-schemes.c b/mm/damon/sysfs-schemes.c index 8c8ab82c8facc9..a4ec5d54cfbd14 100644 --- a/mm/damon/sysfs-schemes.c +++ b/mm/damon/sysfs-schemes.c @@ -2832,27 +2832,14 @@ static int damon_sysfs_add_scheme_filters(struct damos *scheme, } } else if (filter->type == DAMOS_FILTER_TYPE_ADDR) { filter->addr_range = sysfs_filter->addr_range; - if (filter->addr_range.end < - filter->addr_range.start) { - damos_destroy_filter(filter); - return -EINVAL; - } } else if (filter->type == DAMOS_FILTER_TYPE_TARGET) { filter->target_idx = sysfs_filter->target_idx; } else if (filter->type == DAMOS_FILTER_TYPE_HUGEPAGE_SIZE) { filter->sz_range.min = sysfs_filter->range_min; filter->sz_range.max = sysfs_filter->range_max; - if (filter->range_min > filter->range_max) { - damos_destroy_filter(filter); - return -EINVAL; - } } else if (filter->type == DAMOS_FILTER_TYPE_PROBE_HITS_WSUM) { filter->range_min = sysfs_filter->range_min; filter->range_max = sysfs_filter->range_max; - if (filter->range_min > filter->range_max) { - damos_destroy_filter(filter); - return -EINVAL; - } } damos_add_filter(scheme, filter); From 1b527a80bf41aa514bbd012611a30944952425d8 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Mon, 14 Sep 2026 07:23:23 -0700 Subject: [PATCH 1008/1352] mm/damon/sysfs-schemes: use switch-case in add_scheme_filters() damon_sysfs_add_scheeme_filters() has long if-else chains for DAMOS filter types. Convert the code to use switch-case, which would be cleaner and more efficient. Link: https://lore.kernel.org/20260914142327.92510-7-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Usama Arif Cc: --- mm/damon/sysfs-schemes.c | 18 +++++++++++++----- 1 file changed, 13 insertions(+), 5 deletions(-) diff --git a/mm/damon/sysfs-schemes.c b/mm/damon/sysfs-schemes.c index a4ec5d54cfbd14..bfb6f0bc3f2138 100644 --- a/mm/damon/sysfs-schemes.c +++ b/mm/damon/sysfs-schemes.c @@ -2822,7 +2822,8 @@ static int damon_sysfs_add_scheme_filters(struct damos *scheme, if (!filter) return -ENOMEM; - if (filter->type == DAMOS_FILTER_TYPE_MEMCG) { + switch (filter->type) { + case DAMOS_FILTER_TYPE_MEMCG: err = damon_sysfs_memcg_path_to_id( sysfs_filter->memcg_path, &filter->memcg_id); @@ -2830,16 +2831,23 @@ static int damon_sysfs_add_scheme_filters(struct damos *scheme, damos_destroy_filter(filter); return err; } - } else if (filter->type == DAMOS_FILTER_TYPE_ADDR) { + break; + case DAMOS_FILTER_TYPE_ADDR: filter->addr_range = sysfs_filter->addr_range; - } else if (filter->type == DAMOS_FILTER_TYPE_TARGET) { + break; + case DAMOS_FILTER_TYPE_TARGET: filter->target_idx = sysfs_filter->target_idx; - } else if (filter->type == DAMOS_FILTER_TYPE_HUGEPAGE_SIZE) { + break; + case DAMOS_FILTER_TYPE_HUGEPAGE_SIZE: filter->sz_range.min = sysfs_filter->range_min; filter->sz_range.max = sysfs_filter->range_max; - } else if (filter->type == DAMOS_FILTER_TYPE_PROBE_HITS_WSUM) { + break; + case DAMOS_FILTER_TYPE_PROBE_HITS_WSUM: filter->range_min = sysfs_filter->range_min; filter->range_max = sysfs_filter->range_max; + break; + default: + break; } damos_add_filter(scheme, filter); From 0e167006f3b187f88fd694e93cf423401e681205 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Mon, 14 Sep 2026 07:23:24 -0700 Subject: [PATCH 1009/1352] mm/damon/core-kunit: extend damos_commit_filter_for() for wrong input damos_commit_filter_for() supposes damos_commit_filter() to always succeed with given inputs. damos_commit_filter() could return an error for invalid inputs. Existing callers always pass only valid inputs, but they may pass invalid inputs in future, for test purposes. Extend the function to be able to be used for wrong inputs-caused error testing. Link: https://lore.kernel.org/20260914142327.92510-8-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Usama Arif Cc: --- mm/damon/tests/core-kunit.h | 24 +++++++++++++++--------- 1 file changed, 15 insertions(+), 9 deletions(-) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index c01e6a75cadc1e..2b0931cf6fb326 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -1116,9 +1116,15 @@ static void damos_test_commit_dests(struct kunit *test) } static void damos_test_commit_filter_for(struct kunit *test, - struct damos_filter *dst, struct damos_filter *src) + struct damos_filter *dst, struct damos_filter *src, + bool expect_fail) { - damos_commit_filter(dst, src); + int err; + + err = damos_commit_filter(dst, src); + KUNIT_EXPECT_EQ(test, err != 0, expect_fail); + if (expect_fail) + return; KUNIT_EXPECT_EQ(test, dst->type, src->type); KUNIT_EXPECT_EQ(test, dst->matching, src->matching); KUNIT_EXPECT_EQ(test, dst->allow, src->allow); @@ -1157,47 +1163,47 @@ static void damos_test_commit_filter(struct kunit *test) .type = DAMOS_FILTER_TYPE_ANON, .matching = true, .allow = true, - }); + }, false); damos_test_commit_filter_for(test, &dst, &(struct damos_filter){ .type = DAMOS_FILTER_TYPE_MEMCG, .matching = false, .allow = false, .memcg_id = 123, - }); + }, false); damos_test_commit_filter_for(test, &dst, &(struct damos_filter){ .type = DAMOS_FILTER_TYPE_YOUNG, .matching = true, .allow = true, - }); + }, false); damos_test_commit_filter_for(test, &dst, &(struct damos_filter){ .type = DAMOS_FILTER_TYPE_HUGEPAGE_SIZE, .matching = false, .allow = false, .sz_range = {.min = 234, .max = 345}, - }); + }, false); damos_test_commit_filter_for(test, &dst, &(struct damos_filter){ .type = DAMOS_FILTER_TYPE_UNMAPPED, .matching = true, .allow = true, - }); + }, false); damos_test_commit_filter_for(test, &dst, &(struct damos_filter){ .type = DAMOS_FILTER_TYPE_ADDR, .matching = false, .allow = false, .addr_range = {.start = 456, .end = 567}, - }); + }, false); damos_test_commit_filter_for(test, &dst, &(struct damos_filter){ .type = DAMOS_FILTER_TYPE_TARGET, .matching = true, .allow = true, .target_idx = 6, - }); + }, false); } static void damos_test_help_initailize_scheme(struct damos *scheme) From 08731674dd9f0cf628ecafe6d8acfab215ae15eb Mon Sep 17 00:00:00 2001 From: SJ Park Date: Mon, 14 Sep 2026 07:23:25 -0700 Subject: [PATCH 1010/1352] mm/damon/core-kunit: test invalid damos filter commits Add test cases for testing the validation of damos filter arguments in commit time. Link: https://lore.kernel.org/20260914142327.92510-9-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Usama Arif Cc: --- mm/damon/tests/core-kunit.h | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index 2b0931cf6fb326..527abc25706167 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -1184,6 +1184,13 @@ static void damos_test_commit_filter(struct kunit *test) .allow = false, .sz_range = {.min = 234, .max = 345}, }, false); + damos_test_commit_filter_for(test, &dst, + &(struct damos_filter){ + .type = DAMOS_FILTER_TYPE_HUGEPAGE_SIZE, + .matching = false, + .allow = false, + .sz_range = {.min = 456, .max = 123}, + }, true); damos_test_commit_filter_for(test, &dst, &(struct damos_filter){ .type = DAMOS_FILTER_TYPE_UNMAPPED, @@ -1197,6 +1204,13 @@ static void damos_test_commit_filter(struct kunit *test) .allow = false, .addr_range = {.start = 456, .end = 567}, }, false); + damos_test_commit_filter_for(test, &dst, + &(struct damos_filter){ + .type = DAMOS_FILTER_TYPE_ADDR, + .matching = false, + .allow = false, + .addr_range = {.start = 567, .end = 456}, + }, true); damos_test_commit_filter_for(test, &dst, &(struct damos_filter){ .type = DAMOS_FILTER_TYPE_TARGET, From d6753781f3009649970cd2856a1cca10f76ad46f Mon Sep 17 00:00:00 2001 From: "Zenghui Yu (Huawei)" Date: Mon, 14 Sep 2026 07:19:46 -0700 Subject: [PATCH 1011/1352] selftests/damon: stop kdamond on error exits of no-op commit test Patch series "mm/damon: misc improvements in tests and documents". Add various improvements to DAMON tests and documents. Patch 1 and 2 from Zenghui Yu (Huawei) let DAMON selftest to clean up its state and output after running the tests. Patch 3 from Eva Kurchatova makes a few selftest be able to run with Python's safe path mode. Patch 4 from Kunwu Chan improves coverage of DAMON kunit tests. Finally, patches 5 and 6 from Liew Rui Yan clarify and fix typos on documents for DAMOS quota behaviors. Below are notes that can be removed in the final commit message. This is a batched repost of DAMON patches that were individually posted. Please refer to each patch for changelog. This patch (of 6): The sysfs_no_op_commit_break test starts a kdamond via sysfs, but its error paths (e.g., drgn not installed) exit without stopping it. The leaked kdamond then makes subsequent tests, e.g. lru_sort.sh and reclaim.sh, skip with "Another kdamond is running". Wrap the post-start logic in try-finally so that kdamonds.stop() is executed on every exit path. Link: https://lore.kernel.org/20260914141952.91465-1-sj@kernel.org Link: https://lore.kernel.org/20260914141952.91465-2-sj@kernel.org Fixes: 10725cd2b09a ("selftests/damon: test no-op commit broke DAMON status") Signed-off-by: Zenghui Yu (Huawei) Reviewed-by: SJ Park Signed-off-by: SJ Park Signed-off-by: Andrew Morton Assisted-by: GLM-5.3 OpenCode Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Eva Kurchatova Cc: Kunwu Chan Cc: Lian Wang Cc: Liew Rui Yan --- .../damon/sysfs_no_op_commit_break.py | 35 ++++++++++--------- 1 file changed, 18 insertions(+), 17 deletions(-) diff --git a/tools/testing/selftests/damon/sysfs_no_op_commit_break.py b/tools/testing/selftests/damon/sysfs_no_op_commit_break.py index 2c65cffe6b5450..81774ddc55ac45 100755 --- a/tools/testing/selftests/damon/sysfs_no_op_commit_break.py +++ b/tools/testing/selftests/damon/sysfs_no_op_commit_break.py @@ -47,26 +47,27 @@ def main(): print('kdamond start failed: %s' % err) exit(1) - before_commit_status, err = \ - dump_damon_status_dict(kdamonds.kdamonds[0].pid) - if err is not None: - print('before-commit status dump failed: %s' % err) - exit(1) + try: + before_commit_status, err = \ + dump_damon_status_dict(kdamonds.kdamonds[0].pid) + if err is not None: + print('before-commit status dump failed: %s' % err) + exit(1) - kdamonds.kdamonds[0].commit() + kdamonds.kdamonds[0].commit() - after_commit_status, err = \ - dump_damon_status_dict(kdamonds.kdamonds[0].pid) - if err is not None: - print('after-commit status dump failed: %s' % err) - exit(1) - - if before_commit_status != after_commit_status: - print(f'before: {json.dumps(before_commit_status, indent=2)}') - print(f'after: {json.dumps(after_commit_status, indent=2)}') - exit(1) + after_commit_status, err = \ + dump_damon_status_dict(kdamonds.kdamonds[0].pid) + if err is not None: + print('after-commit status dump failed: %s' % err) + exit(1) - kdamonds.stop() + if before_commit_status != after_commit_status: + print(f'before: {json.dumps(before_commit_status, indent=2)}') + print(f'after: {json.dumps(after_commit_status, indent=2)}') + exit(1) + finally: + kdamonds.stop() if __name__ == '__main__': main() From dad020171da7bd7fb66590d682d35bed432e4e73 Mon Sep 17 00:00:00 2001 From: "Zenghui Yu (Huawei)" Date: Mon, 14 Sep 2026 07:19:47 -0700 Subject: [PATCH 1012/1352] selftests/damon: ignore test-generated damon_dump_output sysfs.py and sysfs_no_op_commit_break.py dump the DAMON status collected via drgn_dump_damon_status.py into a damon_dump_output file in the working directory. Ignore it so that running the tests in-tree does not pollute git status. Link: https://lore.kernel.org/20260914141952.91465-3-sj@kernel.org Signed-off-by: Zenghui Yu (Huawei) Reviewed-by: SJ Park Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Eva Kurchatova Cc: Jonathan Corbet Cc: Kunwu Chan Cc: Liam R. Howlett Cc: Lian Wang Cc: Liew Rui Yan Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/damon/.gitignore | 1 + 1 file changed, 1 insertion(+) diff --git a/tools/testing/selftests/damon/.gitignore b/tools/testing/selftests/damon/.gitignore index 2f0297657c8167..226bd7c4e07286 100644 --- a/tools/testing/selftests/damon/.gitignore +++ b/tools/testing/selftests/damon/.gitignore @@ -1,3 +1,4 @@ # SPDX-License-Identifier: GPL-2.0-only access_memory access_memory_even +damon_dump_output From c9527e2168824f7e91b4f41850a75c5adc5d66de Mon Sep 17 00:00:00 2001 From: Eva Kurchatova Date: Mon, 14 Sep 2026 07:19:48 -0700 Subject: [PATCH 1013/1352] selftests/damon: add script dir to sys.path for PYTHONSAFEPATH compatibility Running these tests under Python's safe path mode, either with -P on the interpreter command line or with PYTHONSAFEPATH set in the environment, stops the script's own directory being prepended to sys.path. Some distributions (RHEL, for example) build the tests with -P in the shebang. This breaks all 7 DAMON Python selftests that import the _damon_sysfs helper module located in the same directory: ModuleNotFoundError: No module named '_damon_sysfs' Fix this by explicitly adding the script's directory to sys.path before importing _damon_sysfs, following the same pattern used in commit c3b3eb565bd7 ("tools: ynl: add script dir to sys.path") which fixed the identical issue for the YNL tools. Link: https://lore.kernel.org/20260914141952.91465-4-sj@kernel.org Signed-off-by: Eva Kurchatova Reviewed-by: SJ Park Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Kunwu Chan Cc: Liam R. Howlett Cc: Lian Wang Cc: Liew Rui Yan Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: "Zenghui Yu (Huawei)" --- tools/testing/selftests/damon/damon_nr_regions.py | 3 +++ tools/testing/selftests/damon/damos_apply_interval.py | 3 +++ tools/testing/selftests/damon/damos_quota.py | 3 +++ tools/testing/selftests/damon/damos_quota_goal.py | 3 +++ tools/testing/selftests/damon/damos_tried_regions.py | 3 +++ .../selftests/damon/sysfs_update_schemes_tried_regions_hang.py | 3 +++ .../damon/sysfs_update_schemes_tried_regions_wss_estimation.py | 3 +++ 7 files changed, 21 insertions(+) diff --git a/tools/testing/selftests/damon/damon_nr_regions.py b/tools/testing/selftests/damon/damon_nr_regions.py index 58f3291fed12a4..e55239813c654e 100755 --- a/tools/testing/selftests/damon/damon_nr_regions.py +++ b/tools/testing/selftests/damon/damon_nr_regions.py @@ -1,9 +1,12 @@ #!/usr/bin/env python3 # SPDX-License-Identifier: GPL-2.0 +import os import subprocess +import sys import time +sys.path.append(os.path.dirname(os.path.abspath(__file__))) import _damon_sysfs def test_nr_regions(real_nr_regions, min_nr_regions, max_nr_regions): diff --git a/tools/testing/selftests/damon/damos_apply_interval.py b/tools/testing/selftests/damon/damos_apply_interval.py index 0f2f36584e48cb..0bf7768b2006ac 100755 --- a/tools/testing/selftests/damon/damos_apply_interval.py +++ b/tools/testing/selftests/damon/damos_apply_interval.py @@ -1,9 +1,12 @@ #!/usr/bin/env python3 # SPDX-License-Identifier: GPL-2.0 +import os import subprocess +import sys import time +sys.path.append(os.path.dirname(os.path.abspath(__file__))) import _damon_sysfs def main(): diff --git a/tools/testing/selftests/damon/damos_quota.py b/tools/testing/selftests/damon/damos_quota.py index 57c4937aaed285..879115a499bf8f 100755 --- a/tools/testing/selftests/damon/damos_quota.py +++ b/tools/testing/selftests/damon/damos_quota.py @@ -1,9 +1,12 @@ #!/usr/bin/env python3 # SPDX-License-Identifier: GPL-2.0 +import os import subprocess +import sys import time +sys.path.append(os.path.dirname(os.path.abspath(__file__))) import _damon_sysfs def main(): diff --git a/tools/testing/selftests/damon/damos_quota_goal.py b/tools/testing/selftests/damon/damos_quota_goal.py index 661e4ba4765ae1..fed033a0afdf9c 100755 --- a/tools/testing/selftests/damon/damos_quota_goal.py +++ b/tools/testing/selftests/damon/damos_quota_goal.py @@ -1,9 +1,12 @@ #!/usr/bin/env python3 # SPDX-License-Identifier: GPL-2.0 +import os import subprocess +import sys import time +sys.path.append(os.path.dirname(os.path.abspath(__file__))) import _damon_sysfs def main(): diff --git a/tools/testing/selftests/damon/damos_tried_regions.py b/tools/testing/selftests/damon/damos_tried_regions.py index d6472e6a6e082b..6941f87c10b1ce 100755 --- a/tools/testing/selftests/damon/damos_tried_regions.py +++ b/tools/testing/selftests/damon/damos_tried_regions.py @@ -1,9 +1,12 @@ #!/usr/bin/env python3 # SPDX-License-Identifier: GPL-2.0 +import os import subprocess +import sys import time +sys.path.append(os.path.dirname(os.path.abspath(__file__))) import _damon_sysfs def main(): diff --git a/tools/testing/selftests/damon/sysfs_update_schemes_tried_regions_hang.py b/tools/testing/selftests/damon/sysfs_update_schemes_tried_regions_hang.py index 28c887a0108fde..625761c243b58b 100755 --- a/tools/testing/selftests/damon/sysfs_update_schemes_tried_regions_hang.py +++ b/tools/testing/selftests/damon/sysfs_update_schemes_tried_regions_hang.py @@ -1,9 +1,12 @@ #!/usr/bin/env python3 # SPDX-License-Identifier: GPL-2.0 +import os import subprocess +import sys import time +sys.path.append(os.path.dirname(os.path.abspath(__file__))) import _damon_sysfs def main(): diff --git a/tools/testing/selftests/damon/sysfs_update_schemes_tried_regions_wss_estimation.py b/tools/testing/selftests/damon/sysfs_update_schemes_tried_regions_wss_estimation.py index 16fdc6e7fc566a..36e7ae5f826d82 100755 --- a/tools/testing/selftests/damon/sysfs_update_schemes_tried_regions_wss_estimation.py +++ b/tools/testing/selftests/damon/sysfs_update_schemes_tried_regions_wss_estimation.py @@ -1,9 +1,12 @@ #!/usr/bin/env python3 # SPDX-License-Identifier: GPL-2.0 +import os import subprocess +import sys import time +sys.path.append(os.path.dirname(os.path.abspath(__file__))) import _damon_sysfs def pass_wss_estimation(sz_region): From 0f73dc9133eda87b1e4ed5d9523235f35ad54611 Mon Sep 17 00:00:00 2001 From: Kunwu Chan Date: Mon, 14 Sep 2026 07:19:49 -0700 Subject: [PATCH 1014/1352] mm/damon/tests/core-kunit: improve nr_samples_per_aggr test isolation Test the zero sample interval case with a non-zero aggregation interval, and keep the zero/zero case to cover the zero-result fallback. Also make the overflow case use an explicit sample interval so that it does not depend on the zero sample interval fallback. Link: https://lore.kernel.org/20260914141952.91465-5-sj@kernel.org Signed-off-by: Kunwu Chan Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Lian Wang Reviewed-by: SJ Park Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Eva Kurchatova Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Liew Rui Yan Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: "Zenghui Yu (Huawei)" --- mm/damon/tests/core-kunit.h | 19 +++++++++++++++---- 1 file changed, 15 insertions(+), 4 deletions(-) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index 527abc25706167..5da84caf4124d5 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -627,12 +627,20 @@ static void damon_test_set_regions(struct kunit *test) static void damon_test_nr_samples_per_aggr(struct kunit *test) { - struct damon_attrs attrs = { + struct damon_attrs attrs; + + /* Zero sample interval is treated as one. */ + attrs = (struct damon_attrs){ .sample_interval = 0, - .aggr_interval = 0, + .aggr_interval = 5000, }; + KUNIT_EXPECT_EQ(test, damon_nr_samples_per_aggr(&attrs), 5000); - /* Zero aggregation interval doesn't cause division by zero */ + /* Zero sample and aggregation intervals cover the zero-result fallback. */ + attrs = (struct damon_attrs){ + .sample_interval = 0, + .aggr_interval = 0, + }; KUNIT_EXPECT_EQ(test, damon_nr_samples_per_aggr(&attrs), 1); /* @@ -640,7 +648,10 @@ static void damon_test_nr_samples_per_aggr(struct kunit *test) * overflow */ if (ULONG_MAX > UINT_MAX) { - attrs.aggr_interval = (unsigned long)UINT_MAX + 1; + attrs = (struct damon_attrs){ + .sample_interval = 1, + .aggr_interval = (unsigned long)UINT_MAX + 1, + }; KUNIT_EXPECT_EQ(test, damon_nr_samples_per_aggr(&attrs), UINT_MAX); } From 61909a38aa861a4ee1c937cd34337cda41b5604e Mon Sep 17 00:00:00 2001 From: Liew Rui Yan Date: Mon, 14 Sep 2026 07:19:50 -0700 Subject: [PATCH 1015/1352] Docs/mm/damon/design: clarify when qt_exceeds increases qt_exceeds counts how many times a scheme's quota has been exceeded. When using the temporal auto-tuning algorithm, the effective size quota becomes zero once the goal is [over-]achieved. In this case, the quotas are still set, so qt_exceeds keeps increasing once per quota reset interval while the goal stays achieved. Clarify this behavior in the design document. Link: https://lore.kernel.org/20260914141952.91465-6-sj@kernel.org Signed-off-by: Liew Rui Yan Reviewed-by: SJ Park Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Eva Kurchatova Cc: Jonathan Corbet Cc: Kunwu Chan Cc: Liam R. Howlett Cc: Lian Wang Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: "Zenghui Yu (Huawei)" --- Documentation/mm/damon/design.rst | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index cb116a82ff20b4..ad09416dfb21a4 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -696,7 +696,9 @@ There are two such tuning algorithms that users can select as they need. fast as possible, using maximum allowed quota, but only for a temporal short time. When the quota is under-achieved, this algorithm keeps tuning quota to a maximum allowed one. Once the quota is [over]-achieved, this sets the - quota zero. Useful for deterministic control required environments. + quota zero. Useful for deterministic control required environments. Note + that the zero quota is a valid quota, and therefore ``qt_exceeds`` :ref:`stat + ` will keep increasing in this case. The goal can be specified with five parameters, namely ``target_metric``, ``target_value``, ``current_value``, ``nid`` and ``path``. The auto-tuning From 84fcbc3c5a9329b547ec2b4e15023bc52dea7d03 Mon Sep 17 00:00:00 2001 From: Liew Rui Yan Date: Mon, 14 Sep 2026 07:19:51 -0700 Subject: [PATCH 1016/1352] Docs/mm/damon/design: fix typos in temporal auto-tuning algorithm section Correct incorrect references of "quota" to "goal" in the temporal auto-tuning algorithm description and fix the square bracket formatting. Link: https://lore.kernel.org/20260914141952.91465-7-sj@kernel.org Signed-off-by: Liew Rui Yan Reviewed-by: SJ Park Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Eva Kurchatova Cc: Jonathan Corbet Cc: Kunwu Chan Cc: Liam R. Howlett Cc: Lian Wang Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: "Zenghui Yu (Huawei)" --- Documentation/mm/damon/design.rst | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index ad09416dfb21a4..707170bcd1b33c 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -694,8 +694,8 @@ There are two such tuning algorithms that users can select as they need. This is the default selection. If unsure, use this. - ``temporal``: More straightforward algorithm. Tries to achieve the goal as fast as possible, using maximum allowed quota, but only for a temporal short - time. When the quota is under-achieved, this algorithm keeps tuning quota to - a maximum allowed one. Once the quota is [over]-achieved, this sets the + time. When the goal is under-achieved, this algorithm keeps tuning quota to + a maximum allowed one. Once the goal is [over-]achieved, this sets the quota zero. Useful for deterministic control required environments. Note that the zero quota is a valid quota, and therefore ``qt_exceeds`` :ref:`stat ` will keep increasing in this case. From c7258f5e9d0229e0f0861511142a0499e35c4bd1 Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Sat, 12 Sep 2026 08:28:22 -0400 Subject: [PATCH 1017/1352] proc/task_mmu: handle special PMDs in clear_refs and pagemap mshv_vtl_low can install a special PMD from a PFN supplied in the mmap file offset. That PFN need not have a memmap entry. Reading /proc/PID/pagemap or writing /proc/PID/clear_refs for such a mapping oopses the kernel. With a test driver mapping the PFN at the 1 TiB mark: echo 1 > /proc/self/clear_refs BUG: unable to handle page fault for address: fffffb5b80000008 #PF: supervisor read access in kernel mode #PF: error_code(0x0000) - not-present page RIP: 0010:clear_refs_pte_range+0xb1/0x1e0 Call Trace: walk_pgd_range+0x50a/0xad0 __walk_page_range+0x6a/0x1d0 walk_page_range_mm_unsafe+0x193/0x230 clear_refs_write+0x18e/0x3f0 pread(pagemap_fd, buf, 512 * 8, addr / PAGE_SIZE * 8) BUG: unable to handle page fault for address: fffff57840000008 RIP: 0010:pagemap_pmd_range+0x3cf/0x6b0 Call Trace: walk_pgd_range+0x50a/0xad0 __walk_page_range+0x6a/0x1d0 walk_page_range_mm_unsafe+0x193/0x230 pagemap_read+0x1dc/0x350 clear_refs_pte_range() passes the PMD to pmd_folio(), while pagemap_pmd_range_thp() uses pmd_page() followed by page_folio(). Both paths assume the PFN has a struct page and dereference bad address. Use vm_normal_folio_pmd() and vm_normal_page_pmd() so clear_refs skips folio operations and pagemap skips folio-derived flags for special PMDs. This matches the PTE paths, which already use vm_normal_folio() and vm_normal_page(). User-facing notes: - Pagemap still reports PM_PRESENT and the PFN to a CAP_SYS_ADMIN reader. - For a special PMD backed by a valid memmap entry and for the huge zero PMD, PM_FILE changes from set to clear. - PM_MMAP_EXCLUSIVE is already clear in both cases. - The output for an anonymous THP is unchanged. Tested in QEMU with the test driver on broken and fixed kernels. Both proc operations oops before the fix and return 0 after it. This is technically reachable via mshv_vtl_low, so I tagged it stable. Link: https://lore.kernel.org/20260912122822.3348978-1-gourry@gourry.net Fixes: 3c8e44c9b369 ("mm: mark special bits for huge pfn mappings when inject") Signed-off-by: Gregory Price Signed-off-by: Andrew Morton Reported-by: sashiko-bot Closes: https://sashiko.dev/#/patchset/20260912034833.2952750-1-gourry%40gourry.net Acked-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Jann Horn Cc: Jason Gunthorpe Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Pedro Falcato Cc: Peter Xu Cc: Vlastimil Babka Cc: # v6.19+ --- fs/proc/task_mmu.c | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c index 565e6446bd3127..052e8dc796bcf8 100644 --- a/fs/proc/task_mmu.c +++ b/fs/proc/task_mmu.c @@ -1704,7 +1704,9 @@ static int clear_refs_pte_range(pmd_t *pmd, unsigned long addr, if (!pmd_present(*pmd)) goto out; - folio = pmd_folio(*pmd); + folio = vm_normal_folio_pmd(vma, addr, *pmd); + if (!folio) + goto out; /* Clear accessed and referenced bits. */ pmdp_test_and_clear_young(vma, addr, pmd); @@ -2024,7 +2026,7 @@ static int pagemap_pmd_range_thp(pmd_t *pmdp, unsigned long addr, goto populate_pagemap; if (pmd_present(pmd)) { - page = pmd_page(pmd); + page = vm_normal_page_pmd(vma, addr, pmd); flags |= PM_PRESENT; if (pmd_soft_dirty(pmd)) From c453306925075c91a278d007a58fdbeac1a367cf Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Tue, 15 Sep 2026 10:43:30 +0800 Subject: [PATCH 1018/1352] mm/vmalloc: group xa_init with vbq field initializations Patch series "mm/vmalloc: minor cleanups", v2. Small cleanup series for mm/vmalloc.c, no functional changes: - Group xa_init() with the other vmap_block_queue field initializations in vmalloc_init(), instead of after the unrelated vfree_deferred setup. - Extract vmap_insert_free_area() helper to deduplicate the allocate- and-insert pattern that appeared both inside the loop body and after the loop in vmap_init_free_space(). - Extract show_busy_info() from vmalloc_info_show(), mirroring the existing show_purge_info() pattern, so the top-level show function only orchestrates the two data sources. This patch (of 3): Move xa_init() next to the other vbq field initializations instead of after the unrelated vfree_deferred setup. Link: https://lore.kernel.org/20260915-vmalloc_study-v2-0-cc4dfe635e22@linux.dev Link: https://lore.kernel.org/20260915-vmalloc_study-v2-1-cc4dfe635e22@linux.dev Signed-off-by: Ye Liu Signed-off-by: Andrew Morton Reviewed-by: Uladzislau Rezki (Sony) --- mm/vmalloc.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index aed70e4f8e4e4e..ca4aa340fc3c71 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -5574,10 +5574,11 @@ void __init vmalloc_init(void) vbq = &per_cpu(vmap_block_queue, i); spin_lock_init(&vbq->lock); INIT_LIST_HEAD(&vbq->free); + xa_init(&vbq->vmap_blocks); + p = &per_cpu(vfree_deferred, i); init_llist_head(&p->list); INIT_WORK(&p->wq, delayed_vfree_work); - xa_init(&vbq->vmap_blocks); } /* From e58f49962436d721454981e3012a2014a2a051f8 Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Tue, 15 Sep 2026 10:43:31 +0800 Subject: [PATCH 1019/1352] mm/vmalloc: extract vmap_insert_free_area helper The allocation and insertion of a free vmap_area is duplicated between the loop body and the tail of vmap_init_free_space. Factor it into a small helper so the main function only deals with computing the free gaps between busy regions. Link: https://lore.kernel.org/20260915-vmalloc_study-v2-2-cc4dfe635e22@linux.dev Signed-off-by: Ye Liu Signed-off-by: Andrew Morton Reviewed-by: Uladzislau Rezki (Sony) Reviewed-by: Dev Jain --- mm/vmalloc.c | 41 ++++++++++++++++++----------------------- 1 file changed, 18 insertions(+), 23 deletions(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index ca4aa340fc3c71..f0a1cc07c337b5 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -5432,11 +5432,23 @@ module_init(proc_vmalloc_init); #endif +static void __init vmap_insert_free_area(unsigned long start, unsigned long end) +{ + struct vmap_area *free = kmem_cache_zalloc(vmap_area_cachep, GFP_NOWAIT); + + if (!WARN_ON_ONCE(!free)) { + free->va_start = start; + free->va_end = end; + insert_vmap_area_augment(free, NULL, + &free_vmap_area_root, + &free_vmap_area_list); + } +} + static void __init vmap_init_free_space(void) { unsigned long vmap_start = 1; const unsigned long vmap_end = ULONG_MAX; - struct vmap_area *free; struct vm_struct *busy; /* @@ -5446,32 +5458,15 @@ static void __init vmap_init_free_space(void) * |<--------------------------------->| */ for (busy = vmlist; busy; busy = busy->next) { - if ((unsigned long) busy->addr - vmap_start > 0) { - free = kmem_cache_zalloc(vmap_area_cachep, GFP_NOWAIT); - if (!WARN_ON_ONCE(!free)) { - free->va_start = vmap_start; - free->va_end = (unsigned long) busy->addr; - - insert_vmap_area_augment(free, NULL, - &free_vmap_area_root, - &free_vmap_area_list); - } - } + if ((unsigned long) busy->addr - vmap_start > 0) + vmap_insert_free_area(vmap_start, + (unsigned long) busy->addr); vmap_start = (unsigned long) busy->addr + busy->size; } - if (vmap_end - vmap_start > 0) { - free = kmem_cache_zalloc(vmap_area_cachep, GFP_NOWAIT); - if (!WARN_ON_ONCE(!free)) { - free->va_start = vmap_start; - free->va_end = vmap_end; - - insert_vmap_area_augment(free, NULL, - &free_vmap_area_root, - &free_vmap_area_list); - } - } + if (vmap_end - vmap_start > 0) + vmap_insert_free_area(vmap_start, vmap_end); } static void vmap_init_nodes(void) From 6f3eb03d8343456bda0b88d36677015743f6d723 Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Tue, 15 Sep 2026 10:43:32 +0800 Subject: [PATCH 1020/1352] mm/vmalloc: extract show_busy_info from vmalloc_info_show Extract the busy vmap area iteration into show_busy_info, mirroring the existing show_purge_info pattern. Link: https://lore.kernel.org/20260915-vmalloc_study-v2-3-cc4dfe635e22@linux.dev Signed-off-by: Ye Liu Signed-off-by: Andrew Morton Reviewed-by: Uladzislau Rezki (Sony) --- mm/vmalloc.c | 12 +++++++++--- 1 file changed, 9 insertions(+), 3 deletions(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index f0a1cc07c337b5..2d29b08f263ad3 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -5344,7 +5344,7 @@ static void show_purge_info(struct seq_file *m) } } -static int vmalloc_info_show(struct seq_file *m, void *p) +static void show_busy_info(struct seq_file *m) { struct vmap_node *vn; struct vmap_area *va; @@ -5414,12 +5414,18 @@ static int vmalloc_info_show(struct seq_file *m, void *p) spin_unlock(&vn->busy.lock); } + if (IS_ENABLED(CONFIG_NUMA)) + kfree(counters); +} + +static int vmalloc_info_show(struct seq_file *m, void *p) +{ + show_busy_info(m); + /* * As a final step, dump "unpurged" areas. */ show_purge_info(m); - if (IS_ENABLED(CONFIG_NUMA)) - kfree(counters); return 0; } From 901e33843c34bcdb0ade253291123e362b416e7c Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Sat, 12 Sep 2026 13:48:59 +0200 Subject: [PATCH 1021/1352] mm/secretmem: fix the enable parameter description The module parameter is enable, but its MODULE_PARM_DESC() names the variable behind it, secretmem_enable. secretmem is built in, so the description only reaches modules.builtin.modinfo, where it is filed under secretmem_enable while the parmtype line and /sys/module/secretmem/parameters/ say enable. Use the parameter name in the description. Link: https://lore.kernel.org/20260912114859.88957-1-kmehltretter@gmail.com Fixes: 1507f51255c9 ("mm: introduce memfd_secret system call to create "secret" memory areas") Signed-off-by: Karl Mehltretter Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Assisted-by: LLM --- mm/secretmem.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/secretmem.c b/mm/secretmem.c index 6cbb8efc994a4d..7287a2866897e4 100644 --- a/mm/secretmem.c +++ b/mm/secretmem.c @@ -39,7 +39,7 @@ static bool secretmem_enable __ro_after_init = 1; module_param_named(enable, secretmem_enable, bool, 0400); -MODULE_PARM_DESC(secretmem_enable, +MODULE_PARM_DESC(enable, "Enable secretmem and memfd_secret(2) system call"); static atomic_t secretmem_users; From b8874f7943050c91f5ac9091ed557dbfdc98d5ff Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Sat, 12 Sep 2026 07:08:32 -0400 Subject: [PATCH 1022/1352] mm/madvise: reclaim isolated folios if PTE restart fails MADV_PAGEOUT collects isolated folios on a local list before reclaiming them after the PTE walk. The reschedule path drops the PTE lock and then restarts the mapping with pte_offset_map_lock(). A concurrent operation can remove or replace the PTE table while the lock is dropped, causing pte_offset_map_lock() to return NULL. Returning directly in that case bypasses reclaim_pages(), leaving the collected folios off the LRU with elevated references. This results in a permanent memory leak. nr_isolated_anon increases reliably and does not decrease when the process dies. Route the failure through the existing cleanup path so any isolated folios are reclaimed or put back. Simplest userland pseudo-code reproducer: p = mmap(PMD_SIZE, ANONYMOUS); touch_every_page(p, PMD_SIZE); parallel { while (1) madvise(p, PMD_SIZE, MADV_PAGEOUT); while (1) { madvise(p, PMD_SIZE, MADV_DONTNEED); touch_every_page(p, PMD_SIZE); } } Reproduced in qemu trivially with some explicit widening of the race window. Link: https://lore.kernel.org/20260912110832.3203902-1-gourry@gourry.net Fixes: b2f557a21bc8 ("mm/madvise: add cond_resched() in madvise_cold_or_pageout_pte_range()") Signed-off-by: Gregory Price (Meta) Signed-off-by: Andrew Morton Reported-by: sashiko-bot Closes: https://sashiko.dev/#/patchset/20260821150912.183976-1-gourry@gourry.net Reviewed-by: Lorenzo Stoakes (ARM) Assisted-by: LLM Cc: David Hildenbrand Cc: Jann Horn Cc: Jiexun Wang Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: --- mm/madvise.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/mm/madvise.c b/mm/madvise.c index 80ea991ce350a3..8cd08fc153e6d6 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -463,7 +463,7 @@ static int madvise_cold_or_pageout_pte_range(pmd_t *pmd, restart: start_pte = pte = pte_offset_map_lock(vma->vm_mm, pmd, addr, &ptl); if (!start_pte) - return 0; + goto out; flush_tlb_batched_pending(mm); lazy_mmu_mode_enable(); for (; addr < end; pte += nr, addr += nr * PAGE_SIZE) { @@ -567,6 +567,7 @@ static int madvise_cold_or_pageout_pte_range(pmd_t *pmd, folio_deactivate(folio); } +out: if (start_pte) { lazy_mmu_mode_disable(); pte_unmap_unlock(start_pte, ptl); From 721bf2f00521c5f21c8dc4d2917eac8a31a3d8eb Mon Sep 17 00:00:00 2001 From: "Harry Yoo (Meta)" Date: Mon, 14 Sep 2026 21:36:39 +0100 Subject: [PATCH 1023/1352] selftests/cgroup: ignore memory.reclaim -EAGAIN for zswap writeback test MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The zswap_writeback_enabled test fails when a write to memory.reclaim returns -EAGAIN, which means less than the requested amount was reclaimed. attempt_writeback() propagates the -EAGAIN to the caller, and the test case is marked as failed even when zswap writeback did happen. This heavily depends on the performance of the backing swap device. Reclaim does not wait for writeback (on cgroup v2), does not count pages that are under writeback as reclaimed, and memory.reclaim gives up after MAX_RECLAIM_RETRIES passes without making progress. On a slow device where reclaim does not make any progress before writeback completes, a write to memory.reclaim fails. On a VM with zswap enabled, where IO delay was injected via dm-delay, the success rate of the zswap writeback test drops dramatically once the delay reaches 11 ms: 7% failures at 10 ms and 79% failures at 11 ms, n = 100. When zswap writeback is enabled, ignore -EAGAIN from memory.reclaim and determine pass/fail based on the zswpwb counter because that is what zswap_writeback_enabled actually wants to test. With this change, the test reliably passes even on a slow swap device (tested up to 1000 ms delay). This makes the test resilient against the performance of the swap device. Link: https://lore.kernel.org/20260914-test-zswap-wb-ignore-eagain-v1-1-6fb715c22cd8@kernel.org Fixes: 158863e5d7cc ("selftests: cgroup: add tests to verify the zswap writeback path") Signed-off-by: Harry Yoo (Meta) Signed-off-by: Andrew Morton Reviewed-by: SJ Park Acked-by: Nhat Pham Assisted-by: LLM Cc: Chengming Zhou Cc: Johannes Weiner Cc: Joshua Hahn Cc: Kiryl Shutsemau Cc: Michal Koutný Cc: Shuah Khan Cc: Tejun Heo Cc: Usama Arif --- tools/testing/selftests/cgroup/test_zswap.c | 11 ++++++++++- 1 file changed, 10 insertions(+), 1 deletion(-) diff --git a/tools/testing/selftests/cgroup/test_zswap.c b/tools/testing/selftests/cgroup/test_zswap.c index 1ac77907277570..d50acc1b83ac83 100644 --- a/tools/testing/selftests/cgroup/test_zswap.c +++ b/tools/testing/selftests/cgroup/test_zswap.c @@ -346,7 +346,16 @@ static int attempt_writeback(const char *cgroup, void *arg) * it can't writeback to swap. */ ret = cg_write_numeric(cgroup, "memory.reclaim", memsize); - if (!wb_enabled) + + /* + * When writeback is enabled, memory.reclaim may still fail to reclaim + * the requested amount of memory due to a slow swap device. + * Ignore -EAGAIN here. The caller determines pass/fail based on the + * zswap writeback counter. + */ + if (wb_enabled && ret == -EAGAIN) + ret = 0; + else if (!wb_enabled) ret = (ret == -EAGAIN) ? 0 : -1; out: From df71fe4cc8575613e8be8694fe722664d6226969 Mon Sep 17 00:00:00 2001 From: Dan Carpenter Date: Tue, 15 Sep 2026 19:37:27 +0300 Subject: [PATCH 1024/1352] mm/hmm/test: reject overflowing page counts The number of pages comes directly from userspace. Shifting a value larger than ULONG_MAX >> PAGE_SHIFT discards its high bits before the existing end-address check. A request for a huge number of pages can therefore be accepted and processed as a much smaller request. Reject page counts that cannot be represented as a byte size before performing the shift. So far as ChatGPT and I can tell this doesn't cause an issue in practice but preventing this integer overflow is the correct thing to do. Link: https://lore.kernel.org/6f0d39089e938d4f53dc72632eb84aca34de7015.1789457465.git.error27@gmail.com Fixes: b2ef9f5a5cb3 ("mm/hmm/test: add selftest driver for HMM") Signed-off-by: Dan Carpenter Signed-off-by: Andrew Morton Reviewed-by: Alistair Popple Assisted-by: ChatGPT:gpt-5 Cc: Jason Gunthorpe Cc: Jerome Glisse Cc: Leon Romanovsky Cc: Ralph Campbell Cc: Wei Yongjun --- lib/test_hmm.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/lib/test_hmm.c b/lib/test_hmm.c index 6911daa9f8543c..b409c42bfe6fc4 100644 --- a/lib/test_hmm.c +++ b/lib/test_hmm.c @@ -1624,6 +1624,8 @@ static long dmirror_fops_unlocked_ioctl(struct file *filp, if (cmd.addr & ~PAGE_MASK) return -EINVAL; + if (cmd.npages > ULONG_MAX >> PAGE_SHIFT) + return -EINVAL; if (cmd.addr >= (cmd.addr + (cmd.npages << PAGE_SHIFT))) return -EINVAL; From 163d2848ffbc479dd1da7afce393dc91350a159f Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 15 Sep 2026 07:33:50 -0700 Subject: [PATCH 1025/1352] mm/damon/api: introduce DAMON_FILTER_TYPE_HUGEPAGE_SIZE Patch series "mm/damon: introduce hugepage_size probe filter". Knowing whether a given memory is backed by a hugepage of specific size is useful for efficient utilization of hugepages. For easy monitoring of the information, introduce a new data attribute probe filter type, hugepage_size. It works similar to the DAMOS filter of the same name. It works for memory that is backed by a hugepage of a given size range. Patch 1 introduces the new probe filter type to DAMON API and extends related data structures. Patch 2 updates probe filter commit logic to handle the size range. Patch 3 Updates the filtering logic to support the new type. Patch 4 adds new DAMON sysfs files for the size range. Patch 5 updates DAMON sysfs interface to fully support the new filter type. Patches 6-8 updates design, usage and ABI documents for the new feature. This patch (of 8): Introduce a new data attribute probe filter type, hugepage_size. It will work for memory that is backed by a hugepage of a given size range. Add a new damon_filter_type enum DAMON_FILTER_TYPE_HUGEPAGE_SIZE to identify the type. Add two new fields in the damon_filter struct for saving the size range. Link: https://lore.kernel.org/20260915143359.91472-1-sj@kernel.org Link: https://lore.kernel.org/20260915143359.91472-2-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/damon.h | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/include/linux/damon.h b/include/linux/damon.h index 4be7d1df8e71fa..bbb190b4740150 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -783,12 +783,14 @@ struct damon_prep { * @DAMON_FILTER_TYPE_MEMCG: Specific memcg's pages. * @DAMON_FILTER_TYPE_PGIDLE_UNSET: Pgidle is unset. * @DAMON_FILTER_TYPE_PGIDLE_SET: Pgidle is set. + * @DAMON_FILTER_TYPE_HUGEPAGE_SIZE: Page is part of a hugepage. */ enum damon_filter_type { DAMON_FILTER_TYPE_ANON, DAMON_FILTER_TYPE_MEMCG, DAMON_FILTER_TYPE_PGIDLE_UNSET, DAMON_FILTER_TYPE_PGIDLE_SET, + DAMON_FILTER_TYPE_HUGEPAGE_SIZE, }; /** @@ -798,6 +800,8 @@ enum damon_filter_type { * @matching: Whether this filter is for the type-matching ones. * @allow: Whether the @type-@matching ones should pass this filter. * @memcg_id: Memcg id of the question if @type is DAMON_FILTER_MEMCG. + * @range_min: Minimum value of range arguments. + * @range_max: Maximum value of range arguments. */ struct damon_filter { enum damon_filter_type type; @@ -805,6 +809,10 @@ struct damon_filter { bool allow; union { u64 memcg_id; + struct { + unsigned long range_min; + unsigned long range_max; + }; }; /* private: */ /* Siblings list. */ From 4e163395a543c09663dc8716cf516e93209d78ce Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 15 Sep 2026 07:33:51 -0700 Subject: [PATCH 1026/1352] mm/damon/core: commit hugepage_size type damon filter Extend data attribute probe filters commit logic for the new hugepage_size filter type. Since it needs to carry the size range of the hugepage, update the logic to update the size range fields of the commit destination filter struct. While doing that, validate the given range and propagate an error if it is invalid. Add the error handling in the callers, too. Link: https://lore.kernel.org/20260915143359.91472-3-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/core.c | 28 +++++++++++++++++++++++----- 1 file changed, 23 insertions(+), 5 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 5fdacb9dcee5f6..e1b49c3d72b865 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -1818,7 +1818,7 @@ static int damon_commit_preps(struct damon_probe *dst, struct damon_probe *src) return 0; } -static void damon_commit_filter(struct damon_filter *dst, +static int damon_commit_filter(struct damon_filter *dst, struct damon_filter *src) { dst->type = src->type; @@ -1828,23 +1828,33 @@ static void damon_commit_filter(struct damon_filter *dst, case DAMON_FILTER_TYPE_MEMCG: dst->memcg_id = src->memcg_id; break; + case DAMON_FILTER_TYPE_HUGEPAGE_SIZE: + if (src->range_max < src->range_min) + return -EINVAL; + dst->range_min = src->range_min; + dst->range_max = src->range_max; + break; default: break; } + return 0; } static int damon_commit_filters(struct damon_probe *dst, struct damon_probe *src) { struct damon_filter *dst_filter, *next, *src_filter, *new_filter; - int i = 0, j = 0; + int i = 0, j = 0, err; damon_for_each_filter_safe(dst_filter, next, dst) { src_filter = damon_nth_filter(i++, src); - if (src_filter) - damon_commit_filter(dst_filter, src_filter); - else + if (src_filter) { + err = damon_commit_filter(dst_filter, src_filter); + if (err) + return err; + } else { damon_destroy_filter(dst_filter); + } } damon_for_each_filter_safe(src_filter, next, src) { @@ -1859,6 +1869,14 @@ static int damon_commit_filters(struct damon_probe *dst, case DAMON_FILTER_TYPE_MEMCG: new_filter->memcg_id = src_filter->memcg_id; break; + case DAMON_FILTER_TYPE_HUGEPAGE_SIZE: + if (src_filter->range_max < src_filter->range_min) { + damon_destroy_filter(new_filter); + return -EINVAL; + } + new_filter->range_min = src_filter->range_min; + new_filter->range_max = src_filter->range_max; + break; default: break; } From ce483b351f93b3452498b02f523c4f7eda8ca8d4 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 15 Sep 2026 07:33:52 -0700 Subject: [PATCH 1027/1352] mm/damon/ops-common: support hugepage_size damon filter matching Update ops-common data attribute filter matching logic to support hugepage_size filter type. Link: https://lore.kernel.org/20260915143359.91472-4-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/ops-common.c | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/mm/damon/ops-common.c b/mm/damon/ops-common.c index c36cc39cd2c707..77366f42b3e5bf 100644 --- a/mm/damon/ops-common.c +++ b/mm/damon/ops-common.c @@ -536,6 +536,7 @@ bool damon_ops_filter_match(struct damon_filter *filter, struct folio *folio) { bool matched = false; struct mem_cgroup *memcg; + size_t folio_sz; switch (filter->type) { case DAMON_FILTER_TYPE_ANON: @@ -558,6 +559,15 @@ bool damon_ops_filter_match(struct damon_filter *filter, struct folio *folio) matched = filter->memcg_id == mem_cgroup_id(memcg); rcu_read_unlock(); break; + case DAMON_FILTER_TYPE_HUGEPAGE_SIZE: + if (!folio) { + matched = false; + break; + } + folio_sz = folio_size(folio); + matched = filter->range_min <= folio_sz && + folio_sz <= filter->range_max; + break; default: break; } From a12117a526bfa4fd56c0e84a5e9498aad9be5997 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 15 Sep 2026 07:33:53 -0700 Subject: [PATCH 1028/1352] mm/damon/sysfs: add min,max files under probe filter directory In future, DAMON sysfs interface will support data attribute probe filter types that have range arguments like the newly added hugepage_size type filter. To prepare such supports, add two new DAMON sysfs files, min and max, under the probe filter directory. Link: https://lore.kernel.org/20260915143359.91472-5-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs.c | 48 ++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 48 insertions(+) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index 51fa506c879b02..8e8d89b8ed981e 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -975,6 +975,8 @@ struct damon_sysfs_filter { bool matching; bool allow; char *path; + unsigned long range_min; + unsigned long range_max; }; static struct damon_sysfs_filter *damon_sysfs_filter_alloc(void) @@ -1127,6 +1129,44 @@ static ssize_t path_store(struct kobject *kobj, return count; } +static ssize_t min_show(struct kobject *kobj, + struct kobj_attribute *attr, char *buf) +{ + struct damon_sysfs_filter *filter = container_of(kobj, + struct damon_sysfs_filter, kobj); + + return sysfs_emit(buf, "%lu\n", filter->range_min); +} + +static ssize_t min_store(struct kobject *kobj, + struct kobj_attribute *attr, const char *buf, size_t count) +{ + struct damon_sysfs_filter *filter = container_of(kobj, + struct damon_sysfs_filter, kobj); + int err = kstrtoul(buf, 0, &filter->range_min); + + return err ? err : count; +} + +static ssize_t max_show(struct kobject *kobj, + struct kobj_attribute *attr, char *buf) +{ + struct damon_sysfs_filter *filter = container_of(kobj, + struct damon_sysfs_filter, kobj); + + return sysfs_emit(buf, "%lu\n", filter->range_max); +} + +static ssize_t max_store(struct kobject *kobj, + struct kobj_attribute *attr, const char *buf, size_t count) +{ + struct damon_sysfs_filter *filter = container_of(kobj, + struct damon_sysfs_filter, kobj); + int err = kstrtoul(buf, 0, &filter->range_max); + + return err ? err : count; +} + static void damon_sysfs_filter_release(struct kobject *kobj) { struct damon_sysfs_filter *filter = container_of(kobj, @@ -1148,11 +1188,19 @@ static struct kobj_attribute damon_sysfs_filter_allow_attr = static struct kobj_attribute damon_sysfs_filter_path_attr = __ATTR_RW_MODE(path, 0600); +static struct kobj_attribute damon_sysfs_filter_min_attr = + __ATTR_RW_MODE(min, 0600); + +static struct kobj_attribute damon_sysfs_filter_max_attr = + __ATTR_RW_MODE(max, 0600); + static struct attribute *damon_sysfs_filter_attrs[] = { &damon_sysfs_filter_type_attr.attr, &damon_sysfs_filter_matching_attr.attr, &damon_sysfs_filter_allow_attr.attr, &damon_sysfs_filter_path_attr.attr, + &damon_sysfs_filter_min_attr.attr, + &damon_sysfs_filter_max_attr.attr, NULL, }; ATTRIBUTE_GROUPS(damon_sysfs_filter); From 8eeb76ad04f33dfe727a2e817db751545aceb037 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 15 Sep 2026 07:33:54 -0700 Subject: [PATCH 1029/1352] mm/damon/sysfs: support hugepage_size probe filter Extend DAMON sysfs interface to support hugepage_size probe filter. Allows hugepage_size user string input to the filter type file. Pass the size range argument that users set via min/max files under the probe filter directory to the DAMON core. Link: https://lore.kernel.org/20260915143359.91472-6-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index 8e8d89b8ed981e..43519afb9eb7f3 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -1007,6 +1007,10 @@ damon_sysfs_filter_type_names[] = { .type = DAMON_FILTER_TYPE_PGIDLE_SET, .name = "pgidle_set", }, + { + .type = DAMON_FILTER_TYPE_HUGEPAGE_SIZE, + .name = "hugepage_size", + }, }; static ssize_t type_show(struct kobject *kobj, @@ -2273,6 +2277,9 @@ static int damon_sysfs_set_filters(struct damon_probe *probe, damon_destroy_filter(filter); return err; } + } else if (filter->type == DAMON_FILTER_TYPE_HUGEPAGE_SIZE) { + filter->range_min = sys_filter->range_min; + filter->range_max = sys_filter->range_max; } damon_add_filter(probe, filter); } From 8bce525b644da2e4e9d69255b8285304c4dc4ec4 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 15 Sep 2026 07:33:55 -0700 Subject: [PATCH 1030/1352] Docs/mm/damon/design: update for hugepage_size probe filter Update DAMON design document for the newly added hugepage_size data attribute probe filter type. Link: https://lore.kernel.org/20260915143359.91472-7-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/mm/damon/design.rst | 2 ++ 1 file changed, 2 insertions(+) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index 707170bcd1b33c..0a86792f90a184 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -300,6 +300,8 @@ filter types. Currently below filter types are supported. - ``pgidle_unset``: Matches if the page for the memory is marked as not access-idle. - ``pgidle_set``: Matches if the page for the memory is marked as access-idle. +- ``hugepage_size``: Matches if the page for the memory is a part of a hugepage + of a given size range. If such probes are registered, DAMON executes the probes for each region's sampling memory when it does the access :ref:`sampling From b416db756f423e53dd5ca79deb0c61a349c85637 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 15 Sep 2026 07:33:56 -0700 Subject: [PATCH 1031/1352] Docs/admin-guide/mm/damon/usage: update for hugepage_size Update DAMON usage document for the newly added hugepage_size data attribute filter type. Link: https://lore.kernel.org/20260915143359.91472-8-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/admin-guide/mm/damon/usage.rst | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/Documentation/admin-guide/mm/damon/usage.rst b/Documentation/admin-guide/mm/damon/usage.rst index d3e37400367bdd..6b80bce5d678de 100644 --- a/Documentation/admin-guide/mm/damon/usage.rst +++ b/Documentation/admin-guide/mm/damon/usage.rst @@ -78,7 +78,7 @@ comma (","). │ │ │ │ │ │ │ │ │ 0/prep_action │ │ │ │ │ │ │ │ │ ... │ │ │ │ │ │ │ │ filters/nr_filters - │ │ │ │ │ │ │ │ │ 0/type,matching,allow,path + │ │ │ │ │ │ │ │ │ 0/type,matching,allow,path,min,max │ │ │ │ │ │ │ │ │ ... │ │ │ │ │ │ │ ... │ │ │ │ │ :ref:`targets `/nr_targets @@ -308,6 +308,8 @@ Writing a number (``N``) to the file creates the number of child directories named ``0`` to ``N-1``. Each directory represents each filter and works in a way similar to that for :ref:`DAMOS filter `. When the filter ``type`` is ``memcg``, ``path`` file acts as ``memcg_path`` for :ref:`DAMOS +filter `. When the filter ``type`` is ``hugepage_size``, +``min`` and ``max`` files acts as files of the same names for :ref:`DAMOS filter `. .. _sysfs_targets: From 6d81eebe8f7fc4dcb88b09b7929cbe695bbb366f Mon Sep 17 00:00:00 2001 From: Andrew Morton Date: Tue, 15 Sep 2026 16:33:09 -0700 Subject: [PATCH 1032/1352] docs-admin-guide-mm-damon-usage-update-for-hugepage_size-fix s/files acts/file acts/, per SJ Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: SJ Park Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- Documentation/admin-guide/mm/damon/usage.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Documentation/admin-guide/mm/damon/usage.rst b/Documentation/admin-guide/mm/damon/usage.rst index 6b80bce5d678de..2e191a3dff18a6 100644 --- a/Documentation/admin-guide/mm/damon/usage.rst +++ b/Documentation/admin-guide/mm/damon/usage.rst @@ -309,7 +309,7 @@ named ``0`` to ``N-1``. Each directory represents each filter and works in a way similar to that for :ref:`DAMOS filter `. When the filter ``type`` is ``memcg``, ``path`` file acts as ``memcg_path`` for :ref:`DAMOS filter `. When the filter ``type`` is ``hugepage_size``, -``min`` and ``max`` files acts as files of the same names for :ref:`DAMOS +``min`` and ``max`` file acts as files of the same names for :ref:`DAMOS filter `. .. _sysfs_targets: From 834dacce692a51a74a9ec0e0e9d70aa3e11ae70c Mon Sep 17 00:00:00 2001 From: Andrew Morton Date: Wed, 16 Sep 2026 21:18:41 -0700 Subject: [PATCH 1033/1352] docs-admin-guide-mm-damon-usage-update-for-hugepage_size-fix-fix fix the fix Cc: SJ Park Signed-off-by: Andrew Morton --- Documentation/admin-guide/mm/damon/usage.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Documentation/admin-guide/mm/damon/usage.rst b/Documentation/admin-guide/mm/damon/usage.rst index 2e191a3dff18a6..ba47255448564b 100644 --- a/Documentation/admin-guide/mm/damon/usage.rst +++ b/Documentation/admin-guide/mm/damon/usage.rst @@ -309,7 +309,7 @@ named ``0`` to ``N-1``. Each directory represents each filter and works in a way similar to that for :ref:`DAMOS filter `. When the filter ``type`` is ``memcg``, ``path`` file acts as ``memcg_path`` for :ref:`DAMOS filter `. When the filter ``type`` is ``hugepage_size``, -``min`` and ``max`` file acts as files of the same names for :ref:`DAMOS +``min`` and ``max`` files act as files of the same names for :ref:`DAMOS filter `. .. _sysfs_targets: From c237e23020ee20ccbf2fd2a1b9a087122b882346 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 15 Sep 2026 07:33:57 -0700 Subject: [PATCH 1034/1352] Docs/ABI/damon: update for hugepage_size probe filter For the newly added hugepage_size data attribute probe filter, two new sysfs files are added for the size range. Update DAMON ABI document for the new files. Link: https://lore.kernel.org/20260915143359.91472-9-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/ABI/testing/sysfs-kernel-mm-damon | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/Documentation/ABI/testing/sysfs-kernel-mm-damon b/Documentation/ABI/testing/sysfs-kernel-mm-damon index ad21f58f3c9126..55df688ea596ff 100644 --- a/Documentation/ABI/testing/sysfs-kernel-mm-damon +++ b/Documentation/ABI/testing/sysfs-kernel-mm-damon @@ -206,6 +206,20 @@ Description: If 'memcg' is written to the 'type' file, writing to and reading from this file sets and gets the path to the memory cgroup of the interest. +What: /sys/kernel/mm/damon/admin/kdamonds//contexts//monitoring_attrs/probes/

/filters//min +Date: Sep 2026 +Contact: SJ Park +Description: If 'hugepage_size' is written to the 'type' file, writing to and + reading from this file sets and gets the minimum size of the + huge page of the interest. + +What: /sys/kernel/mm/damon/admin/kdamonds//contexts//monitoring_attrs/probes/

/filters//max +Date: Sep 2026 +Contact: SJ Park +Description: If 'hugepage_size' is written to the 'type' file, writing to and + reading from this file sets and gets the maximum size of the + huge page of the interest. + What: /sys/kernel/mm/damon/admin/kdamonds//contexts//monitoring_attrs/probes/

/filters//matching Date: May 2026 Contact: SJ Park From aa849ec301e2551d1dd06c492ed9e90ed805889a Mon Sep 17 00:00:00 2001 From: Lance Yang Date: Thu, 17 Sep 2026 13:40:15 +0800 Subject: [PATCH 1035/1352] mm/huge_memory: simplify pgtable deposit detection Whether a PMD has a deposited PTE page table only depends on whether the architecture requires deposits or the VMA is anonymous. Implement this rule directly in vma_has_deposited_pgtable(), avoiding mistaking a non-anonymous raw PFN PMD for one with a deposit and attempting to withdraw a page table that was never deposited. There is no known in-tree workload that triggers this. Mostly a defensive/simplifying change. Link: https://lore.kernel.org/20260917054015.23553-1-lance.yang@linux.dev Signed-off-by: Lance Yang Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand Reviewed-by: Kiryl Shutsemau (Meta) Acked-by: David Hildenbrand (Arm) Acked-by: Zi Yan Reviewed-by: Baolin Wang Cc: Barry Song Cc: Dev Jain Cc: Liam R. Howlett Cc: Ryan Roberts --- mm/huge_memory.c | 22 +++++----------------- 1 file changed, 5 insertions(+), 17 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index ec37a63b8a2ec7..d2e990da86b0d9 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -2515,25 +2515,13 @@ static struct folio *normal_or_softleaf_folio_pmd(struct vm_area_struct *vma, return pmd_to_softleaf_folio(pmdval); } -static bool has_deposited_pgtable(struct vm_area_struct *vma, pmd_t pmdval, - struct folio *folio) +static bool vma_has_deposited_pgtable(struct vm_area_struct *vma) { - /* Some architectures require unconditional depositing. */ - if (arch_needs_pgtable_deposit()) - return true; - - /* - * Huge zero always deposited except for DAX which handles itself, see - * set_huge_zero_folio(). - */ - if (is_huge_zero_pmd(pmdval)) - return !vma_is_dax(vma); - /* - * Otherwise, only anonymous folios are deposited, see - * __do_huge_pmd_anonymous_page(). + * PMDs in anonymous VMAs always have a deposited page table. PMDs in + * other VMAs only have one when required by the architecture. */ - return folio && folio_test_anon(folio); + return arch_needs_pgtable_deposit() || vma_is_anonymous(vma); } /** @@ -2573,7 +2561,7 @@ bool zap_huge_pmd(struct mmu_gather *tlb, struct vm_area_struct *vma, is_present = pmd_present(orig_pmd); folio = normal_or_softleaf_folio_pmd(vma, addr, orig_pmd, is_present); - has_deposit = has_deposited_pgtable(vma, orig_pmd, folio); + has_deposit = vma_has_deposited_pgtable(vma); if (folio) zap_huge_pmd_folio(mm, vma, orig_pmd, folio, is_present); if (has_deposit) From ae107e1971b04b3ee8cecdf30b64c2b92b5fea78 Mon Sep 17 00:00:00 2001 From: Sebastian Andrzej Siewior Date: Fri, 18 Sep 2026 12:50:13 +0200 Subject: [PATCH 1036/1352] mm/vmalloc: use %p for pointer formatting Commit 45ec16908e84e ("mm: use %pK for /proc/vmallocinfo") introduced the %pK in order not to leak kernel pointers. Since commit ad67b74d2469d ("printk: hash addresses printed with %p") pointers are hashed by default and the behaviour can be controlled by `hash_pointers' boot argument. The policy on %p is to not introduce new ones. Removing %pK makes it possible to remove its handling from the library. Looking at the output, having the pointer in the output makes it possible to distinguish the individual entries. Use %p instead %pK. /proc/vmallocinfo has mode 0400, so this change doesn't make kernel pointer information available to unprivileged userspace. Link: https://lore.kernel.org/20260918105013.UpdykT6j@linutronix.de Signed-off-by: Sebastian Andrzej Siewior Signed-off-by: Andrew Morton Reviewed-by: SJ Park Reviewed-by: Uladzislau Rezki (Sony) --- mm/vmalloc.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index 2d29b08f263ad3..fad918765f054e 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -5336,7 +5336,7 @@ static void show_purge_info(struct seq_file *m) for_each_vmap_node(vn) { spin_lock(&vn->lazy.lock); list_for_each_entry(va, &vn->lazy.head, list) { - seq_printf(m, "0x%pK-0x%pK %7ld unpurged vm_area\n", + seq_printf(m, "0x%p-0x%p %7ld unpurged vm_area\n", (void *)va->va_start, (void *)va->va_end, va_size(va)); } @@ -5359,7 +5359,7 @@ static void show_busy_info(struct seq_file *m) list_for_each_entry(va, &vn->busy.head, list) { if (!va->vm) { if (va->flags & VMAP_RAM) - seq_printf(m, "0x%pK-0x%pK %7ld vm_map_ram\n", + seq_printf(m, "0x%p-0x%p %7ld vm_map_ram\n", (void *)va->va_start, (void *)va->va_end, va_size(va)); @@ -5373,7 +5373,7 @@ static void show_busy_info(struct seq_file *m) /* Pair with smp_wmb() in clear_vm_uninitialized_flag() */ smp_rmb(); - seq_printf(m, "0x%pK-0x%pK %7ld", + seq_printf(m, "0x%p-0x%p %7ld", v->addr, v->addr + v->size, v->size); if (v->caller) From eaf809b1d597812e6a814c6288ac9bc0cac66735 Mon Sep 17 00:00:00 2001 From: Sarthak Sharma Date: Tue, 15 Sep 2026 15:55:24 +0530 Subject: [PATCH 1037/1352] mm/gup_test: safely calculate GUP batch size __gup_test_ioctl() calculates the end of a GUP batch using: next = addr + nr * PAGE_SIZE; If nr is too large, it can cause next to overflow and wrap around. If it wraps, the next > end check is bypassed and a large value of nr is passed to the GUP call, even though the pages array was allocated according to gup->size. This can lead to out of bounds writes. Also, when fewer than PAGE_SIZE bytes remain, the calculated batch contains zero pages. The code still calls GUP functions with pages + i, which can point past the allocated array. Calculate nr by taking the minimum of the number of pages per call and the pages remaining in the address range. Reject zero sized and non page aligned gup->size values and zero nr_pages_per_call value. Link: https://lore.kernel.org/20260915102524.125758-1-sarthak.sharma@arm.com Fixes: 64c349f4ae78 ("mm: add infrastructure for get_user_pages_fast() benchmarking") Signed-off-by: Sarthak Sharma Signed-off-by: Andrew Morton Reviewed-by: Kiryl Shutsemau (Meta) Acked-by: David Hildenbrand (Arm) Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu --- mm/gup_test.c | 9 ++++----- 1 file changed, 4 insertions(+), 5 deletions(-) diff --git a/mm/gup_test.c b/mm/gup_test.c index 185ba3bb8ed10b..ba74bf3f410486 100644 --- a/mm/gup_test.c +++ b/mm/gup_test.c @@ -116,7 +116,8 @@ static int __gup_test_ioctl(unsigned int cmd, bool needs_mmap_lock = cmd != GUP_FAST_BENCHMARK && cmd != PIN_FAST_BENCHMARK; - if (gup->addr > ULONG_MAX || gup->size > ULONG_MAX) + if (gup->addr > ULONG_MAX || gup->size > ULONG_MAX || !gup->size || + !gup->nr_pages_per_call || !PAGE_ALIGNED(gup->size)) return -EINVAL; if (check_add_overflow((unsigned long)gup->addr, (unsigned long)gup->size, &end)) @@ -139,11 +140,9 @@ static int __gup_test_ioctl(unsigned int cmd, if (nr != gup->nr_pages_per_call) break; + nr = min_t(unsigned long, nr, (end - addr) / PAGE_SIZE); + next = addr + nr * PAGE_SIZE; - if (next > end) { - next = end; - nr = (next - addr) / PAGE_SIZE; - } switch (cmd) { case GUP_FAST_BENCHMARK: From b12b5b718dcf6cda8a8014f829cb82cfcd410572 Mon Sep 17 00:00:00 2001 From: "Barry Song (Xiaomi)" Date: Tue, 15 Sep 2026 18:15:56 +0800 Subject: [PATCH 1038/1352] mm/mglru: restore accidentally removed seq < max_seq check Since commit 798c0330c2ca ("mm/mglru: rework aging feedback"), the following sanity check was accidentally removed: if (seq < max_seq) return 0; That means we can perform aging for any value less than or equal to max_gen_nr. This has been inconsistent with Documentation/admin-guide/mm/multigen_lru.rst, which states: Users can write the following command to ``lru_gen`` to create a new generation ``max_gen_nr+1``: ``+ memcg_id node_id max_gen_nr [can_swap [force_scan]]`` The correct semantics are that writing a value smaller than max_gen_nr should return 0, since the requested generation already exists. Link: https://lore.kernel.org/20260915101556.50467-1-baohua@kernel.org Fixes: 798c0330c2ca ("mm/mglru: rework aging feedback") Signed-off-by: Barry Song (Xiaomi) Signed-off-by: Andrew Morton Reported-by: Chuanhua Han Reviewed-by: Kairui Song Reviewed-by: Baolin Wang Cc: Axel Rasmussen Cc: Baoquan He Cc: David Hildenbrand Cc: Johannes Weiner Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie Cc: Yu Zhao --- mm/vmscan.c | 3 +++ 1 file changed, 3 insertions(+) diff --git a/mm/vmscan.c b/mm/vmscan.c index 836f50814ffae5..f2e641e9cf7de9 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -5840,6 +5840,9 @@ static int run_aging(struct lruvec *lruvec, unsigned long seq, { DEFINE_MAX_SEQ(lruvec); + if (seq < max_seq) + return 0; + if (seq > max_seq) return -EINVAL; From aed002156f34c4193e875a614b044073b961d8dd Mon Sep 17 00:00:00 2001 From: Zhenghui Hao Date: Tue, 15 Sep 2026 16:20:29 +0800 Subject: [PATCH 1039/1352] mm/hugetlb: fix misspelled parameter names in comment The comment above hugepages_setup() refers to "hugepagsz" and "default_hugepagsz", but the parameters parsed by this code are "hugepagesz" and "default_hugepagesz". Fix the spelling and also drop the redundant spaces so the comment reads normally. No functional change. Link: https://lore.kernel.org/tencent_034C6FC23D4817C40657E5F17F64E260A009@qq.com Signed-off-by: Zhenghui Hao Signed-off-by: Andrew Morton Acked-by: Muchun Song Reviewed-by: SJ Park Cc: David Hildenbrand Cc: Oscar Salvador --- mm/hugetlb.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 1b53ba991d360a..9d05fecf21326d 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -4330,9 +4330,9 @@ static __init void hugetlb_parse_params(void) /* * hugepages command line processing - * hugepages normally follows a valid hugepagsz or default_hugepagsz - * specification. If not, ignore the hugepages value. hugepages can also - * be the first huge page command line option in which case it implicitly + * hugepages normally follows a valid hugepagesz or default_hugepagesz + * specification. If not, ignore the hugepages value. hugepages can also + * be the first huge page command line option in which case it implicitly * specifies the number of huge pages for the default size. */ static int __init hugepages_setup(char *s) From 7ef4732a446712c10234918e5ad8347355a8d8fa Mon Sep 17 00:00:00 2001 From: Qiqi Liu Date: Tue, 15 Sep 2026 15:49:28 +0800 Subject: [PATCH 1040/1352] mm/page_alloc: apply per-task GFP context in bulk allocator alloc_pages_bulk_noprof() does not call current_gfp_context(), so per-task scoped allocation constraints (PF_MEMALLOC_NOIO, PF_MEMALLOC_NOFS, PF_MEMALLOC_PIN) are not applied on the bulk fast path. By ignoring PF_MEMALLOC_NOIO and PF_MEMALLOC_NOFS, the allocation can theoretically result in a deadlock. Ignoring PF_MEMALLOC_PIN also has consequences: without clearing __GFP_MOVABLE, prepare_alloc_pages() selects MIGRATE_MOVABLE for the PCP list, and a task with PF_MEMALLOC_PIN set receives movable pages from the bulk allocator. Once pinned, these pages can no longer be migrated but remain in MOVABLE pageblocks, violating the mobility contract. This can increase fragmentation and interfere with compaction or contiguous-memory allocations, eventually surfacing as higher allocation latency or allocation failures under memory pressure. Found via review of the bulk allocation tracepoint hooks [1]. Link: https://sashiko.dev/#/patchset/20260907120949.418450-1-liuqiqi%40kylinos.cn Link: https://lore.kernel.org/all/20260907120949.418450-1-liuqiqi@kylinos.cn/ [1] Link: https://lore.kernel.org/20260915074928.327471-1-liuqiqi@kylinos.cn Fixes: 387ba26fb1cb ("mm/page_alloc: add a bulk page allocator") Signed-off-by: Qiqi Liu Signed-off-by: Andrew Morton Reviewed-by: Vlastimil Babka (SUSE) Cc: Brendan Jackman Cc: Johannes Weiner Cc: Michal Hocko Cc: Suren Baghdasaryan Cc: Zi Yan --- mm/page_alloc.c | 1 + 1 file changed, 1 insertion(+) diff --git a/mm/page_alloc.c b/mm/page_alloc.c index c9318a97f69460..d62670b8f9610c 100644 --- a/mm/page_alloc.c +++ b/mm/page_alloc.c @@ -5231,6 +5231,7 @@ unsigned long alloc_pages_bulk_noprof(gfp_t gfp, int preferred_nid, /* May set ALLOC_NOFRAGMENT, fragmentation will return 1 page. */ gfp &= gfp_allowed_mask; + gfp = current_gfp_context(gfp); if (!prepare_alloc_pages(gfp, 0, preferred_nid, nodemask, &ac, &gfp, &alloc_flags)) goto out; From f93d7564dfc3b84d91fb51d9f5baa49efb91f087 Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Sat, 12 Sep 2026 07:05:40 -0400 Subject: [PATCH 1041/1352] mm/madvise: use folio_trylock() in the cold/pageout PMD split MADV_COLD or MADV_PAGEOUT over part of a PMD splits the THP in madvise_cold_or_pageout_pte_range(). Two threads doing that to the same THP create spurious failures. CPU0 CPU1 ---- ---- folio_get() spin_unlock(ptl) folio_lock() folio_get() spin_unlock(ptl) folio_lock() <- blocks, keeps its ref split_folio() folio_expected_ref_count(folio) != folio_ref_count(folio) - 1 -EAGAIN CPU1 cannot drop its reference until it gets the lock CPU0 holds, so CPU0's split always fails. folio_trylock() makes CPU1 leave without ever taking a reference. The PTE branch of this same function already does this, as do madvise_free_pte_range() and madvise_free_huge_pmd(). Reproducer: 400 rounds of eight threads calling MADV_COLD on half of each of eight THPs, re-formed with MADV_COLLAPSE between rounds. From /proc/vmstat: thp_split_page thp_split_page_failed before 3186 860 after 3200 0 The short before count is rounds where every thread failed and the advice was dropped for that THP entirely. On failure the walker returns 0 and nothing retries. The PMD path becomes best effort when the folio lock is held elsewhere - same as the PTE path. Link: https://lore.kernel.org/20260912110540.3203010-1-gourry@gourry.net Signed-off-by: Gregory Price (Meta) Signed-off-by: Andrew Morton Reported-by: sashiko-bot Closes: https://sashiko.dev/#/patchset/20260817220810.1175596-1-gourry%40gourry.net Acked-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Jann Horn Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: --- mm/madvise.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/mm/madvise.c b/mm/madvise.c index 8cd08fc153e6d6..32a28b9bb6880a 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -418,9 +418,10 @@ static int madvise_cold_or_pageout_pte_range(pmd_t *pmd, if (next - addr != HPAGE_PMD_SIZE) { int err; + if (!folio_trylock(folio)) + goto huge_unlock; folio_get(folio); spin_unlock(ptl); - folio_lock(folio); err = split_folio(folio); folio_unlock(folio); folio_put(folio); From c373ef242cf6508b3b9501e1e4c94d775b76cd04 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 15 Sep 2026 19:20:05 -0400 Subject: [PATCH 1042/1352] mm: memcontrol: take a const folio in folio_memcg() and friends Patch series "mm: memcontrol: constify the read side of the read side of the memcg API", v3. The memcg accessors, lruvec helpers, and stat readers only read from the memcg, folio, or lruvec they are given, but take non-const pointers. Constify them, along with the page_counter readers and the swap I/O blkg helpers along the way. This patch (of 11): The folio_memcg() family only reads from the folio, and everything it calls already takes a const folio. Constify it, along with the page wrappers built on top of it and get_obj_cgroup_from_folio(), which only reads the folio's objcg. Link: https://lore.kernel.org/20260915-folio_memcg-const-v3-0-c239a6010b58@columbia.edu Link: https://lore.kernel.org/20260915-folio_memcg-const-v3-1-c239a6010b58@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Reviewed-by: Muchun Song Acked-by: Shakeel Butt Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Roman Gushchin Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie --- include/linux/memcontrol.h | 50 +++++++++++++++++++------------------- mm/memcontrol.c | 8 +++--- 2 files changed, 29 insertions(+), 29 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 46bf724cae7af9..5a5ca6814782a3 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -410,7 +410,7 @@ static inline struct mem_cgroup *obj_cgroup_memcg(struct obj_cgroup *objcg) * or NULL. This function assumes that the folio is known to have a * proper object cgroup pointer. */ -static inline struct obj_cgroup *folio_objcg(struct folio *folio) +static inline struct obj_cgroup *folio_objcg(const struct folio *folio) { unsigned long memcg_data = folio->memcg_data; @@ -448,7 +448,7 @@ static inline struct obj_cgroup *folio_objcg(struct folio *folio) * Note: The caller should hold an rcu read lock or cgroup_mutex to protect * memcg associated with a folio from being released. */ -static inline struct mem_cgroup *folio_memcg(struct folio *folio) +static inline struct mem_cgroup *folio_memcg(const struct folio *folio) { struct obj_cgroup *objcg = folio_objcg(folio); @@ -461,7 +461,7 @@ static inline struct mem_cgroup *folio_memcg(struct folio *folio) * * Returns true if folio is charged to a memory cgroup, otherwise returns false. */ -static inline bool folio_memcg_charged(struct folio *folio) +static inline bool folio_memcg_charged(const struct folio *folio) { return folio->memcg_data != 0; } @@ -481,7 +481,7 @@ static inline bool folio_memcg_charged(struct folio *folio) * A caller should hold an rcu read lock to protect memcg associated with a * page from being released. */ -static inline struct mem_cgroup *folio_memcg_check(struct folio *folio) +static inline struct mem_cgroup *folio_memcg_check(const struct folio *folio) { /* * Because folio->memcg_data might be changed asynchronously @@ -498,11 +498,11 @@ static inline struct mem_cgroup *folio_memcg_check(struct folio *folio) return obj_cgroup_memcg(objcg); } -static inline struct mem_cgroup *page_memcg_check(struct page *page) +static inline struct mem_cgroup *page_memcg_check(const struct page *page) { if (PageTail(page)) return NULL; - return folio_memcg_check((struct folio *)page); + return folio_memcg_check((const struct folio *)page); } static inline struct mem_cgroup *get_mem_cgroup_from_objcg(struct obj_cgroup *objcg) @@ -527,14 +527,14 @@ static inline struct mem_cgroup *get_mem_cgroup_from_objcg(struct obj_cgroup *ob * that the folio has an associated memory cgroup. It's not safe to call * this function against some types of folios, e.g. slab folios. */ -static inline bool folio_memcg_kmem(struct folio *folio) +static inline bool folio_memcg_kmem(const struct folio *folio) { VM_BUG_ON_PGFLAGS(PageTail(&folio->page), &folio->page); VM_BUG_ON_FOLIO(folio->memcg_data & MEMCG_DATA_OBJEXTS, folio); return folio->memcg_data & MEMCG_DATA_KMEM; } -static inline bool PageMemcgKmem(struct page *page) +static inline bool PageMemcgKmem(const struct page *page) { return folio_memcg_kmem(page_folio(page)); } @@ -777,7 +777,7 @@ struct mem_cgroup *get_mem_cgroup_from_mm(struct mm_struct *mm); struct mem_cgroup *get_mem_cgroup_from_current(void); -struct mem_cgroup *get_mem_cgroup_from_folio(struct folio *folio); +struct mem_cgroup *get_mem_cgroup_from_folio(const struct folio *folio); struct lruvec *folio_lruvec_lock(struct folio *folio); struct lruvec *folio_lruvec_lock_irq(struct folio *folio); @@ -905,8 +905,8 @@ static inline bool mm_match_cgroup(struct mm_struct *mm, return match; } -struct cgroup_subsys_state *get_mem_cgroup_css_from_folio(struct folio *folio); -ino_t page_cgroup_ino(struct page *page); +struct cgroup_subsys_state *get_mem_cgroup_css_from_folio(const struct folio *folio); +ino_t page_cgroup_ino(const struct page *page); static inline bool mem_cgroup_online(struct mem_cgroup *memcg) { @@ -956,7 +956,7 @@ void mem_cgroup_print_oom_group(struct mem_cgroup *memcg); void mod_memcg_state(struct mem_cgroup *memcg, enum memcg_stat_item idx, int val); -static inline void mod_memcg_page_state(struct page *page, +static inline void mod_memcg_page_state(const struct page *page, enum memcg_stat_item idx, int val) { struct mem_cgroup *memcg; @@ -990,7 +990,7 @@ void mod_lruvec_kmem_state(void *p, enum node_stat_item idx, int val); void count_memcg_events(struct mem_cgroup *memcg, enum vm_event_item idx, unsigned long count); -static inline void count_memcg_folio_events(struct folio *folio, +static inline void count_memcg_folio_events(const struct folio *folio, enum vm_event_item idx, unsigned long nr) { struct mem_cgroup *memcg; @@ -1084,22 +1084,22 @@ static inline struct mem_cgroup *obj_cgroup_memcg(struct obj_cgroup *objcg) #define root_mem_cgroup (NULL) -static inline struct mem_cgroup *folio_memcg(struct folio *folio) +static inline struct mem_cgroup *folio_memcg(const struct folio *folio) { return NULL; } -static inline bool folio_memcg_charged(struct folio *folio) +static inline bool folio_memcg_charged(const struct folio *folio) { return false; } -static inline struct mem_cgroup *folio_memcg_check(struct folio *folio) +static inline struct mem_cgroup *folio_memcg_check(const struct folio *folio) { return NULL; } -static inline struct mem_cgroup *page_memcg_check(struct page *page) +static inline struct mem_cgroup *page_memcg_check(const struct page *page) { return NULL; } @@ -1109,12 +1109,12 @@ static inline struct mem_cgroup *get_mem_cgroup_from_objcg(struct obj_cgroup *ob return NULL; } -static inline bool folio_memcg_kmem(struct folio *folio) +static inline bool folio_memcg_kmem(const struct folio *folio) { return false; } -static inline bool PageMemcgKmem(struct page *page) +static inline bool PageMemcgKmem(const struct page *page) { return false; } @@ -1248,7 +1248,7 @@ static inline struct mem_cgroup *get_mem_cgroup_from_current(void) return NULL; } -static inline struct mem_cgroup *get_mem_cgroup_from_folio(struct folio *folio) +static inline struct mem_cgroup *get_mem_cgroup_from_folio(const struct folio *folio) { return NULL; } @@ -1406,7 +1406,7 @@ static inline void mod_memcg_state(struct mem_cgroup *memcg, { } -static inline void mod_memcg_page_state(struct page *page, +static inline void mod_memcg_page_state(const struct page *page, enum memcg_stat_item idx, int val) { } @@ -1471,7 +1471,7 @@ static inline void count_memcg_events(struct mem_cgroup *memcg, { } -static inline void count_memcg_folio_events(struct folio *folio, +static inline void count_memcg_folio_events(const struct folio *folio, enum vm_event_item idx, unsigned long nr) { } @@ -1767,7 +1767,7 @@ void __memcg_kmem_uncharge_page(struct page *page, int order); * needs to be used outside of the local scope. */ struct obj_cgroup *current_obj_cgroup(void); -struct obj_cgroup *get_obj_cgroup_from_folio(struct folio *folio); +struct obj_cgroup *get_obj_cgroup_from_folio(const struct folio *folio); static inline struct obj_cgroup *get_obj_cgroup_from_current(void) { @@ -1870,7 +1870,7 @@ static inline void __memcg_kmem_uncharge_page(struct page *page, int order) { } -static inline struct obj_cgroup *get_obj_cgroup_from_folio(struct folio *folio) +static inline struct obj_cgroup *get_obj_cgroup_from_folio(const struct folio *folio) { return NULL; } @@ -1901,7 +1901,7 @@ static inline void count_objcg_events(struct obj_cgroup *objcg, { } -static inline ino_t page_cgroup_ino(struct page *page) +static inline ino_t page_cgroup_ino(const struct page *page) { return 0; } diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 791e536efaebe8..72c0e9358cfce2 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -348,7 +348,7 @@ EXPORT_SYMBOL(memcg_bpf_enabled_key); * If memcg is bound to a traditional hierarchy, the css of root_mem_cgroup * is returned. */ -struct cgroup_subsys_state *get_mem_cgroup_css_from_folio(struct folio *folio) +struct cgroup_subsys_state *get_mem_cgroup_css_from_folio(const struct folio *folio) { struct mem_cgroup *memcg; @@ -373,7 +373,7 @@ struct cgroup_subsys_state *get_mem_cgroup_css_from_folio(struct folio *folio) * after page_cgroup_ino() returns, so it only should be used by callers that * do not care (such as procfs interfaces). */ -ino_t page_cgroup_ino(struct page *page) +ino_t page_cgroup_ino(const struct page *page) { struct mem_cgroup *memcg; unsigned long ino = 0; @@ -1256,7 +1256,7 @@ struct mem_cgroup *get_mem_cgroup_from_current(void) * * See folio_memcg() for folio->objcg/memcg binding rules. */ -struct mem_cgroup *get_mem_cgroup_from_folio(struct folio *folio) +struct mem_cgroup *get_mem_cgroup_from_folio(const struct folio *folio) { struct mem_cgroup *memcg; @@ -3154,7 +3154,7 @@ __always_inline struct obj_cgroup *current_obj_cgroup(void) return rcu_dereference_check(root_mem_cgroup->nodeinfo[nid]->objcg, 1); } -struct obj_cgroup *get_obj_cgroup_from_folio(struct folio *folio) +struct obj_cgroup *get_obj_cgroup_from_folio(const struct folio *folio) { struct obj_cgroup *objcg; From ea4b2183077801434d51ee125433a24623a7b257 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 15 Sep 2026 19:20:06 -0400 Subject: [PATCH 1043/1352] mm: memcontrol: constify obj_cgroup_memcg() and friends obj_cgroup_memcg() and get_mem_cgroup_from_objcg() only read from the objcg. Constify them. Link: https://lore.kernel.org/20260915-folio_memcg-const-v3-2-c239a6010b58@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Reviewed-by: Muchun Song Acked-by: Shakeel Butt Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Roman Gushchin Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie --- include/linux/memcontrol.h | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 5a5ca6814782a3..65cb45f4be3204 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -396,7 +396,7 @@ enum objext_flags { * * The caller must ensure that the returned memcg won't be released. */ -static inline struct mem_cgroup *obj_cgroup_memcg(struct obj_cgroup *objcg) +static inline struct mem_cgroup *obj_cgroup_memcg(const struct obj_cgroup *objcg) { lockdep_assert_once(rcu_read_lock_held() || lockdep_is_held(&cgroup_mutex)); return objcg ? READ_ONCE(objcg->memcg) : NULL; @@ -505,7 +505,7 @@ static inline struct mem_cgroup *page_memcg_check(const struct page *page) return folio_memcg_check((const struct folio *)page); } -static inline struct mem_cgroup *get_mem_cgroup_from_objcg(struct obj_cgroup *objcg) +static inline struct mem_cgroup *get_mem_cgroup_from_objcg(const struct obj_cgroup *objcg) { struct mem_cgroup *memcg; @@ -1075,7 +1075,7 @@ void mem_cgroup_flush_workqueue(void); extern int mem_cgroup_init(void); #else /* CONFIG_MEMCG */ -static inline struct mem_cgroup *obj_cgroup_memcg(struct obj_cgroup *objcg) +static inline struct mem_cgroup *obj_cgroup_memcg(const struct obj_cgroup *objcg) { return NULL; } @@ -1104,7 +1104,7 @@ static inline struct mem_cgroup *page_memcg_check(const struct page *page) return NULL; } -static inline struct mem_cgroup *get_mem_cgroup_from_objcg(struct obj_cgroup *objcg) +static inline struct mem_cgroup *get_mem_cgroup_from_objcg(const struct obj_cgroup *objcg) { return NULL; } From bc632faf46c27ec818ad114e3b1db296ea260c24 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 15 Sep 2026 19:20:07 -0400 Subject: [PATCH 1044/1352] mm: memcontrol: constify the lruvec helpers The lruvec lookup helpers only read from the memcg, folio, or lruvec they are given. Constify them, along with lruvec_pgdat() and the folio argument of the folio_lruvec_relock_irq() helpers. mem_cgroup_lruvec() may update lruvec->pgdat for a newly onlined node, but that lives in the per-node structure, not the memcg. Similarly, lruvec_pgdat() still returns a non-const pgdat and does not use container_of_const(). Use container_of_const() in lruvec_memcg() and mem_cgroup_get_zone_lru_size(), so the const isn't silently cast away. Link: https://lore.kernel.org/20260915-folio_memcg-const-v3-3-c239a6010b58@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Reviewed-by: Muchun Song Acked-by: Shakeel Butt Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Roman Gushchin Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie --- include/linux/memcontrol.h | 46 +++++++++++++++++++------------------- include/linux/mmzone.h | 2 +- mm/memcontrol.c | 6 ++--- 3 files changed, 27 insertions(+), 27 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 65cb45f4be3204..436934030d796b 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -722,7 +722,7 @@ void mem_cgroup_migrate(struct folio *old, struct folio *new); * @pgdat combination. This can be the node lruvec, if the memory * controller is disabled. */ -static inline struct lruvec *mem_cgroup_lruvec(struct mem_cgroup *memcg, +static inline struct lruvec *mem_cgroup_lruvec(const struct mem_cgroup *memcg, struct pglist_data *pgdat) { struct mem_cgroup_per_node *mz; @@ -763,7 +763,7 @@ static inline struct lruvec *mem_cgroup_lruvec(struct mem_cgroup *memcg, * their binding is stable if the returned lruvec matches the one the caller has * locked. Useful for lock batching. */ -static inline struct lruvec *folio_lruvec(struct folio *folio) +static inline struct lruvec *folio_lruvec(const struct folio *folio) { struct mem_cgroup *memcg = folio_memcg(folio); @@ -779,9 +779,9 @@ struct mem_cgroup *get_mem_cgroup_from_current(void); struct mem_cgroup *get_mem_cgroup_from_folio(const struct folio *folio); -struct lruvec *folio_lruvec_lock(struct folio *folio); -struct lruvec *folio_lruvec_lock_irq(struct folio *folio); -struct lruvec *folio_lruvec_lock_irqsave(struct folio *folio, +struct lruvec *folio_lruvec_lock(const struct folio *folio); +struct lruvec *folio_lruvec_lock_irq(const struct folio *folio); +struct lruvec *folio_lruvec_lock_irqsave(const struct folio *folio, unsigned long *flags); static inline @@ -861,14 +861,14 @@ static inline struct mem_cgroup *mem_cgroup_from_seq(struct seq_file *m) return mem_cgroup_from_css(seq_css(m)); } -static inline struct mem_cgroup *lruvec_memcg(struct lruvec *lruvec) +static inline struct mem_cgroup *lruvec_memcg(const struct lruvec *lruvec) { - struct mem_cgroup_per_node *mz; + const struct mem_cgroup_per_node *mz; if (mem_cgroup_disabled()) return NULL; - mz = container_of(lruvec, struct mem_cgroup_per_node, lruvec); + mz = container_of_const(lruvec, struct mem_cgroup_per_node, lruvec); return mz->memcg; } @@ -919,13 +919,13 @@ void mem_cgroup_update_lru_size(struct lruvec *lruvec, enum lru_list lru, int zid, long nr_pages); static inline -unsigned long mem_cgroup_get_zone_lru_size(struct lruvec *lruvec, +unsigned long mem_cgroup_get_zone_lru_size(const struct lruvec *lruvec, enum lru_list lru, int zone_idx) { long val; - struct mem_cgroup_per_node *mz; + const struct mem_cgroup_per_node *mz; - mz = container_of(lruvec, struct mem_cgroup_per_node, lruvec); + mz = container_of_const(lruvec, struct mem_cgroup_per_node, lruvec); val = READ_ONCE(mz->lru_zone_size[zone_idx][lru]); if (WARN_ON_ONCE(val < 0)) return 0; @@ -1215,13 +1215,13 @@ static inline void mem_cgroup_migrate(struct folio *old, struct folio *new) { } -static inline struct lruvec *mem_cgroup_lruvec(struct mem_cgroup *memcg, +static inline struct lruvec *mem_cgroup_lruvec(const struct mem_cgroup *memcg, struct pglist_data *pgdat) { return &pgdat->__lruvec; } -static inline struct lruvec *folio_lruvec(struct folio *folio) +static inline struct lruvec *folio_lruvec(const struct folio *folio) { struct pglist_data *pgdat = folio_pgdat(folio); return &pgdat->__lruvec; @@ -1281,7 +1281,7 @@ static inline void mem_cgroup_put(struct mem_cgroup *memcg) { } -static inline struct lruvec *folio_lruvec_lock(struct folio *folio) +static inline struct lruvec *folio_lruvec_lock(const struct folio *folio) { struct pglist_data *pgdat = folio_pgdat(folio); @@ -1290,7 +1290,7 @@ static inline struct lruvec *folio_lruvec_lock(struct folio *folio) return &pgdat->__lruvec; } -static inline struct lruvec *folio_lruvec_lock_irq(struct folio *folio) +static inline struct lruvec *folio_lruvec_lock_irq(const struct folio *folio) { struct pglist_data *pgdat = folio_pgdat(folio); @@ -1299,7 +1299,7 @@ static inline struct lruvec *folio_lruvec_lock_irq(struct folio *folio) return &pgdat->__lruvec; } -static inline struct lruvec *folio_lruvec_lock_irqsave(struct folio *folio, +static inline struct lruvec *folio_lruvec_lock_irqsave(const struct folio *folio, unsigned long *flagsp) { struct pglist_data *pgdat = folio_pgdat(folio); @@ -1354,7 +1354,7 @@ static inline struct mem_cgroup *mem_cgroup_from_seq(struct seq_file *m) return NULL; } -static inline struct mem_cgroup *lruvec_memcg(struct lruvec *lruvec) +static inline struct mem_cgroup *lruvec_memcg(const struct lruvec *lruvec) { return NULL; } @@ -1365,7 +1365,7 @@ static inline bool mem_cgroup_online(struct mem_cgroup *memcg) } static inline -unsigned long mem_cgroup_get_zone_lru_size(struct lruvec *lruvec, +unsigned long mem_cgroup_get_zone_lru_size(const struct lruvec *lruvec, enum lru_list lru, int zone_idx) { return 0; @@ -1505,7 +1505,7 @@ static inline void mem_cgroup_flush_workqueue(void) { } static inline int mem_cgroup_init(void) { return 0; } #endif /* CONFIG_MEMCG */ -static inline struct lruvec *parent_lruvec(struct lruvec *lruvec) +static inline struct lruvec *parent_lruvec(const struct lruvec *lruvec) { struct mem_cgroup *memcg; @@ -1568,15 +1568,15 @@ static inline void lruvec_unlock_irqrestore(struct lruvec *lruvec, unsigned long } /* Test requires a stable folio->memcg binding, see folio_memcg() */ -static inline bool folio_matches_lruvec(struct folio *folio, - struct lruvec *lruvec) +static inline bool folio_matches_lruvec(const struct folio *folio, + const struct lruvec *lruvec) { return lruvec_pgdat(lruvec) == folio_pgdat(folio) && lruvec_memcg(lruvec) == folio_memcg(folio); } /* Don't lock again iff page's lruvec locked */ -static inline struct lruvec *folio_lruvec_relock_irq(struct folio *folio, +static inline struct lruvec *folio_lruvec_relock_irq(const struct folio *folio, struct lruvec *locked_lruvec) { if (locked_lruvec) { @@ -1590,7 +1590,7 @@ static inline struct lruvec *folio_lruvec_relock_irq(struct folio *folio, } /* Don't lock again iff folio's lruvec locked */ -static inline void folio_lruvec_relock_irqsave(struct folio *folio, +static inline void folio_lruvec_relock_irqsave(const struct folio *folio, struct lruvec **lruvecp, unsigned long *flags) { if (*lruvecp) { diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 65de3bb13eb3ae..8e4e0bda3b586b 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -1638,7 +1638,7 @@ extern void init_currently_empty_zone(struct zone *zone, unsigned long start_pfn extern void lruvec_init(struct lruvec *lruvec); -static inline struct pglist_data *lruvec_pgdat(struct lruvec *lruvec) +static inline struct pglist_data *lruvec_pgdat(const struct lruvec *lruvec) { #ifdef CONFIG_MEMCG return lruvec->pgdat; diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 72c0e9358cfce2..294827ef54a54c 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -1477,7 +1477,7 @@ void mem_cgroup_scan_tasks(struct mem_cgroup *memcg, * * Return: The lruvec this folio is on with its lock held and rcu read lock held. */ -struct lruvec *folio_lruvec_lock(struct folio *folio) +struct lruvec *folio_lruvec_lock(const struct folio *folio) { struct lruvec *lruvec; @@ -1505,7 +1505,7 @@ struct lruvec *folio_lruvec_lock(struct folio *folio) * Return: The lruvec this folio is on with its lock held and interrupts * disabled and rcu read lock held. */ -struct lruvec *folio_lruvec_lock_irq(struct folio *folio) +struct lruvec *folio_lruvec_lock_irq(const struct folio *folio) { struct lruvec *lruvec; @@ -1534,7 +1534,7 @@ struct lruvec *folio_lruvec_lock_irq(struct folio *folio) * Return: The lruvec this folio is on with its lock held and interrupts * disabled and rcu read lock held. */ -struct lruvec *folio_lruvec_lock_irqsave(struct folio *folio, +struct lruvec *folio_lruvec_lock_irqsave(const struct folio *folio, unsigned long *flags) { struct lruvec *lruvec; From 6f2a068608588a74397632669680a844bb1fc16e Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 15 Sep 2026 19:20:08 -0400 Subject: [PATCH 1045/1352] mm/page_io: take a const folio in bio_associate_blkg_from_folio() bio_associate_blkg_from_folio() and its helpers only read from the folio. Constify their folio arguments. Link: https://lore.kernel.org/20260915-folio_memcg-const-v3-4-c239a6010b58@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Reviewed-by: Muchun Song Acked-by: Shakeel Butt Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Roman Gushchin Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie --- mm/page_io.c | 14 +++++++++----- 1 file changed, 9 insertions(+), 5 deletions(-) diff --git a/mm/page_io.c b/mm/page_io.c index 1da4ff484f0971..5f7756e370f7a4 100644 --- a/mm/page_io.c +++ b/mm/page_io.c @@ -256,12 +256,13 @@ int swap_writeout(struct swap_io_ctx *ctx, struct folio *folio) } #if defined(CONFIG_MEMCG) && defined(CONFIG_BLK_CGROUP) -static struct cgroup_subsys_state *folio_memcg_blkg_css(struct folio *folio) +static struct cgroup_subsys_state *folio_memcg_blkg_css(const struct folio *folio) { return cgroup_e_css(folio_memcg(folio)->css.cgroup, &io_cgrp_subsys); } -static bool folio_blkg_can_merge(struct folio *folio, struct folio *prev_folio) +static bool folio_blkg_can_merge(const struct folio *folio, + const struct folio *prev_folio) { bool can_merge = true; @@ -277,7 +278,8 @@ static bool folio_blkg_can_merge(struct folio *folio, struct folio *prev_folio) return can_merge; } -static void bio_associate_blkg_from_folio(struct bio *bio, struct folio *folio) +static void bio_associate_blkg_from_folio(struct bio *bio, + const struct folio *folio) { struct cgroup_subsys_state *css; @@ -294,11 +296,13 @@ static void bio_associate_blkg_from_folio(struct bio *bio, struct folio *folio) css_put(css); } #else -static bool folio_blkg_can_merge(struct folio *folio, struct folio *prev_folio) +static bool folio_blkg_can_merge(const struct folio *folio, + const struct folio *prev_folio) { return true; } -static void bio_associate_blkg_from_folio(struct bio *bio, struct folio *folio) +static void bio_associate_blkg_from_folio(struct bio *bio, + const struct folio *folio) { } #endif /* CONFIG_MEMCG && CONFIG_BLK_CGROUP */ From bea8ee5e88d41dc312ca4e20d04687149d49ada5 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 15 Sep 2026 19:20:09 -0400 Subject: [PATCH 1046/1352] mm: memcontrol: constify the mem_cgroup accessors mem_cgroup_id(), parent_mem_cgroup(), mem_cgroup_is_root(), mem_cgroup_is_descendant(), memcg_kmem_id(), and the other memcg accessors only read from the memcg. Constify them, along with mem_cgroup_shrink_is_root()'s shrink_control. mem_cgroup_print_oom_context()'s task stays non-const, as task_cgroup() takes a non-const task. Link: https://lore.kernel.org/20260915-folio_memcg-const-v3-5-c239a6010b58@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: Muchun Song Acked-by: Shakeel Butt Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Roman Gushchin Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie --- include/linux/memcontrol.h | 41 +++++++++++++++++++------------------- mm/memcontrol.c | 3 ++- 2 files changed, 23 insertions(+), 21 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 436934030d796b..01f74413fc79e9 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -539,7 +539,7 @@ static inline bool PageMemcgKmem(const struct page *page) return folio_memcg_kmem(page_folio(page)); } -static inline bool mem_cgroup_is_root(struct mem_cgroup *memcg) +static inline bool mem_cgroup_is_root(const struct mem_cgroup *memcg) { return (memcg == root_mem_cgroup); } @@ -555,7 +555,7 @@ static inline bool mem_cgroup_is_root(struct mem_cgroup *memcg) * and do not honour sc->memcg can use this to early-return 0 in per-memcg * contexts. */ -static inline bool mem_cgroup_shrink_is_root(struct shrink_control *sc) +static inline bool mem_cgroup_shrink_is_root(const struct shrink_control *sc) { return !sc->memcg || mem_cgroup_is_root(sc->memcg); } @@ -840,7 +840,7 @@ void mem_cgroup_iter_break(struct mem_cgroup *, struct mem_cgroup *); void mem_cgroup_scan_tasks(struct mem_cgroup *memcg, int (*)(struct task_struct *, void *), void *arg); -static inline unsigned short mem_cgroup_private_id(struct mem_cgroup *memcg) +static inline unsigned short mem_cgroup_private_id(const struct mem_cgroup *memcg) { if (mem_cgroup_disabled()) return 0; @@ -849,7 +849,7 @@ static inline unsigned short mem_cgroup_private_id(struct mem_cgroup *memcg) } struct mem_cgroup *mem_cgroup_from_private_id(unsigned short id); -static inline u64 mem_cgroup_id(struct mem_cgroup *memcg) +static inline u64 mem_cgroup_id(const struct mem_cgroup *memcg) { return memcg ? cgroup_id(memcg->css.cgroup) : 0; } @@ -878,13 +878,13 @@ static inline struct mem_cgroup *lruvec_memcg(const struct lruvec *lruvec) * * Returns the parent memcg, or NULL if this is the root. */ -static inline struct mem_cgroup *parent_mem_cgroup(struct mem_cgroup *memcg) +static inline struct mem_cgroup *parent_mem_cgroup(const struct mem_cgroup *memcg) { return mem_cgroup_from_css(memcg->css.parent); } -static inline bool mem_cgroup_is_descendant(struct mem_cgroup *memcg, - struct mem_cgroup *root) +static inline bool mem_cgroup_is_descendant(const struct mem_cgroup *memcg, + const struct mem_cgroup *root) { if (root == memcg) return true; @@ -892,7 +892,7 @@ static inline bool mem_cgroup_is_descendant(struct mem_cgroup *memcg, } static inline bool mm_match_cgroup(struct mm_struct *mm, - struct mem_cgroup *memcg) + const struct mem_cgroup *memcg) { struct mem_cgroup *task_memcg; bool match = false; @@ -943,7 +943,7 @@ static inline void mem_cgroup_handle_over_high(gfp_t gfp_mask) unsigned long mem_cgroup_get_max(struct mem_cgroup *memcg); -void mem_cgroup_print_oom_context(struct mem_cgroup *memcg, +void mem_cgroup_print_oom_context(const struct mem_cgroup *memcg, struct task_struct *p); void mem_cgroup_print_oom_meminfo(struct mem_cgroup *memcg); @@ -1119,12 +1119,12 @@ static inline bool PageMemcgKmem(const struct page *page) return false; } -static inline bool mem_cgroup_is_root(struct mem_cgroup *memcg) +static inline bool mem_cgroup_is_root(const struct mem_cgroup *memcg) { return true; } -static inline bool mem_cgroup_shrink_is_root(struct shrink_control *sc) +static inline bool mem_cgroup_shrink_is_root(const struct shrink_control *sc) { return true; } @@ -1227,13 +1227,13 @@ static inline struct lruvec *folio_lruvec(const struct folio *folio) return &pgdat->__lruvec; } -static inline struct mem_cgroup *parent_mem_cgroup(struct mem_cgroup *memcg) +static inline struct mem_cgroup *parent_mem_cgroup(const struct mem_cgroup *memcg) { return NULL; } static inline bool mm_match_cgroup(struct mm_struct *mm, - struct mem_cgroup *memcg) + const struct mem_cgroup *memcg) { return true; } @@ -1327,7 +1327,7 @@ static inline void mem_cgroup_scan_tasks(struct mem_cgroup *memcg, { } -static inline unsigned short mem_cgroup_private_id(struct mem_cgroup *memcg) +static inline unsigned short mem_cgroup_private_id(const struct mem_cgroup *memcg) { return 0; } @@ -1339,7 +1339,7 @@ static inline struct mem_cgroup *mem_cgroup_from_private_id(unsigned short id) return NULL; } -static inline u64 mem_cgroup_id(struct mem_cgroup *memcg) +static inline u64 mem_cgroup_id(const struct mem_cgroup *memcg) { return 0; } @@ -1377,7 +1377,8 @@ static inline unsigned long mem_cgroup_get_max(struct mem_cgroup *memcg) } static inline void -mem_cgroup_print_oom_context(struct mem_cgroup *memcg, struct task_struct *p) +mem_cgroup_print_oom_context(const struct mem_cgroup *memcg, + struct task_struct *p) { } @@ -1813,7 +1814,7 @@ static inline void memcg_kmem_uncharge_page(struct page *page, int order) * A helper for accessing memcg's kmem_id, used for getting * corresponding LRU lists. */ -static inline int memcg_kmem_id(struct mem_cgroup *memcg) +static inline int memcg_kmem_id(const struct mem_cgroup *memcg) { return memcg ? memcg->kmemcg_id : -1; } @@ -1885,7 +1886,7 @@ static inline bool memcg_kmem_online(void) return false; } -static inline int memcg_kmem_id(struct mem_cgroup *memcg) +static inline int memcg_kmem_id(const struct mem_cgroup *memcg) { return -1; } @@ -1962,7 +1963,7 @@ static inline bool mem_cgroup_zswap_writeback_enabled(struct mem_cgroup *memcg) #ifdef CONFIG_MEMCG_V1 bool mem_cgroup_oom_synchronize(bool wait); -static inline bool task_in_memcg_oom(struct task_struct *p) +static inline bool task_in_memcg_oom(const struct task_struct *p) { return p->memcg_in_oom; } @@ -1980,7 +1981,7 @@ static inline void mem_cgroup_exit_user_fault(void) } #else /* CONFIG_MEMCG_V1 */ -static inline bool task_in_memcg_oom(struct task_struct *p) +static inline bool task_in_memcg_oom(const struct task_struct *p) { return false; } diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 294827ef54a54c..adc93d28dbe72b 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -1832,7 +1832,8 @@ static void memory_stat_format(struct mem_cgroup *memcg, struct seq_buf *s) * NOTE: @memcg and @p's mem_cgroup can be different when hierarchy is * enabled */ -void mem_cgroup_print_oom_context(struct mem_cgroup *memcg, struct task_struct *p) +void mem_cgroup_print_oom_context(const struct mem_cgroup *memcg, + struct task_struct *p) { rcu_read_lock(); From dc6074d31ccff9ed8031122570c61b1418e97024 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 15 Sep 2026 19:20:10 -0400 Subject: [PATCH 1047/1352] mm: page_counter: constify page_counter_read() and page_counter_margin() Both only read the counter. Constify them so that users can read counters from const memcgs. Link: https://lore.kernel.org/20260915-folio_memcg-const-v3-6-c239a6010b58@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Reviewed-by: Muchun Song Acked-by: Shakeel Butt Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Roman Gushchin Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie --- include/linux/page_counter.h | 4 ++-- mm/page_counter.c | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/include/linux/page_counter.h b/include/linux/page_counter.h index 07b7cb12249c7c..2baf7a2b29b2e1 100644 --- a/include/linux/page_counter.h +++ b/include/linux/page_counter.h @@ -63,12 +63,12 @@ static inline void page_counter_init(struct page_counter *counter, counter->track_failcnt = false; } -static inline unsigned long page_counter_read(struct page_counter *counter) +static inline unsigned long page_counter_read(const struct page_counter *counter) { return atomic_long_read(&counter->usage); } -long page_counter_margin(struct page_counter *counter); +long page_counter_margin(const struct page_counter *counter); void page_counter_cancel(struct page_counter *counter, unsigned long nr_pages); void page_counter_charge(struct page_counter *counter, unsigned long nr_pages); bool page_counter_try_charge(struct page_counter *counter, diff --git a/mm/page_counter.c b/mm/page_counter.c index 98322803941a70..9167ffd1380c82 100644 --- a/mm/page_counter.c +++ b/mm/page_counter.c @@ -54,7 +54,7 @@ static void propagate_protected_usage(struct page_counter *c, * Return: The minimum value of max minus usage across @counter and all of * its ancestors. The value may be negative during a concurrent charge. */ -long page_counter_margin(struct page_counter *counter) +long page_counter_margin(const struct page_counter *counter) { long margin = PAGE_COUNTER_MAX; From 0cc02533665932e02ae258669d912421a13df552 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 15 Sep 2026 19:20:11 -0400 Subject: [PATCH 1048/1352] mm: memcontrol: constify the reclaim protection helpers mem_cgroup_protection(), mem_cgroup_unprotected(), mem_cgroup_below_low(), and mem_cgroup_below_min() only read the protection state. Constify them. Link: https://lore.kernel.org/20260915-folio_memcg-const-v3-7-c239a6010b58@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: Muchun Song Acked-by: Shakeel Butt Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Roman Gushchin Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie --- include/linux/memcontrol.h | 32 ++++++++++++++++---------------- 1 file changed, 16 insertions(+), 16 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 01f74413fc79e9..22067899eb6c34 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -570,8 +570,8 @@ static inline bool mem_cgroup_disabled(void) return !cgroup_subsys_enabled(memory_cgrp_subsys); } -static inline void mem_cgroup_protection(struct mem_cgroup *root, - struct mem_cgroup *memcg, +static inline void mem_cgroup_protection(const struct mem_cgroup *root, + const struct mem_cgroup *memcg, unsigned long *min, unsigned long *low, unsigned long *usage) @@ -625,8 +625,8 @@ static inline void mem_cgroup_protection(struct mem_cgroup *root, void mem_cgroup_calculate_protection(struct mem_cgroup *root, struct mem_cgroup *memcg); -static inline bool mem_cgroup_unprotected(struct mem_cgroup *target, - struct mem_cgroup *memcg) +static inline bool mem_cgroup_unprotected(const struct mem_cgroup *target, + const struct mem_cgroup *memcg) { /* * The root memcg doesn't account charges, and doesn't support @@ -637,8 +637,8 @@ static inline bool mem_cgroup_unprotected(struct mem_cgroup *target, memcg == target; } -static inline bool mem_cgroup_below_low(struct mem_cgroup *target, - struct mem_cgroup *memcg) +static inline bool mem_cgroup_below_low(const struct mem_cgroup *target, + const struct mem_cgroup *memcg) { if (mem_cgroup_unprotected(target, memcg)) return false; @@ -647,8 +647,8 @@ static inline bool mem_cgroup_below_low(struct mem_cgroup *target, page_counter_read(&memcg->memory); } -static inline bool mem_cgroup_below_min(struct mem_cgroup *target, - struct mem_cgroup *memcg) +static inline bool mem_cgroup_below_min(const struct mem_cgroup *target, + const struct mem_cgroup *memcg) { if (mem_cgroup_unprotected(target, memcg)) return false; @@ -1149,8 +1149,8 @@ static inline void memcg_memory_event_mm(struct mm_struct *mm, { } -static inline void mem_cgroup_protection(struct mem_cgroup *root, - struct mem_cgroup *memcg, +static inline void mem_cgroup_protection(const struct mem_cgroup *root, + const struct mem_cgroup *memcg, unsigned long *min, unsigned long *low, unsigned long *usage) @@ -1163,19 +1163,19 @@ static inline void mem_cgroup_calculate_protection(struct mem_cgroup *root, { } -static inline bool mem_cgroup_unprotected(struct mem_cgroup *target, - struct mem_cgroup *memcg) +static inline bool mem_cgroup_unprotected(const struct mem_cgroup *target, + const struct mem_cgroup *memcg) { return true; } -static inline bool mem_cgroup_below_low(struct mem_cgroup *target, - struct mem_cgroup *memcg) +static inline bool mem_cgroup_below_low(const struct mem_cgroup *target, + const struct mem_cgroup *memcg) { return false; } -static inline bool mem_cgroup_below_min(struct mem_cgroup *target, - struct mem_cgroup *memcg) +static inline bool mem_cgroup_below_min(const struct mem_cgroup *target, + const struct mem_cgroup *memcg) { return false; } From fd45b427db7357bb88ef3556f053c9d60bc4bdd9 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 15 Sep 2026 19:20:12 -0400 Subject: [PATCH 1049/1352] mm: memcontrol: constify the memcg and lruvec stat readers The memcg_page_state(), memcg_events(), and lruvec_page_state() families only read counters. Constify them. Use container_of_const() in the lruvec_page_state() family while at it, so the const isn't silently cast away. Link: https://lore.kernel.org/20260915-folio_memcg-const-v3-8-c239a6010b58@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: Muchun Song Acked-by: Shakeel Butt Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Roman Gushchin Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie --- include/linux/memcontrol.h | 23 ++++++++++++----------- mm/memcontrol-v1.h | 6 +++--- mm/memcontrol.c | 30 +++++++++++++++--------------- 3 files changed, 30 insertions(+), 29 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 22067899eb6c34..9beb065c087935 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -971,15 +971,16 @@ static inline void mod_memcg_page_state(const struct page *page, rcu_read_unlock(); } -unsigned long memcg_events(struct mem_cgroup *memcg, int event); -unsigned long memcg_page_state(struct mem_cgroup *memcg, int idx); -unsigned long memcg_page_state_output(struct mem_cgroup *memcg, int item); +unsigned long memcg_events(const struct mem_cgroup *memcg, int event); +unsigned long memcg_page_state(const struct mem_cgroup *memcg, int idx); +unsigned long memcg_page_state_output(const struct mem_cgroup *memcg, int item); bool memcg_stat_item_valid(int idx); bool memcg_vm_event_item_valid(enum vm_event_item idx); -unsigned long lruvec_page_state(struct lruvec *lruvec, enum node_stat_item idx); -unsigned long lruvec_page_state_monotonic(struct lruvec *lruvec, +unsigned long lruvec_page_state(const struct lruvec *lruvec, + enum node_stat_item idx); +unsigned long lruvec_page_state_monotonic(const struct lruvec *lruvec, enum node_stat_item idx); -unsigned long lruvec_page_state_local(struct lruvec *lruvec, +unsigned long lruvec_page_state_local(const struct lruvec *lruvec, enum node_stat_item idx); void mem_cgroup_flush_stats(struct mem_cgroup *memcg); @@ -1412,12 +1413,12 @@ static inline void mod_memcg_page_state(const struct page *page, { } -static inline unsigned long memcg_page_state(struct mem_cgroup *memcg, int idx) +static inline unsigned long memcg_page_state(const struct mem_cgroup *memcg, int idx) { return 0; } -static inline unsigned long memcg_page_state_output(struct mem_cgroup *memcg, int item) +static inline unsigned long memcg_page_state_output(const struct mem_cgroup *memcg, int item) { return 0; } @@ -1432,19 +1433,19 @@ static inline bool memcg_vm_event_item_valid(enum vm_event_item idx) return false; } -static inline unsigned long lruvec_page_state(struct lruvec *lruvec, +static inline unsigned long lruvec_page_state(const struct lruvec *lruvec, enum node_stat_item idx) { return node_page_state(lruvec_pgdat(lruvec), idx); } -static inline unsigned long lruvec_page_state_monotonic(struct lruvec *lruvec, +static inline unsigned long lruvec_page_state_monotonic(const struct lruvec *lruvec, enum node_stat_item idx) { return node_page_state_monotonic(lruvec_pgdat(lruvec), idx); } -static inline unsigned long lruvec_page_state_local(struct lruvec *lruvec, +static inline unsigned long lruvec_page_state_local(const struct lruvec *lruvec, enum node_stat_item idx) { return node_page_state(lruvec_pgdat(lruvec), idx); diff --git a/mm/memcontrol-v1.h b/mm/memcontrol-v1.h index 2cd37e1792d79e..0952b2a783e524 100644 --- a/mm/memcontrol-v1.h +++ b/mm/memcontrol-v1.h @@ -37,9 +37,9 @@ static inline bool do_memsw_account(void) return !cgroup_subsys_on_dfl(memory_cgrp_subsys); } -unsigned long memcg_events_local(struct mem_cgroup *memcg, int event); -unsigned long memcg_page_state_local(struct mem_cgroup *memcg, int idx); -unsigned long memcg_page_state_local_output(struct mem_cgroup *memcg, int item); +unsigned long memcg_events_local(const struct mem_cgroup *memcg, int event); +unsigned long memcg_page_state_local(const struct mem_cgroup *memcg, int idx); +unsigned long memcg_page_state_local_output(const struct mem_cgroup *memcg, int item); bool memcg1_alloc_events(struct mem_cgroup *memcg); void memcg1_free_events(struct mem_cgroup *memcg); diff --git a/mm/memcontrol.c b/mm/memcontrol.c index adc93d28dbe72b..1b0e511a18638b 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -506,9 +506,9 @@ struct lruvec_stats { long state_pending[NR_MEMCG_NODE_STAT_ITEMS]; }; -unsigned long lruvec_page_state(struct lruvec *lruvec, enum node_stat_item idx) +unsigned long lruvec_page_state(const struct lruvec *lruvec, enum node_stat_item idx) { - struct mem_cgroup_per_node *pn; + const struct mem_cgroup_per_node *pn; long x; int i; @@ -519,7 +519,7 @@ unsigned long lruvec_page_state(struct lruvec *lruvec, enum node_stat_item idx) if (WARN_ONCE(BAD_STAT_IDX(i), "%s: missing stat item %d\n", __func__, idx)) return 0; - pn = container_of(lruvec, struct mem_cgroup_per_node, lruvec); + pn = container_of_const(lruvec, struct mem_cgroup_per_node, lruvec); x = READ_ONCE(pn->lruvec_stats->state[i]); #ifdef CONFIG_SMP if (x < 0) @@ -547,10 +547,10 @@ unsigned long lruvec_page_state(struct lruvec *lruvec, enum node_stat_item idx) * monotonically-incremented event counters are stored in * enum node_stat_item. */ -unsigned long lruvec_page_state_monotonic(struct lruvec *lruvec, +unsigned long lruvec_page_state_monotonic(const struct lruvec *lruvec, enum node_stat_item idx) { - struct mem_cgroup_per_node *pn; + const struct mem_cgroup_per_node *pn; int i; if (mem_cgroup_disabled()) @@ -560,14 +560,14 @@ unsigned long lruvec_page_state_monotonic(struct lruvec *lruvec, if (WARN_ONCE(BAD_STAT_IDX(i), "%s: missing stat item %d\n", __func__, idx)) return 0; - pn = container_of(lruvec, struct mem_cgroup_per_node, lruvec); + pn = container_of_const(lruvec, struct mem_cgroup_per_node, lruvec); return (unsigned long)READ_ONCE(pn->lruvec_stats->state[i]); } -unsigned long lruvec_page_state_local(struct lruvec *lruvec, +unsigned long lruvec_page_state_local(const struct lruvec *lruvec, enum node_stat_item idx) { - struct mem_cgroup_per_node *pn; + const struct mem_cgroup_per_node *pn; long x; int i; @@ -578,7 +578,7 @@ unsigned long lruvec_page_state_local(struct lruvec *lruvec, if (WARN_ONCE(BAD_STAT_IDX(i), "%s: missing stat item %d\n", __func__, idx)) return 0; - pn = container_of(lruvec, struct mem_cgroup_per_node, lruvec); + pn = container_of_const(lruvec, struct mem_cgroup_per_node, lruvec); x = READ_ONCE(pn->lruvec_stats->state_local[i]); #ifdef CONFIG_SMP if (x < 0) @@ -841,7 +841,7 @@ static void flush_memcg_stats_dwork(struct work_struct *w) queue_delayed_work(system_dfl_wq, &stats_flush_dwork, FLUSH_TIME); } -unsigned long memcg_page_state(struct mem_cgroup *memcg, int idx) +unsigned long memcg_page_state(const struct mem_cgroup *memcg, int idx) { long x; int i = memcg_stats_index(idx); @@ -964,7 +964,7 @@ void mod_memcg_state(struct mem_cgroup *memcg, enum memcg_stat_item idx, #ifdef CONFIG_MEMCG_V1 /* idx can be of type enum memcg_stat_item or node_stat_item. */ -unsigned long memcg_page_state_local(struct mem_cgroup *memcg, int idx) +unsigned long memcg_page_state_local(const struct mem_cgroup *memcg, int idx) { long x; int i = memcg_stats_index(idx); @@ -1127,7 +1127,7 @@ void count_memcg_events(struct mem_cgroup *memcg, enum vm_event_item idx, put_cpu(); } -unsigned long memcg_events(struct mem_cgroup *memcg, int event) +unsigned long memcg_events(const struct mem_cgroup *memcg, int event) { int i = memcg_events_index(event); @@ -1146,7 +1146,7 @@ bool memcg_vm_event_item_valid(enum vm_event_item idx) } #ifdef CONFIG_MEMCG_V1 -unsigned long memcg_events_local(struct mem_cgroup *memcg, int event) +unsigned long memcg_events_local(const struct mem_cgroup *memcg, int event) { int i = memcg_events_index(event); @@ -1729,14 +1729,14 @@ static int memcg_page_state_output_unit(int item) } } -unsigned long memcg_page_state_output(struct mem_cgroup *memcg, int item) +unsigned long memcg_page_state_output(const struct mem_cgroup *memcg, int item) { return memcg_page_state(memcg, item) * memcg_page_state_output_unit(item); } #ifdef CONFIG_MEMCG_V1 -unsigned long memcg_page_state_local_output(struct mem_cgroup *memcg, int item) +unsigned long memcg_page_state_local_output(const struct mem_cgroup *memcg, int item) { return memcg_page_state_local(memcg, item) * memcg_page_state_output_unit(item); From e20aeee6fee5dd438ad7b5f6eb23a4382c8eaa82 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 15 Sep 2026 19:20:13 -0400 Subject: [PATCH 1050/1352] mm: memcontrol: constify the swap accounting helpers mem_cgroup_get_nr_swap_pages(), mem_cgroup_get_folio_swap_margin(), and mem_cgroup_swap_full() only read swap counters and limits. Constify them. Remove externs from function declarations while at it. Link: https://lore.kernel.org/20260915-folio_memcg-const-v3-9-c239a6010b58@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: Muchun Song Acked-by: Shakeel Butt Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Roman Gushchin Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie --- include/linux/swap.h | 12 ++++++------ mm/memcontrol.c | 6 +++--- 2 files changed, 9 insertions(+), 9 deletions(-) diff --git a/include/linux/swap.h b/include/linux/swap.h index 61005501888c53..82f0bfec611fc7 100644 --- a/include/linux/swap.h +++ b/include/linux/swap.h @@ -527,9 +527,9 @@ static inline void mem_cgroup_uncharge_swap(unsigned short id, unsigned int nr_p __mem_cgroup_uncharge_swap(id, nr_pages); } -long mem_cgroup_get_folio_swap_margin(struct folio *folio); -extern long mem_cgroup_get_nr_swap_pages(struct mem_cgroup *memcg); -extern bool mem_cgroup_swap_full(struct folio *folio); +long mem_cgroup_get_folio_swap_margin(const struct folio *folio); +long mem_cgroup_get_nr_swap_pages(const struct mem_cgroup *memcg); +bool mem_cgroup_swap_full(const struct folio *folio); #else static inline int mem_cgroup_try_charge_swap(struct folio *folio) { @@ -541,17 +541,17 @@ static inline void mem_cgroup_uncharge_swap(unsigned short id, { } -static inline long mem_cgroup_get_folio_swap_margin(struct folio *folio) +static inline long mem_cgroup_get_folio_swap_margin(const struct folio *folio) { return PAGE_COUNTER_MAX; } -static inline long mem_cgroup_get_nr_swap_pages(struct mem_cgroup *memcg) +static inline long mem_cgroup_get_nr_swap_pages(const struct mem_cgroup *memcg) { return get_nr_swap_pages(); } -static inline bool mem_cgroup_swap_full(struct folio *folio) +static inline bool mem_cgroup_swap_full(const struct folio *folio) { return vm_swap_full(); } diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 1b0e511a18638b..d954ffb43a4323 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -6017,7 +6017,7 @@ void __mem_cgroup_uncharge_swap(unsigned short id, unsigned int nr_pages) rcu_read_unlock(); } -long mem_cgroup_get_nr_swap_pages(struct mem_cgroup *memcg) +long mem_cgroup_get_nr_swap_pages(const struct mem_cgroup *memcg) { long nr_swap_pages = get_nr_swap_pages(); @@ -6033,7 +6033,7 @@ long mem_cgroup_get_nr_swap_pages(struct mem_cgroup *memcg) * * Return: Remaining chargeable pages in the folio's memcg hierarchy. */ -long mem_cgroup_get_folio_swap_margin(struct folio *folio) +long mem_cgroup_get_folio_swap_margin(const struct folio *folio) { struct mem_cgroup *memcg; long margin; @@ -6050,7 +6050,7 @@ long mem_cgroup_get_folio_swap_margin(struct folio *folio) return margin; } -bool mem_cgroup_swap_full(struct folio *folio) +bool mem_cgroup_swap_full(const struct folio *folio) { struct mem_cgroup *memcg; bool ret = false; From f26be9fe5a588cd22d340232ec17af7a48358538 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 15 Sep 2026 19:20:14 -0400 Subject: [PATCH 1051/1352] mm: memcontrol: constify mem_cgroup_swappiness() and mem_cgroup_get_max() Both only read swappiness and the memory limits. Constify them. Link: https://lore.kernel.org/20260915-folio_memcg-const-v3-10-c239a6010b58@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: Muchun Song Acked-by: Shakeel Butt Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Roman Gushchin Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie --- include/linux/memcontrol.h | 4 ++-- mm/memcontrol.c | 2 +- mm/swap.h | 2 +- 3 files changed, 4 insertions(+), 4 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 9beb065c087935..f118990854734e 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -941,7 +941,7 @@ static inline void mem_cgroup_handle_over_high(gfp_t gfp_mask) __mem_cgroup_handle_over_high(gfp_mask); } -unsigned long mem_cgroup_get_max(struct mem_cgroup *memcg); +unsigned long mem_cgroup_get_max(const struct mem_cgroup *memcg); void mem_cgroup_print_oom_context(const struct mem_cgroup *memcg, struct task_struct *p); @@ -1372,7 +1372,7 @@ unsigned long mem_cgroup_get_zone_lru_size(const struct lruvec *lruvec, return 0; } -static inline unsigned long mem_cgroup_get_max(struct mem_cgroup *memcg) +static inline unsigned long mem_cgroup_get_max(const struct mem_cgroup *memcg) { return 0; } diff --git a/mm/memcontrol.c b/mm/memcontrol.c index d954ffb43a4323..36101129679fd0 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -1898,7 +1898,7 @@ void mem_cgroup_print_oom_meminfo(struct mem_cgroup *memcg) /* * Return the memory (and swap, if configured) limit for a memcg. */ -unsigned long mem_cgroup_get_max(struct mem_cgroup *memcg) +unsigned long mem_cgroup_get_max(const struct mem_cgroup *memcg) { unsigned long max = READ_ONCE(memcg->memory.max); diff --git a/mm/swap.h b/mm/swap.h index b3b54c28929a19..1957960dc60d63 100644 --- a/mm/swap.h +++ b/mm/swap.h @@ -82,7 +82,7 @@ enum swap_cluster_flags { extern int vm_swappiness; -static inline int mem_cgroup_swappiness(struct mem_cgroup *memcg) +static inline int mem_cgroup_swappiness(const struct mem_cgroup *memcg) { #ifdef CONFIG_MEMCG_V1 if (!cgroup_subsys_on_dfl(memory_cgrp_subsys) && From b8a331f509394ccc63f8d8812a3e96e13ddaa814 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 15 Sep 2026 19:20:15 -0400 Subject: [PATCH 1052/1352] mm: memcontrol: constify the zswap and socket pressure helpers mem_cgroup_zswap_writeback_enabled() only reads the zswap_writeback flags, and mem_cgroup_get_socket_pressure() only reads the socket pressure timestamp. Constify them. Link: https://lore.kernel.org/20260915-folio_memcg-const-v3-11-c239a6010b58@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: Muchun Song Acked-by: Shakeel Butt Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Roman Gushchin Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie --- include/linux/memcontrol.h | 8 ++++---- mm/memcontrol.c | 2 +- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index f118990854734e..64c183be8cbfe7 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -1684,7 +1684,7 @@ static inline void mem_cgroup_set_socket_pressure(struct mem_cgroup *memcg) write_sequnlock_irqrestore(&memcg->socket_pressure_seqlock, flags); } -static inline u64 mem_cgroup_get_socket_pressure(struct mem_cgroup *memcg) +static inline u64 mem_cgroup_get_socket_pressure(const struct mem_cgroup *memcg) { unsigned int seq; u64 val; @@ -1702,7 +1702,7 @@ static inline void mem_cgroup_set_socket_pressure(struct mem_cgroup *memcg) WRITE_ONCE(memcg->socket_pressure, jiffies + HZ); } -static inline u64 mem_cgroup_get_socket_pressure(struct mem_cgroup *memcg) +static inline u64 mem_cgroup_get_socket_pressure(const struct mem_cgroup *memcg) { return READ_ONCE(memcg->socket_pressure); } @@ -1937,7 +1937,7 @@ static inline void mem_cgroup_calculate_protection_path(struct mem_cgroup *root, bool obj_cgroup_may_zswap(struct obj_cgroup *objcg); void obj_cgroup_charge_zswap(struct obj_cgroup *objcg, size_t size); void obj_cgroup_uncharge_zswap(struct obj_cgroup *objcg, size_t size); -bool mem_cgroup_zswap_writeback_enabled(struct mem_cgroup *memcg); +bool mem_cgroup_zswap_writeback_enabled(const struct mem_cgroup *memcg); #else static inline bool obj_cgroup_may_zswap(struct obj_cgroup *objcg) { @@ -1951,7 +1951,7 @@ static inline void obj_cgroup_uncharge_zswap(struct obj_cgroup *objcg, size_t size) { } -static inline bool mem_cgroup_zswap_writeback_enabled(struct mem_cgroup *memcg) +static inline bool mem_cgroup_zswap_writeback_enabled(const struct mem_cgroup *memcg) { /* if zswap is disabled, do not block pages going to the swapping device */ return true; diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 36101129679fd0..4d00748c8a5b87 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -6318,7 +6318,7 @@ void obj_cgroup_uncharge_zswap(struct obj_cgroup *objcg, size_t size) rcu_read_unlock(); } -bool mem_cgroup_zswap_writeback_enabled(struct mem_cgroup *memcg) +bool mem_cgroup_zswap_writeback_enabled(const struct mem_cgroup *memcg) { /* if zswap is disabled, do not block pages going to the swapping device */ if (!zswap_is_enabled()) From 398978878de7af5e26b9aa6d2e3ac65c3ab502a5 Mon Sep 17 00:00:00 2001 From: Usama Arif Date: Wed, 16 Sep 2026 05:51:22 -0700 Subject: [PATCH 1053/1352] mm: filemap: move lruvec accounting outside the xarray lock __filemap_add_folio() inserts a folio and updates mapping->nrpages while holding mapping->i_pages.xa_lock with interrupts disabled. The XArray insertion and nrpages update require the lock, but the lruvec statistic updates do not. With CONFIG_MEMCG, those calls also update per-CPU memcg and lruvec counters and notify cgroup rstat, extending the critical section. Move the lruvec accounting after a successful XArray insertion and after xas_unlock_irq(). The page-cache references pin the folio, while the folio lock keeps folio->mapping stable and prevents removal until accounting is complete. This moves one lruvec update for ordinary folios and a second for PMD-mappable folios out of the serialized section. In a 30-second system-wide perf lock contention -ab capture on a production host, the hottest caller-stack record attributed to __filemap_add_folio() had 20,867 contentions and 557.930 ms total wait. That was 14% of the 3.998 seconds of aggregate lock wait in the capture. Moving lruvec accuting outside of critical section should help optimize it. Link: https://lore.kernel.org/20260916125122.2696271-1-usama.arif@linux.dev Signed-off-by: Usama Arif Signed-off-by: Andrew Morton Reviewed-by: Shakeel Butt Acked-by: Muchun Song Reviewed-by: Vishal Moola (Fractile) Reviewed-by: Jan Kara Cc: David Hildenbrand Cc: Johannes Weiner Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Rik van Riel Cc: Roman Gushchin --- mm/filemap.c | 15 +++++++-------- 1 file changed, 7 insertions(+), 8 deletions(-) diff --git a/mm/filemap.c b/mm/filemap.c index 00fd89cf6f5509..4720bbfc1a6636 100644 --- a/mm/filemap.c +++ b/mm/filemap.c @@ -918,14 +918,6 @@ noinline int __filemap_add_folio(struct address_space *mapping, mapping->nrpages += nr; - /* hugetlb pages do not participate in page cache accounting */ - if (!huge) { - lruvec_stat_mod_folio(folio, NR_FILE_PAGES, nr); - if (folio_test_pmd_mappable(folio)) - lruvec_stat_mod_folio(folio, - NR_FILE_THPS, nr); - } - unlock: xas_unlock_irq(&xas); @@ -942,6 +934,13 @@ noinline int __filemap_add_folio(struct address_space *mapping, if (xas_error(&xas)) goto error; + /* hugetlb pages do not participate in page cache accounting */ + if (!huge) { + lruvec_stat_mod_folio(folio, NR_FILE_PAGES, nr); + if (folio_test_pmd_mappable(folio)) + lruvec_stat_mod_folio(folio, NR_FILE_THPS, nr); + } + trace_mm_filemap_add_to_page_cache(folio); return 0; error: From dda256bf4092f12dbd6dba2cad8e0f480e5ef65d Mon Sep 17 00:00:00 2001 From: Yuanhe Shu Date: Wed, 16 Sep 2026 19:25:45 +0800 Subject: [PATCH 1054/1352] mm/page_alloc: do not boost watermarks in kdump capture kernels A watermark boost is not confined to one watermark: wmark_pages() adds it to min, low and high alike, so every watermark check sees it, including should_reclaim_retry() and the last ditch ALLOC_WMARK_HIGH attempt in __alloc_pages_may_oom(). Once the boost exceeds the memory still free the allocator gives up and invokes the OOM killer, and a capture kernel that is still booting has nothing to kill: the boot panics and the vmcore is lost. Seen on an arm64 machine with 64K pages, CONFIG_PAGE_BLOCK_MAX_ORDER=10 (pageblock = 64M) and crashkernel=512M, running a distribution kernel based on 7.0.14. A high order UNMOVABLE allocation fell back to a MOVABLE pageblock while the capture kernel was still in do_initcalls(): Node 0 DMA free:68096kB boost:65536kB min:68160kB low:68800kB high:69440kB managed:479168kB Out of memory and no killable processes... Kernel panic - not syncing: System is deadlocked on memory The zone was not short of memory. Subtracting the boost gives min:2624kB low:3264kB high:3904kB, so the 68096kB still free sat 17 times above the high watermark and the allocator would not even have entered its slow path. The boost supplied 65536kB of the 68160kB min and by itself put the zone 64kB under water. It is that large because boost_watermark() clamps it with max(pageblock_nr_pages, max_boost); watermark_boost_factor alone would have allowed 5824kB. Commit 14f69140ff9c ("mm: limit boost_watermark on small zones") already tried to protect capture kernels, but it infers them from the zone size and skips the boost only below four pageblocks. arm64 64K pageblocks were 512M then, so the guard reached zones up to 2G; CONFIG_PAGE_BLOCK_MAX_ORDER can cap them at 64M, which shrinks the guard to zones under 256M and lets this 468M zone through. kdump is a property of the kernel, not of the zone, so test for it directly. A capture kernel exits within seconds and never uses the fragmentation avoidance the boost buys. Normal kernels are unaffected: the size based check still covers their genuinely tiny zones. Passing sysctl.vm.watermark_boost_factor=0 to the capture kernel does not cover this window: sysctl.* parameters are written through procfs by do_sysctl_args(), which runs after do_initcalls() where the panic above happened, and watermark_boost_factor has no early_param of its own. Link: https://lore.kernel.org/20260916112545.3707893-1-xiangzao@linux.alibaba.com Fixes: 1c30844d2dfe ("mm: reclaim small amounts of memory when an external fragmentation event occurs") Signed-off-by: Yuanhe Shu Signed-off-by: Andrew Morton Reviewed-by: Vlastimil Babka (SUSE) Reviewed-by: Johannes Weiner Cc: Henry Willard Cc: Brendan Jackman Cc: David Hildenbrand Cc: Mel Gorman Cc: Michal Hocko Cc: Suren Baghdasaryan Cc: Zi Yan Cc: # v5.0 --- mm/page_alloc.c | 12 ++++++++++-- 1 file changed, 10 insertions(+), 2 deletions(-) diff --git a/mm/page_alloc.c b/mm/page_alloc.c index d62670b8f9610c..67af64352855a2 100644 --- a/mm/page_alloc.c +++ b/mm/page_alloc.c @@ -37,6 +37,7 @@ #include #include #include +#include #include #include #include @@ -2160,10 +2161,17 @@ static inline bool boost_watermark(struct zone *zone) if (!watermark_boost_factor) return false; + + /* + * A kdump capture kernel exits before a boost can pay off, while + * the raised watermark can exceed the memory left for the dump. + */ + if (is_kdump_kernel()) + return false; + /* * Don't bother in zones that are unlikely to produce results. - * On small machines, including kdump capture kernels running - * in a small area, boosting the watermark can cause an out of + * On small machines, boosting the watermark can cause an out of * memory situation immediately. */ if ((pageblock_nr_pages * 4) > zone_managed_pages(zone)) From 1033b619f5725d09b8a934580702ea23f3acc740 Mon Sep 17 00:00:00 2001 From: Kefeng Wang Date: Wed, 16 Sep 2026 12:31:52 +0800 Subject: [PATCH 1055/1352] mm: mincore: use per-vma lock during page table walk do_mincore() performs a read-only, per-VMA residency query, making it a good candidate for per-VMA locking. Convert it to acquire the per-VMA lock, thereby reducing contention on the per-MM mmap_lock. Link: https://lore.kernel.org/20260916043153.2631696-1-wangkefeng.wang@huawei.com Signed-off-by: Kefeng Wang Signed-off-by: Andrew Morton Reviewed-by: Pedro Falcato Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Cc: Jann Horn Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Zi Yan --- mm/mincore.c | 29 +++++++++++++++++------------ 1 file changed, 17 insertions(+), 12 deletions(-) diff --git a/mm/mincore.c b/mm/mincore.c index c086836bc4bcc5..0fe50f8a7e6291 100644 --- a/mm/mincore.c +++ b/mm/mincore.c @@ -235,24 +235,22 @@ static const struct mm_walk_ops mincore_walk_ops = { .pmd_entry = mincore_pte_range, .pte_hole = mincore_unmapped_range, .hugetlb_entry = mincore_hugetlb, - .walk_lock = PGWALK_RDLOCK, + .walk_lock = PGWALK_VMA_RDLOCK_VERIFY, }; /* * Do a chunk of "sys_mincore()". We've already checked - * all the arguments, we hold the mmap semaphore: we should + * all the arguments, we hold the VMA read lock: we should * just return the amount of info we're asked for. */ -static long do_mincore(unsigned long addr, unsigned long pages, unsigned char *vec) +static long do_mincore(struct vm_area_struct *vma, unsigned long addr, + unsigned long pages, unsigned char *vec) { - struct vm_area_struct *vma; - unsigned long end; + unsigned long end = min(vma->vm_end, addr + (pages << PAGE_SHIFT)); int err; - vma = vma_lookup(current->mm, addr); - if (!vma) - return -ENOMEM; - end = min(vma->vm_end, addr + (pages << PAGE_SHIFT)); + vma_assert_locked(vma); + if (!can_do_mincore(vma)) { unsigned long pages = DIV_ROUND_UP(end - addr, PAGE_SIZE); memset(vec, 1, pages); @@ -319,13 +317,20 @@ SYSCALL_DEFINE3(mincore, unsigned long, start, size_t, len, retval = 0; while (pages) { + struct vm_area_struct *vma; + + vma = vma_start_read_unlocked(current->mm, start); + if (!vma) { + retval = -ENOMEM; + break; + } + /* * Do at most PAGE_SIZE entries per iteration, due to * the temporary buffer size. */ - mmap_read_lock(current->mm); - retval = do_mincore(start, min(pages, PAGE_SIZE), tmp); - mmap_read_unlock(current->mm); + retval = do_mincore(vma, start, min(pages, PAGE_SIZE), tmp); + vma_end_read(vma); if (retval <= 0) break; From 4a54776f9d9dfb6778a61001346b9899c0b935ef Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 15 Sep 2026 19:42:15 -0400 Subject: [PATCH 1056/1352] vmcore: convert mmap_vmcore_fault() to use folios Use a folio for the page cache page the s390 fault handler reads the old kernel's memory into. This removes four compound_head() calls and one of the last callers of find_or_create_page(). The vmcore mapping only has order-0 folios, so the logic is unchanged. Compile tested for s390. Link: https://lore.kernel.org/20260915-vmcore-fault-folio-v1-1-a0cb6278670f@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: Pratyush Yadav Cc: Baoquan He Cc: Dave Young Cc: Mike Rapoport Cc: Pasha Tatashin --- fs/proc/vmcore.c | 27 ++++++++++++++------------- 1 file changed, 14 insertions(+), 13 deletions(-) diff --git a/fs/proc/vmcore.c b/fs/proc/vmcore.c index 44d15436439fd9..2dc3803aa11dcd 100644 --- a/fs/proc/vmcore.c +++ b/fs/proc/vmcore.c @@ -474,29 +474,30 @@ static vm_fault_t mmap_vmcore_fault(struct vm_fault *vmf) pgoff_t index = vmf->pgoff; struct iov_iter iter; struct kvec kvec; - struct page *page; + struct folio *folio; loff_t offset; int rc; - page = find_or_create_page(mapping, index, GFP_KERNEL); - if (!page) + folio = __filemap_get_folio(mapping, index, + FGP_LOCK | FGP_ACCESSED | FGP_CREAT, GFP_KERNEL); + if (IS_ERR(folio)) return VM_FAULT_OOM; - if (!PageUptodate(page)) { - offset = (loff_t) index << PAGE_SHIFT; - kvec.iov_base = page_address(page); - kvec.iov_len = PAGE_SIZE; - iov_iter_kvec(&iter, ITER_DEST, &kvec, 1, PAGE_SIZE); + if (!folio_test_uptodate(folio)) { + offset = folio_pos(folio); + kvec.iov_base = folio_address(folio); + kvec.iov_len = folio_size(folio); + iov_iter_kvec(&iter, ITER_DEST, &kvec, 1, folio_size(folio)); rc = __read_vmcore(&iter, &offset); if (rc < 0) { - unlock_page(page); - put_page(page); + folio_unlock(folio); + folio_put(folio); return vmf_error(rc); } - SetPageUptodate(page); + folio_mark_uptodate(folio); } - unlock_page(page); - vmf->page = page; + folio_unlock(folio); + vmf->page = folio_file_page(folio, index); return 0; #else return VM_FAULT_SIGBUS; From f7c18de77672df355190dadf4477bc1f19ad8bd2 Mon Sep 17 00:00:00 2001 From: Alexandre Ghiti Date: Mon, 21 Sep 2026 17:13:02 +0200 Subject: [PATCH 1057/1352] mm: swap: move LRU insertion out of the swap cache allocator Patch series "mm: zswap: free cold writeback folios promptly", v6. When zswap writes an entry back, it allocates an order-0 swap cache folio, decompresses into it, and issues the write. The folio is cold by construction, yet today it is left on the LRU for page reclaim to find and free later. That wastes a reclaim scan and keeps cold memory resident longer than necessary. Rather than implement this in zswap, extend the existing dropbehind mechanism to swap cache folios and have zswap opt into it (Yosry). A PG_dropbehind folio is already dropped from its cache once writeback completes instead of being left for reclaim; for a swap cache folio that "drop" is removing it from the swap cache. Patch 1 - move LRU insertion out of the swap cache allocator into its callers, so zswap writeback can allocate off the LRU. Patch 2 - drop dropbehind swap cache folios on writeback completion. Patch 3 - zswap allocates its writeback folio off the LRU and marks it dropbehind, opting into the mechanism above. This patch (of 3): This is a preparatory patch. swap_cache_alloc_folio() adds the new folio to the LRU itself, which leaves its callers no way to act on the folio before it becomes visible to reclaim. Two users need exactly that: - moving the refault evaluation out of the swap cache folio allocation requires it to happen before folio_add_lru(): that consumes PG_active to file the folio on the inactive or the active list, and under MGLRU it also reads PG_workingset to pick the generation. Setting either flag afterwards does not move the folio; - zswap writeback dropbehind needs the buffer folio to stay off the LRU entirely, as the per-CPU LRU batch would hold a reference on it and keep remove_mapping() from freeing it once writeback completes. Defer the LRU insertion to the callers and rename the helper to __swap_cache_alloc_folio(): each caller adds the folio right after the allocation, so there is no functional change intended. Link: https://lore.kernel.org/20260921151306.625134-1-alex@ghiti.fr Link: https://lore.kernel.org/20260921151306.625134-2-alex@ghiti.fr Signed-off-by: Alexandre Ghiti Signed-off-by: Andrew Morton Suggested-by: Kairui Song Reviewed-by: Kairui Song Reviewed-by: Nhat Pham Reviewed-by: Kunwu Chan Acked-by: Usama Arif Reviewed-by: Barry Song Cc: Al Viro Cc: Axel Rasmussen Cc: Baoquan He Cc: Chengming Zhou Cc: Chis Li (Google) Cc: Christian Brauner (Amutable) Cc: "David Hildenbrand (arm)" Cc: Jan Kara Cc: Johannes Weiner Cc: Kemeng Shi Cc: Lorenzo Stoakes (ARM) Cc: Matthew Wilcox Cc: Michal Hocko Cc: Qi Zheng Cc: Shakeel Butt Cc: Tal Zussman Cc: Wei Xu Cc: Yosry Ahmed Cc: Youngjun Park Cc: Yuanchu Xie --- mm/swap.h | 6 +++--- mm/swap_state.c | 20 ++++++++++++-------- mm/swapfile.c | 2 +- mm/zswap.c | 5 +++-- 4 files changed, 19 insertions(+), 14 deletions(-) diff --git a/mm/swap.h b/mm/swap.h index 1957960dc60d63..d5bf21f517dcea 100644 --- a/mm/swap.h +++ b/mm/swap.h @@ -312,9 +312,9 @@ bool swap_cache_has_folio(swp_entry_t entry); struct folio *swap_cache_get_folio(swp_entry_t entry); void *swap_cache_get_shadow(swp_entry_t entry); void swap_cache_del_folio(struct folio *folio); -struct folio *swap_cache_alloc_folio(swp_entry_t target_entry, gfp_t gfp_mask, - unsigned long orders, struct vm_fault *vmf, - struct mempolicy *mpol, pgoff_t ilx); +struct folio *__swap_cache_alloc_folio(swp_entry_t target_entry, gfp_t gfp_mask, + unsigned long orders, struct vm_fault *vmf, + struct mempolicy *mpol, pgoff_t ilx); /* Below helpers require the caller to lock and pass in the swap cluster. */ void __swap_cache_add_folio(struct swap_cluster_info *ci, struct folio *folio, swp_entry_t entry); diff --git a/mm/swap_state.c b/mm/swap_state.c index cef44aadee6158..87790369dd9aa3 100644 --- a/mm/swap_state.c +++ b/mm/swap_state.c @@ -497,13 +497,11 @@ static struct folio *__swap_cache_alloc(struct swap_cluster_info *ci, node_stat_mod_folio(folio, NR_FILE_PAGES, nr_pages); lruvec_stat_mod_folio(folio, NR_SWAPCACHE, nr_pages); - /* Caller will initiate read into locked new_folio */ - folio_add_lru(folio); return folio; } /** - * swap_cache_alloc_folio - Allocate folio for swapped out slot in swap cache. + * __swap_cache_alloc_folio - Allocate folio for swapped out slot in swap cache. * @targ_entry: swap entry indicating the target slot * @gfp: memory allocation flags * @orders: allocation orders, must be non zero @@ -515,13 +513,17 @@ static struct folio *__swap_cache_alloc(struct swap_cluster_info *ci, * doing IO (e.g. swap in or zswap writeback). The swap slot indicated by * @targ_entry must have a non-zero swap count (swapped out). * + * The returned folio is locked and is NOT on the LRU. The caller must either + * add it to the LRU with folio_add_lru() so page reclaim can find it, or free + * it directly once done; a folio left off the LRU is unreclaimable and leaks. + * * Context: Caller must protect the swap device with reference count or locks. * Return: Returns the folio if allocation succeeded and folio is in the swap * cache. Returns error code if failed due to race, OOM or invalid arguments. */ -struct folio *swap_cache_alloc_folio(swp_entry_t targ_entry, gfp_t gfp, - unsigned long orders, struct vm_fault *vmf, - struct mempolicy *mpol, pgoff_t ilx) +struct folio *__swap_cache_alloc_folio(swp_entry_t targ_entry, gfp_t gfp, + unsigned long orders, struct vm_fault *vmf, + struct mempolicy *mpol, pgoff_t ilx) { int order, err; struct folio *ret; @@ -657,12 +659,13 @@ static struct folio *swap_cache_read_folio(struct swap_io_ctx *ctx, folio = swap_cache_get_folio(entry); if (folio) return folio; - folio = swap_cache_alloc_folio(entry, gfp, BIT(0), NULL, mpol, ilx); + folio = __swap_cache_alloc_folio(entry, gfp, BIT(0), NULL, mpol, ilx); } while (PTR_ERR(folio) == -EEXIST); if (IS_ERR_OR_NULL(folio)) return NULL; + folio_add_lru(folio); swap_read_folio(ctx, folio); if (readahead) { folio_set_readahead(folio); @@ -698,12 +701,13 @@ struct folio *swapin_sync(swp_entry_t entry, gfp_t gfp, unsigned long orders, folio = swap_cache_get_folio(entry); if (folio) return folio; - folio = swap_cache_alloc_folio(entry, gfp, orders, vmf, mpol, ilx); + folio = __swap_cache_alloc_folio(entry, gfp, orders, vmf, mpol, ilx); } while (PTR_ERR(folio) == -EEXIST); if (IS_ERR(folio)) return folio; + folio_add_lru(folio); swap_read_folio(&ctx, folio); swap_read_submit(&ctx); return folio; diff --git a/mm/swapfile.c b/mm/swapfile.c index c1c5fbb3c909d3..0901cb8fa7291c 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -1887,7 +1887,7 @@ void folio_put_swap(struct folio *folio, struct page *page) * CPU1 CPU2 * do_swap_page() * ... swapoff+swapon - * swap_cache_alloc_folio() + * __swap_cache_alloc_folio() * // check swap_map * // verify PTE not changed * diff --git a/mm/zswap.c b/mm/zswap.c index 584dd306376943..c79cca61abf933 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -1019,8 +1019,8 @@ static int zswap_writeback_entry(struct zswap_entry *entry, return -ENOENT; mpol = get_task_policy(current); - folio = swap_cache_alloc_folio(swpentry, GFP_KERNEL, BIT(0), NULL, mpol, - NO_INTERLEAVE_INDEX); + folio = __swap_cache_alloc_folio(swpentry, GFP_KERNEL, BIT(0), NULL, mpol, + NO_INTERLEAVE_INDEX); put_swap_device(si); /* @@ -1032,6 +1032,7 @@ static int zswap_writeback_entry(struct zswap_entry *entry, */ if (IS_ERR(folio)) return PTR_ERR(folio); + folio_add_lru(folio); /* * folio is locked, and the swapcache is now secured against From 76a51489bceced283cf6bf959e8c5cbc188d2192 Mon Sep 17 00:00:00 2001 From: Alexandre Ghiti Date: Mon, 21 Sep 2026 17:13:03 +0200 Subject: [PATCH 1058/1352] mm: swap: drop dropbehind swap cache folios on writeback completion A PG_dropbehind folio is dropped from its cache once writeback completes rather than left for reclaim to find later; this is implemented for file folios in folio_end_dropbehind(). Extend it to swap cache folios. The drop blocks on the folio lock, so it cannot run in interrupt context. Set BIO_COMPLETE_IN_TASK on the write, as the file dropbehind paths do, and drop the folio directly from folio_end_writeback(). It has to block rather than trylock: the folio is off the LRU, so skipping it would leave it in the swap cache with nothing able to reclaim it, and it cannot be put back while another thread holds its lock. Link: https://lore.kernel.org/20260921151306.625134-3-alex@ghiti.fr Signed-off-by: Alexandre Ghiti Signed-off-by: Andrew Morton Suggested-by: Yosry Ahmed Suggested-by: Johannes Weiner Suggested-by: Nhat Pham Reviewed-by: Nhat Pham Reviewed-by: Kunwu Chan Reviewed-by: Barry Song Cc: Al Viro Cc: Axel Rasmussen Cc: Baoquan He Cc: Chengming Zhou Cc: Chis Li Cc: Christian Brauner (Amutable) Cc: "David Hildenbrand (arm)" Cc: Jan Kara Cc: Kairui Song Cc: Kemeng Shi Cc: Lorenzo Stoakes (ARM) Cc: Matthew Wilcox Cc: Michal Hocko Cc: Qi Zheng Cc: Shakeel Butt Cc: Tal Zussman Cc: Usama Arif Cc: Wei Xu Cc: Youngjun Park Cc: Yuanchu Xie --- include/linux/swap.h | 6 ++++++ mm/filemap.c | 19 +++++++++++++++++ mm/page_io.c | 9 ++++++++ mm/swap_state.c | 42 +++++++++++++++++++++++++++++++++++++ mm/vmscan.c | 49 +++++++++++++++++++++++++++++++++++--------- 5 files changed, 115 insertions(+), 10 deletions(-) diff --git a/include/linux/swap.h b/include/linux/swap.h index 82f0bfec611fc7..cb434cccd653a6 100644 --- a/include/linux/swap.h +++ b/include/linux/swap.h @@ -336,6 +336,9 @@ static inline bool lru_cache_disabled(void) extern unsigned long shrink_all_memory(unsigned long nr_pages); long remove_mapping(struct address_space *mapping, struct folio *folio); +long remove_mapping_set_shadow(struct address_space *mapping, + struct folio *folio, + struct mem_cgroup *target_memcg); #if defined(CONFIG_SYSFS) && defined(CONFIG_NUMA) extern int reclaim_register_node(struct node *node); @@ -421,6 +424,8 @@ void swap_put_entries_direct(swp_entry_t entry, int nr); */ bool folio_free_swap(struct folio *folio); +void swap_writeback_dropbehind_folio(struct folio *folio); + /* Allocate / free (hibernation) exclusive entries */ swp_entry_t swap_alloc_hibernation_slot(int type); void swap_free_hibernation_slot(swp_entry_t entry); @@ -431,6 +436,7 @@ static inline void put_swap_device(struct swap_info_struct *si) } #else /* CONFIG_SWAP */ +static inline void swap_writeback_dropbehind_folio(struct folio *folio) {} static inline struct swap_info_struct *get_swap_device(swp_entry_t entry) { return NULL; diff --git a/mm/filemap.c b/mm/filemap.c index 4720bbfc1a6636..b74bc1e5015c6f 100644 --- a/mm/filemap.c +++ b/mm/filemap.c @@ -1685,6 +1685,8 @@ EXPORT_SYMBOL_GPL(folio_end_writeback_no_dropbehind); */ void folio_end_writeback(struct folio *folio) { + bool swap_dropbehind; + VM_BUG_ON_FOLIO(!folio_test_writeback(folio), folio); /* @@ -1694,7 +1696,24 @@ void folio_end_writeback(struct folio *folio) * reused before the folio_wake_bit(). */ folio_get(folio); + + /* + * Sample this before folio_end_writeback_no_dropbehind() clears + * PG_writeback: until then a racing swapin cannot remove the folio from + * the swap cache. Afterwards it can, and the drop below then finds a + * non-swapcache folio and puts it back on the LRU instead. The + * reference taken above keeps the folio alive across that window. + */ + swap_dropbehind = folio_test_swapcache(folio) && + folio_test_dropbehind(folio); + folio_end_writeback_no_dropbehind(folio); + + if (swap_dropbehind) { + swap_writeback_dropbehind_folio(folio); + return; + } + folio_end_dropbehind(folio); folio_put(folio); } diff --git a/mm/page_io.c b/mm/page_io.c index 5f7756e370f7a4..0808808f0309ce 100644 --- a/mm/page_io.c +++ b/mm/page_io.c @@ -608,6 +608,15 @@ static void swap_bdev_submit_write(struct swap_io_ctx *ctx) submit_bio_wait(bio); end_swap_bio_write(bio); } else { + int p; + + for (p = 0; p < sio->nr_bvecs; p++) { + if (folio_test_dropbehind(bvec_folio(&sio->bvecs[p]))) { + bio_set_flag(bio, BIO_COMPLETE_IN_TASK); + break; + } + } + bio->bi_end_io = end_swap_bio_write; submit_bio(bio); } diff --git a/mm/swap_state.c b/mm/swap_state.c index 87790369dd9aa3..2475ba29126dca 100644 --- a/mm/swap_state.c +++ b/mm/swap_state.c @@ -551,6 +551,48 @@ struct folio *__swap_cache_alloc_folio(swp_entry_t targ_entry, gfp_t gfp, return ret; } +/** + * swap_writeback_dropbehind_folio - drop a dropbehind swap cache folio + * @folio: the off-LRU folio whose writeback has completed + * + * Context: task context, with the reference taken by folio_end_writeback() + * donated to us. + */ +void swap_writeback_dropbehind_folio(struct folio *folio) +{ + struct mem_cgroup *memcg; + + folio_lock(folio); + + /* The folio was allocated off the LRU and nothing re-adds it here. */ + VM_WARN_ON_ONCE_FOLIO(folio_test_lru(folio), folio); + + rcu_read_lock(); + memcg = folio_memcg(folio); + if (!mem_cgroup_tryget(memcg)) + memcg = NULL; + rcu_read_unlock(); + + /* + * Gate remove_mapping_set_shadow() on folio_test_swapcache(): a racing + * swapin may have freed the swap slot (folio_free_swap()) and dropped the + * folio from the cache, and it must not run on a non-swapcache folio (it + * would trip __remove_mapping()'s mapping == folio_mapping() check). + */ + if (!folio_test_swapcache(folio) || folio_test_writeback(folio) || + !remove_mapping_set_shadow(swap_address_space(folio->swap), folio, + memcg)) { + /* Raced: the folio is now owned by the swapin; put it back. */ + folio_clear_dropbehind(folio); + folio_add_lru(folio); + } + + mem_cgroup_put(memcg); + + folio_unlock(folio); + folio_put(folio); +} + /* * If we are the only user, then try to free up the swap cache. * diff --git a/mm/vmscan.c b/mm/vmscan.c index f2e641e9cf7de9..e200ce3eb056b1 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -854,6 +854,22 @@ static int __remove_mapping(struct address_space *mapping, struct folio *folio, return 0; } +static long __remove_mapping_unfreeze(struct address_space *mapping, + struct folio *folio, bool reclaimed, + struct mem_cgroup *target_memcg) +{ + if (__remove_mapping(mapping, folio, reclaimed, target_memcg)) { + /* + * Unfreezing the refcount with 1 effectively + * drops the pagecache ref for us without requiring another + * atomic operation. + */ + folio_ref_unfreeze(folio, 1); + return folio_nr_pages(folio); + } + return 0; +} + /** * remove_mapping() - Attempt to remove a folio from its mapping. * @mapping: The address space. @@ -868,16 +884,29 @@ static int __remove_mapping(struct address_space *mapping, struct folio *folio, */ long remove_mapping(struct address_space *mapping, struct folio *folio) { - if (__remove_mapping(mapping, folio, false, NULL)) { - /* - * Unfreezing the refcount with 1 effectively - * drops the pagecache ref for us without requiring another - * atomic operation. - */ - folio_ref_unfreeze(folio, 1); - return folio_nr_pages(folio); - } - return 0; + return __remove_mapping_unfreeze(mapping, folio, false, NULL); +} + +/** + * remove_mapping_set_shadow() - Remove a folio and record an eviction shadow. + * @mapping: The address space. + * @folio: The folio to remove. + * @target_memcg: The memcg to charge the eviction shadow to; the caller must + * keep it alive across the call. + * + * Like remove_mapping(), but stores a workingset eviction shadow the way page + * reclaim does, so that a later refault can be detected and the folio + * re-activated. + * Return: The number of pages removed from the mapping. 0 if the folio + * could not be removed. + * Context: The caller should have a single refcount on the folio and + * hold its lock. + */ +long remove_mapping_set_shadow(struct address_space *mapping, + struct folio *folio, + struct mem_cgroup *target_memcg) +{ + return __remove_mapping_unfreeze(mapping, folio, true, target_memcg); } /** From 33ab77d7a4063750d334ee18ff033b69d082cdd6 Mon Sep 17 00:00:00 2001 From: Alexandre Ghiti Date: Mon, 21 Sep 2026 17:13:04 +0200 Subject: [PATCH 1059/1352] mm: zswap: drop cold writeback folios via swap dropbehind zswap writeback decompresses an entry into a fresh swap cache folio and writes it back. The folio is cold by construction, yet it is left on the LRU for reclaim to find and free later, wasting a reclaim scan and keeping cold memory resident longer than necessary. Allocate the folio off the LRU and mark it PG_dropbehind so the swap dropbehind path frees it from the swap cache once writeback completes. __swap_cache_alloc_folio() evaluates a refault on the new folio, and workingset_refault() sets PG_active when it looks recent. Until now folio_add_lru() consumed that flag and __page_cache_release() cleared it once the folio left the LRU. This folio never reaches the LRU, so nothing would clear PG_active and the folio would be freed with a PAGE_FLAGS_CHECK_AT_FREE flag set, tripping bad_page() under CONFIG_DEBUG_VM. Clear it after allocation. That is a workaround: the refault should not be evaluated on a writeback buffer at all. A fix for that is on the mailing list [1]. Link: https://lore.kernel.org/20260921151306.625134-4-alex@ghiti.fr Link: https://lore.kernel.org/linux-mm/20260911092012.92399-1-alex@ghiti.fr/ [1] Signed-off-by: Alexandre Ghiti Signed-off-by: Andrew Morton Suggested-by: Johannes Weiner Suggested-by: Nhat Pham Reviewed-by: Nhat Pham Reviewed-by: Kunwu Chan Cc: Al Viro Cc: Axel Rasmussen Cc: Baoquan He Cc: Barry Song Cc: Chengming Zhou Cc: Chis Li Cc: Christian Brauner (Amutable) Cc: "David Hildenbrand (arm)" Cc: Jan Kara Cc: Kairui Song Cc: Kemeng Shi Cc: Lorenzo Stoakes (ARM) Cc: Matthew Wilcox Cc: Michal Hocko Cc: Qi Zheng Cc: Shakeel Butt Cc: Tal Zussman Cc: Usama Arif Cc: Wei Xu Cc: Yosry Ahmed Cc: Youngjun Park Cc: Yuanchu Xie --- mm/zswap.c | 33 +++++++++++++++++++++++---------- 1 file changed, 23 insertions(+), 10 deletions(-) diff --git a/mm/zswap.c b/mm/zswap.c index c79cca61abf933..ae19e301fced7e 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -1032,7 +1032,8 @@ static int zswap_writeback_entry(struct zswap_entry *entry, */ if (IS_ERR(folio)) return PTR_ERR(folio); - folio_add_lru(folio); + + folio_clear_active(folio); /* * folio is locked, and the swapcache is now secured against @@ -1046,12 +1047,12 @@ static int zswap_writeback_entry(struct zswap_entry *entry, tree = swap_zswap_tree(swpentry); if (entry != xa_load(tree, offset)) { ret = -ENOMEM; - goto out; + goto err; } if (!zswap_decompress(entry, folio)) { ret = -EIO; - goto out; + goto err; } xa_erase(tree, offset); @@ -1065,18 +1066,30 @@ static int zswap_writeback_entry(struct zswap_entry *entry, /* folio is up to date */ folio_mark_uptodate(folio); - /* move it to the tail of the inactive list after end_writeback */ - folio_set_reclaim(folio); + folio_set_dropbehind(folio); + + /* + * Drop our reference before starting writeback so the swap cache holds + * the only one: the drop in folio_end_writeback() needs that for + * remove_mapping_set_shadow() to succeed, otherwise the folio is + * handed back to reclaim instead. + * + * Nothing can free the folio in the meantime: we hold the folio lock + * until writeback starts, PG_writeback then blocks swap cache removal, + * and folio_end_writeback() takes its own reference before clearing + * PG_writeback and donates it to the drop. + */ + folio_put(folio); /* start writeback */ __swap_writeout(&ctx, folio); swap_write_submit(&ctx); -out: - if (ret) { - swap_cache_del_folio(folio); - folio_unlock(folio); - } + return 0; + +err: + swap_cache_del_folio(folio); + folio_unlock(folio); folio_put(folio); return ret; } From c6ce9d08fd10f0c22b71c6ab921e023ad61288ab Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 20:47:41 +0100 Subject: [PATCH 1060/1352] mm/vma: const-ify vma_assert_stabilised() and associated functions Patch series "mm: implement and use vma_has_anon_rmap(), silence KCSAN". Provide a function to abstract the common task of checking whether a VMA has an anonymous reverse mapping associated with it. In the first patch, const-ify vma_assert_stabilised() and related functions so that vma_has_anon_rmap() can reference a const vma pointer. In the second patch, introduce vma_has_anon_rmap(). Finally in the third patch, update comments referencing anon_vma to instead reference the anon rmap to abstract this conceptually to avoid confusion. There are still other functions which reference anon_vma directly, those can be addressed in a follow up. This patch (of 3): The vma pointers are not modified in any of these functions so make it official by const-ify them. Propagate this throughout the call stack. No functional change intended. Link: https://lore.kernel.org/20260917-vma-is-faulted-v3-0-5c22314a72e7@kernel.org Link: https://lore.kernel.org/20260917-vma-is-faulted-v3-1-5c22314a72e7@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Acked-by: David Hildenbrand (Arm) Reviewed-by: Pedro Falcato Reviewed-by: Kiryl Shutsemau (Meta) Reviewed-by: Lance Yang Cc: Alistair Popple Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Byungchul Park Cc: Chengming Zhou Cc: Chris Li Cc: Dev Jain Cc: Gregory Price Cc: Guilherme Giacomo Simoes Cc: Harry Yoo Cc: "Huang, Ying" Cc: Jann Horn Cc: Joshua Hahn Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Brost Cc: Michal Hocko Cc: Mike Rapoport Cc: Muchun Song Cc: Nhat Pham Cc: Oscar Salvador Cc: Peter Xu Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Shakeel Butt Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: xu xin --- include/linux/mmap_lock.h | 12 ++++++------ tools/testing/vma/include/dup.h | 2 +- 2 files changed, 7 insertions(+), 7 deletions(-) diff --git a/include/linux/mmap_lock.h b/include/linux/mmap_lock.h index 28e3696ce9ff37..e5553f4a414cd1 100644 --- a/include/linux/mmap_lock.h +++ b/include/linux/mmap_lock.h @@ -273,7 +273,7 @@ static inline void vma_end_read(struct vm_area_struct *vma) vma_refcount_put(vma); } -static inline unsigned int __vma_raw_mm_seqnum(struct vm_area_struct *vma) +static inline unsigned int __vma_raw_mm_seqnum(const struct vm_area_struct *vma) { const struct mm_struct *mm = vma->vm_mm; @@ -288,7 +288,7 @@ static inline unsigned int __vma_raw_mm_seqnum(struct vm_area_struct *vma) * * Returns true if write-locked, otherwise false. */ -static inline bool __is_vma_write_locked(struct vm_area_struct *vma) +static inline bool __is_vma_write_locked(const struct vm_area_struct *vma) { /* * current task is holding mmap_write_lock, both vma->vm_lock_seq and @@ -344,7 +344,7 @@ int vma_start_write_killable(struct vm_area_struct *vma) * vma_assert_write_locked() - assert that @vma holds a VMA write lock. * @vma: The VMA to assert. */ -static inline void vma_assert_write_locked(struct vm_area_struct *vma) +static inline void vma_assert_write_locked(const struct vm_area_struct *vma) { if (!IS_ENABLED(CONFIG_MMU)) { mmap_assert_write_locked(vma->vm_mm); @@ -359,7 +359,7 @@ static inline void vma_assert_write_locked(struct vm_area_struct *vma) * lock and is not detached. * @vma: The VMA to assert. */ -static inline void vma_assert_locked(struct vm_area_struct *vma) +static inline void vma_assert_locked(const struct vm_area_struct *vma) { unsigned int refcnt; @@ -410,7 +410,7 @@ static inline void vma_assert_locked(struct vm_area_struct *vma) * With lockdep disabled we may sometimes race with other threads acquiring the * mmap read lock simultaneous with our VMA read lock. */ -static inline void vma_assert_stabilised(struct vm_area_struct *vma) +static inline void vma_assert_stabilised(const struct vm_area_struct *vma) { /* * If another thread owns an mmap lock, it may go away at any time, and @@ -445,7 +445,7 @@ static inline void vma_assert_stabilised(struct vm_area_struct *vma) vma_assert_locked(vma); } -static inline bool vma_is_attached(struct vm_area_struct *vma) +static inline bool vma_is_attached(const struct vm_area_struct *vma) { return refcount_read(&vma->vm_refcnt); } diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index c21f67decab58d..a4e3d30b2fc171 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -1178,7 +1178,7 @@ static inline struct vm_area_struct *vma_next(struct vma_iterator *vmi) return mas_find(&vmi->mas, ULONG_MAX); } -static inline bool vma_is_attached(struct vm_area_struct *vma) +static inline bool vma_is_attached(const struct vm_area_struct *vma) { return refcount_read(&vma->vm_refcnt); } From a81803850594f6262a674301b622fdb5db9f7e8c Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 20:47:42 +0100 Subject: [PATCH 1061/1352] mm: implement and use vma_has_anon_rmap(), silence KCSAN Provide a function to abstract the common task of checking whether a VMA has an anonymous reverse mapping associated with it. If the VMA is attached, a VMA or mmap lock must be held when calling this function. For an attached, anonymous, VMA: Transition | VMA/mmap Lock state -----------------------------|------------------------------------------- No anon rmap to anon rmap | Write lock/read lock + mm->page_table_lock Anon rmap to no anon rmap | Write lock vma_has_anon_rmap() never provides a false positive (the lock precludes it), but if only a read lock is held, a negative result must be re-checked with mm->page_table_lock held. A VMA obtains an anonymous reverse mapping when first faulted or forked and it is removed when it is freed. Detached VMAs cannot be concurrently manipulated as they are removed from the maple tree so require no guarantees. Use data_race() to silence KCSAN about non-existent data races between concurrent vma->anon_vma read/write on optimistic fault tests. Update the core VMA merge/split, rmap, mremap, KSM, fork, khugepaged and fault preparation callers which test vma->anon_vma directly to use vma_has_anon_rmap() instead. Finally, update comments that reference anon_vma to reference the anon rmap instead. Since the lockless read in reusable_anon_vma() is doing more than checking whether the VMA has anon rmap - it is returning the anon_vma to be used on fault - do not alter it. There is one odd one out - file_backed_vma_is_retractable() - which holds neither a VMA nor mmap lock and is stabilised by the file rmap lock only, so simply add a comment to explain why it's necessary. Link: https://lore.kernel.org/20260917-vma-is-faulted-v3-2-5c22314a72e7@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reported-by: Guilherme Giacomo Simoes Closes: https://lore.kernel.org/all/20260829100034.423064-1-trintaeoitogc@gmail.com/ Closes: https://lore.kernel.org/all/20260909115723.528501-1-trintaeoitogc@gmail.com/ Reviewed-by: Pedro Falcato Reviewed-by: Kiryl Shutsemau (Meta) Acked-by: David Hildenbrand (Arm) Reviewed-by: Lance Yang Cc: Alistair Popple Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Byungchul Park Cc: Chengming Zhou Cc: Chris Li Cc: Dev Jain Cc: Gregory Price Cc: Harry Yoo Cc: "Huang, Ying" Cc: Jann Horn Cc: Joshua Hahn Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Brost Cc: Michal Hocko Cc: Mike Rapoport Cc: Muchun Song Cc: Nhat Pham Cc: Oscar Salvador Cc: Peter Xu Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Shakeel Butt Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: xu xin Cc: Zi Yan --- mm/huge_memory.c | 4 ++-- mm/internal.h | 2 +- mm/khugepaged.c | 5 ++++- mm/ksm.c | 10 +++++----- mm/madvise.c | 4 ++-- mm/memory.c | 4 ++-- mm/mprotect.c | 2 +- mm/mremap.c | 4 ++-- mm/rmap.c | 22 +++++++++++----------- mm/swapfile.c | 2 +- mm/userfaultfd.c | 2 +- mm/vma.c | 29 +++++++++++++++-------------- mm/vma.h | 29 ++++++++++++++++++++++++++++- tools/testing/vma/include/stubs.h | 4 ++++ 14 files changed, 79 insertions(+), 44 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index d2e990da86b0d9..e0e252cfbf9580 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -259,7 +259,7 @@ unsigned long __thp_vma_allowable_orders(struct vm_area_struct *vma, * Allow page fault since anon_vma may be not initialized until * the first page fault. */ - if (!vma->anon_vma) + if (!vma_has_anon_rmap(vma)) return (smaps || in_pf) ? orders : 0; return orders; @@ -2171,7 +2171,7 @@ vm_fault_t do_huge_pmd_wp_page(struct vm_fault *vmf) pmd_t orig_pmd = vmf->orig_pmd; vmf->ptl = pmd_lockptr(vma->vm_mm, vmf->pmd); - VM_BUG_ON_VMA(!vma->anon_vma, vma); + VM_BUG_ON_VMA(!vma_has_anon_rmap(vma), vma); if (is_huge_zero_pmd(orig_pmd)) { vm_fault_t ret = do_huge_zero_wp_pmd(vmf); diff --git a/mm/internal.h b/mm/internal.h index 3b9fdb826162df..0434dfcfc36f14 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -339,7 +339,7 @@ void unlink_anon_vmas(struct vm_area_struct *vma); static inline int anon_vma_prepare(struct vm_area_struct *vma) { - if (likely(vma->anon_vma)) + if (likely(vma_has_anon_rmap(vma))) return 0; return __anon_vma_prepare(vma); diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 8a5c7f38096ef7..3e8dbd2dd46d8d 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -1045,7 +1045,7 @@ enum scan_result collapse_vma_revalidate(struct mm_struct *mm, unsigned long add * thp_vma_allowable_orders() may return true for qualified file * vmas. */ - if (expect_anon && (!(*vmap)->anon_vma || !vma_is_anonymous(*vmap))) + if (expect_anon && (!vma_has_anon_rmap(vma) || !vma_is_anonymous(vma))) return SCAN_PAGE_ANON; return SCAN_SUCCEED; } @@ -2079,6 +2079,9 @@ static bool file_backed_vma_is_retractable(struct vm_area_struct *vma) * Check vma->anon_vma to exclude MAP_PRIVATE mappings that * got written to. These VMAs are likely not worth removing * page tables from, as PMD-mapping is likely to be split later. + * + * Can't use vma_has_anon_rmap() here as the VMA may be stabilised + * by the file rmap lock. */ if (READ_ONCE(vma->anon_vma)) return false; diff --git a/mm/ksm.c b/mm/ksm.c index f80372bfd4b2fc..bcf5799bfe371f 100644 --- a/mm/ksm.c +++ b/mm/ksm.c @@ -775,7 +775,7 @@ static struct vm_area_struct *find_mergeable_vma(struct mm_struct *mm, if (ksm_test_exit(mm)) return NULL; vma = vma_lookup(mm, addr); - if (!vma || !(vma->vm_flags & VM_MERGEABLE) || !vma->anon_vma) + if (!vma || !(vma->vm_flags & VM_MERGEABLE) || !vma_has_anon_rmap(vma)) return NULL; return vma; } @@ -1239,7 +1239,7 @@ static int unmerge_and_remove_all_rmap_items(void) goto mm_exiting; for_each_vma(vmi, vma) { - if (!(vma->vm_flags & VM_MERGEABLE) || !vma->anon_vma) + if (!(vma->vm_flags & VM_MERGEABLE) || !vma_has_anon_rmap(vma)) continue; err = break_ksm(vma, vma->vm_start, vma->vm_end, false); if (err) @@ -2689,7 +2689,7 @@ static struct ksm_rmap_item *scan_get_next_rmap_item(struct page **page) continue; if (ksm_scan.address < vma->vm_start) ksm_scan.address = vma->vm_start; - if (!vma->anon_vma) + if (!vma_has_anon_rmap(vma)) ksm_scan.address = vma->vm_end; while (ksm_scan.address < vma->vm_end) { @@ -2880,7 +2880,7 @@ static int __ksm_del_vma(struct vm_area_struct *vma) if (!(vma->vm_flags & VM_MERGEABLE)) return 0; - if (vma->anon_vma) { + if (vma_has_anon_rmap(vma)) { err = break_ksm(vma, vma->vm_start, vma->vm_end, true); if (err) return err; @@ -3032,7 +3032,7 @@ int ksm_madvise(struct vm_area_struct *vma, unsigned long start, if (!(*vm_flags & VM_MERGEABLE)) return 0; /* just ignore the advice */ - if (vma->anon_vma) { + if (vma_has_anon_rmap(vma)) { err = break_ksm(vma, start, end, true); if (err) return err; diff --git a/mm/madvise.c b/mm/madvise.c index 32a28b9bb6880a..010ad3d47f5bb3 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -1331,7 +1331,7 @@ static long madvise_guard_install(struct madvise_behavior *madv_behavior) * as part of the VMA lock logic. */ if (vma_is_anonymous(vma)) { - VM_WARN_ON_ONCE(!vma->anon_vma && + VM_WARN_ON_ONCE(!vma_has_anon_rmap(vma) && madv_behavior->lock_mode != MADVISE_MMAP_READ_LOCK); err = anon_vma_prepare(vma); @@ -1793,7 +1793,7 @@ static bool is_vma_lock_sufficient(struct vm_area_struct *vma, * check overly paranoid which is safe. */ if (vma_is_anonymous(vma) && - prepares_anon_vma(madv_behavior->behavior) && !vma->anon_vma) + prepares_anon_vma(madv_behavior->behavior) && !vma_has_anon_rmap(vma)) return false; return true; diff --git a/mm/memory.c b/mm/memory.c index 338fce99e71197..544a9e5068f09d 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -1536,7 +1536,7 @@ vma_needs_copy(struct vm_area_struct *dst_vma, struct vm_area_struct *src_vma) * The presence of an anon_vma indicates an anonymous VMA has page * tables which naturally cannot be reconstituted on page fault. */ - if (src_vma->anon_vma) + if (vma_has_anon_rmap(src_vma)) return true; /* @@ -4011,7 +4011,7 @@ vm_fault_t __vmf_anon_prepare(struct vm_fault *vmf) struct vm_area_struct *vma = vmf->vma; vm_fault_t ret = 0; - if (likely(vma->anon_vma)) + if (likely(vma_has_anon_rmap(vma))) return 0; if (vmf->flags & FAULT_FLAG_VMA_LOCK) { if (!mmap_read_trylock(vma->vm_mm)) diff --git a/mm/mprotect.c b/mm/mprotect.c index a1b6d29bf03908..4b1296f0d502a2 100644 --- a/mm/mprotect.c +++ b/mm/mprotect.c @@ -816,7 +816,7 @@ mprotect_fixup(struct vma_iterator *vmi, struct mmu_gather *tlb, vma_flags_set(&new_vma_flags, VMA_ACCOUNT_BIT); } } else if (vma_flags_test(&old_vma_flags, VMA_ACCOUNT_BIT) && - vma_is_anonymous(vma) && !vma->anon_vma) { + vma_is_anonymous(vma) && !vma_has_anon_rmap(vma)) { vma_flags_clear(&new_vma_flags, VMA_ACCOUNT_BIT); } diff --git a/mm/mremap.c b/mm/mremap.c index 5c72545db1752c..ef1eb8a47f3b22 100644 --- a/mm/mremap.c +++ b/mm/mremap.c @@ -144,13 +144,13 @@ static void take_rmap_locks(struct vm_area_struct *vma) { if (vma->vm_file) i_mmap_lock_write(vma->vm_file->f_mapping); - if (vma->anon_vma) + if (vma_has_anon_rmap(vma)) anon_vma_lock_write(vma->anon_vma); } static void drop_rmap_locks(struct vm_area_struct *vma) { - if (vma->anon_vma) + if (vma_has_anon_rmap(vma)) anon_vma_unlock_write(vma->anon_vma); if (vma->vm_file) i_mmap_unlock_write(vma->vm_file->f_mapping); diff --git a/mm/rmap.c b/mm/rmap.c index 6661bc11ce658b..fbd66a2823b7b3 100644 --- a/mm/rmap.c +++ b/mm/rmap.c @@ -208,7 +208,7 @@ int __anon_vma_prepare(struct vm_area_struct *vma) anon_vma_lock_write(anon_vma); /* page_table_lock to protect against threads */ spin_lock(&mm->page_table_lock); - if (likely(!vma->anon_vma)) { + if (likely(!vma_has_anon_rmap(vma))) { /* * Make anon_vma fields visible before anon_vma is published. * Paired with an address dependency in reusable_anon_vma(). @@ -246,21 +246,21 @@ static void check_anon_vma_clone(struct vm_area_struct *dst, VM_WARN_ON_ONCE(operation != VMA_OP_FORK && dst->vm_mm != src->vm_mm); /* If we have anything to do src->anon_vma must be provided. */ - VM_WARN_ON_ONCE(!src->anon_vma && !list_empty(&src->anon_vma_chain)); - VM_WARN_ON_ONCE(!src->anon_vma && dst->anon_vma); + VM_WARN_ON_ONCE(!vma_has_anon_rmap(src) && !list_empty(&src->anon_vma_chain)); + VM_WARN_ON_ONCE(!vma_has_anon_rmap(src) && vma_has_anon_rmap(dst)); /* We are establishing a new anon_vma_chain. */ VM_WARN_ON_ONCE(!list_empty(&dst->anon_vma_chain)); /* * On fork, dst->anon_vma is set NULL (temporarily). Otherwise, anon_vma * must be the same across dst and src. */ - VM_WARN_ON_ONCE(dst->anon_vma && dst->anon_vma != src->anon_vma); + VM_WARN_ON_ONCE(vma_has_anon_rmap(dst) && dst->anon_vma != src->anon_vma); /* * Essentially equivalent to above - if not a no-op, we should expect * dst->anon_vma to be set for everything except a fork. */ - VM_WARN_ON_ONCE(operation != VMA_OP_FORK && src->anon_vma && - !dst->anon_vma); + VM_WARN_ON_ONCE(operation != VMA_OP_FORK && vma_has_anon_rmap(src) && + !vma_has_anon_rmap(dst)); /* For the anon_vma to be compatible, it can only be singular. */ VM_WARN_ON_ONCE(operation == VMA_OP_MERGE_UNFAULTED && !list_is_singular(&src->anon_vma_chain)); @@ -273,7 +273,7 @@ static void maybe_reuse_anon_vma(struct vm_area_struct *dst, struct anon_vma *anon_vma) { /* If already populated, nothing to do.*/ - if (dst->anon_vma) + if (vma_has_anon_rmap(dst)) return; /* @@ -327,7 +327,7 @@ int anon_vma_clone(struct vm_area_struct *dst, struct vm_area_struct *src, check_anon_vma_clone(dst, src, operation); - if (!active_anon_vma) + if (!vma_has_anon_rmap(src)) return 0; /* @@ -384,7 +384,7 @@ int anon_vma_fork(struct vm_area_struct *vma, struct vm_area_struct *pvma) int rc; /* Don't bother if the parent process has no anon_vma here. */ - if (!pvma->anon_vma) + if (!vma_has_anon_rmap(pvma)) return 0; /* Drop inherited anon_vma, we'll reuse existing or allocate new. */ @@ -405,7 +405,7 @@ int anon_vma_fork(struct vm_area_struct *vma, struct vm_area_struct *pvma) */ rc = anon_vma_clone(vma, pvma, VMA_OP_FORK); /* An error arose or an existing anon_vma was reused, all done then. */ - if (rc || vma->anon_vma) { + if (rc || vma_has_anon_rmap(vma)) { put_anon_vma(anon_vma); anon_vma_chain_free(avc); return rc; @@ -864,7 +864,7 @@ unsigned long page_address_in_vma(const struct folio *folio, * Note: swapoff's unuse_vma() is more efficient with this * check, and needs it to match anon_vma when KSM is active. */ - if (!vma->anon_vma || !anon_vma || + if (!vma_has_anon_rmap(vma) || !anon_vma || vma->anon_vma->root != anon_vma->root) return -EFAULT; /* KSM folios don't reach here because of the !anon_vma check */ diff --git a/mm/swapfile.c b/mm/swapfile.c index 0901cb8fa7291c..6187c02ec5ec73 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -2707,7 +2707,7 @@ static int unuse_mm(struct mm_struct *mm, unsigned int type) if (check_stable_address_space(mm)) goto unlock; for_each_vma(vmi, vma) { - if (vma->anon_vma && !vma_is_hugetlb(vma)) { + if (vma_has_anon_rmap(vma) && !vma_is_hugetlb(vma)) { ret = unuse_vma(vma, type); if (ret) break; diff --git a/mm/userfaultfd.c b/mm/userfaultfd.c index 17ecbb0ceddf11..666cc18902639b 100644 --- a/mm/userfaultfd.c +++ b/mm/userfaultfd.c @@ -145,7 +145,7 @@ static struct vm_area_struct *uffd_lock_vma(struct mm_struct *mm, * We know we're going to need to use anon_vma, so check * that early. */ - if (!(vma->vm_flags & VM_SHARED) && unlikely(!vma->anon_vma)) + if (!(vma->vm_flags & VM_SHARED) && unlikely(!vma_has_anon_rmap(vma))) vma_end_read(vma); else return vma; diff --git a/mm/vma.c b/mm/vma.c index ac3908d6967ab8..34c3a681864232 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -100,7 +100,8 @@ static bool vma_is_fork_child(struct vm_area_struct *vma) * parents. This can improve scalability caused by the anon_vma root * lock. */ - return vma && vma->anon_vma && !list_is_singular(&vma->anon_vma_chain); + return vma && vma_has_anon_rmap(vma) && + !list_is_singular(&vma->anon_vma_chain); } static inline bool is_mergeable_vma(struct vma_merge_struct *vmg, bool merge_next) @@ -140,7 +141,7 @@ static bool is_mergeable_anon_vma(struct vma_merge_struct *vmg, bool merge_next) VM_WARN_ON(src && src_anon != src->anon_vma); /* Case 1 - we will dup_anon_vma() from src into tgt. */ - if (!tgt_anon && src_anon) { + if (!vma_has_anon_rmap(tgt) && src_anon) { struct vm_area_struct *copied_from = vmg->copied_from; if (vma_is_fork_child(src)) @@ -151,7 +152,7 @@ static bool is_mergeable_anon_vma(struct vma_merge_struct *vmg, bool merge_next) return true; } /* Case 2 - we will simply use tgt's anon_vma. */ - if (tgt_anon && !src_anon) + if (vma_has_anon_rmap(tgt) && !src_anon) return !vma_is_fork_child(tgt); /* Case 3 - the anon_vma's are already shared. */ return src_anon == tgt_anon; @@ -190,10 +191,10 @@ static void init_multi_vma_prep(struct vma_prepare *vp, adjust = NULL; vp->adj_next = adjust; - if (!vp->anon_vma && adjust) + if (!vma_has_anon_rmap(vma) && adjust) vp->anon_vma = adjust->anon_vma; - VM_WARN_ON(vp->anon_vma && adjust && adjust->anon_vma && + VM_WARN_ON(vma_has_anon_rmap(vma) && adjust && vma_has_anon_rmap(adjust) && vp->anon_vma != adjust->anon_vma); vp->file = vma->vm_file; @@ -430,7 +431,7 @@ static void vma_complete(struct vma_prepare *vp, struct vma_iterator *vmi, vp->remove->vm_end); fput(vp->file); } - if (vp->remove->anon_vma) + if (vma_has_anon_rmap(vp->remove)) unlink_anon_vmas(vp->remove); mm->map_count--; mpol_put(vma_policy(vp->remove)); @@ -500,7 +501,7 @@ static bool can_vma_merge_right(struct vma_merge_struct *vmg, * We therefore check this in addition to mergeability to either side. */ prev = vmg->prev; - return !prev->anon_vma || !next->anon_vma || + return !vma_has_anon_rmap(prev) || !vma_has_anon_rmap(next) || prev->anon_vma == next->anon_vma; } @@ -670,7 +671,7 @@ static int dup_anon_vma(struct vm_area_struct *dst, * that is it is unfaulted, we need to ensure that the newly merged * range is referenced by the anon_vma's of the source. */ - if (src->anon_vma && !dst->anon_vma) { + if (vma_has_anon_rmap(src) && !vma_has_anon_rmap(dst)) { int ret; vma_assert_write_locked(dst); @@ -720,7 +721,7 @@ void validate_mm(struct mm_struct *mm) } #ifdef CONFIG_DEBUG_VM_RB - if (anon_vma) { + if (vma_has_anon_rmap(vma)) { anon_vma_lock_read(anon_vma); list_for_each_entry(avc, &vma->anon_vma_chain, same_vma) anon_rmap_tree_verify(avc); @@ -1020,7 +1021,7 @@ static __must_check struct vm_area_struct *vma_merge_existing_range( * simply a case of, if prev has no anon_vma object, which of * next or middle contains the anon_vma we must duplicate. */ - err = dup_anon_vma(prev, next->anon_vma ? next : middle, + err = dup_anon_vma(prev, vma_has_anon_rmap(next) ? next : middle, &anon_dup); } else if (merge_left) { /* @@ -1960,7 +1961,7 @@ struct vm_area_struct *copy_vma(struct vm_area_struct **vmap, * If a vma has not yet been faulted, update its anonymous pgoff to * match the new location to increase its chance of merging. */ - if (!vma->anon_vma) { + if (!vma_has_anon_rmap(vma)) { anon_pgoff = addr >> PAGE_SHIFT; if (vma_is_anonymous(vma)) { @@ -2387,7 +2388,7 @@ int mm_take_all_locks(struct mm_struct *mm) for_each_vma(vmi, vma) { if (signal_pending(current)) goto out_unlock; - if (vma->anon_vma) + if (vma_has_anon_rmap(vma)) list_for_each_entry(avc, &vma->anon_vma_chain, same_vma) vm_lock_anon_vma(mm, avc->anon_vma); } @@ -2449,7 +2450,7 @@ void mm_drop_all_locks(struct mm_struct *mm) BUG_ON(!mutex_is_locked(&mm_all_locks_mutex)); for_each_vma(vmi, vma) { - if (vma->anon_vma) + if (vma_has_anon_rmap(vma)) list_for_each_entry(avc, &vma->anon_vma_chain, same_vma) vm_unlock_anon_vma(avc->anon_vma); if (vma->vm_file && vma->vm_file->f_mapping) @@ -3597,7 +3598,7 @@ int insert_vm_struct(struct mm_struct *mm, struct vm_area_struct *vma) * Similarly in do_mmap and in do_brk_flags. */ if (vma_is_anonymous(vma)) { - WARN_ON_ONCE(vma->anon_vma); + WARN_ON_ONCE(vma_has_anon_rmap(vma)); vma_set_pgoff(vma, vma->vm_start >> PAGE_SHIFT); } vma_set_anon_pgoff(vma, vma->vm_start >> PAGE_SHIFT); diff --git a/mm/vma.h b/mm/vma.h index b9b99fa02a861d..7a683272c0a82a 100644 --- a/mm/vma.h +++ b/mm/vma.h @@ -255,6 +255,33 @@ static inline pgoff_t vmg_end_pgoff(const struct vma_merge_struct *vmg) return vmg_start_pgoff(vmg) + vmg_pages(vmg); } +/** + * vma_has_anon_rmap() - does @vma possess an anonymous reverse mapping? + * @vma: The VMA to be checked. + * + * If the VMA is attached, a VMA or mmap lock must be held. + * + * This state is only possible for CoW mappings, see the comment for + * vma_flags_is_cow_mapping() for details. + * + * Importantly, a VMA which possesses an anonymous rmap may map anonymous + * folios. + * + * This function will not result in a false positive. + * + * However, if only a read lock is held, it may give a false negative, in which + * case it should be re-checked with mm->page_table_lock held. + * + * Returns: true if @vma has an anonymous reverse mapping, otherwise false. + */ +static inline bool vma_has_anon_rmap(const struct vm_area_struct *vma) +{ + if (vma_is_attached(vma)) + vma_assert_stabilised(vma); + /* KCSAN gets confused about the optimistic check. Silence it. */ + return data_race(vma->anon_vma); +} + static inline void assert_sane_pgoff(struct vm_area_struct *vma, pgoff_t pgoff) { /* nommu doesn't set a virtual pgoff for anon VMAs. */ @@ -268,7 +295,7 @@ static inline void assert_sane_pgoff(struct vm_area_struct *vma, pgoff_t pgoff) if (!vma_is_anonymous(vma)) return; /* If faulted in, could have been remapped. */ - if (vma->anon_vma) + if (vma_has_anon_rmap(vma)) return; /* OK this is really an anon VMA - expect virtual page offset. */ VM_WARN_ON_ONCE(pgoff != vma->vm_start >> PAGE_SHIFT); diff --git a/tools/testing/vma/include/stubs.h b/tools/testing/vma/include/stubs.h index 48d1dc53df42cb..e4acc6f1fe7bab 100644 --- a/tools/testing/vma/include/stubs.h +++ b/tools/testing/vma/include/stubs.h @@ -302,6 +302,10 @@ static inline void vma_assert_write_locked(struct vm_area_struct *vma) { } +static inline void vma_assert_stabilised(const struct vm_area_struct *vma) +{ +} + static inline void ksm_add_vma(struct vm_area_struct *vma) { } From 7147f6e04e0c086ac23aed33a5a6390c6bdc7dc6 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 20:47:43 +0100 Subject: [PATCH 1062/1352] mm: update comments to refer to anon rmap rather than anon_vma Now that vma_has_anon_rmap() abstracts whether a VMA has an anonymous reverse mapping, remove references to anon_vma and instead reference the anon rmap. The anon_vma is an implementation detail and should be treated as such. Do not update mm/rmap.c which implements the anon_vma mechanism as it is reasonable to directly reference it there. No functional change intended. Link: https://lore.kernel.org/20260917-vma-is-faulted-v3-3-5c22314a72e7@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Pedro Falcato Reviewed-by: Lance Yang Cc: Alistair Popple Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Byungchul Park Cc: Chengming Zhou Cc: Chris Li Cc: Dev Jain Cc: Gregory Price Cc: Guilherme Giacomo Simoes Cc: Harry Yoo Cc: "Huang, Ying" Cc: Jann Horn Cc: Joshua Hahn Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Liam R. Howlett Cc: Matthew Brost Cc: Michal Hocko Cc: Mike Rapoport Cc: Muchun Song Cc: Nhat Pham Cc: Oscar Salvador Cc: Peter Xu Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Shakeel Butt Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: xu xin Cc: Zi Yan --- mm/huge_memory.c | 8 ++-- mm/hugetlb.c | 2 +- mm/khugepaged.c | 12 +++--- mm/ksm.c | 6 +-- mm/madvise.c | 6 +-- mm/memory.c | 12 +++--- mm/migrate.c | 12 +++--- mm/mmap.c | 6 +-- mm/mprotect.c | 4 +- mm/mremap.c | 6 +-- mm/pgtable-generic.c | 2 +- mm/userfaultfd.c | 10 ++--- mm/vma.c | 98 ++++++++++++++++++++++---------------------- 13 files changed, 92 insertions(+), 92 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index e0e252cfbf9580..ba5e20bbfd3bc3 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -254,10 +254,10 @@ unsigned long __thp_vma_allowable_orders(struct vm_area_struct *vma, /* * THPeligible bit of smaps should show 1 for proper VMAs even - * though anon_vma is not initialized yet. + * though they don't have an anon rmap yet. * - * Allow page fault since anon_vma may be not initialized until - * the first page fault. + * Allow page fault since the VMA may not have an anon rmap until the + * first page fault. */ if (!vma_has_anon_rmap(vma)) return (smaps || in_pf) ? orders : 0; @@ -4370,7 +4370,7 @@ static int __folio_split(struct folio *folio, unsigned int new_order, * THP pages in the middle of migration, due to allocation issues on either * side. * - * anon_vma_lock is not required to be held, mmap_read_lock() or + * The anon rmap lock is not required to be held, mmap_read_lock() or * mmap_write_lock() should be held. @folio is expected to be locked by the * caller. device-private and non device-private folios are supported along * with folios that are in the swapcache. @folio should also be unmapped and diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 9d05fecf21326d..da980377d35339 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -5677,7 +5677,7 @@ static vm_fault_t hugetlb_wp(struct vm_fault *vmf) /* * When the original hugepage is shared one, it does not have - * anon_vma prepared. + * an anon rmap prepared. */ ret = __vmf_anon_prepare(vmf); if (unlikely(ret)) diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 3e8dbd2dd46d8d..913086eaf17ba0 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -1315,7 +1315,7 @@ static enum scan_result collapse_huge_page(struct mm_struct *mm, /* * Prevent all access to pagetables with the exception of * gup_fast later handled by the pmdp_collapse_flush() and the VM - * handled by the anon_vma lock + folio lock. + * handled by the anon rmap lock + folio lock. * * UFFDIO_MOVE is prevented to race as well thanks to the * mmap_lock. @@ -1382,8 +1382,8 @@ static enum scan_result collapse_huge_page(struct mm_struct *mm, } /* - * For PMD collapse all pages are isolated and locked so anon_vma - * rmap can't run anymore. For mTHP collapse the PMD entry has been + * For PMD collapse all pages are isolated and locked so the anon + * rmap walk can't run anymore. For mTHP collapse the PMD entry has been * removed and not all pages are isolated and locked, so we must hold * the lock to prevent neighboring folios from attempting to access * this PMD until its reinstalled. @@ -2168,9 +2168,9 @@ static void retract_page_tables(struct address_space *mapping, pgoff_t pgoff) /* * Huge page lock is still held, so normally the page table must - * remain empty; and we have already skipped anon_vma and - * userfaultfd_wp() vmas. But since the mmap_lock is not held, - * it is still possible for a racing userfaultfd_ioctl() or + * remain empty; and we have already skipped vmas with an anon + * rmap and userfaultfd_wp() vmas. But since the mmap_lock is not + * held, it is still possible for a racing userfaultfd_ioctl() or * madvise() to have inserted ptes or markers. Now that we hold * ptlock, repeating the retractable checks protects us from * races against the prior checks. diff --git a/mm/ksm.c b/mm/ksm.c index bcf5799bfe371f..fbeae63ced2043 100644 --- a/mm/ksm.c +++ b/mm/ksm.c @@ -793,7 +793,7 @@ static void break_cow(struct ksm_rmap_item *rmap_item) /* * It is not an accident that whenever we want to break COW - * to undo, we also need to drop a reference to the anon_vma. + * to undo, we also need to drop a reference to the anon rmap. */ put_anon_vma(rmap_item->anon_vma); /* @@ -1413,7 +1413,7 @@ static int replace_page(struct vm_area_struct *vma, struct page *page, goto out; /* * Some THP functions use the sequence pmdp_huge_clear_flush(), set_pmd_at() - * without holding anon_vma lock for write. So when looking for a + * without holding the anon rmap lock for write. So when looking for a * genuine pmde (in which to find pte), test present and !THP together. */ pmde = pmdp_get_lockless(pmd); @@ -1617,7 +1617,7 @@ static int try_to_merge_with_ksm_page(struct ksm_rmap_item *rmap_item, /* * We can consider the VMA only while still holding the mmap lock, - * so lock, so reference the anon_vma and calculate the linear + * so lock, so reference the anon rmap and calculate the linear * page index early, before stable_tree_append(). If anything goes * wrong that prevents the rmap_item from being added to the * stable_tree, break_cow() will clean it up. diff --git a/mm/madvise.c b/mm/madvise.c index 010ad3d47f5bb3..20135275cb5550 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -1325,7 +1325,7 @@ static long madvise_guard_install(struct madvise_behavior *madv_behavior) /* * If anonymous and we are establishing page tables the VMA ought to - * have an anon_vma associated with it. + * have an anon rmap associated with it. * * We will hold an mmap read lock if this is necessary, this is checked * as part of the VMA lock logic. @@ -1789,8 +1789,8 @@ static bool is_vma_lock_sufficient(struct vm_area_struct *vma, * anon_vma_prepare() explicitly requires an mmap lock for * serialisation, so we cannot use a VMA lock in this case. * - * Note we might race with anon_vma being set, however this makes this - * check overly paranoid which is safe. + * Note we might race with the anon rmap being assigned, however this + * makes this check overly paranoid which is safe. */ if (vma_is_anonymous(vma) && prepares_anon_vma(madv_behavior->behavior) && !vma_has_anon_rmap(vma)) diff --git a/mm/memory.c b/mm/memory.c index 544a9e5068f09d..6349ef676549a8 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -1291,8 +1291,8 @@ copy_pte_range(struct vm_area_struct *dst_vma, struct vm_area_struct *src_vma, * copy_pmd_range()'s prior pmd_none_or_clear_bad(src_pmd), and the * error handling here, assume that exclusive mmap_lock on dst and src * protects anon from unexpected THP transitions; with shmem and file - * protected by mmap_lock-less collapse skipping areas with anon_vma - * (whereas vma_needs_copy() skips areas without anon_vma). A rework + * protected by mmap_lock-less collapse skipping areas with an anon rmap + * (whereas vma_needs_copy() skips areas without one). A rework * can remove such assumptions later, but this is good enough for now. */ dst_pte = pte_alloc_map_lock(dst_mm, dst_pmd, addr, &dst_ptl); @@ -1533,8 +1533,8 @@ vma_needs_copy(struct vm_area_struct *dst_vma, struct vm_area_struct *src_vma) if (dst_vma->vm_flags & VM_COPY_ON_FORK) return true; /* - * The presence of an anon_vma indicates an anonymous VMA has page - * tables which naturally cannot be reconstituted on page fault. + * The presence of an anon rmap indicates the VMA may map anonymous + * folios which naturally cannot be reconstituted on page fault. */ if (vma_has_anon_rmap(src_vma)) return true; @@ -3997,10 +3997,10 @@ static inline vm_fault_t vmf_can_call_fault(const struct vm_fault *vmf) * * When preparing to insert an anonymous page into a VMA from a * fault handler, call this function rather than anon_vma_prepare(). - * If this vma does not already have an associated anon_vma and we are + * If this vma does not already have an anon rmap and we are * only protected by the per-VMA lock, the caller must retry with the * mmap_lock held. __anon_vma_prepare() will look at adjacent VMAs to - * determine if this VMA can share its anon_vma, and that's not safe to + * determine if this VMA can share its anon rmap, and that's not safe to * do with only the per-VMA lock held for this VMA. * * Return: 0 if fault handling can proceed. Any other value should be diff --git a/mm/migrate.c b/mm/migrate.c index 7e3a81f0697442..7bdcdb57652f8f 100644 --- a/mm/migrate.c +++ b/mm/migrate.c @@ -1174,7 +1174,7 @@ static void migrate_folio_undo_src(struct folio *src, int was_mapped, { if (was_mapped) remove_migration_ptes(src, src, 0); - /* Drop an anon_vma reference if we took one */ + /* Drop an anon rmap reference if we took one */ if (anon_vma) put_anon_vma(anon_vma); if (locked) @@ -1281,15 +1281,15 @@ static int migrate_folio_unmap(new_folio_t get_new_folio, /* * By try_to_migrate(), src->mapcount goes down to 0 here. In this case, - * we cannot notice that anon_vma is freed while we migrate a page. - * This get_anon_vma() delays freeing anon_vma pointer until the end + * we cannot notice that the anon rmap is freed while we migrate a page. + * This get_anon_vma() delays freeing the anon rmap until the end * of migration. File cache pages are no problem because of page_lock() * File Caches may use write_page() or lock_page() in migration, then, * just care Anon page here. * * Only folio_get_anon_vma() understands the subtleties of - * getting a hold on an anon_vma from outside one of its mms. - * But if we cannot get anon_vma, then we won't need it anyway, + * getting a hold on an anon rmap from outside one of its mms. + * But if we cannot get the anon rmap, then we won't need it anyway, * because that implies that the anon page is no longer mapped * (and cannot be remapped so long as we hold the page lock). */ @@ -1432,7 +1432,7 @@ static int migrate_folio_move(free_folio_t put_new_folio, unsigned long private, * and will be freed. */ list_del(&src->lru); - /* Drop an anon_vma reference if we took one */ + /* Drop an anon rmap reference if we took one */ if (anon_vma) put_anon_vma(anon_vma); folio_unlock(src); diff --git a/mm/mmap.c b/mm/mmap.c index 98449f364af1c4..148ba01c03739f 100644 --- a/mm/mmap.c +++ b/mm/mmap.c @@ -547,7 +547,7 @@ unsigned long do_mmap(struct file *file, unsigned long addr, } case MAP_PRIVATE: /* - * Set pgoff according to addr for anon_vma. + * Set pgoff according to addr for the anon rmap. */ pgoff = addr >> PAGE_SHIFT; break; @@ -1774,8 +1774,8 @@ __latent_entropy int dup_mmap(struct mm_struct *mm, struct mm_struct *oldmm) if (vma_test(tmp, VMA_WIPEONFORK_BIT)) { /* * VMA_WIPEONFORK_BIT gets a clean slate in the child. - * Don't prepare anon_vma until fault since we don't - * copy page for current vma. + * Don't prepare the anon rmap until fault since we + * don't copy pages for the current vma. */ tmp->anon_vma = NULL; } else if (anon_vma_fork(tmp, mpnt)) diff --git a/mm/mprotect.c b/mm/mprotect.c index 4b1296f0d502a2..e59c69cb5a2388 100644 --- a/mm/mprotect.c +++ b/mm/mprotect.c @@ -796,8 +796,8 @@ mprotect_fixup(struct vma_iterator *vmi, struct mmu_gather *tlb, /* * If we make a private mapping writable we increase our commit; * but (without finer accounting) cannot reduce our commit if we - * make it unwritable again except in the anonymous case where no - * anon_vma has yet to be assigned. + * make it unwritable again except in the anonymous case where the + * VMA's anon rmap has yet to be assigned. * * hugetlb mapping were accounted for even if read-only so there is * no need to account for them here. diff --git a/mm/mremap.c b/mm/mremap.c index ef1eb8a47f3b22..2e1ee879f70fd9 100644 --- a/mm/mremap.c +++ b/mm/mremap.c @@ -214,7 +214,7 @@ static int move_ptes(struct pagetable_move_control *pmc, int err = 0; /* - * When need_rmap_locks is true, we take the i_mmap_rwsem and anon_vma + * When need_rmap_locks is true, we take the i_mmap_rwsem and anon rmap * locks to ensure that rmap will always observe either the old or the * new ptes. This is the easiest way to avoid races with * truncate_pagecache(), page migration, etc... @@ -1373,8 +1373,8 @@ static void dontunmap_complete(struct vma_remap_struct *vrm, vma_clear_flags_mask(vma, VMA_LOCKED_MASK); /* - * anon_vma links of the old vma is no longer needed after its page - * table has been moved. + * The anon rmap links of the old vma are no longer needed after its + * page table has been moved. */ if (start == old_start && end == old_end) { const pgoff_t pgoff_unfaulted = vma->vm_start >> PAGE_SHIFT; diff --git a/mm/pgtable-generic.c b/mm/pgtable-generic.c index 26643d76bfb00e..6e83ce3801b8af 100644 --- a/mm/pgtable-generic.c +++ b/mm/pgtable-generic.c @@ -349,7 +349,7 @@ pte_t *pte_offset_map_rw_nolock(struct mm_struct *mm, pmd_t *pmd, * pte_offset_map_lock(mm, pmd, addr, ptlp) is usually called with the pmd * pointer for addr, reached by walking down the mm's pgd, p4d, pud for addr: * either while holding mmap_lock or vma lock for read or for write; or in - * truncate or rmap context, while holding file's i_mmap_lock or anon_vma lock + * truncate or rmap context, while holding file's i_mmap_lock or anon rmap lock * for read (or for write). In a few cases, it may be used with pmd pointing to * a pmd_t already copied to or constructed on the stack. * diff --git a/mm/userfaultfd.c b/mm/userfaultfd.c index 666cc18902639b..3c7fd39deb1376 100644 --- a/mm/userfaultfd.c +++ b/mm/userfaultfd.c @@ -130,7 +130,7 @@ struct vm_area_struct *find_vma_and_prepare_anon(struct mm_struct *mm, * Should be called without holding mmap_lock. * * Return: A locked vma containing @address, -ENOENT if no vma is found, - * -ENOMEM if anon_vma couldn't be allocated, or -EAGAIN if vma refcount + * -ENOMEM if the anon rmap couldn't be allocated, or -EAGAIN if vma refcount * overflow happened due to high number of readers and the caller should * retry later. */ @@ -142,8 +142,8 @@ static struct vm_area_struct *uffd_lock_vma(struct mm_struct *mm, vma = lock_vma_under_rcu(mm, address); if (vma) { /* - * We know we're going to need to use anon_vma, so check - * that early. + * We know we're going to need an anon rmap, so check that + * early. */ if (!(vma->vm_flags & VM_SHARED) && unlikely(!vma_has_anon_rmap(vma))) vma_end_read(vma); @@ -1681,7 +1681,7 @@ static long move_pages_ptes(struct mm_struct *mm, pmd_t *dst_pmd, pmd_t *src_pmd /* * Verify the existence of the swapcache. If present, the folio's * index and mapping must be updated even when the PTE is a swap - * entry. The anon_vma lock is not taken during this process since + * entry. The anon rmap lock is not taken during this process since * the folio has already been unmapped, and the swap entry is * exclusive, preventing rmap walks. * @@ -1918,7 +1918,7 @@ static void uffd_move_unlock(struct vm_area_struct *dst_vma, * * move_pages() remaps arbitrary anonymous pages atomically in zero * copy. It only works on non shared anonymous pages because those can - * be relocated without generating non linear anon_vmas in the rmap + * be relocated without generating non linear anon rmaps in the rmap * code. * * It provides a zero copy mechanism to handle userspace page faults. diff --git a/mm/vma.c b/mm/vma.c index 34c3a681864232..42e4e674f783ff 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -97,7 +97,7 @@ static bool vma_is_fork_child(struct vm_area_struct *vma) { /* * The list_is_singular() test is to avoid merging VMA cloned from - * parents. This can improve scalability caused by the anon_vma root + * parents. This can improve scalability caused by the anon rmap root * lock. */ return vma && vma_has_anon_rmap(vma) && @@ -135,7 +135,7 @@ static bool is_mergeable_anon_vma(struct vma_merge_struct *vmg, bool merge_next) /* * We _can_ have !src, vmg->anon_vma via copy_vma(). In this instance we - * will remove the existing VMA's anon_vma's so there's no scalability + * will remove the existing VMA's anon rmap so there's no scalability * concerns. */ VM_WARN_ON(src && src_anon != src->anon_vma); @@ -151,10 +151,10 @@ static bool is_mergeable_anon_vma(struct vma_merge_struct *vmg, bool merge_next) return true; } - /* Case 2 - we will simply use tgt's anon_vma. */ + /* Case 2 - we will simply use tgt's anon rmap. */ if (vma_has_anon_rmap(tgt) && !src_anon) return !vma_is_fork_child(tgt); - /* Case 3 - the anon_vma's are already shared. */ + /* Case 3 - src and tgt already share an anon rmap. */ return src_anon == tgt_anon; } @@ -228,8 +228,8 @@ static bool needs_adjacent_anon_pgoff(const struct vma_merge_struct *vmg) * Return true if we can merge this (vma_flags,anon_vma,file,vm_pgoff) * in front of (at a lower virtual address and file offset than) the vma. * - * We cannot merge two vmas if they have differently assigned (non-NULL) - * anon_vmas, nor if same anon_vma is assigned but offsets incompatible. + * We cannot merge two vmas if they have differently assigned anon rmaps, + * nor if the same anon rmap is assigned but offsets incompatible. * * We don't check here for the merged mmap wrapping around the end of pagecache * indices (16TB on ia32) because do_mmap() does not permit mmap's which @@ -255,8 +255,8 @@ static bool can_vma_merge_before(struct vma_merge_struct *vmg) * Return true if we can merge this (vma_flags,anon_vma,file,vm_pgoff) * beyond (at a higher virtual address and file offset than) the vma. * - * We cannot merge two vmas if they have differently assigned (non-NULL) - * anon_vmas, nor if same anon_vma is assigned but offsets incompatible. + * We cannot merge two vmas if they have differently assigned anon rmaps, + * nor if the same anon rmap is assigned but offsets incompatible. * * We assume that vma is not removed as part of the merge. */ @@ -300,18 +300,18 @@ static void __remove_shared_vm_struct(struct vm_area_struct *vma, } /* - * vma has some anon_vma assigned, and is already inserted on that - * anon_vma's interval trees. + * vma has an anon rmap assigned, and is already inserted on its interval + * trees. * * Before updating the vma's vm_start / vm_end / vm_pgoff fields, the - * vma must be removed from the anon_vma's interval trees using + * vma must be removed from the anon rmap's interval trees using * anon_rmap_tree_pre_update_vma(). * * After the update, the vma will be reinserted using * anon_rmap_tree_post_update_vma(). * * The entire update must be protected by exclusive mmap_lock and by - * the root anon_vma's mutex. + * the anon rmap root lock. */ static void anon_rmap_tree_pre_update_vma(struct vm_area_struct *vma) @@ -479,7 +479,7 @@ static bool can_vma_merge_left(struct vma_merge_struct *vmg) * account the end position of the proposed range. * * In addition, if we can merge with the left VMA, ensure that left and right - * anon_vma's are also compatible. + * anon rmaps are also compatible. */ static bool can_vma_merge_right(struct vma_merge_struct *vmg, bool can_merge_left) @@ -495,7 +495,7 @@ static bool can_vma_merge_right(struct vma_merge_struct *vmg, /* * If we can merge with prev (left) and next (right), indicating that - * each VMA's anon_vma is compatible with the proposed anon_vma, this + * each VMA's anon rmap is compatible with the proposed anon rmap, this * does not mean prev and next are compatible with EACH OTHER. * * We therefore check this in addition to mergeability to either side. @@ -645,8 +645,8 @@ int split_vma(struct vma_iterator *vmi, struct vm_area_struct *vma, } /* - * dup_anon_vma() - Helper function to duplicate anon_vma on VMA merge in the - * instance that the destination VMA has no anon_vma but the source does. + * dup_anon_vma() - Helper function to duplicate the anon rmap on VMA merge in + * the instance that the destination VMA has no anon rmap but the source does. * * @dst: The destination VMA * @src: The source VMA @@ -659,17 +659,17 @@ static int dup_anon_vma(struct vm_area_struct *dst, { /* * There are three cases to consider for correctly propagating - * anon_vma's on merge. + * anon rmaps on merge. * - * The first is trivial - neither VMA has anon_vma, we need not do + * The first is trivial - neither VMA has an anon rmap, we need not do * anything. * - * The second where both have anon_vma is also a no-op, as they must + * The second where both have an anon rmap is also a no-op, as they must * then be the same, so there is simply nothing to copy. * - * Here we cover the third - if the destination VMA has no anon_vma, + * Here we cover the third - if the destination VMA has no anon rmap, * that is it is unfaulted, we need to ensure that the newly merged - * range is referenced by the anon_vma's of the source. + * range is referenced by the anon rmap of the source. */ if (vma_has_anon_rmap(src) && !vma_has_anon_rmap(dst)) { int ret; @@ -1017,9 +1017,9 @@ static __must_check struct vm_area_struct *vma_merge_existing_range( vmg->anon_pgoff = vma_start_anon_pgoff(prev); /* - * We already ensured anon_vma compatibility above, so now it's - * simply a case of, if prev has no anon_vma object, which of - * next or middle contains the anon_vma we must duplicate. + * We already ensured anon rmap compatibility above, so now it's + * simply a case of, if prev has no anon rmap, which of next or + * middle contains the anon rmap we must duplicate. */ err = dup_anon_vma(prev, vma_has_anon_rmap(next) ? next : middle, &anon_dup); @@ -1085,7 +1085,7 @@ static __must_check struct vm_area_struct *vma_merge_existing_range( unlink_anon_vmas(anon_dup); /* - * This means we have failed to clone anon_vma's correctly, but no + * This means we have failed to clone the anon rmap correctly, but no * actual changes to VMAs have occurred, so no harm no foul - if the * user doesn't want this reported and instead just wants to give up on * the merge, allow it. @@ -1282,7 +1282,7 @@ int vma_expand(struct vma_merge_struct *vmg) /* * If we are removing the next VMA or copying from a VMA - * (e.g. mremap()'ing), we must propagate anon_vma state. + * (e.g. mremap()'ing), we must propagate anon rmap state. * * Note that, by convention, callers ignore OOM for this case, so * we don't need to account for vmg->give_up_on_mm here. @@ -2059,16 +2059,16 @@ struct vm_area_struct *copy_vma(struct vm_area_struct **vmap, /* * Rough compatibility check to quickly see if it's even worth looking - * at sharing an anon_vma. + * at sharing an anon rmap. * * They need to have the same vm_file, and the flags can only differ * in things that mprotect may change. * - * NOTE! The fact that we share an anon_vma doesn't _have_ to mean that + * NOTE! The fact that we share an anon rmap doesn't _have_ to mean that * we can merge the two vma's. For example, we refuse to merge a vma if * there is a vm_ops->close() function, because that indicates that the * driver is doing some kind of reference counting. But that doesn't - * really matter for the anon_vma sharing case. + * really matter for the anon rmap sharing case. */ static int anon_vma_compatible(struct vm_area_struct *a, struct vm_area_struct *b) { @@ -2101,13 +2101,13 @@ static int anon_vma_compatible(struct vm_area_struct *a, struct vm_area_struct * } /* - * Do some basic sanity checking to see if we can re-use the anon_vma + * Do some basic sanity checking to see if we can re-use the anon rmap * from 'old'. The 'a'/'b' vma's are in VM order - one of them will be * the same as 'old', the other will be the new one that is trying - * to share the anon_vma. + * to share the anon rmap. * * NOTE! This runs with mmap_lock held for reading, so it is possible that - * the anon_vma of 'old' is concurrently in the process of being set up + * the anon rmap of 'old' is concurrently in the process of being set up * by another page fault trying to merge _that_. But that's ok: if it * is being set up, that automatically means that it will be a singleton * acceptable for merging, so we can do all of this optimistically. But @@ -2121,8 +2121,8 @@ static int anon_vma_compatible(struct vm_area_struct *a, struct vm_area_struct * * accessing an uninitialised anon_vma's fields may result in a UAF. * * IOW: that the "list_is_singular()" test on the anon_vma_chain only - * matters for the 'stable anon_vma' case (ie the thing we want to avoid - * is to return an anon_vma that is "complex" due to having gone through + * matters for the 'stable anon rmap' case (ie the thing we want to avoid + * is to return an anon rmap that is "complex" due to having gone through * a fork). * * We also make sure that the two vma's are compatible (adjacent, @@ -2145,10 +2145,10 @@ static struct anon_vma *reusable_anon_vma(struct vm_area_struct *old, /* * find_mergeable_anon_vma is used by anon_vma_prepare, to check - * neighbouring vmas for a suitable anon_vma, before it goes off - * to allocate a new anon_vma. It checks because a repetitive + * neighbouring vmas for a suitable anon rmap, before it goes off + * to allocate a new anon rmap. It checks because a repetitive * sequence of mprotects and faults may otherwise lead to distinct - * anon_vmas being allocated, preventing vma merge in subsequent + * anon rmaps being allocated, preventing vma merge in subsequent * mprotect. */ struct anon_vma *find_mergeable_anon_vma(struct vm_area_struct *vma) @@ -2174,13 +2174,13 @@ struct anon_vma *find_mergeable_anon_vma(struct vm_area_struct *vma) /* * We might reach here with anon_vma == NULL if we can't find - * any reusable anon_vma. + * any reusable anon rmap. * There's no absolute need to look only at touching neighbours: - * we could search further afield for "compatible" anon_vmas. + * we could search further afield for "compatible" anon rmaps. * But it would probably just be a waste of time searching, - * or lead to too many vmas hanging off the same anon_vma. + * or lead to too many vmas hanging off the same anon rmap. * We're trying to allow mprotect remerging later on, - * not trying to minimize memory used for anon_vmas. + * not trying to minimize memory used for anon rmaps. */ return anon_vma; } @@ -2276,7 +2276,7 @@ static void vm_lock_anon_vma(struct mm_struct *mm, struct anon_vma *anon_vma) /* * We can safely modify head.next after taking the * anon_vma->root->rwsem. If some other vma in this mm shares - * the same anon_vma we won't take it again. + * the same anon rmap we won't take it again. * * No need of atomic instructions here, head.next * can't change from under us thanks to the @@ -2318,14 +2318,14 @@ static void vm_lock_mapping(struct mm_struct *mm, struct address_space *mapping) * mmap_lock in write mode is required in order to block all operations * that could modify pagetables and free pages without need of * altering the vma layout. It's also needed in write mode to avoid new - * anon_vmas to be associated with existing vmas. + * anon rmaps being associated with existing vmas. * * A single task can't take more than one mm_take_all_locks() in a row * or it would deadlock. * * The LSB in anon_vma->rb_root.rb_node and the AS_MM_ALL_LOCKS bitflag in * mapping->flags avoid to take the same lock twice, if more than one - * vma in this mm is backed by the same anon_vma or address_space. + * vma in this mm is backed by the same anon rmap or address_space. * * We take locks in following order, accordingly to comment at beginning * of mm/rmap.c: @@ -3419,7 +3419,7 @@ int expand_upwards(struct vm_area_struct *vma, unsigned long address) if (next && vma_is_accessible(next)) { if (!vma_test(next, VMA_GROWSUP_BIT)) return -ENOMEM; - /* Check that both stack segments have the same anon_vma? */ + /* Check that both stack segments have the same anon rmap? */ } if (next) @@ -3429,7 +3429,7 @@ int expand_upwards(struct vm_area_struct *vma, unsigned long address) if (vma_iter_prealloc(&vmi, vma)) return -ENOMEM; - /* We must make sure the anon_vma is allocated. */ + /* We must make sure the anon rmap is allocated. */ if (unlikely(anon_vma_prepare(vma))) { vma_iter_free(&vmi); return -ENOMEM; @@ -3492,7 +3492,7 @@ int expand_downwards(struct vm_area_struct *vma, unsigned long address) /* Enforce stack_guard_gap */ prev = vma_prev(&vmi); - /* Check that both stack segments have the same anon_vma? */ + /* Check that both stack segments have the same anon rmap? */ if (prev) { if (!vma_test(prev, VMA_GROWSDOWN_BIT) && vma_is_accessible(prev) && @@ -3507,7 +3507,7 @@ int expand_downwards(struct vm_area_struct *vma, unsigned long address) if (vma_iter_prealloc(&vmi, vma)) return -ENOMEM; - /* We must make sure the anon_vma is allocated. */ + /* We must make sure the anon rmap is allocated. */ if (unlikely(anon_vma_prepare(vma))) { vma_iter_free(&vmi); return -ENOMEM; @@ -3587,7 +3587,7 @@ int insert_vm_struct(struct mm_struct *mm, struct vm_area_struct *vma) /* * The vm_pgoff of a purely anonymous vma should be irrelevant - * until its first write fault, when page's anon_vma and index + * until its first write fault, when page's anon rmap and index * are set. But now set the vm_pgoff it will almost certainly * end up with (unless mremap moves it elsewhere before that * first wfault), so /proc/pid/maps tells a consistent story. From e8021e4c02fceb609de78469c64d78cd1643ed45 Mon Sep 17 00:00:00 2001 From: Andrew Morton Date: Thu, 17 Sep 2026 16:39:24 -0700 Subject: [PATCH 1063/1352] mm-update-comments-to-refer-to-anon-rmap-rather-than-anon_vma-fix fix comment, per Zi Yan. Cc: "Lorenzo Stoakes (ARM)" Cc: Zi Yan Signed-off-by: Andrew Morton --- mm/ksm.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/ksm.c b/mm/ksm.c index fbeae63ced2043..f33523a84a1a13 100644 --- a/mm/ksm.c +++ b/mm/ksm.c @@ -1617,7 +1617,7 @@ static int try_to_merge_with_ksm_page(struct ksm_rmap_item *rmap_item, /* * We can consider the VMA only while still holding the mmap lock, - * so lock, so reference the anon rmap and calculate the linear + * so reference the anon rmap and calculate the linear * page index early, before stable_tree_append(). If anything goes * wrong that prevents the rmap_item from being added to the * stable_tree, break_cow() will clean it up. From a3b90e961e64208202ff4d2070066fdd54d47b35 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 17 Sep 2026 07:21:59 -0700 Subject: [PATCH 1064/1352] mm/damon/api: remove NR_DAMOS_FILTER_TYPES Patch series "mm/damon: improve readability, clarity and test coverage". Yet another batch of miscellaneous DAMON minor improvements. Mostly focused on readability and clarity of code and document, and unit/self test coverage. No user-visible behavioral change is intended. This patch (of 10): Nobody uses NR_DAMOS_FILTER_TYPES. Remove it. Link: https://lore.kernel.org/20260917142210.90829-1-sj@kernel.org Link: https://lore.kernel.org/20260917142210.90829-2-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Asier Gutierrez Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/damon.h | 2 -- 1 file changed, 2 deletions(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index bbb190b4740150..836353c4ab9aab 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -400,7 +400,6 @@ struct damos_stat { * @DAMOS_FILTER_TYPE_ADDR: Address range. * @DAMOS_FILTER_TYPE_TARGET: Data Access Monitoring target. * @DAMOS_FILTER_TYPE_PROBE_HITS_WSUM: probe_hits weighted sum range. - * @NR_DAMOS_FILTER_TYPES: Number of filter types. * * All types except &DAMOS_FILTER_TYPE_ADDR, &DAMOS_FILTER_TYPE_TARGET and * &DAMOS_FILTER_TYPE_PROBE_HITS_WSUM are handled by the underlying &struct @@ -422,7 +421,6 @@ enum damos_filter_type { DAMOS_FILTER_TYPE_ADDR, DAMOS_FILTER_TYPE_TARGET, DAMOS_FILTER_TYPE_PROBE_HITS_WSUM, - NR_DAMOS_FILTER_TYPES, }; /** From 5d0fbe2865a8f09dc0eb116fb32448850330c25c Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 17 Sep 2026 07:22:00 -0700 Subject: [PATCH 1065/1352] mm/damon/core: use abs_diff() in damon_feed_loop_next_input() damon_feed_loop_next_input() is open-coding absolute diff calculation instead of the dedicated helper, abs_diff(), for no good reason. Use the dedicated helper. Link: https://lore.kernel.org/20260917142210.90829-3-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Asier Gutierrez Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/core.c | 5 +---- 1 file changed, 1 insertion(+), 4 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index e1b49c3d72b865..e49bbf7c07e85e 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2926,10 +2926,7 @@ static unsigned long damon_feed_loop_next_input(unsigned long last_input, if (score >= goal * 2) return min_input; - if (over_achieving) - score_goal_diff = score - goal; - else - score_goal_diff = goal - score; + score_goal_diff = abs_diff(score, goal); if (last_input < ULONG_MAX / score_goal_diff) compensation = last_input * score_goal_diff / goal; From 50e38f43a316570741a837ec7d2a51e7424bf9fb Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 17 Sep 2026 07:22:01 -0700 Subject: [PATCH 1066/1352] mm/damon/core: use mult_frac() in damon_feed_loop_next_input() damon_feed_loop_next_input() does its best effort overflow protection. score_goal_diff is always smaller than goal (10,000). Hence the calculation can be replaced to use mult_frac() without concerning the overflow. Use mult_frac(). Link: https://lore.kernel.org/20260917142210.90829-4-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Asier Gutierrez Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/core.c | 6 +----- 1 file changed, 1 insertion(+), 5 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index e49bbf7c07e85e..f3ccfc9ad0f086 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2927,11 +2927,7 @@ static unsigned long damon_feed_loop_next_input(unsigned long last_input, return min_input; score_goal_diff = abs_diff(score, goal); - - if (last_input < ULONG_MAX / score_goal_diff) - compensation = last_input * score_goal_diff / goal; - else - compensation = last_input / goal * score_goal_diff; + compensation = mult_frac(last_input, score_goal_diff, goal); if (over_achieving) return max(last_input - compensation, min_input); From f777b1f45e2e24a8fc8ef4585b6bb7f1c9f8b09d Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 17 Sep 2026 07:22:02 -0700 Subject: [PATCH 1067/1352] mm/damon/core: set damon_ctx->walk_control_obsolete in damon_new_ctx() damos_walk() should be called for a damon_ctx context that has successfully started at least once. That's because damon_ctx->walk_control_obsolete is initialized when kdamond starts. If the rule is violated, an indefinite wait can happen. There is no existing violation of the rule. damon_call() had a similar rule, and it turned out keeping the rule is not easy for damon_call()'s case. Hence, commit 8023b5f47e09 ("mm/damon/core: set ctx->call_controls_obsolete in damon_new_ctx()") added the initialization in damon_new_ctx() and removed the rule. Keeping the rule for damos_walk() is relatively easier. But having slightly different rules for similar functions could be confusing. Sashiko, for example, repeatedly asked questions about this. Do the initialization of walk_control_obsolete in damon_new_ctx() for consistency. Link: https://lore.kernel.org/20260917142210.90829-5-sj@kernel.org Link: https://lore.kernel.org/20260915011614.102342-1-sj@kernel.org [1] Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Asier Gutierrez Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/core.c | 5 +---- 1 file changed, 1 insertion(+), 4 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index f3ccfc9ad0f086..327277ba365818 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -939,6 +939,7 @@ struct damon_ctx *damon_new_ctx(void) INIT_LIST_HEAD(&ctx->schemes); ctx->call_controls_obsolete = true; + ctx->walk_control_obsolete = true; prandom_seed_state(&ctx->rnd_state, get_random_u64()); return ctx; @@ -2308,10 +2309,6 @@ int damon_call(struct damon_ctx *ctx, struct damon_call_control *control) * passed at least one &damos->apply_interval_us, kdamond marks the request as * completed so that damos_walk() can wakeup and return. * - * Note that this function should be called only after damon_start() with the - * @ctx has succeeded. Otherwise, this function could fall into an indefinite - * wait. - * * Return: 0 on success, negative error code otherwise. */ int damos_walk(struct damon_ctx *ctx, struct damos_walk_control *control) From 15d6f251a456a995894135e3a34cbf3073116e2e Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 17 Sep 2026 07:22:03 -0700 Subject: [PATCH 1068/1352] mm/damon/core: document damon_call()/damon_start() race hang issue Let's suppose damon_start() and damon_call() are executed in parallel for the same DAMON context. Then, damon_call() could show ctx->damon_calls_obsolete set while ctx->kdamond is unset. If damon_start() sets ctx->kdamond before damon_call() starts the cancelling, damon_call() can indefinitely hang. No DAMON API caller does such parallel execution of damon_start() and damon_call(), so the issue doesn't exist. But who knows what will happen in future. Add a clarification comment for caution. Link: https://lore.kernel.org/20260917142210.90829-6-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Asier Gutierrez Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/core.c | 3 +++ 1 file changed, 3 insertions(+) diff --git a/mm/damon/core.c b/mm/damon/core.c index 327277ba365818..add1b7afb957ac 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2258,6 +2258,9 @@ int damon_kdamond_pid(struct damon_ctx *ctx) * * When this function is failed, the @ctx is guaranteed to be stopped. * + * This function should not be called in parallel to damon_start() for the + * @ctx. In the case, this function could indefinitely hang. + * * Return: 0 on success, negative error code otherwise. */ int damon_call(struct damon_ctx *ctx, struct damon_call_control *control) From 5b6e38dfad28a188033e978a3acb1227e501de3c Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 17 Sep 2026 07:22:04 -0700 Subject: [PATCH 1069/1352] mm/damon/paddr: remove pa parameter from damon_pa_filter_pass() damon_pa_filter_pass() receives the 'pa' parameter, but doesn't use it. Remove the parameter from the function signature. Link: https://lore.kernel.org/20260917142210.90829-7-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Asier Gutierrez Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/paddr.c | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c index 5abfabaa339e0e..2cfdc356b41572 100644 --- a/mm/damon/paddr.c +++ b/mm/damon/paddr.c @@ -163,8 +163,7 @@ static bool damon_pa_filter_match(struct damon_filter *filter, return matched == filter->matching; } -static bool damon_pa_filter_pass(phys_addr_t pa, struct folio *folio, - struct damon_probe *p) +static bool damon_pa_filter_pass(struct folio *folio, struct damon_probe *p) { struct damon_filter *f; bool pass = true; @@ -200,7 +199,7 @@ static unsigned int damon_pa_apply_probes(struct damon_ctx *ctx, ctx->addr_unit); folio = damon_get_folio(PHYS_PFN(pa)); damon_for_each_probe(p, ctx) { - if (damon_pa_filter_pass(pa, folio, p)) + if (damon_pa_filter_pass(folio, p)) r->probe_hits[i]++; i++; } From 4df17304d6cb8bcba40bcc085086a079377f3049 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 17 Sep 2026 07:22:05 -0700 Subject: [PATCH 1070/1352] mm/damon/tests/core-kunit: test eligible_mem_bp commitment There was a DAMOS quota goal commit bug [1] that doesn't update the nid field for DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP metric goal. Add a kunit test case for confirming nid commitment. Link: https://lore.kernel.org/20260917142210.90829-8-sj@kernel.org Link: https://lore.kkernel.org/20260827045035.94611-1-sj@kernel.org [1] Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Asier Gutierrez Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/tests/core-kunit.h | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index 5da84caf4124d5..1f19fefdd98c39 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -835,6 +835,9 @@ static void damos_test_commit_quota_goal_for(struct kunit *test, KUNIT_EXPECT_EQ(test, dst->nid, src->nid); KUNIT_EXPECT_EQ(test, dst->memcg_id, src->memcg_id); break; + case DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP: + KUNIT_EXPECT_EQ(test, dst->nid, src->nid); + break; default: break; } @@ -898,6 +901,13 @@ static void damos_test_commit_quota_goal(struct kunit *test) .current_value = 345, .last_psi_total = 567, }); + damos_test_commit_quota_goal_for(test, &dst, + &(struct damos_quota_goal){ + .metric = DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP, + .target_value = 12, + .current_value = 345, + .nid = 6, + }); } static void damos_test_commit_quota_goals_for(struct kunit *test, From c676acab41a77fee6903370881fdd473790a47af Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 17 Sep 2026 07:22:06 -0700 Subject: [PATCH 1071/1352] mm/damon/tests/core-kunit: add probe_hits_wsum damos filter commit test DAMOS filter commit kunit test lacks test cases for probe_hits_wsum filter type. Add test cases for probe_hits_wsum type DAMOS filter commit. Link: https://lore.kernel.org/20260917142210.90829-9-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Asier Gutierrez Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/tests/core-kunit.h | 20 ++++++++++++++++++++ 1 file changed, 20 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index 1f19fefdd98c39..5ff0436c58441f 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -1166,6 +1166,10 @@ static void damos_test_commit_filter_for(struct kunit *test, KUNIT_EXPECT_EQ(test, dst->sz_range.min, src->sz_range.min); KUNIT_EXPECT_EQ(test, dst->sz_range.max, src->sz_range.max); break; + case DAMOS_FILTER_TYPE_PROBE_HITS_WSUM: + KUNIT_EXPECT_EQ(test, dst->range_min, src->range_min); + KUNIT_EXPECT_EQ(test, dst->range_max, src->range_max); + break; default: break; } @@ -1239,6 +1243,22 @@ static void damos_test_commit_filter(struct kunit *test) .allow = true, .target_idx = 6, }, false); + damos_test_commit_filter_for(test, &dst, + &(struct damos_filter){ + .type = DAMOS_FILTER_TYPE_PROBE_HITS_WSUM, + .matching = false, + .allow = true, + .range_min = 12, + .range_max = 34, + }, false); + damos_test_commit_filter_for(test, &dst, + &(struct damos_filter){ + .type = DAMOS_FILTER_TYPE_PROBE_HITS_WSUM, + .matching = false, + .allow = true, + .range_min = 34, + .range_max = 12, + }, true); } static void damos_test_help_initailize_scheme(struct damos *scheme) From f679f214cd98819bf471aa87663ae744841bf2cf Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 17 Sep 2026 07:22:07 -0700 Subject: [PATCH 1072/1352] selftests/damon/sysfs_memcg_path_leak: fail only for real DAMON leak The selftest can fail for any leak if it happens while the test is running. Remove the false positive test failures by further checking if the expected leaking function is called out on the report. Link: https://lore.kernel.org/20260917142210.90829-10-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Asier Gutierrez Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/damon/sysfs_memcg_path_leak.sh | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/tools/testing/selftests/damon/sysfs_memcg_path_leak.sh b/tools/testing/selftests/damon/sysfs_memcg_path_leak.sh index 33a7ff43ed6cc7..34c37129c49fe1 100755 --- a/tools/testing/selftests/damon/sysfs_memcg_path_leak.sh +++ b/tools/testing/selftests/damon/sysfs_memcg_path_leak.sh @@ -41,5 +41,12 @@ if [ "$kmemleak_report" = "" ] then exit 0 fi +if ! echo "$kmemleak_report" | grep "memcg_path_store" --quiet +then + echo "[WARN] memleak found; apparently not from DAMON, though" + echo "$kmemleak_report" + exit 0 +fi + echo "$kmemleak_report" exit 1 From 901d6b3ab9ece7f33b3c7dbbb9bdb842a493b775 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 17 Sep 2026 07:22:08 -0700 Subject: [PATCH 1073/1352] Docs/mm/damon/design: clarify bp is basis point DAMON design document uses "bp" for "basis point" in multiple places. Because it is not clearly mentioned, it is difficult to understand what "bp" stands for. Add the clarification. Link: https://lore.kernel.org/20260917142210.90829-11-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reported-by: Randy Dunlap Closes: https://lore.kernel.org/107dc6ba-697e-4b25-ba3e-8ce2499cac9a@infradead.org Acked-by: Randy Dunlap Acked-by: Zenghui Yu (Huawei) Cc: Asier Gutierrez Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/mm/damon/design.rst | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index 0a86792f90a184..e82390e77a70ae 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -445,7 +445,7 @@ users to set the aimed amount of access events to observe via DAMON within given time interval. The target can be specified by the user as a ratio of DAMON-observed access events to the theoretical maximum amount of the events (``access_bp``) that measured within a given number of aggregations -(``aggrs``). +(``aggrs``). The ratio is in basis point (bp or 1/10,000). The DAMON-observed access events are calculated in byte granularity based on DAMON :ref:`region assumption `. For @@ -717,7 +717,8 @@ mechanism tries to make ``current_value`` of ``target_metric`` be same to in microseconds that measured from last quota reset to next quota reset. DAMOS does the measurement on its own, so only ``target_value`` need to be set by users at the initial time. In other words, DAMOS does self-feedback. -- ``node_mem_used_bp``: Specific NUMA node's used memory ratio in bp (1/10,000). +- ``node_mem_used_bp``: Specific NUMA node's used memory ratio in basis point + (bp or 1/10,000). - ``node_mem_free_bp``: Specific NUMA node's free memory ratio in bp (1/10,000). - ``node_memcg_used_bp``: Specific cgroup's node used memory ratio for a specific NUMA node, in bp (1/10,000). From b760579514ffe61df29189299d71ddbe69070eda Mon Sep 17 00:00:00 2001 From: Breno Leitao Date: Thu, 17 Sep 2026 06:47:35 -0700 Subject: [PATCH 1074/1352] Documentation: kmemleak: describe the metadata pool, not the early log Patch series "kmemleak: fix stale documentation and raise the verbose default". The first two patches fix statements in Documentation/dev-tools/kmemleak.rst that do not match mm/kmemleak.c. The third requires one more consecutive unreferenced scan before a CONFIG_DEBUG_KMEMLEAK_VERBOSE kernel reports a leak on the console, to avoid the last false positives I am seeing when running CONFIG_DEBUG_KMEMLEAK_VERBOSE on a daily basis. This patch (of 3): commit c5665868183f ("mm: kmemleak: use the memory pool for early allocations") removed the early log buffer in favour of a static pool of kmemleak_object structures, but the documentation still describes the old mechanism. Fix the documentation by describing what the pool actually is, matching the Kconfig help text. Link: https://lore.kernel.org/20260917142210.90829-1-sj@kernel.org Link: https://lore.kernel.org/20260917-b4-kmemleak-doc-v1-1-84fde6d1f749@debian.org Fixes: c5665868183f ("mm: kmemleak: use the memory pool for early allocations") Signed-off-by: Breno Leitao Signed-off-by: Andrew Morton Reviewed-by: Catalin Marinas Cc: Jonathan Corbet Cc: Randy Dunlap --- Documentation/dev-tools/kmemleak.rst | 11 ++++++++--- 1 file changed, 8 insertions(+), 3 deletions(-) diff --git a/Documentation/dev-tools/kmemleak.rst b/Documentation/dev-tools/kmemleak.rst index d1b690b1716967..8dad7647742d31 100644 --- a/Documentation/dev-tools/kmemleak.rst +++ b/Documentation/dev-tools/kmemleak.rst @@ -66,9 +66,14 @@ Memory scanning parameters can be modified at run-time by writing to the Kmemleak can also be disabled at boot-time by passing ``kmemleak=off`` on the kernel command line. -Memory may be allocated or freed before kmemleak is initialised and -these actions are stored in an early log buffer. The size of this buffer -is configured via the CONFIG_DEBUG_KMEMLEAK_MEM_POOL_SIZE option. +Memory may be allocated or freed before kmemleak is initialised, so a +static pool of metadata objects is used to track those allocations. Once +kmemleak is fully initialised the pool becomes an emergency reserve, used +whenever a metadata object cannot be allocated from the slab. The number +of objects in the pool is configured via the +CONFIG_DEBUG_KMEMLEAK_MEM_POOL_SIZE option. Exhausting it at run time +prints "Cannot allocate a kmemleak_object structure" and disables +kmemleak. If CONFIG_DEBUG_KMEMLEAK_DEFAULT_OFF are enabled, the kmemleak is disabled by default. Passing ``kmemleak=on`` on the kernel command From 5e7fdb68b955c05fe3e19969f861403d90ce683f Mon Sep 17 00:00:00 2001 From: Breno Leitao Date: Thu, 17 Sep 2026 06:47:36 -0700 Subject: [PATCH 1075/1352] Documentation: kmemleak: fix stale statements about scanning Three statements in the "false positives/negatives" and "Limitations" sections have never matched the code: - task stack scanning is on by default (kmemleak_stack_scan = 1), as the parameter list earlier in the same document already states; - MSECS_MIN_AGE has been 5000, not 1000, since the initial commit; - scanning is done by a periodic kthread. Reading the debugfs file only lists what the last scan found; kmemleak_open() calls seq_open() and never scans. Link: https://lore.kernel.org/20260917-b4-kmemleak-doc-v1-2-84fde6d1f749@debian.org Fixes: 04f70336c80c ("kmemleak: Add documentation on the memory leak detector") Fixes: e0a2a1601bec ("kmemleak: Enable task stacks scanning by default") Signed-off-by: Breno Leitao Signed-off-by: Andrew Morton Reviewed-by: Catalin Marinas Cc: Jonathan Corbet Cc: Randy Dunlap --- Documentation/dev-tools/kmemleak.rst | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/Documentation/dev-tools/kmemleak.rst b/Documentation/dev-tools/kmemleak.rst index 8dad7647742d31..b5fe7e671d0f82 100644 --- a/Documentation/dev-tools/kmemleak.rst +++ b/Documentation/dev-tools/kmemleak.rst @@ -190,7 +190,8 @@ reported by kmemleak because values found during the memory scanning point to such objects. To reduce the number of false negatives, kmemleak provides the kmemleak_ignore, kmemleak_scan_area, kmemleak_no_scan and kmemleak_erase functions (see above). The task stacks also increase the -amount of false negatives and their scanning is not enabled by default. +amount of false negatives and their scanning is enabled by default; it +can be turned off with ``stack=off``. The false positives are objects wrongly reported as being memory leaks (orphan). For objects known not to be leaks, kmemleak provides the @@ -200,7 +201,7 @@ longer be scanned. Some of the reported leaks are only transient, especially on SMP systems, because of pointers temporarily stored in CPU registers or -stacks. Kmemleak defines MSECS_MIN_AGE (defaulting to 1000) representing +stacks. Kmemleak defines MSECS_MIN_AGE (defaulting to 5000) representing the minimum age of an object to be reported as a memory leak. The ``min_unref_scans`` module parameter requires an object to be seen @@ -217,8 +218,10 @@ Limitations and Drawbacks ------------------------- The main drawback is the reduced performance of memory allocation and -freeing. To avoid other penalties, the memory scanning is only performed -when the /sys/kernel/debug/kmemleak file is read. Anyway, this tool is +freeing. To avoid other penalties, the memory scanning is performed by a +periodic thread rather than on every allocation. Reading the +/sys/kernel/debug/kmemleak file only lists the objects found by the last +scan; writing ``scan`` to it triggers a new one. Anyway, this tool is intended for debugging purposes where the performance might not be the most important requirement. From ccd00a107d3e75c8b991fcf07fabdc068322bb34 Mon Sep 17 00:00:00 2001 From: Breno Leitao Date: Thu, 17 Sep 2026 06:47:37 -0700 Subject: [PATCH 1076/1352] mm: kmemleak: raise min_unref_scans to 3 for verbose auto-scan CONFIG_DEBUG_KMEMLEAK_VERBOSE sends every report to the console, so a transient false positive there is broadcast to whatever collects the kernel log rather than sitting in the debugfs file until someone looks. That asymmetry justifies being more conservative than the general case. Require one more consecutive unreferenced scan before reporting. The only cost is that a genuine leak is reported one scan interval later (600s by default); the value stays writable at run time through the module parameter. Kernels without CONFIG_DEBUG_KMEMLEAK_VERBOSE keep reporting on the first unreferenced scan. I've been running constant upstream kernel with CONFIG_DEBUG_KMEMLEAK_VERBOSE set, and I am still seeing some rare false positive, that goes away with min_unref_scans=3, so, making it the default based on my heuristic. Link: https://lore.kernel.org/20260917-b4-kmemleak-doc-v1-3-84fde6d1f749@debian.org Signed-off-by: Breno Leitao Signed-off-by: Andrew Morton Reviewed-by: Catalin Marinas Cc: Jonathan Corbet Cc: Randy Dunlap --- Documentation/dev-tools/kmemleak.rst | 2 +- mm/kmemleak.c | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/Documentation/dev-tools/kmemleak.rst b/Documentation/dev-tools/kmemleak.rst index b5fe7e671d0f82..c0d32937234251 100644 --- a/Documentation/dev-tools/kmemleak.rst +++ b/Documentation/dev-tools/kmemleak.rst @@ -206,7 +206,7 @@ the minimum age of an object to be reported as a memory leak. The ``min_unref_scans`` module parameter requires an object to be seen unreferenced in that many consecutive scans before it is reported. It -defaults to 2 when CONFIG_DEBUG_KMEMLEAK_VERBOSE is enabled, where the +defaults to 3 when CONFIG_DEBUG_KMEMLEAK_VERBOSE is enabled, where the periodic scan thread confirms a leak on its own, and to 1 otherwise. A value of 1 preserves the historical behaviour; higher values filter the transient false positives described above, at the cost of delaying genuine diff --git a/mm/kmemleak.c b/mm/kmemleak.c index 8fa409a4f9fb26..5d0daea93c471f 100644 --- a/mm/kmemleak.c +++ b/mm/kmemleak.c @@ -238,7 +238,7 @@ static struct task_struct *scan_thread; static unsigned long jiffies_min_age; /* consecutive scans an object must stay unreferenced before reporting */ static unsigned int min_unref_scans = - IS_ENABLED(CONFIG_DEBUG_KMEMLEAK_VERBOSE) ? 2 : 1; + IS_ENABLED(CONFIG_DEBUG_KMEMLEAK_VERBOSE) ? 3 : 1; module_param(min_unref_scans, uint, 0644); static unsigned long jiffies_last_scan; /* delay between automatic memory scannings */ From f70bcbe359fcf28b3c1774afca118603ba508104 Mon Sep 17 00:00:00 2001 From: Lisa Wang Date: Thu, 17 Sep 2026 20:43:47 +0000 Subject: [PATCH 1077/1352] mm: memory_failure: clarify the MF_DELAYED definition Patch series "mm: Fix MF_DELAYED handling on memory failure", v6. This series addresses an issue in the memory failure handling path where MF_DELAYED is incorrectly treated as an error. This issue was discovered while testing memory failure handling for guest_memfd. The proposed solution involves - 1. Clarifying the definition of MF_DELAYED to mean that memory failure handling is only partially completed, and that the metadata for the memory that failed (as in struct page/folio) is still referenced. 2. Updating shmems handling to align with the clarified definition. 3. Updating how the result of .error_remove_folio() is interpreted. This patch (of 5): This patch clarifies the definition of MF_DELAYED to represent cases where a folio's removal is initiated but not immediately completed (e.g., due to remaining metadata references). Link: https://lore.kernel.org/20260917-memory-failure-mf-delayed-fix-v6-0-4b00856b5364@google.com Link: https://lore.kernel.org/20260917-memory-failure-mf-delayed-fix-v6-1-4b00856b5364@google.com Signed-off-by: Lisa Wang Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Miaohe Lin Reviewed-by: Ackerley Tng Cc: Andi Kleen Cc: Baolin Wang Cc: Dave Hansen Cc: David Rientjes Cc: Fuad Tabba Cc: Hidehiro Kawai Cc: Hugh Dickins Cc: Isaku Yamahata Cc: Jiaqi Yan Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michael Roth Cc: Michal Hocko Cc: Mike Rapoport Cc: Naoya Horiguchi Cc: Paolo Bonzini Cc: Rik van Riel Cc: Sean Christopherson Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vishal Annapurve Cc: Vlastimil Babka Cc: Xiaoyao Li Cc: Yu Zhang --- mm/memory-failure.c | 15 ++++++++------- 1 file changed, 8 insertions(+), 7 deletions(-) diff --git a/mm/memory-failure.c b/mm/memory-failure.c index a8b03e2920ba8a..057cb9537db292 100644 --- a/mm/memory-failure.c +++ b/mm/memory-failure.c @@ -847,24 +847,25 @@ static int kill_accessing_process(struct task_struct *p, unsigned long pfn, } /* - * MF_IGNORED - The m-f() handler marks the page as PG_hwpoisoned'ed. + * MF_IGNORED - The m-f() handler marks the page as PG_hwpoison'ed. * But it could not do more to isolate the page from being accessed again, * nor does it kill the process. This is extremely rare and one of the * potential causes is that the page state has been changed due to * underlying race condition. This is the most severe outcomes. * - * MF_FAILED - The m-f() handler marks the page as PG_hwpoisoned'ed. + * MF_FAILED - The m-f() handler marks the page as PG_hwpoison'ed. * It should have killed the process, but it can't isolate the page, * due to conditions such as extra pin, unmap failure, etc. Accessing * the page again may trigger another MCE and the process will be killed * by the m-f() handler immediately. * - * MF_DELAYED - The m-f() handler marks the page as PG_hwpoisoned'ed. - * The page is unmapped, and is removed from the LRU or file mapping. - * An attempt to access the page again will trigger page fault and the - * PF handler will kill the process. + * MF_DELAYED - The m-f() handler marks the page as PG_hwpoison'ed. + * It means the page was unmapped and partially isolated (e.g. removed from + * file mapping or the LRU) but full cleanup is deferred (e.g. the metadata + * for the memory, as in struct page/folio, is still referenced). Any + * further access to the page will result in the process being killed. * - * MF_RECOVERED - The m-f() handler marks the page as PG_hwpoisoned'ed. + * MF_RECOVERED - The m-f() handler marks the page as PG_hwpoison'ed. * The page has been completely isolated, that is, unmapped, taken out of * the buddy system, or hole-punched out of the file mapping. */ From 145e0623847788cdd9564df026ad7cb7acc0beeb Mon Sep 17 00:00:00 2001 From: Lisa Wang Date: Thu, 17 Sep 2026 20:43:48 +0000 Subject: [PATCH 1078/1352] mm: memory_failure: Allow truncate_error_folio to return MF_DELAYED The .error_remove_folio a_ops is used by different filesystems to handle folio truncation upon discovery of a memory failure in the memory associated with the given folio. Currently, MF_DELAYED is treated as an error, causing "Failed to punch page" to be written to the console. MF_DELAYED is then relayed to the caller of truncate_error_folio() as MF_FAILED. This further causes memory_failure() to return -EBUSY, which then always causes a SIGBUS. This is also implies that regardless of whether the thread's memory corruption kill policy is PR_MCE_KILL_EARLY or PR_MCE_KILL_LATE, a memory failure with MF_DELAYED will always cause a SIGBUS. Update truncate_error_folio() to return MF_DELAYED to the caller if the .error_remove_folio() callback reports MF_DELAYED. Link: https://lore.kernel.org/20260917-memory-failure-mf-delayed-fix-v6-2-4b00856b5364@google.com Fixes: 6a46079cf57a ("HWPOISON: The high level memory error handler in the VM v7") Fixes: a7800aa80ea4 ("KVM: Add KVM_CREATE_GUEST_MEMFD ioctl() for guest-specific backing memory") Signed-off-by: Lisa Wang Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Miaohe Lin Reviewed-by: Ackerley Tng Cc: Andi Kleen Cc: Baolin Wang Cc: Dave Hansen Cc: David Rientjes Cc: Fuad Tabba Cc: Hidehiro Kawai Cc: Hugh Dickins Cc: Isaku Yamahata Cc: Jiaqi Yan Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michael Roth Cc: Michal Hocko Cc: Mike Rapoport Cc: Naoya Horiguchi Cc: Paolo Bonzini Cc: Rik van Riel Cc: Sean Christopherson Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vishal Annapurve Cc: Vlastimil Babka Cc: Xiaoyao Li Cc: Yu Zhang --- mm/memory-failure.c | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/mm/memory-failure.c b/mm/memory-failure.c index 057cb9537db292..83ab0fce22a61a 100644 --- a/mm/memory-failure.c +++ b/mm/memory-failure.c @@ -939,10 +939,12 @@ static int truncate_error_folio(struct folio *folio, unsigned long pfn, if (mapping->a_ops->error_remove_folio) { int err = mapping->a_ops->error_remove_folio(mapping, folio); - if (err != 0) + if (err == MF_DELAYED) + ret = err; + else if (err != 0) pr_info("%#lx: Failed to punch page: %d\n", pfn, err); else if (!filemap_release_folio(folio, GFP_NOIO)) - pr_info("%#lx: failed to release buffers\n", pfn); + pr_info("%#lx: Failed to release buffers\n", pfn); else ret = MF_RECOVERED; } else { From 26a3f4ccada643697f68e60b17136149e2b9f335 Mon Sep 17 00:00:00 2001 From: Lisa Wang Date: Thu, 17 Sep 2026 20:43:49 +0000 Subject: [PATCH 1079/1352] mm: shmem: Update shmem handler to the MF_DELAYED definition To align with the definition of MF_DELAYED, update shmem_error_remove_folio() to return MF_DELAYED. shmem handles memory failures but defers the actual file truncation. The function's return value should therefore be MF_DELAYED to accurately reflect the state. Currently, this logical error does not cause a bug, because: - For shmem folios, folio->private is not set. - As a result, filemap_release_folio() is a no-op and returns true. - This, in turn, causes truncate_error_folio() to incorrectly return MF_RECOVERED. - The caller then treats MF_RECOVERED as a success condition, masking the issue. The previous patch relays MF_DELAYED to the caller of truncate_error_folio() before any logging, so returning MF_DELAYED from shmem_error_remove_folio() will retain the original behavior of not adding any logs. The return value of truncate_error_folio() is consumed in action_result(), which treats MF_DELAYED the same way as MF_RECOVERED, hence action_result() also returns the same thing after this change. Link: https://lore.kernel.org/20260917-memory-failure-mf-delayed-fix-v6-3-4b00856b5364@google.com Signed-off-by: Lisa Wang Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Miaohe Lin Reviewed-by: Ackerley Tng Cc: Andi Kleen Cc: Baolin Wang Cc: Dave Hansen Cc: David Rientjes Cc: Fuad Tabba Cc: Hidehiro Kawai Cc: Hugh Dickins Cc: Isaku Yamahata Cc: Jiaqi Yan Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michael Roth Cc: Michal Hocko Cc: Mike Rapoport Cc: Naoya Horiguchi Cc: Paolo Bonzini Cc: Rik van Riel Cc: Sean Christopherson Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vishal Annapurve Cc: Vlastimil Babka Cc: Xiaoyao Li Cc: Yu Zhang --- mm/shmem.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/shmem.c b/mm/shmem.c index 05bc7c3aa52548..f33dbf5af2cb79 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -5363,7 +5363,7 @@ static void __init shmem_destroy_inodecache(void) static int shmem_error_remove_folio(struct address_space *mapping, struct folio *folio) { - return 0; + return MF_DELAYED; } static const struct address_space_operations shmem_aops = { From 203b7865d09c16cb972c8f23fdcd35a6bd414f0d Mon Sep 17 00:00:00 2001 From: Lisa Wang Date: Thu, 17 Sep 2026 20:43:50 +0000 Subject: [PATCH 1080/1352] mm: memory_failure: Generalize extra_pins handling to all MF_DELAYED cases Generalize extra_pins handling to all MF_DELAYED cases not only shmem_mapping. If MF_DELAYED is returned, the filemap continues to hold refcounts on the folio. Hence, take that into account when checking for extra refcounts. As clarified in an earlier patch, a return value of MF_DELAYED implies that the page still has elevated refcounts. Hence, set extra_pins to true if the return value is MF_DELAYED. This is aligned with the implementation in me_swapcache_dirty(), where, if a folio is still in the swap cache, ret is set to MF_DELAYED and extra_pins is set to true. Link: https://lore.kernel.org/20260917-memory-failure-mf-delayed-fix-v6-4-4b00856b5364@google.com Signed-off-by: Lisa Wang Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Miaohe Lin Reviewed-by: Ackerley Tng Cc: Andi Kleen Cc: Baolin Wang Cc: Dave Hansen Cc: David Rientjes Cc: Fuad Tabba Cc: Hidehiro Kawai Cc: Hugh Dickins Cc: Isaku Yamahata Cc: Jiaqi Yan Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michael Roth Cc: Michal Hocko Cc: Mike Rapoport Cc: Naoya Horiguchi Cc: Paolo Bonzini Cc: Rik van Riel Cc: Sean Christopherson Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vishal Annapurve Cc: Vlastimil Babka Cc: Xiaoyao Li Cc: Yu Zhang --- mm/memory-failure.c | 8 ++------ 1 file changed, 2 insertions(+), 6 deletions(-) diff --git a/mm/memory-failure.c b/mm/memory-failure.c index 83ab0fce22a61a..d237f556b3a09e 100644 --- a/mm/memory-failure.c +++ b/mm/memory-failure.c @@ -1039,18 +1039,14 @@ static int me_pagecache_clean(struct page_state *ps, struct page *p) goto out; } - /* - * The shmem page is kept in page cache instead of truncating - * so is expected to have an extra refcount after error-handling. - */ - extra_pins = shmem_mapping(mapping); - /* * Truncation is a bit tricky. Enable it per file system for now. * * Open: to take i_rwsem or not for this? Right now we don't. */ ret = truncate_error_folio(folio, page_to_pfn(p), mapping); + + extra_pins = ret == MF_DELAYED; if (has_extra_refcount(ps, p, extra_pins)) ret = MF_FAILED; From 6724186916c157111e595a69c89298d57e04456c Mon Sep 17 00:00:00 2001 From: Lisa Wang Date: Thu, 17 Sep 2026 20:43:51 +0000 Subject: [PATCH 1081/1352] mm: selftests: Add shmem into memory failure test Add a shmem memory failure selftest to test the shmem memory failure is correct after modifying shmem return value. Specifically, test the expected behavior under various scenarios combining page dirtiness (dirty vs clean) and failure types (hard vs soft): + Dirty + Hard: Trigger a SIGBUS on injection, and trigger another SIGBUS when reading the page again. + Dirty + Soft: No SIGBUS is triggered, and the original value can be read successfully. + Clean + Hard: No SIGBUS is triggered on injection, but trigger a SIGBUS when trying to read the page again. + Clean + Soft: No SIGBUS is triggered, and the page can be read successfully. Link: https://lore.kernel.org/20260917-memory-failure-mf-delayed-fix-v6-5-4b00856b5364@google.com Signed-off-by: Lisa Wang Signed-off-by: Andrew Morton Acked-by: Miaohe Lin Cc: Ackerley Tng Cc: Andi Kleen Cc: Baolin Wang Cc: Dave Hansen Cc: David Hildenbrand (Arm) Cc: David Rientjes Cc: Fuad Tabba Cc: Hidehiro Kawai Cc: Hugh Dickins Cc: Isaku Yamahata Cc: Jiaqi Yan Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michael Roth Cc: Michal Hocko Cc: Mike Rapoport Cc: Naoya Horiguchi Cc: Paolo Bonzini Cc: Rik van Riel Cc: Sean Christopherson Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vishal Annapurve Cc: Vlastimil Babka Cc: Xiaoyao Li Cc: Yu Zhang --- tools/testing/selftests/mm/memory-failure.c | 114 +++++++++++++++++++- 1 file changed, 111 insertions(+), 3 deletions(-) diff --git a/tools/testing/selftests/mm/memory-failure.c b/tools/testing/selftests/mm/memory-failure.c index f3cb578b160962..1f22869e94670a 100644 --- a/tools/testing/selftests/mm/memory-failure.c +++ b/tools/testing/selftests/mm/memory-failure.c @@ -29,9 +29,14 @@ enum result_type { MADV_HARD_ANON, MADV_HARD_CLEAN_PAGECACHE, MADV_HARD_DIRTY_PAGECACHE, + MADV_HARD_CLEAN_SHMEM, + MADV_HARD_DIRTY_SHMEM, MADV_SOFT_ANON, MADV_SOFT_CLEAN_PAGECACHE, MADV_SOFT_DIRTY_PAGECACHE, + MADV_SOFT_CLEAN_SHMEM, + MADV_SOFT_DIRTY_SHMEM, + READ_ERROR, }; static jmp_buf signal_jmp_buf; @@ -157,17 +162,22 @@ static void check(struct __test_metadata *_metadata, FIXTURE_DATA(memory_failure case MADV_HARD_CLEAN_PAGECACHE: case MADV_SOFT_CLEAN_PAGECACHE: case MADV_SOFT_DIRTY_PAGECACHE: - /* It is not expected to receive a SIGBUS signal. */ - ASSERT_EQ(setjmp, 0); - + case MADV_SOFT_DIRTY_SHMEM: /* The page content should remain unchanged. */ ASSERT_TRUE(check_memory(vaddr, self->page_size)); + /* FALLTHORUGH */ + case MADV_HARD_CLEAN_SHMEM: + case MADV_SOFT_CLEAN_SHMEM: + /* It is not expected to receive a SIGBUS signal. */ + ASSERT_EQ(setjmp, 0); /* The backing pfn of addr should have changed. */ ASSERT_NE(pagemap_get_pfn(self->pagemap_fd, vaddr), self->pfn); break; case MADV_HARD_ANON: case MADV_HARD_DIRTY_PAGECACHE: + case MADV_HARD_DIRTY_SHMEM: + case READ_ERROR: /* The SIGBUS signal should have been received. */ ASSERT_EQ(setjmp, 1); @@ -263,6 +273,20 @@ static int prepare_file(const char *fname, unsigned long size) return fd; } +static int prepare_shmem(const char *fname, unsigned long size) +{ + int fd; + + fd = memfd_create(fname, 0); + if (fd < 0) + return -1; + if (ftruncate(fd, size) < 0) { + close(fd); + return -1; + } + return fd; +} + /* Borrowed from mm/gup_longterm.c. */ static int get_fs_type(int fd) { @@ -365,4 +389,88 @@ TEST_F(memory_failure, dirty_pagecache) ASSERT_EQ(close(fd), 0); } +TEST_F(memory_failure, dirty_shmem) +{ + int fd; + char *addr; + int ret; + + fd = prepare_shmem("shmem-file", self->page_size); + if (fd < 0) + SKIP(return, "failed to open test shmem-file.\n"); + + addr = mmap(0, self->page_size, PROT_READ | PROT_WRITE, + MAP_SHARED, fd, 0); + if (addr == MAP_FAILED) { + close(fd); + SKIP(return, "mmap failed, not enough memory.\n"); + } + memset(addr, 0xce, self->page_size); + + prepare(_metadata, self, addr); + + ret = sigsetjmp(signal_jmp_buf, 1); + if (!ret && !self->injection_attempted) { + self->injection_attempted = true; + ASSERT_EQ(variant->inject(self, addr), 0); + } + + if (variant->type == MADV_HARD) { + check(_metadata, self, addr, MADV_HARD_DIRTY_SHMEM, ret); + ret = sigsetjmp(signal_jmp_buf, 1); + if (ret == 0) + FORCE_READ(*addr); + check(_metadata, self, addr, READ_ERROR, ret); + } else { + check(_metadata, self, addr, MADV_SOFT_DIRTY_SHMEM, ret); + } + + ASSERT_EQ(munmap(addr, self->page_size), 0); + + ASSERT_EQ(close(fd), 0); +} + +TEST_F(memory_failure, clean_shmem) +{ + int fd; + char *addr; + int ret; + + fd = prepare_shmem("shmem-file", self->page_size); + if (fd < 0) + SKIP(return, "failed to open test shmem-file.\n"); + + addr = mmap(0, self->page_size, PROT_READ | PROT_WRITE, + MAP_SHARED, fd, 0); + if (addr == MAP_FAILED) { + close(fd); + SKIP(return, "mmap failed, not enough memory.\n"); + } + FORCE_READ(*addr); + + prepare(_metadata, self, addr); + + ret = sigsetjmp(signal_jmp_buf, 1); + if (!ret && !self->injection_attempted) { + self->injection_attempted = true; + ASSERT_EQ(variant->inject(self, addr), 0); + } + + if (variant->type == MADV_HARD) { + check(_metadata, self, addr, MADV_HARD_CLEAN_SHMEM, ret); + ret = sigsetjmp(signal_jmp_buf, 1); + if (ret == 0) + FORCE_READ(*addr); + check(_metadata, self, addr, READ_ERROR, ret); + } else { + /* Test the address accessability without check_memory(). */ + FORCE_READ(*addr); + check(_metadata, self, addr, MADV_SOFT_CLEAN_SHMEM, ret); + } + + ASSERT_EQ(munmap(addr, self->page_size), 0); + + ASSERT_EQ(close(fd), 0); +} + TEST_HARNESS_MAIN From 09fe37538ba027d889440620d78c3f5279d145d0 Mon Sep 17 00:00:00 2001 From: Andrew Morton Date: Thu, 17 Sep 2026 17:06:00 -0700 Subject: [PATCH 1082/1352] mm-selftests-add-shmem-into-memory-failure-test-fix fix commant typo, per Lisa Cc: Lisa Wang Cc: Ackerley Tng Signed-off-by: Andrew Morton --- tools/testing/selftests/mm/memory-failure.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tools/testing/selftests/mm/memory-failure.c b/tools/testing/selftests/mm/memory-failure.c index 1f22869e94670a..135a2e8069ba5c 100644 --- a/tools/testing/selftests/mm/memory-failure.c +++ b/tools/testing/selftests/mm/memory-failure.c @@ -165,7 +165,7 @@ static void check(struct __test_metadata *_metadata, FIXTURE_DATA(memory_failure case MADV_SOFT_DIRTY_SHMEM: /* The page content should remain unchanged. */ ASSERT_TRUE(check_memory(vaddr, self->page_size)); - /* FALLTHORUGH */ + /* FALLTHROUGH */ case MADV_HARD_CLEAN_SHMEM: case MADV_SOFT_CLEAN_SHMEM: /* It is not expected to receive a SIGBUS signal. */ From 4cdef481129b41c82da1a8ea8fca10da1d89a0a4 Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Fri, 18 Sep 2026 10:56:10 +0800 Subject: [PATCH 1083/1352] mm/hugetlb_cgroup: move per-node usage on cross node migration Patch series "mm/hugetlb_cgroup: move the per-node usage along with the folio", v2. The per-node usage reported by hugetlb..numa_stat is accounted against folio_nid() in __hugetlb_cgroup_commit_charge() and __hugetlb_cgroup_uncharge_folio(), so it is only correct while a folio stays charged on the same node and in the same hugetlb_cgroup. Two paths move a folio which stays charged, and neither moves the usage with it. hugetlb_cgroup_migrate() only moves the hugetlb_cgroup pointers of a folio migrated to another node, and hugetlb_cgroup_move_parent() only moves the page_counter charges and the hugetlb_cgroup pointer of the folios of a dying cgroup. In both cases the node (or cgroup) which was charged keeps a usage which never goes away, while the node (or cgroup) which ends up uncharging the folio underflows as soon as the folio is freed. Patch 1/2 moves the usage along with the folio on cross node migration, patch 2/2 does the same for the folios a dying cgroup reparents. This patch (of 2): hugetlb..numa_stat uses folio_nid() to account usage in __hugetlb_cgroup_commit_charge() and __hugetlb_cgroup_uncharge_folio(). hugetlb_cgroup_migrate() only moves hugetlb_cgroup pointers, leaving per-node usage behind on the source node during cross-node migration. When the migrated folio gets uncharged, we subtract usage from the destination node counter. This creates stale usage on the source node and unsigned long counter underflow on the destination node. The hugetlb..numa_stat interface exposes these incorrect per-node usage values to userspace. Add a hugetlb_cgroup_move_usage() helper which moves the usage from the old node to the new node, and call it from hugetlb_cgroup_migrate(). Link: https://lore.kernel.org/20260918-for-hugetlb-charge-v2-0-2b6d8c2bdc36@kylinos.cn Link: https://lore.kernel.org/20260918-for-hugetlb-charge-v2-1-2b6d8c2bdc36@kylinos.cn Fixes: f47761999052 ("hugetlb: add hugetlb.*.numa_stat file") Signed-off-by: Hongfu Li Signed-off-by: Andrew Morton Acked-by: Muchun Song Cc: Colin Ian King Cc: David Hildenbrand Cc: Kees Cook Cc: Mina Almasry Cc: Oscar Salvador Cc: Shakeel Butt Cc: --- mm/hugetlb_cgroup.c | 31 +++++++++++++++++++++++++++++++ 1 file changed, 31 insertions(+) diff --git a/mm/hugetlb_cgroup.c b/mm/hugetlb_cgroup.c index ecb6e0b7819a0d..7cf7c18119b432 100644 --- a/mm/hugetlb_cgroup.c +++ b/mm/hugetlb_cgroup.c @@ -179,6 +179,34 @@ static void hugetlb_cgroup_css_free(struct cgroup_subsys_state *css) hugetlb_cgroup_free(hugetlb_cgroup_from_css(css)); } +static void hugetlb_cgroup_move_usage(struct hugetlb_cgroup *from, + struct hugetlb_cgroup *to, + struct folio *from_folio, + struct folio *to_folio) +{ + int idx = hstate_index(folio_hstate(from_folio)); + unsigned long nr_pages = folio_nr_pages(from_folio); + int from_nid = folio_nid(from_folio); + int to_nid = folio_nid(to_folio); + unsigned long usage; + + lockdep_assert_held(&hugetlb_lock); + + if (!from || !to) + return; + + if (from == to && from_nid == to_nid) + return; + + usage = from->nodeinfo[from_nid]->usage[idx]; + if (WARN_ON_ONCE(usage < nr_pages)) + return; + WRITE_ONCE(from->nodeinfo[from_nid]->usage[idx], usage - nr_pages); + + usage = to->nodeinfo[to_nid]->usage[idx]; + WRITE_ONCE(to->nodeinfo[to_nid]->usage[idx], usage + nr_pages); +} + /* * Should be called with hugetlb_lock held. * Since we are holding hugetlb_lock, pages cannot get moved from @@ -906,6 +934,9 @@ void hugetlb_cgroup_migrate(struct folio *old_folio, struct folio *new_folio) /* move the h_cg details to new cgroup */ set_hugetlb_cgroup(new_folio, h_cg); set_hugetlb_cgroup_rsvd(new_folio, h_cg_rsvd); + + hugetlb_cgroup_move_usage(h_cg, h_cg, old_folio, new_folio); + list_move(&new_folio->lru, &h->hugepage_activelist); spin_unlock_irq(&hugetlb_lock); } From f683aa437777ee32ddd7a3cb3bd0cf1a62865a54 Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Fri, 18 Sep 2026 10:56:11 +0800 Subject: [PATCH 1084/1352] mm/hugetlb_cgroup: move per-node usage on cgroup reparenting hugetlb_cgroup_css_offline() hands the folios of a dying cgroup over to its parent with hugetlb_cgroup_move_parent(), which moves the page_counter charges and the hugetlb_cgroup pointer of the folio but not its per-node usage. The parent's hugetlb..numa_stat is short by that usage while they are charged, and underflows once they are freed, exposing incorrect per-node usage values to userspace. Move the per-node usage to the parent as well. The folios keep their node here, so only the cgroup which holds the usage changes. Link: https://lore.kernel.org/20260918-for-hugetlb-charge-v2-2-2b6d8c2bdc36@kylinos.cn Fixes: f47761999052 ("hugetlb: add hugetlb.*.numa_stat file") Signed-off-by: Hongfu Li Signed-off-by: Andrew Morton Acked-by: Muchun Song Cc: Colin Ian King Cc: David Hildenbrand Cc: Kees Cook Cc: Mina Almasry Cc: Oscar Salvador Cc: Shakeel Butt Cc: --- mm/hugetlb_cgroup.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/mm/hugetlb_cgroup.c b/mm/hugetlb_cgroup.c index 7cf7c18119b432..3fb41311e4c7dc 100644 --- a/mm/hugetlb_cgroup.c +++ b/mm/hugetlb_cgroup.c @@ -241,6 +241,8 @@ static void hugetlb_cgroup_move_parent(int idx, struct hugetlb_cgroup *h_cg, /* Take the pages off the local counter */ page_counter_cancel(counter, nr_pages); + hugetlb_cgroup_move_usage(h_cg, parent, folio, folio); + set_hugetlb_cgroup(folio, parent); out: return; From 0b26606ee2d1d5874d35535e0c49742867c87a62 Mon Sep 17 00:00:00 2001 From: Suren Baghdasaryan Date: Fri, 18 Sep 2026 08:33:12 -0700 Subject: [PATCH 1085/1352] proc/task_mmu: remove unnecessary helpers Patch series "read proc/pid/smaps_rollup under per-vma lock", v5. proc/pid/smaps_rollup can be read using the combination of RCU and VMA read locks, similar to proc/pid/{maps|smaps|numa_maps}. RCU is required to safely traverse the VMA tree and VMA lock stabilizes the VMA being processed and the pagetable walk. Note that we have to keep the logic to drop mmap_lock on contention because even when using per-VMA locks we might have to fall back to holding the mmap_lock. The first 5 patches are cleanups making later change simpler. The main change is in patch 6. Patch 7 extends existing proc-maps-race tearing test to verify smaps_rollup content. This patch (of 7): When per-vma locks were behind a config option, a number of helper functions were needed to simplify the locking code. Now that these locks are universally available, we can do a little cleanup. Remove lock_vma_range(), unlock_vma_range(), query_vma_setup(), query_vma_teardown() helpers. No functional change intended. Link: https://lore.kernel.org/20260918153318.758387-1-surenb@google.com Link: https://lore.kernel.org/20260918153318.758387-2-surenb@google.com Signed-off-by: Suren Baghdasaryan Signed-off-by: Andrew Morton Reviewed-by: Liam R. Howlett (Oracle) Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: Usama Arif Acked-by: David Hildenbrand (Arm) Cc: Jann Horn Cc: Matthew Wilcox (Oracle) Cc: "Paul E . McKenney" Cc: Pedro Falcato Cc: Vlastimil Babka --- fs/proc/task_mmu.c | 67 ++++++++++++---------------------------------- 1 file changed, 17 insertions(+), 50 deletions(-) diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c index 052e8dc796bcf8..191a054d7e4093 100644 --- a/fs/proc/task_mmu.c +++ b/fs/proc/task_mmu.c @@ -160,25 +160,6 @@ static void unlock_ctx_vma(struct proc_maps_locking_ctx *lock_ctx) } } -static inline bool lock_vma_range(struct seq_file *m, - struct proc_maps_locking_ctx *lock_ctx) -{ - rcu_read_lock(); - reset_lock_ctx(lock_ctx); - - return true; -} - -static inline void unlock_vma_range(struct proc_maps_locking_ctx *lock_ctx) -{ - if (lock_ctx->mmap_locked) { - unlock_ctx_mm(lock_ctx); - } else { - unlock_ctx_vma(lock_ctx); - rcu_read_unlock(); - } -} - static struct vm_area_struct *get_next_vma(struct proc_maps_private *priv, loff_t last_pos) { @@ -286,13 +267,8 @@ static void *m_start(struct seq_file *m, loff_t *ppos) return NULL; } - if (!lock_vma_range(m, lock_ctx)) { - mmput(mm); - put_task_struct(priv->task); - priv->task = NULL; - return ERR_PTR(-EINTR); - } - + rcu_read_lock(); + reset_lock_ctx(lock_ctx); /* * Reset current position if last_addr was set before * and it's not a sentinel. @@ -325,7 +301,12 @@ static void m_stop(struct seq_file *m, void *v) return; release_task_mempolicy(priv); - unlock_vma_range(&priv->lock_ctx); + if (priv->lock_ctx.mmap_locked) { + unlock_ctx_mm(&priv->lock_ctx); + } else { + unlock_ctx_vma(&priv->lock_ctx); + rcu_read_unlock(); + } mmput(mm); put_task_struct(priv->task); priv->task = NULL; @@ -518,21 +499,6 @@ static int pid_maps_open(struct inode *inode, struct file *file) PROCMAP_QUERY_VMA_FLAGS \ ) -static int query_vma_setup(struct proc_maps_locking_ctx *lock_ctx) -{ - reset_lock_ctx(lock_ctx); - - return 0; -} - -static void query_vma_teardown(struct proc_maps_locking_ctx *lock_ctx) -{ - if (lock_ctx->mmap_locked) - unlock_ctx_mm(lock_ctx); - else - unlock_ctx_vma(lock_ctx); -} - static struct vm_area_struct *query_vma_find_by_addr(struct proc_maps_locking_ctx *lock_ctx, unsigned long addr) { @@ -653,12 +619,7 @@ static int do_procmap_query(struct mm_struct *mm, void __user *uarg) if (!mm || !mmget_not_zero(mm)) return -ESRCH; - err = query_vma_setup(&lock_ctx); - if (err) { - mmput(mm); - return err; - } - + reset_lock_ctx(&lock_ctx); vma = query_matching_vma(&lock_ctx, karg.query_addr, karg.query_flags); if (IS_ERR(vma)) { err = PTR_ERR(vma); @@ -732,7 +693,10 @@ static int do_procmap_query(struct mm_struct *mm, void __user *uarg) vm_file = get_file(vma->vm_file); /* unlock vma or mmap_lock, and put mm_struct before copying data to user */ - query_vma_teardown(&lock_ctx); + if (lock_ctx.mmap_locked) + unlock_ctx_mm(&lock_ctx); + else + unlock_ctx_vma(&lock_ctx); mmput(mm); if (karg.build_id_size) { @@ -773,7 +737,10 @@ static int do_procmap_query(struct mm_struct *mm, void __user *uarg) return 0; out: - query_vma_teardown(&lock_ctx); + if (lock_ctx.mmap_locked) + unlock_ctx_mm(&lock_ctx); + else + unlock_ctx_vma(&lock_ctx); mmput(mm); out_file: if (vm_file) From 4ab87d0c9ee4e345c0f3467fdb8add0b1e1e20e8 Mon Sep 17 00:00:00 2001 From: Suren Baghdasaryan Date: Fri, 18 Sep 2026 08:33:13 -0700 Subject: [PATCH 1086/1352] proc/task_mmu: remove unnecessary inlines in function definitions It was pointed out in the previous reviews of this code that many functions are specified as inline, which is unnecessary as the compiler can make that decision by itself. Cleanup these definitions. No change in the resulting binary file size with gcc v15.2.0. No functional change intended. Link: https://lore.kernel.org/20260918153318.758387-3-surenb@google.com Signed-off-by: Suren Baghdasaryan Signed-off-by: Andrew Morton Reviewed-by: Liam R. Howlett (Oracle) Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: Usama Arif Acked-by: David Hildenbrand (Arm) Cc: Jann Horn Cc: Matthew Wilcox (Oracle) Cc: "Paul E . McKenney" Cc: Pedro Falcato Cc: Vlastimil Babka --- fs/proc/task_mmu.c | 34 +++++++++++++++++++--------------- 1 file changed, 19 insertions(+), 15 deletions(-) diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c index 191a054d7e4093..8a72dc0dc9452b 100644 --- a/fs/proc/task_mmu.c +++ b/fs/proc/task_mmu.c @@ -130,7 +130,8 @@ static void release_task_mempolicy(struct proc_maps_private *priv) } #endif -static inline int lock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx) +#ifdef CONFIG_PROC_PAGE_MONITOR +static int lock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx) { int ret = mmap_read_lock_killable(lock_ctx->mm); @@ -139,8 +140,9 @@ static inline int lock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx) return ret; } +#endif -static inline void unlock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx) +static void unlock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx) { mmap_read_unlock(lock_ctx->mm); lock_ctx->mmap_locked = false; @@ -177,8 +179,8 @@ static struct vm_area_struct *get_next_vma(struct proc_maps_private *priv, return vma; } -static inline bool fallback_to_mmap_lock(struct proc_maps_private *priv, - loff_t pos) +static bool fallback_to_mmap_lock(struct proc_maps_private *priv, + loff_t pos) { struct proc_maps_locking_ctx *lock_ctx = &priv->lock_ctx; @@ -194,7 +196,8 @@ static inline bool fallback_to_mmap_lock(struct proc_maps_private *priv, return true; } -static inline void drop_rcu(struct proc_maps_private *priv) +#ifdef CONFIG_PROC_PAGE_MONITOR +static void drop_rcu(struct proc_maps_private *priv) { if (priv->lock_ctx.mmap_locked) return; @@ -202,7 +205,7 @@ static inline void drop_rcu(struct proc_maps_private *priv) rcu_read_unlock(); } -static inline void reacquire_rcu(struct proc_maps_private *priv) +static void reacquire_rcu(struct proc_maps_private *priv) { if (priv->lock_ctx.mmap_locked) return; @@ -211,6 +214,7 @@ static inline void reacquire_rcu(struct proc_maps_private *priv) /* Reinitialize the iterator. */ vma_iter_set(&priv->iter, priv->lock_ctx.locked_vma->vm_end); } +#endif static struct vm_area_struct *proc_get_vma(struct seq_file *m, loff_t *ppos) { @@ -1230,7 +1234,7 @@ static const struct mm_walk_ops smaps_shmem_walk_vma_lock_ops = { .walk_lock = PGWALK_VMA_RDLOCK_VERIFY, }; -static inline const struct mm_walk_ops * +static const struct mm_walk_ops * get_smaps_walk_ops(struct proc_maps_private *priv) { if (priv->lock_ctx.mmap_locked) @@ -1238,7 +1242,7 @@ get_smaps_walk_ops(struct proc_maps_private *priv) return &smaps_walk_vma_lock_ops; } -static inline const struct mm_walk_ops * +static const struct mm_walk_ops * get_smaps_shmem_walk_ops(struct proc_maps_private *priv) { if (priv->lock_ctx.mmap_locked) @@ -1572,7 +1576,7 @@ struct clear_refs_private { enum clear_refs_types type; }; -static inline bool pte_is_pinned(struct vm_area_struct *vma, unsigned long addr, pte_t pte) +static bool pte_is_pinned(struct vm_area_struct *vma, unsigned long addr, pte_t pte) { struct folio *folio; @@ -1588,8 +1592,8 @@ static inline bool pte_is_pinned(struct vm_area_struct *vma, unsigned long addr, return folio_maybe_dma_pinned(folio); } -static inline void clear_soft_dirty(struct vm_area_struct *vma, - unsigned long addr, pte_t *pte) +static void clear_soft_dirty(struct vm_area_struct *vma, unsigned long addr, + pte_t *pte) { if (!pgtable_supports_soft_dirty()) return; @@ -1620,7 +1624,7 @@ static inline void clear_soft_dirty(struct vm_area_struct *vma, } #if defined(CONFIG_TRANSPARENT_HUGEPAGE) -static inline void clear_soft_dirty_pmd(struct vm_area_struct *vma, +static void clear_soft_dirty_pmd(struct vm_area_struct *vma, unsigned long addr, pmd_t *pmdp) { pmd_t old, pmd = *pmdp; @@ -1646,7 +1650,7 @@ static inline void clear_soft_dirty_pmd(struct vm_area_struct *vma, } } #else -static inline void clear_soft_dirty_pmd(struct vm_area_struct *vma, +static void clear_soft_dirty_pmd(struct vm_area_struct *vma, unsigned long addr, pmd_t *pmdp) { } @@ -1848,7 +1852,7 @@ struct pagemapread { #define PM_END_OF_BUFFER 1 -static inline pagemap_entry_t make_pme(u64 frame, u64 flags) +static pagemap_entry_t make_pme(u64 frame, u64 flags) { return (pagemap_entry_t) { .pme = (frame & PM_PFRAME_MASK) | flags }; } @@ -3390,7 +3394,7 @@ static const struct mm_walk_ops show_numa_vma_lock_ops = { .walk_lock = PGWALK_VMA_RDLOCK_VERIFY, }; -static inline const struct mm_walk_ops * +static const struct mm_walk_ops * get_show_numa_ops(struct proc_maps_private *priv) { if (priv->lock_ctx.mmap_locked) From d7fdcb6336b3d15e366e70e93c0cf909f4e11426 Mon Sep 17 00:00:00 2001 From: Andrew Morton Date: Sun, 27 Sep 2026 10:59:53 -0700 Subject: [PATCH 1087/1352] proc-task_mmu-remove-unnecessary-inlines-in-function-definitions-fix fix CONFIG_NUMA=y && CONFIG_PROC_PAGE_MONITOR=n build show_numa_map() uses drop_rcu() and reacquire_rcu(), but those helpers are currently built only when CONFIG_PROC_PAGE_MONITOR is enabled. A kernel with CONFIG_NUMA=y and CONFIG_PROC_PAGE_MONITOR=n therefore fails to build with implicit declarations for both helpers. Cc: Suren Baghdasaryan Signed-off-by: Andrew Morton --- fs/proc/task_mmu.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c index 8a72dc0dc9452b..4ed3b76d00c871 100644 --- a/fs/proc/task_mmu.c +++ b/fs/proc/task_mmu.c @@ -196,7 +196,7 @@ static bool fallback_to_mmap_lock(struct proc_maps_private *priv, return true; } -#ifdef CONFIG_PROC_PAGE_MONITOR +#if defined(CONFIG_PROC_PAGE_MONITOR) || defined(CONFIG_NUMA) static void drop_rcu(struct proc_maps_private *priv) { if (priv->lock_ctx.mmap_locked) From c52fd13e66f121be1561207ba59a32b23bccc9b3 Mon Sep 17 00:00:00 2001 From: Suren Baghdasaryan Date: Fri, 18 Sep 2026 08:33:14 -0700 Subject: [PATCH 1088/1352] proc/task_mmu: clarify shmem mapping walk conditions in smap_gather_stats() smap_gather_stats() optimizes stats gathering by skipping the walk for shmem mappings in certain conditions. Update the comment to clarify these conditions and use vma_is_cow_mapping() for CoW identification instead of open-coding it. No functional change intended. Link: https://lore.kernel.org/20260918153318.758387-4-surenb@google.com Suggested-by: David Hildenbrand (Arm) Signed-off-by: Suren Baghdasaryan Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) Cc: Jann Horn Cc: Liam R. Howlett (Oracle) Cc: Matthew Wilcox (Oracle) Cc: "Paul E . McKenney" Cc: Pedro Falcato Cc: Usama Arif Cc: Vlastimil Babka --- fs/proc/task_mmu.c | 23 +++++++++-------------- 1 file changed, 9 insertions(+), 14 deletions(-) diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c index 4ed3b76d00c871..dd9cccf8a4892d 100644 --- a/fs/proc/task_mmu.c +++ b/fs/proc/task_mmu.c @@ -1274,23 +1274,18 @@ static void smap_gather_stats(struct proc_maps_private *priv, if (vma->vm_file && shmem_mapping(vma->vm_file->f_mapping)) { /* - * For shared or readonly shmem mappings we know that all - * swapped out pages belong to the shmem object, and we can - * obtain the swap value much more efficiently. For private - * writable mappings, we might have COW pages that are - * not affected by the parent swapped out pages of the shmem - * object, so we have to distinguish them during the page walk. - * Unless we know that the shmem object (or the part mapped by - * our VMA) has no swapped out pages at all. + * CoW mappings might map anon folios that do not belong to + * shmem. Perform a less efficient page table walk in this + * situation, unless we know that the shmem object (or the + * part mapped by our VMA) has no swapped out pages at all. */ - unsigned long shmem_swapped = shmem_swap_usage(vma); + const unsigned long shmem_swapped = shmem_swap_usage(vma); + const bool is_cow = vma_is_cow_mapping(vma); - if (!start && (!shmem_swapped || (vma->vm_flags & VM_SHARED) || - !(vma->vm_flags & VM_WRITE))) { - mss->swap += shmem_swapped; - } else { + if (start || (shmem_swapped && is_cow)) ops = get_smaps_shmem_walk_ops(priv); - } + else + mss->swap += shmem_swapped; } if (!start) From 780b27fb8efdd32f935749ead7dd7efdfaa43dcf Mon Sep 17 00:00:00 2001 From: Suren Baghdasaryan Date: Fri, 18 Sep 2026 08:33:15 -0700 Subject: [PATCH 1089/1352] proc/task_mmu: remove special-casing of smap_gather_stats() start parameter smap_gather_stats() interprets its start parameter to mean vma->vm_start when it's set to 0. Eliminate this special interpretation and provide two separate functions for a partial and complete VMA walk. Since smap_gather_stats() operates within a single VMA, we can replace walk_page_vma()/walk_page_range() calls with walk_page_range_vma() which is simpler and also can be called while holding per-VMA lock. No functional change intended. Link: https://lore.kernel.org/20260918153318.758387-5-surenb@google.com Suggested-by: Lorenzo Stoakes Signed-off-by: Suren Baghdasaryan Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) Cc: Jann Horn Cc: Liam R. Howlett (Oracle) Cc: Matthew Wilcox (Oracle) Cc: "Paul E . McKenney" Cc: Pedro Falcato Cc: Usama Arif Cc: Vlastimil Babka --- fs/proc/task_mmu.c | 52 +++++++++++++++++++++++++++++++--------------- 1 file changed, 35 insertions(+), 17 deletions(-) diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c index dd9cccf8a4892d..f62359ffc3f74b 100644 --- a/fs/proc/task_mmu.c +++ b/fs/proc/task_mmu.c @@ -1250,20 +1250,26 @@ get_smaps_shmem_walk_ops(struct proc_maps_private *priv) return &smaps_shmem_walk_vma_lock_ops; } -/* - * Gather mem stats from @vma with the indicated beginning - * address @start, and keep them in @mss. +/** + * smap_gather_stats_range() - Gather mem stats from a portion of the @vma. + * @priv: proc maps private state. + * @vma: The VMA to gather stats for. + * @mss: The accumulated stats. + * @start: The address from which to start. * - * Use vm_start of @vma as the beginning address if @start is 0. + * This gathers stats for the portion of the VMA starting at the @start + * address. */ -static void smap_gather_stats(struct proc_maps_private *priv, - struct vm_area_struct *vma, - struct mem_size_stats *mss, unsigned long start) +static void smap_gather_stats_range(struct proc_maps_private *priv, + struct vm_area_struct *vma, + struct mem_size_stats *mss, + unsigned long start) { const struct mm_walk_ops *ops = get_smaps_walk_ops(priv); + const bool is_partial = start > vma->vm_start; /* Invalid start */ - if (start >= vma->vm_end) + if (start < vma->vm_start || start >= vma->vm_end) return; if (vma == get_gate_vma(priv->lock_ctx.mm)) @@ -1282,20 +1288,31 @@ static void smap_gather_stats(struct proc_maps_private *priv, const unsigned long shmem_swapped = shmem_swap_usage(vma); const bool is_cow = vma_is_cow_mapping(vma); - if (start || (shmem_swapped && is_cow)) + if (is_partial || (shmem_swapped && is_cow)) ops = get_smaps_shmem_walk_ops(priv); else mss->swap += shmem_swapped; } - if (!start) - walk_page_vma(vma, ops, mss); - else - walk_page_range(vma->vm_mm, start, vma->vm_end, ops, mss); + walk_page_range_vma(vma, start, vma->vm_end, ops, mss); reacquire_rcu(priv); } +/** + * smap_gather_stats() - Gather mem stats from the entire @vma. + * @priv: proc maps private state. + * @vma: The VMA to gather stats for. + * @mss: The accumulated stats. + * + * This gathers stats for the whole of the VMA. + */ +static void smap_gather_stats(struct proc_maps_private *priv, + struct vm_area_struct *vma, struct mem_size_stats *mss) +{ + smap_gather_stats_range(priv, vma, mss, vma->vm_start); +} + #define SEQ_PUT_DEC(str, val) \ seq_put_decimal_ull_width(m, str, (val) >> 10, 8) @@ -1346,7 +1363,7 @@ static int show_smap(struct seq_file *m, void *v) struct vm_area_struct *vma = v; struct mem_size_stats mss = {}; - smap_gather_stats(priv, vma, &mss, 0); + smap_gather_stats(priv, vma, &mss); show_map_vma(m, vma); @@ -1399,7 +1416,7 @@ static int show_smaps_rollup(struct seq_file *m, void *v) vma_start = vma->vm_start; do { - smap_gather_stats(priv, vma, &mss, 0); + smap_gather_stats(priv, vma, &mss); last_vma_end = vma->vm_end; /* @@ -1458,14 +1475,15 @@ static int show_smaps_rollup(struct seq_file *m, void *v) /* Case 1 and 2 above */ if (vma->vm_start >= last_vma_end) { - smap_gather_stats(priv, vma, &mss, 0); + smap_gather_stats(priv, vma, &mss); last_vma_end = vma->vm_end; continue; } /* Case 4 above */ if (vma->vm_end > last_vma_end) { - smap_gather_stats(priv, vma, &mss, last_vma_end); + smap_gather_stats_range(priv, vma, &mss, + last_vma_end); last_vma_end = vma->vm_end; } } From 443cd8e89907ac54986ddf0b261deea2d2243f0b Mon Sep 17 00:00:00 2001 From: Suren Baghdasaryan Date: Fri, 18 Sep 2026 08:33:16 -0700 Subject: [PATCH 1090/1352] proc/task_mmu: change proc_get_vma() to stop returning gate VMA at the end proc_get_vma() returning gate VMA at the end is desirable for the its current m_start/m_next callers, as they need to report a gate VMA at the end of the address space. This behavior is very specific to these callers and makes proc_get_vma() hard to use for other purposes. Move this usage-specific behavior into the callers themselves so that proc_get_vma() returns either a valid VMA, an error or a NULL when no more VMAs are available. This makes it more generic, simpler and usable in the later patches. Link: https://lore.kernel.org/20260918153318.758387-6-surenb@google.com Signed-off-by: Suren Baghdasaryan Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) Cc: Jann Horn Cc: Liam R. Howlett (Oracle) Cc: Matthew Wilcox (Oracle) Cc: "Paul E . McKenney" Cc: Pedro Falcato Cc: Usama Arif Cc: Vlastimil Babka --- fs/proc/task_mmu.c | 28 +++++++++++++++++++++++----- 1 file changed, 23 insertions(+), 5 deletions(-) diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c index f62359ffc3f74b..2a29bfb41ab52b 100644 --- a/fs/proc/task_mmu.c +++ b/fs/proc/task_mmu.c @@ -240,9 +240,6 @@ static struct vm_area_struct *proc_get_vma(struct seq_file *m, loff_t *ppos) * found the extended vma with the same vm_start. */ *ppos = vma->vm_end; - } else { - *ppos = SENTINEL_VMA_GATE; - vma = get_gate_vma(priv->lock_ctx.mm); } return vma; @@ -252,6 +249,7 @@ static void *m_start(struct seq_file *m, loff_t *ppos) { struct proc_maps_private *priv = m->private; struct proc_maps_locking_ctx *lock_ctx; + struct vm_area_struct *vma; loff_t last_addr = *ppos; struct mm_struct *mm; @@ -281,19 +279,39 @@ static void *m_start(struct seq_file *m, loff_t *ppos) *ppos = last_addr = priv->last_pos; vma_iter_init(&priv->iter, mm, (unsigned long)last_addr); hold_task_mempolicy(priv); + /* + * If seq_file had to flush its collected data right after m_next() set + * position to SENTINEL_VMA_GATE, m_start() will get that sentinel and + * should return gate_vma without calling proc_get_vma(). + */ if (last_addr == SENTINEL_VMA_GATE) return get_gate_vma(mm); - return proc_get_vma(m, ppos); + vma = proc_get_vma(m, ppos); + if (vma) + return vma; + + /* Return gate VMA at the end */ + *ppos = SENTINEL_VMA_GATE; + return get_gate_vma(mm); } static void *m_next(struct seq_file *m, void *v, loff_t *ppos) { + struct proc_maps_private *priv = m->private; + struct vm_area_struct *vma; + if (*ppos == SENTINEL_VMA_GATE) { *ppos = SENTINEL_VMA_END; return NULL; } - return proc_get_vma(m, ppos); + vma = proc_get_vma(m, ppos); + if (vma) + return vma; + + /* Return gate VMA at the end */ + *ppos = SENTINEL_VMA_GATE; + return get_gate_vma(priv->lock_ctx.mm); } static void m_stop(struct seq_file *m, void *v) From 147d4d28c93bf538c47320b85c366dc1c7a5ba1c Mon Sep 17 00:00:00 2001 From: Suren Baghdasaryan Date: Fri, 18 Sep 2026 08:33:17 -0700 Subject: [PATCH 1091/1352] proc/task_mmu: read proc/pid/smaps_rollup under per-vma lock proc/pid/smaps_rollup can be read using the combination of RCU and VMA read locks, similar to proc/pid/{maps|smaps|numa_maps}. RCU is required to safely traverse the VMA tree and VMA lock stabilizes the VMA being processed and the pagetable walk. Note that we have to keep the logic to drop mmap_lock on contention because even when using per-VMA locks we might have to fall back to holding the mmap_lock. Running Paul's contention benchmark [1] shows considerable improvement both in median and in the worst case latencies: Execution command: run-proc-vs-map.sh --nsamples 20 --rawdata -- \ --busyduration 2 --procfile smaps_rollup Baseline: Median Minimum Maximum 0.174 0.161 2.553 0.174 0.164 2.663 0.174 0.165 2.664 0.174 0.166 2.679 0.174 0.167 2.691 0.174 0.168 2.704 0.174 0.169 2.729 0.174 0.172 2.741 0.174 0.174 2.745 0.174 0.174 2.755 0.174 0.175 2.790 0.174 0.177 2.809 0.174 0.179 3.096 0.174 0.183 3.144 0.174 0.184 3.158 0.174 0.185 3.175 0.174 0.185 4.568 0.174 0.198 4.821 0.174 0.214 5.143 0.174 0.251 5.220 Patched: Median Minimum Maximum 0.007 0.007 1.952 0.007 0.007 1.955 0.007 0.007 1.955 0.007 0.007 1.955 0.007 0.007 1.957 0.007 0.007 1.969 0.007 0.007 2.065 0.007 0.007 2.075 0.007 0.007 2.146 0.007 0.007 2.195 0.007 0.007 2.223 0.007 0.007 2.259 0.007 0.007 2.488 0.007 0.007 2.562 0.007 0.007 2.599 0.007 0.007 2.697 0.007 0.007 3.030 0.007 0.007 3.075 0.007 0.007 3.145 0.007 0.007 3.225 Remove now unused lock_ctx_mm() and move unlock_ctx_vma() next to unlock_ctx_mm() as they are logically related. Remove a long comment about 4 cases that we handle when dropping the mmap lock in the middle of VMA walk due to contention. The first 3 cases explained there are handled naturally and only case 4 needs to be handled in a special way, which is done in smap_gather_stats() by gathering stats from the portion of the VMA that has not yet been processed. For posterity, moving this comment here: After dropping the lock, there are four cases to consider. See the following example for explanation. +------+------+-----------+ | VMA1 | VMA2 | VMA3 | +------+------+-----------+ | | | | 4k 8k 16k 400k Suppose we drop the lock after reading VMA2 due to contention, then we get: last_vma_end = 16k 1) VMA2 is freed, but VMA3 exists: vma_next(vmi) will return VMA3. In this case, just continue from VMA3. 2) VMA2 still exists: vma_next(vmi) will return VMA3. In this case, just continue from VMA3. 3) No more VMAs can be found: vma_next(vmi) will return NULL. No more things to do, just break. 4) (last_vma_end - 1) is the middle of a vma (VMA'): vma_next(vmi) will return VMA' whose range contains last_vma_end. Iterate VMA' from last_vma_end. Link: https://lore.kernel.org/20260918153318.758387-7-surenb@google.com Link: https://github.com/paulmckrcu/proc-mmap_sem-test [1] Signed-off-by: Suren Baghdasaryan Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Cc: David Hildenbrand (Arm) Cc: Jann Horn Cc: Liam R. Howlett (Oracle) Cc: Matthew Wilcox (Oracle) Cc: "Paul E . McKenney" Cc: Pedro Falcato Cc: Usama Arif Cc: Vlastimil Babka --- fs/proc/task_mmu.c | 159 ++++++++++++++++++--------------------------- 1 file changed, 63 insertions(+), 96 deletions(-) diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c index 2a29bfb41ab52b..44147ba0b899da 100644 --- a/fs/proc/task_mmu.c +++ b/fs/proc/task_mmu.c @@ -130,30 +130,12 @@ static void release_task_mempolicy(struct proc_maps_private *priv) } #endif -#ifdef CONFIG_PROC_PAGE_MONITOR -static int lock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx) -{ - int ret = mmap_read_lock_killable(lock_ctx->mm); - - if (!ret) - lock_ctx->mmap_locked = true; - - return ret; -} -#endif - static void unlock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx) { mmap_read_unlock(lock_ctx->mm); lock_ctx->mmap_locked = false; } -static void reset_lock_ctx(struct proc_maps_locking_ctx *lock_ctx) -{ - lock_ctx->locked_vma = NULL; - lock_ctx->mmap_locked = false; -} - static void unlock_ctx_vma(struct proc_maps_locking_ctx *lock_ctx) { if (lock_ctx->locked_vma) { @@ -162,6 +144,12 @@ static void unlock_ctx_vma(struct proc_maps_locking_ctx *lock_ctx) } } +static void reset_lock_ctx(struct proc_maps_locking_ctx *lock_ctx) +{ + lock_ctx->locked_vma = NULL; + lock_ctx->mmap_locked = false; +} + static struct vm_area_struct *get_next_vma(struct proc_maps_private *priv, loff_t last_pos) { @@ -1406,12 +1394,14 @@ static int show_smap(struct seq_file *m, void *v) static int show_smaps_rollup(struct seq_file *m, void *v) { struct proc_maps_private *priv = m->private; + struct proc_maps_locking_ctx *lock_ctx = &priv->lock_ctx; + struct mm_struct *mm = lock_ctx->mm; struct mem_size_stats mss = {}; - struct mm_struct *mm = priv->lock_ctx.mm; + unsigned long last_vma_end = 0; + unsigned long vma_start = 0; struct vm_area_struct *vma; - unsigned long vma_start = 0, last_vma_end = 0; + loff_t pos = 0; int ret = 0; - VMA_ITERATOR(vmi, mm, 0); priv->task = get_proc_task(priv->inode); if (!priv->task) @@ -1422,90 +1412,63 @@ static int show_smaps_rollup(struct seq_file *m, void *v) goto out_put_task; } - ret = lock_ctx_mm(&priv->lock_ctx); - if (ret) - goto out_put_mm; - hold_task_mempolicy(priv); - vma = vma_next(&vmi); + rcu_read_lock(); + reset_lock_ctx(lock_ctx); + vma_iter_init(&priv->iter, mm, 0); + vma = proc_get_vma(m, &pos); if (unlikely(!vma)) goto empty_set; - vma_start = vma->vm_start; - do { - smap_gather_stats(priv, vma, &mss); + if (!IS_ERR(vma)) + vma_start = vma->vm_start; + + while (vma) { + if (IS_ERR(vma)) { + ret = PTR_ERR(vma); + goto out_unlock; + } + + if (vma->vm_start < last_vma_end) { + /* + * After retaking the lock, already reported VMA grew + * or got merged with the next one and we found it + * again. Gather stats for the remaining portion by + * starting at last_vma_end. + */ + smap_gather_stats_range(priv, vma, &mss, last_vma_end); + } else { + /* Found next unreported VMA, start from its beginning */ + smap_gather_stats(priv, vma, &mss); + } last_vma_end = vma->vm_end; /* - * Release mmap_lock temporarily if someone wants to - * access it for write request. + * If the VMA lock is not taken, we hold the often contended + * mmap lock. This can happen if we had to fall back to the + * mmap lock. + * + * To relieve pressure, check if it is indeed contended, then + * temporarily release it. */ - if (mmap_lock_is_contended(mm)) { - vma_iter_invalidate(&vmi); - unlock_ctx_mm(&priv->lock_ctx); - ret = lock_ctx_mm(&priv->lock_ctx); - if (ret) { - release_task_mempolicy(priv); - goto out_put_mm; - } - + if (lock_ctx->mmap_locked && + mmap_lock_is_contended(lock_ctx->mm)) { + unlock_ctx_mm(lock_ctx); /* - * After dropping the lock, there are four cases to - * consider. See the following example for explanation. - * - * +------+------+-----------+ - * | VMA1 | VMA2 | VMA3 | - * +------+------+-----------+ - * | | | | - * 4k 8k 16k 400k - * - * Suppose we drop the lock after reading VMA2 due to - * contention, then we get: - * - * last_vma_end = 16k - * - * 1) VMA2 is freed, but VMA3 exists: - * - * vma_next(vmi) will return VMA3. - * In this case, just continue from VMA3. - * - * 2) VMA2 still exists: - * - * vma_next(vmi) will return VMA3. - * In this case, just continue from VMA3. - * - * 3) No more VMAs can be found: - * - * vma_next(vmi) will return NULL. - * No more things to do, just break. - * - * 4) (last_vma_end - 1) is the middle of a vma (VMA'): - * - * vma_next(vmi) will return VMA' whose range - * contains last_vma_end. - * Iterate VMA' from last_vma_end. + * Even though we previously fell back to mmap lock, + * we try taking VMA lock for the next VMA, since it + * might not be under modification. In the worst case + * we will fall back to mmap lock again. */ - vma = vma_next(&vmi); - /* Case 3 above */ - if (!vma) - break; - - /* Case 1 and 2 above */ - if (vma->vm_start >= last_vma_end) { - smap_gather_stats(priv, vma, &mss); - last_vma_end = vma->vm_end; - continue; - } - - /* Case 4 above */ - if (vma->vm_end > last_vma_end) { - smap_gather_stats_range(priv, vma, &mss, - last_vma_end); - last_vma_end = vma->vm_end; - } + rcu_read_lock(); + reset_lock_ctx(lock_ctx); + /* Resume from the last position. */ + pos = last_vma_end; + vma_iter_init(&priv->iter, mm, pos); } - } for_each_vma(vmi, vma); + vma = proc_get_vma(m, &pos); + } empty_set: show_vma_header_prefix(m, vma_start, last_vma_end, 0, 0, 0, 0); @@ -1514,10 +1477,14 @@ static int show_smaps_rollup(struct seq_file *m, void *v) __show_smap(m, &mss, true); +out_unlock: + if (lock_ctx->mmap_locked) { + unlock_ctx_mm(lock_ctx); + } else { + unlock_ctx_vma(lock_ctx); + rcu_read_unlock(); + } release_task_mempolicy(priv); - unlock_ctx_mm(&priv->lock_ctx); - -out_put_mm: mmput(mm); out_put_task: put_task_struct(priv->task); From 8e933fa2c80730ce3d8b46955ae45644af0021a0 Mon Sep 17 00:00:00 2001 From: Suren Baghdasaryan Date: Fri, 18 Sep 2026 08:33:18 -0700 Subject: [PATCH 1092/1352] selftests/proc: add /proc/pid/smaps_rollup tearing tests During tearing tests, smaps_rollup Pss* metrics should stay constant. Extend /proc/pid/smaps tearing tests to also check for smaps_rollup consistency. Link: https://lore.kernel.org/20260918153318.758387-8-surenb@google.com Signed-off-by: Suren Baghdasaryan Signed-off-by: Andrew Morton Acked-by: Lorenzo Stoakes (ARM) Cc: David Hildenbrand (Arm) Cc: Jann Horn Cc: Liam R. Howlett (Oracle) Cc: Matthew Wilcox (Oracle) Cc: "Paul E . McKenney" Cc: Pedro Falcato Cc: Usama Arif Cc: Vlastimil Babka --- tools/testing/selftests/proc/proc-maps-race.c | 186 +++++++++++++++++- 1 file changed, 181 insertions(+), 5 deletions(-) diff --git a/tools/testing/selftests/proc/proc-maps-race.c b/tools/testing/selftests/proc/proc-maps-race.c index 415eccb7046848..bf4c5073f6fc86 100644 --- a/tools/testing/selftests/proc/proc-maps-race.c +++ b/tools/testing/selftests/proc/proc-maps-race.c @@ -80,6 +80,61 @@ enum maps_file { struct vma_modifier_info; +enum smaps_rollup_stat { + Rss, + Pss, + Pss_Dirty, + Pss_Anon, + Pss_File, + Pss_Shmem, + Shared_Clean, + Shared_Dirty, + Private_Clean, + Private_Dirty, + Referenced, + Anonymous, + KSM, + LazyFree, + AnonHugePages, + ShmemPmdMapped, + FilePmdMapped, + Shared_Hugetlb, + Private_Hugetlb, + Swap, + SwapPss, + Locked, + RollupFieldCount +}; + +static const char *smaps_rollup_stat_names[RollupFieldCount] = { + "Rss", + "Pss", + "Pss_Dirty", + "Pss_Anon", + "Pss_File", + "Pss_Shmem", + "Shared_Clean", + "Shared_Dirty", + "Private_Clean", + "Private_Dirty", + "Referenced", + "Anonymous", + "KSM", + "LazyFree", + "AnonHugePages", + "ShmemPmdMapped", + "FilePmdMapped", + "Shared_Hugetlb", + "Private_Hugetlb", + "Swap", + "SwapPss", + "Locked", +}; + +struct smaps_rollup_stats { + unsigned long values[RollupFieldCount]; +}; + FIXTURE(proc_maps_race) { struct vma_modifier_info *mod_info; @@ -91,6 +146,7 @@ FIXTURE(proc_maps_race) enum maps_file maps_file; int shared_mem_size; int skip_pages; + int rollup_fd; int page_size; int vma_count; bool verbose; @@ -132,12 +188,12 @@ struct vma_modifier_info { void *child_mapped_addr[]; }; -static bool read_page(FIXTURE_DATA(proc_maps_race) *self, +static bool read_page(FIXTURE_DATA(proc_maps_race) *self, int fd, struct page_content *page) { ssize_t bytes_read; - bytes_read = read(self->maps_fd, page->data, self->page_size); + bytes_read = read(fd, page->data, self->page_size); if (bytes_read <= 0) return false; @@ -175,7 +231,7 @@ static int locate_containing_page(FIXTURE_DATA(proc_maps_race) *self, char *curr_pos; char *end_pos; - if (!read_page(self, &self->page1)) + if (!read_page(self, self->maps_fd, &self->page1)) return -1; curr_pos = self->page1.data; @@ -205,10 +261,11 @@ static bool read_two_pages(FIXTURE_DATA(proc_maps_race) *self) return false; for (int i = 0; i < self->skip_pages; i++) - if (!read_page(self, &self->page1)) + if (!read_page(self, self->maps_fd, &self->page1)) return false; - return read_page(self, &self->page1) && read_page(self, &self->page2); + return read_page(self, self->maps_fd, &self->page1) && + read_page(self, self->maps_fd, &self->page2); } static void copy_line(const char *line_start, const char *line_end, @@ -317,6 +374,61 @@ static bool read_boundary_lines(FIXTURE_DATA(proc_maps_race) *self, &first_line->end_addr) == 2; } +static bool parse_smaps_rollup(FIXTURE_DATA(proc_maps_race) *self, + struct smaps_rollup_stats *stats) +{ + unsigned int dev_maj, dev_min, inode; + unsigned long start, end, offs; + unsigned long value; + char name[32], perm[5]; + char *curr_pos; + char *end_pos; + char *line_end; + + if (lseek(self->rollup_fd, 0, SEEK_SET) < 0) + return false; + + if (!read_page(self, self->rollup_fd, &self->page1)) + return false; + + curr_pos = self->page1.data; + end_pos = self->page1.data + self->page1.size; + + line_end = strchr(curr_pos, '\n'); + if (!line_end) + return false; + + if (sscanf(curr_pos, "%lx-%lx %4s %lx %u:%u %u %31s", + &start, &end, perm, &offs, &dev_maj, &dev_min, &inode, name) != 8) + return false; + + if (strcmp(name, "[rollup]")) + return false; + + for (int stat = 0; stat < ARRAY_SIZE(smaps_rollup_stat_names); stat++) { + int len; + + curr_pos = line_end + 1; + if (curr_pos >= end_pos) + return false; + + line_end = strchr(curr_pos, '\n'); + if (!line_end) + return false; + + if (sscanf(curr_pos, "%31s %lu kB", name, &value) != 2) + return false; + + len = strlen(name); + if (name[len - 1] != ':' || strncmp(name, smaps_rollup_stat_names[stat], len - 1)) + return false; + + stats->values[stat] = value; + } + + return true; +} + /* Thread synchronization routines */ static void wait_for_state(struct vma_modifier_info *mod_info, enum test_state state) { @@ -397,6 +509,40 @@ static bool print_boundaries_on(bool condition, const char *title, return condition; } +static void print_smaps_rollup_stats(const char *title, FIXTURE_DATA(proc_maps_race) *self, + struct smaps_rollup_stats *stats) +{ + printf("%s", title); + for (int stat = 0; stat < ARRAY_SIZE(smaps_rollup_stat_names); stat++) + printf("%64s %lu kB\n", smaps_rollup_stat_names[stat], stats->values[stat]); +} + +static bool cmp_smaps_rollup_stat(struct smaps_rollup_stats *s1, + struct smaps_rollup_stats *s2, enum smaps_rollup_stat stat) +{ + return s1->values[stat] == s2->values[stat]; +} + +static bool compare_smaps_rollup(FIXTURE_DATA(proc_maps_race) *self, + struct smaps_rollup_stats *expected, + struct smaps_rollup_stats *actual) +{ + /* + * Clean/dirty metrics might change but Pss-related ones + * should stay constant. + */ + if (cmp_smaps_rollup_stat(expected, actual, Pss) && + cmp_smaps_rollup_stat(expected, actual, Pss_Anon) && + cmp_smaps_rollup_stat(expected, actual, Pss_File) && + cmp_smaps_rollup_stat(expected, actual, Pss_Shmem)) + return true; + + print_smaps_rollup_stats("Expected stats:", self, expected); + print_smaps_rollup_stats("Actual stats:", self, actual); + + return false; +} + static void report_test_start(const char *name, bool verbose) { if (verbose) @@ -572,6 +718,7 @@ FIXTURE_SETUP(proc_maps_race) unsigned long first_map_addr; unsigned long last_map_addr; unsigned long duration_sec; + char rollup_fname[32]; char fname[32]; self->page_size = (unsigned long)sysconf(_SC_PAGESIZE); @@ -649,6 +796,9 @@ FIXTURE_SETUP(proc_maps_race) break; case SMAPS: sprintf(fname, "/proc/%d/smaps", self->pid); + sprintf(rollup_fname, "/proc/%d/smaps_rollup", self->pid); + self->rollup_fd = open(rollup_fname, O_RDONLY); + ASSERT_NE(self->rollup_fd, -1); break; default: ksft_exit_fail(); @@ -711,6 +861,8 @@ FIXTURE_TEARDOWN(proc_maps_race) for (int i = 0; i < self->vma_count; i++) munmap(self->mod_info->child_mapped_addr[i], self->page_size); close(self->maps_fd); + if (self->maps_file == SMAPS) + close(self->rollup_fd); waitpid(self->pid, &status, 0); munmap(self->mod_info, self->shared_mem_size); } @@ -723,6 +875,7 @@ TEST_F(proc_maps_race, test_maps_tearing_from_split) struct line_content split_first_line; struct line_content restored_last_line; struct line_content restored_first_line; + struct smaps_rollup_stats orig_stats; wait_for_state(mod_info, SETUP_READY); @@ -736,6 +889,8 @@ TEST_F(proc_maps_race, test_maps_tearing_from_split) report_test_start("Tearing from split", self->verbose); ASSERT_TRUE(capture_mod_pattern(self, &split_last_line, &split_first_line, &restored_last_line, &restored_first_line)); + if (self->maps_file == SMAPS) + ASSERT_TRUE(parse_smaps_rollup(self, &orig_stats)); /* Now start concurrent modifications for self->duration_sec */ signal_state(mod_info, TEST_READY); @@ -799,6 +954,11 @@ TEST_F(proc_maps_race, test_maps_tearing_from_split) vma_end == self->last_line.end_addr) || (vma_start == split_first_line.start_addr && vma_end == split_first_line.end_addr)); + } else { + struct smaps_rollup_stats stats; + + ASSERT_TRUE(parse_smaps_rollup(self, &stats)); + ASSERT_TRUE(compare_smaps_rollup(self, &orig_stats, &stats)); } clock_gettime(CLOCK_MONOTONIC_COARSE, &end_ts); end_test_iteration(&end_ts, self->verbose); @@ -817,6 +977,7 @@ TEST_F(proc_maps_race, test_maps_tearing_from_resize) struct line_content shrunk_first_line; struct line_content restored_last_line; struct line_content restored_first_line; + struct smaps_rollup_stats orig_stats; wait_for_state(mod_info, SETUP_READY); @@ -830,6 +991,8 @@ TEST_F(proc_maps_race, test_maps_tearing_from_resize) report_test_start("Tearing from resize", self->verbose); ASSERT_TRUE(capture_mod_pattern(self, &shrunk_last_line, &shrunk_first_line, &restored_last_line, &restored_first_line)); + if (self->maps_file == SMAPS) + ASSERT_TRUE(parse_smaps_rollup(self, &orig_stats)); /* Now start concurrent modifications for self->duration_sec */ signal_state(mod_info, TEST_READY); @@ -880,6 +1043,11 @@ TEST_F(proc_maps_race, test_maps_tearing_from_resize) ASSERT_TRUE(vma_start == self->last_line.start_addr && (vma_end - vma_start == self->page_size * 3 || vma_end - vma_start == self->page_size)); + } else { + struct smaps_rollup_stats stats; + + ASSERT_TRUE(parse_smaps_rollup(self, &stats)); + ASSERT_TRUE(compare_smaps_rollup(self, &orig_stats, &stats)); } clock_gettime(CLOCK_MONOTONIC_COARSE, &end_ts); end_test_iteration(&end_ts, self->verbose); @@ -898,6 +1066,7 @@ TEST_F(proc_maps_race, test_maps_tearing_from_remap) struct line_content remapped_first_line; struct line_content restored_last_line; struct line_content restored_first_line; + struct smaps_rollup_stats orig_stats; wait_for_state(mod_info, SETUP_READY); @@ -911,6 +1080,8 @@ TEST_F(proc_maps_race, test_maps_tearing_from_remap) report_test_start("Tearing from remap", self->verbose); ASSERT_TRUE(capture_mod_pattern(self, &remapped_last_line, &remapped_first_line, &restored_last_line, &restored_first_line)); + if (self->maps_file == SMAPS) + ASSERT_TRUE(parse_smaps_rollup(self, &orig_stats)); /* Now start concurrent modifications for self->duration_sec */ signal_state(mod_info, TEST_READY); @@ -963,6 +1134,11 @@ TEST_F(proc_maps_race, test_maps_tearing_from_remap) vma_end - vma_start == self->page_size * 3) || (vma_start == self->last_line.start_addr + self->page_size && vma_end - vma_start == self->page_size)); + } else { + struct smaps_rollup_stats stats; + + ASSERT_TRUE(parse_smaps_rollup(self, &stats)); + ASSERT_TRUE(compare_smaps_rollup(self, &orig_stats, &stats)); } clock_gettime(CLOCK_MONOTONIC_COARSE, &end_ts); end_test_iteration(&end_ts, self->verbose); From e4b19c7f28081ce4eb499f245d2c66ce359f093b Mon Sep 17 00:00:00 2001 From: Sang-Heon Jeon Date: Sat, 19 Sep 2026 01:56:40 +0900 Subject: [PATCH 1093/1352] mm/swapops: remove unused is_hwpoison_entry() Since commit 93976a20345b ("mm: eliminate further swapops predicates"), is_hwpoison_entry() has no callers. So remove it. No functional change. Link: https://lore.kernel.org/20260918165642.1014988-1-ekffu200098@gmail.com Signed-off-by: Sang-Heon Jeon Signed-off-by: Andrew Morton Reviewed-by: SJ Park Reviewed-by: Zenghui Yu (Huawei) Reviewed-by: Barry Song Acked-by: Chris Li Cc: Baoquan He Cc: Kairui Song Cc: Kemeng Shi Cc: Nhat Pham Cc: Youngjun Park --- include/linux/swapops.h | 9 --------- 1 file changed, 9 deletions(-) diff --git a/include/linux/swapops.h b/include/linux/swapops.h index e7d0d529f3e0f9..603f9e3f909cd1 100644 --- a/include/linux/swapops.h +++ b/include/linux/swapops.h @@ -261,11 +261,6 @@ static inline swp_entry_t make_hwpoison_entry(struct page *page) return swp_entry(SWP_HWPOISON, page_to_pfn(page)); } -static inline int is_hwpoison_entry(swp_entry_t entry) -{ - return swp_type(entry) == SWP_HWPOISON; -} - #else static inline swp_entry_t make_hwpoison_entry(struct page *page) @@ -273,10 +268,6 @@ static inline swp_entry_t make_hwpoison_entry(struct page *page) return swp_entry(0, 0); } -static inline int is_hwpoison_entry(swp_entry_t swp) -{ - return 0; -} #endif typedef unsigned long pte_marker; From ba82a0b13cfe8d7181704f6f99856387a5ddeb41 Mon Sep 17 00:00:00 2001 From: Sarthak Sharma Date: Fri, 18 Sep 2026 16:52:29 +0530 Subject: [PATCH 1094/1352] selftests/mm: make file helpers return errors Patch series "selftests/mm: separate GUP microbenchmarking from functional testing", v11. gup_test.c currently serves two separate purposes: benchmarking (GUP_FAST_BENCHMARK, PIN_FAST_BENCHMARK and PIN_LONGTERM_BENCHMARK) and functional testing (GUP_BASIC_TEST, PIN_BASIC_TEST and DUMP_USER_PAGES_TEST). Keeping both in one program makes the functional tests harder to run and report individually, while run_vmtests.sh has to invoke the program repeatedly with different options. Separate these roles into tools/mm/gup_bench for benchmarking and tools/testing/selftests/mm/gup for functional testing. Move the shared file and hugepage helpers to tools/lib/mm/ so both programs can use them without duplicating the implementation. Patch 1 makes read_file(), write_file(), read_num(), write_num() and write_num_ignore_einval() return errors to their callers instead of exiting. It also makes read_num() reject negative and malformed values and updates the existing callers to handle failures. Patch 2 moves these file helpers from vm_util.c to tools/lib/mm/. It keeps them available to the mm selftests through vm_util.h and adjusts the selftests build accordingly. Patch 3 moves hugepage_settings.[ch] from selftests/mm to tools/lib/mm/. It also removes its kselftest dependency while preserving TAP-compatible diagnostics for selftest users. Patch 4 moves the existing gup_test implementation from selftests/mm to tools/mm as gup_bench. This keeps the code movement separate from the subsequent changes and makes it easier to review. Patch 5 removes the functional test modes and kselftest dependency from gup_bench. When run without arguments, it performs one GUP_FAST benchmark using the existing defaults instead of running the whole matrix. Other benchmark configurations can be selected through command-line options. Patch 6 adds a new harness-based GUP selftest. It covers THP, non-THP and HugeTLB mappings across private/shared and read/write variants. For each variant, it tests get_user_pages(), get_user_pages_fast(), pin_user_pages(), pin_user_pages_fast() and long-term pinning modes using four batch sizes. The HugeTLB variants share a one-time setup of two hugeTLB pages. This patch (of 6): Change read_file(), write_file(), read_num(), write_num() and write_num_ignore_einval() in vm_util.c to report failures to callers instead of exiting from the helper. Make read_file() return a negative errno on failure and 0 on success, so callers can distinguish a successful read from an I/O error. Also make read_num() reject negative and malformed values. Keep write_num_ignore_einval() silent for -EINVAL while returning other errors to its caller. Update callers to print diagnostics and fail wherever required. Modify a comment which implies write_num() uses ksft_exit_fail_msg(). Also add a helper print_file_access_error() in hugepage_settings.c to print TAP-compatible errors without a kselftest dependency. This prepares the helpers to be moved to tools/lib/mm without a kselftest dependency. Link: https://lore.kernel.org/20260918112234.195857-1-sarthak.sharma@arm.com Link: https://lore.kernel.org/20260918112234.195857-2-sarthak.sharma@arm.com Signed-off-by: Sarthak Sharma Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Tested-by: Muhammad Usama Anjum Cc: Anshuman Khandual Cc: Baolin Wang Cc: Barry Song Cc: Dev Jain Cc: Jason Gunthorpe Cc: John Hubbard Cc: Jonathan Corbet Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Mark Brown Cc: Michal Hocko Cc: Nico Pache Cc: Peter Xu Cc: Ryan Roberts Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Zi Yan --- .../testing/selftests/mm/hugepage_settings.c | 98 +++++++++++--- .../selftests/mm/hugetlb-soft-offline.c | 16 ++- tools/testing/selftests/mm/khugepaged.c | 14 +- .../selftests/mm/split_huge_page_test.c | 5 +- tools/testing/selftests/mm/vm_util.c | 122 ++++++++++++------ tools/testing/selftests/mm/vm_util.h | 8 +- 6 files changed, 192 insertions(+), 71 deletions(-) diff --git a/tools/testing/selftests/mm/hugepage_settings.c b/tools/testing/selftests/mm/hugepage_settings.c index 584054736ce99f..9a63420d0744fb 100644 --- a/tools/testing/selftests/mm/hugepage_settings.c +++ b/tools/testing/selftests/mm/hugepage_settings.c @@ -8,6 +8,7 @@ #include #include #include +#include #include "vm_util.h" #include "hugepage_settings.h" @@ -48,6 +49,11 @@ static const char * const shmem_enabled_strings[] = { NULL }; +static void print_file_access_error(const char *path, int ret) +{ + printf("# %s: %s (%d)\n", path, strerror(-ret), -ret); +} + int thp_read_string(const char *name, const char * const strings[]) { char path[PATH_MAX]; @@ -61,8 +67,9 @@ int thp_read_string(const char *name, const char * const strings[]) exit(EXIT_FAILURE); } - if (!read_file(path, buf, sizeof(buf))) { - perror(path); + ret = read_file(path, buf, sizeof(buf)); + if (ret) { + print_file_access_error(path, ret); exit(EXIT_FAILURE); } @@ -103,12 +110,17 @@ void thp_write_string(const char *name, const char *val) printf("%s: Pathname is too long\n", __func__); exit(EXIT_FAILURE); } - write_file(path, val, strlen(val) + 1); + ret = write_file(path, val, strlen(val) + 1); + if (ret) { + print_file_access_error(path, ret); + exit(EXIT_FAILURE); + } } unsigned long thp_read_num(const char *name) { char path[PATH_MAX]; + unsigned long num; int ret; ret = snprintf(path, PATH_MAX, THP_SYSFS "%s", name); @@ -116,7 +128,13 @@ unsigned long thp_read_num(const char *name) printf("%s: Pathname is too long\n", __func__); exit(EXIT_FAILURE); } - return read_num(path); + ret = read_num(path, &num); + if (ret) { + print_file_access_error(path, ret); + exit(EXIT_FAILURE); + } + + return num; } void thp_write_num(const char *name, unsigned long num) @@ -129,7 +147,11 @@ void thp_write_num(const char *name, unsigned long num) printf("%s: Pathname is too long\n", __func__); exit(EXIT_FAILURE); } - write_num(path, num); + ret = write_num(path, num); + if (ret) { + print_file_access_error(path, ret); + exit(EXIT_FAILURE); + } } void thp_read_settings(struct thp_settings *settings) @@ -157,8 +179,15 @@ void thp_read_settings(struct thp_settings *settings) .max_ptes_shared = thp_read_num("khugepaged/max_ptes_shared"), .pages_to_scan = thp_read_num("khugepaged/pages_to_scan"), }; - if (dev_queue_read_ahead_path[0]) - settings->read_ahead_kb = read_num(dev_queue_read_ahead_path); + if (dev_queue_read_ahead_path[0]) { + int ret = read_num(dev_queue_read_ahead_path, + &settings->read_ahead_kb); + + if (ret) { + print_file_access_error(dev_queue_read_ahead_path, ret); + exit(EXIT_FAILURE); + } + } for (i = 0; i < NR_ORDERS; i++) { if (!((1 << i) & orders)) { @@ -208,8 +237,15 @@ void thp_write_settings(struct thp_settings *settings) thp_write_num("khugepaged/max_ptes_shared", khugepaged->max_ptes_shared); thp_write_num("khugepaged/pages_to_scan", khugepaged->pages_to_scan); - if (dev_queue_read_ahead_path[0]) - write_num(dev_queue_read_ahead_path, settings->read_ahead_kb); + if (dev_queue_read_ahead_path[0]) { + int ret = write_num(dev_queue_read_ahead_path, + settings->read_ahead_kb); + + if (ret) { + print_file_access_error(dev_queue_read_ahead_path, ret); + exit(EXIT_FAILURE); + } + } for (i = 0; i < NR_ORDERS; i++) { if (!((1 << i) & orders)) @@ -307,8 +343,15 @@ static unsigned long __thp_supported_orders(bool is_shmem) } ret = read_file(path, buf, sizeof(buf)); - if (ret) - orders |= 1UL << i; + if (ret) { + if (ret != -ENOENT) { + print_file_access_error(path, ret); + exit(EXIT_FAILURE); + } + continue; + } + + orders |= 1UL << i; } return orders; @@ -382,8 +425,7 @@ int detect_hugetlb_page_sizes(unsigned long sizes[], int max) if (sscanf(entry->d_name, "hugepages-%zukB", &kb) != 1) continue; sizes[count++] = kb * 1024; - ksft_print_msg("[INFO] detected hugetlb page size: %zu KiB\n", - kb); + printf("# [INFO] detected hugetlb page size: %zu KiB\n", kb); } closedir(dir); return count; @@ -425,28 +467,49 @@ static void hugetlb_sysfs_path(char *buf, size_t buflen, unsigned long hugetlb_nr_pages(unsigned long size) { char path[PATH_MAX]; + unsigned long nr; + int ret; hugetlb_sysfs_path(path, sizeof(path), size, "nr_hugepages"); - return read_num(path); + ret = read_num(path, &nr); + if (ret) { + print_file_access_error(path, ret); + exit(EXIT_FAILURE); + } + + return nr; } void hugetlb_set_nr_pages(unsigned long size, unsigned long nr) { char path[PATH_MAX]; + int ret; hugetlb_sysfs_path(path, sizeof(path), size, "nr_hugepages"); - write_num_ignore_einval(path, nr); + ret = write_num_ignore_einval(path, nr); + if (ret) { + print_file_access_error(path, ret); + exit(EXIT_FAILURE); + } } unsigned long hugetlb_free_pages(unsigned long size) { char path[PATH_MAX]; + unsigned long nr; + int ret; hugetlb_sysfs_path(path, sizeof(path), size, "free_hugepages"); - return read_num(path); + ret = read_num(path, &nr); + if (ret) { + print_file_access_error(path, ret); + exit(EXIT_FAILURE); + } + + return nr; } unsigned long hugetlb_nr_resv_pages(unsigned long size) @@ -511,7 +574,8 @@ unsigned long hugetlb_setup(unsigned long nr, unsigned long sizes[], return 0; if (nr_enabled > max) { - ksft_print_msg("detected %d huge page sizes, will only test %d\n", nr_enabled, max); + printf("# detected %d huge page sizes, will only test %d\n", + nr_enabled, max); nr_enabled = max; } diff --git a/tools/testing/selftests/mm/hugetlb-soft-offline.c b/tools/testing/selftests/mm/hugetlb-soft-offline.c index 4af9d3db7b5b6f..ffc85b958c6920 100644 --- a/tools/testing/selftests/mm/hugetlb-soft-offline.c +++ b/tools/testing/selftests/mm/hugetlb-soft-offline.c @@ -85,8 +85,7 @@ static unsigned long orig_enable_soft_offline = -1UL; /* * Runs from an atexit handler, so it must not call anything that - * exits on failure: write_num() would re-enter exit() through - * ksft_exit_fail_msg(). + * exits on failure. */ static void restore_enable_soft_offline(void) { @@ -152,7 +151,10 @@ static void test_soft_offline_common(int enable_soft_offline) hugepagesize_kb = file_stat.f_bsize / 1024; ksft_print_msg("Hugepagesize is %ldkB\n", hugepagesize_kb); - write_num(ENABLE_SOFT_OFFLINE_PATH, enable_soft_offline); + ret = write_num(ENABLE_SOFT_OFFLINE_PATH, enable_soft_offline); + if (ret) + ksft_exit_fail_msg("Failed to write to %s: %s\n", + ENABLE_SOFT_OFFLINE_PATH, strerror(-ret)); nr_hugepages_before = hugetlb_nr_default_pages(); @@ -189,6 +191,8 @@ static void test_soft_offline_common(int enable_soft_offline) int main(int argc, char **argv) { + int ret; + ksft_print_header(); if (!hugetlb_setup_default(8)) @@ -196,7 +200,11 @@ int main(int argc, char **argv) ksft_set_plan(2); - orig_enable_soft_offline = read_num(ENABLE_SOFT_OFFLINE_PATH); + ret = read_num(ENABLE_SOFT_OFFLINE_PATH, &orig_enable_soft_offline); + if (ret) + ksft_exit_fail_msg("Failed to read %s: %s\n", + ENABLE_SOFT_OFFLINE_PATH, strerror(-ret)); + atexit(restore_enable_soft_offline); test_soft_offline_common(1); diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index f82673f5f6b47e..6daa22f6da2f3f 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -122,6 +122,7 @@ static void get_finfo(const char *dir) char buf[1 << 10]; char path[PATH_MAX]; char *str, *end; + int ret; finfo.dir = dir; if (stat(finfo.dir, &path_stat)) @@ -142,8 +143,9 @@ static void get_finfo(const char *dir) major(path_stat.st_dev), minor(path_stat.st_dev)) >= sizeof(path)) ksft_exit_fail_msg("%s: Pathname is too long\n", __func__); - if (!read_file(path, buf, sizeof(buf))) - ksft_exit_fail_perror("read_file(uevent)"); + ret = read_file(path, buf, sizeof(buf)); + if (ret) + ksft_exit_fail_msg("read_file(%s): %s\n", path, strerror(-ret)); if (strstr(buf, "DEVTYPE=disk")) { /* Found it */ if (snprintf(finfo.dev_queue_read_ahead_path, @@ -324,7 +326,7 @@ static void *file_setup_area_common(int nr_hpages, enum file_setup_ops setup) { const int open_opt = setup == FILE_SETUP_READ_ONLY_FS ? O_RDONLY : O_RDWR; const int mmap_prot = setup == FILE_SETUP_READ_ONLY_FS ? PROT_READ : (PROT_READ | PROT_WRITE); - int fd; + int fd, ret; void *p; unsigned long size; @@ -362,7 +364,11 @@ static void *file_setup_area_common(int nr_hpages, enum file_setup_ops setup) ksft_exit_fail_perror("mmap()"); /* Drop page cache */ - write_file("/proc/sys/vm/drop_caches", "3", 2); + ret = write_file("/proc/sys/vm/drop_caches", "3", 2); + if (ret) + ksft_exit_fail_msg("write_file(drop_caches): %s\n", + strerror(-ret)); + success("OK"); return p; } diff --git a/tools/testing/selftests/mm/split_huge_page_test.c b/tools/testing/selftests/mm/split_huge_page_test.c index c01d227d7fd6dd..a30927514b4f6c 100644 --- a/tools/testing/selftests/mm/split_huge_page_test.c +++ b/tools/testing/selftests/mm/split_huge_page_test.c @@ -145,7 +145,10 @@ static void write_debugfs(const char *fmt, ...) if (ret >= INPUT_MAX) ksft_exit_fail_msg("%s: Debugfs input is too long\n", __func__); - write_file(SPLIT_DEBUGFS, input, ret + 1); + ret = write_file(SPLIT_DEBUGFS, input, ret + 1); + if (ret) + ksft_exit_fail_msg("write_file(%s): %s\n", SPLIT_DEBUGFS, + strerror(-ret)); } static char *allocate_zero_filled_hugepage(size_t len) diff --git a/tools/testing/selftests/mm/vm_util.c b/tools/testing/selftests/mm/vm_util.c index 4821a356303631..d04c904e76c35e 100644 --- a/tools/testing/selftests/mm/vm_util.c +++ b/tools/testing/selftests/mm/vm_util.c @@ -887,109 +887,149 @@ int unpoison_memory(unsigned long pfn) int read_file(const char *path, char *buf, size_t buflen) { - int fd; + int fd, err; ssize_t numread; fd = open(path, O_RDONLY); if (fd == -1) - return 0; + return -errno; numread = read(fd, buf, buflen - 1); if (numread < 1) { + err = numread ? errno : ENODATA; close(fd); - return 0; + return -err; } buf[numread] = '\0'; close(fd); - return (unsigned int) numread; + return 0; } -static void __write_file(const char *path, const char *buf, size_t buflen, bool ignore_einval) +int write_file(const char *path, const char *buf, size_t buflen) { int fd, saved_errno; ssize_t numwritten; if (buflen < 2) - ksft_exit_fail_msg("Incorrect buffer len: %zu\n", buflen); + return -EINVAL; fd = open(path, O_WRONLY); if (fd == -1) - ksft_exit_fail_msg("%s open failed: %s\n", path, strerror(errno)); + return -errno; numwritten = write(fd, buf, buflen - 1); saved_errno = errno; close(fd); - errno = saved_errno; - if (numwritten < 0) { - if (ignore_einval && errno == EINVAL) - return; - ksft_exit_fail_msg("%s write(%.*s) failed: %s\n", path, (int)(buflen - 1), - buf, strerror(errno)); - } - if (numwritten != buflen - 1) - ksft_exit_fail_msg("%s write(%.*s) is truncated, expected %zu bytes, got %zd bytes\n", - path, (int)(buflen - 1), buf, buflen - 1, numwritten); -} -void write_file(const char *path, const char *buf, size_t buflen) -{ - __write_file(path, buf, buflen, /* ignore_einval = */ false); + if (numwritten < 0) + return -saved_errno; + + if (numwritten != (ssize_t)(buflen - 1)) + return -EIO; + + return 0; } -unsigned long read_num(const char *path) +int read_num(const char *path, unsigned long *num) { + unsigned long val; + int ret; char buf[21]; + char *end; - if (!read_file(path, buf, sizeof(buf))) - ksft_exit_fail_perror("read_file()"); + if (!num) + return -EINVAL; - return strtoul(buf, NULL, 10); + ret = read_file(path, buf, sizeof(buf)); + if (ret) + return ret; + + /* Reject signs and leading whitespace that are accepted by strtoul() */ + if (buf[0] < '0' || buf[0] > '9') + return -EINVAL; + + errno = 0; + val = strtoul(buf, &end, 10); + if (errno) + return -errno; + + /* Only allow a newline after the number */ + if (*end == '\n') + end++; + + if (*end != '\0') + return -EINVAL; + + *num = val; + return 0; } -static void __write_num(const char *path, unsigned long num, bool ignore_einval) +int write_num(const char *path, unsigned long num) { char buf[21]; sprintf(buf, "%lu", num); - __write_file(path, buf, strlen(buf) + 1, ignore_einval); + return write_file(path, buf, strlen(buf) + 1); } -void write_num(const char *path, unsigned long num) +int write_num_ignore_einval(const char *path, unsigned long num) { - return __write_num(path, num, /* ignore_einval = */ false); -} + int ret; -void write_num_ignore_einval(const char *path, unsigned long num) -{ - return __write_num(path, num, /* ignore_einval = */ true); + ret = write_num(path, num); + return ret == -EINVAL ? 0 : ret; } static unsigned long shmall, shmmax; void __shm_limits_restore(void) { - if (shmmax) - write_num("/proc/sys/kernel/shmmax", shmmax); - if (shmall) - write_num("/proc/sys/kernel/shmall", shmall); + int ret; + + if (shmmax) { + ret = write_num("/proc/sys/kernel/shmmax", shmmax); + if (ret < 0) + ksft_exit_fail_msg("Failed to restore shmmax: %s\n", + strerror(-ret)); + } + if (shmall) { + ret = write_num("/proc/sys/kernel/shmall", shmall); + if (ret < 0) + ksft_exit_fail_msg("Failed to restore shmall: %s\n", + strerror(-ret)); + } } void shm_limits_prepare(unsigned long length) { unsigned long nr = length / psize(); unsigned long val; + int ret; + + ret = read_num("/proc/sys/kernel/shmmax", &val); + if (ret < 0) + ksft_exit_fail_msg("Failed to read /proc/sys/kernel/shmmax: %s\n", + strerror(-ret)); - val = read_num("/proc/sys/kernel/shmmax"); if (val < length) { - write_num("/proc/sys/kernel/shmmax", length); + ret = write_num("/proc/sys/kernel/shmmax", length); + if (ret < 0) + ksft_exit_fail_msg("Failed to write %lu to /proc/sys/kernel/shmmax: %s\n", + length, strerror(-ret)); shmmax = val; } - val = read_num("/proc/sys/kernel/shmall"); + ret = read_num("/proc/sys/kernel/shmall", &val); + if (ret < 0) + ksft_exit_fail_msg("Failed to read /proc/sys/kernel/shmall: %s\n", + strerror(-ret)); if (val < nr) { - write_num("/proc/sys/kernel/shmall", nr); + ret = write_num("/proc/sys/kernel/shmall", nr); + if (ret < 0) + ksft_exit_fail_msg("Failed to write %lu to /proc/sys/kernel/shmall: %s\n", + nr, strerror(-ret)); shmall = val; } } diff --git a/tools/testing/selftests/mm/vm_util.h b/tools/testing/selftests/mm/vm_util.h index 9a49af88702e4c..62f6f5b4264924 100644 --- a/tools/testing/selftests/mm/vm_util.h +++ b/tools/testing/selftests/mm/vm_util.h @@ -166,11 +166,11 @@ int unpoison_memory(unsigned long pfn); #define PAGEMAP_PRESENT(ent) (((ent) & (1ull << 63)) != 0) #define PAGEMAP_PFN(ent) ((ent) & ((1ull << 55) - 1)) -void write_file(const char *path, const char *buf, size_t buflen); +int write_file(const char *path, const char *buf, size_t buflen); int read_file(const char *path, char *buf, size_t buflen); -unsigned long read_num(const char *path); -void write_num(const char *path, unsigned long num); -void write_num_ignore_einval(const char *path, unsigned long num); +int read_num(const char *path, unsigned long *num); +int write_num(const char *path, unsigned long num); +int write_num_ignore_einval(const char *path, unsigned long num); void shm_limits_prepare(unsigned long length); void __shm_limits_restore(void); From d30f0158542492e2b4e75999710508414f622642 Mon Sep 17 00:00:00 2001 From: Sarthak Sharma Date: Fri, 18 Sep 2026 17:40:50 +0530 Subject: [PATCH 1095/1352] selftests-mm-make-file-helpers-return-errors-fix Convert hugetlb_nr_resv_pages(), which was missed when read_num() changed to return an error and store the parsed value through an output pointer. Link: https://lore.kernel.org/937939c3-ae9a-4148-a601-0f8876216423@arm.com Signed-off-by: Sarthak Sharma Signed-off-by: Andrew Morton Cc: Anshuman Khandual Cc: Baolin Wang Cc: Barry Song Cc: David Hildenbrand (Arm) Cc: Dev Jain Cc: Jason Gunthorpe Cc: John Hubbard Cc: Jonathan Corbet Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Mark Brown Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Muhammad Usama Anjum Cc: Nico Pache Cc: Peter Xu Cc: Ryan Roberts Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Zi Yan --- tools/testing/selftests/mm/hugepage_settings.c | 10 +++++++++- 1 file changed, 9 insertions(+), 1 deletion(-) diff --git a/tools/testing/selftests/mm/hugepage_settings.c b/tools/testing/selftests/mm/hugepage_settings.c index 9a63420d0744fb..6f3abd35738592 100644 --- a/tools/testing/selftests/mm/hugepage_settings.c +++ b/tools/testing/selftests/mm/hugepage_settings.c @@ -515,10 +515,18 @@ unsigned long hugetlb_free_pages(unsigned long size) unsigned long hugetlb_nr_resv_pages(unsigned long size) { char path[PATH_MAX]; + unsigned long nr; + int ret; hugetlb_sysfs_path(path, sizeof(path), size, "resv_hugepages"); - return read_num(path); + ret = read_num(path, &nr); + if (ret) { + print_file_access_error(path, ret); + exit(EXIT_FAILURE); + } + + return nr; } static bool __hugetlb_setup(unsigned long size, unsigned long nr) From 120f7c712d53d138726d55dbc712255fe5676814 Mon Sep 17 00:00:00 2001 From: Sarthak Sharma Date: Fri, 18 Sep 2026 16:52:30 +0530 Subject: [PATCH 1096/1352] tools/lib/mm: add shared file helpers Move read_file(), write_file(), read_num(), write_num() and write_num_ignore_einval() out of tools/testing/selftests/mm/vm_util.c into a new shared helper under tools/lib/mm/. These helpers are used by mm selftests today and will also be needed by shared hugepage helpers in subsequent patches. Move them to a generic location so they can be reused outside selftests as well. Keep the helpers exposed to mm selftests through vm_util.h by including the new shared header there, and link the new helper into the selftests/mm build. Update the explicit x86 protection_keys 32-bit and 64-bit build rules to preserve prerequisite paths, now that file_utils.c is built from tools/lib/mm. Add tools/lib/mm/ to the MEMORY MANAGEMENT - MISC entry in MAINTAINERS. Link: https://lore.kernel.org/20260918112234.195857-3-sarthak.sharma@arm.com Signed-off-by: Sarthak Sharma Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Tested-by: Muhammad Usama Anjum Cc: Anshuman Khandual Cc: Baolin Wang Cc: Barry Song Cc: Dev Jain Cc: Jason Gunthorpe Cc: John Hubbard Cc: Jonathan Corbet Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Mark Brown Cc: Michal Hocko Cc: Nico Pache Cc: Peter Xu Cc: Ryan Roberts Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Zi Yan --- MAINTAINERS | 1 + tools/lib/mm/file_utils.c | 106 +++++++++++++++++++++++++++ tools/lib/mm/file_utils.h | 13 ++++ tools/testing/selftests/mm/Makefile | 11 +-- tools/testing/selftests/mm/vm_util.c | 97 ------------------------ tools/testing/selftests/mm/vm_util.h | 7 +- 6 files changed, 127 insertions(+), 108 deletions(-) create mode 100644 tools/lib/mm/file_utils.c create mode 100644 tools/lib/mm/file_utils.h diff --git a/MAINTAINERS b/MAINTAINERS index 4689a021006002..dfd9f948390708 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -17289,6 +17289,7 @@ F: mm/mapping_dirty_helpers.c F: mm/page_idle.c F: mm/pgalloc-track.h F: mm/process_vm_access.c +F: tools/lib/mm/ F: tools/testing/selftests/mm/ MEMORY MANAGEMENT - NUMA MEMBLOCKS AND NUMA EMULATION diff --git a/tools/lib/mm/file_utils.c b/tools/lib/mm/file_utils.c new file mode 100644 index 00000000000000..9b2237e9823e98 --- /dev/null +++ b/tools/lib/mm/file_utils.c @@ -0,0 +1,106 @@ +// SPDX-License-Identifier: GPL-2.0 +#include +#include +#include +#include +#include +#include + +#include "file_utils.h" + +int read_file(const char *path, char *buf, size_t buflen) +{ + int fd, err; + ssize_t numread; + + fd = open(path, O_RDONLY); + if (fd == -1) + return -errno; + + numread = read(fd, buf, buflen - 1); + if (numread < 1) { + err = numread ? errno : ENODATA; + close(fd); + return -err; + } + + buf[numread] = '\0'; + close(fd); + + return 0; +} + +int write_file(const char *path, const char *buf, size_t buflen) +{ + int fd, saved_errno; + ssize_t numwritten; + + if (buflen < 2) + return -EINVAL; + + fd = open(path, O_WRONLY); + if (fd == -1) + return -errno; + + numwritten = write(fd, buf, buflen - 1); + saved_errno = errno; + close(fd); + + if (numwritten < 0) + return -saved_errno; + + if (numwritten != (ssize_t)(buflen - 1)) + return -EIO; + + return 0; +} + +int read_num(const char *path, unsigned long *num) +{ + unsigned long val; + int ret; + char buf[21]; + char *end; + + if (!num) + return -EINVAL; + + ret = read_file(path, buf, sizeof(buf)); + if (ret) + return ret; + + /* Reject signs and leading whitespace that are accepted by strtoul() */ + if (buf[0] < '0' || buf[0] > '9') + return -EINVAL; + + errno = 0; + val = strtoul(buf, &end, 10); + if (errno) + return -errno; + + /* Only allow a newline after the number */ + if (*end == '\n') + end++; + + if (*end != '\0') + return -EINVAL; + + *num = val; + return 0; +} + +int write_num(const char *path, unsigned long num) +{ + char buf[21]; + + sprintf(buf, "%lu", num); + return write_file(path, buf, strlen(buf) + 1); +} + +int write_num_ignore_einval(const char *path, unsigned long num) +{ + int ret; + + ret = write_num(path, num); + return ret == -EINVAL ? 0 : ret; +} diff --git a/tools/lib/mm/file_utils.h b/tools/lib/mm/file_utils.h new file mode 100644 index 00000000000000..50daa82c2b2b48 --- /dev/null +++ b/tools/lib/mm/file_utils.h @@ -0,0 +1,13 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +#ifndef __MM_FILE_UTILS_H__ +#define __MM_FILE_UTILS_H__ + +#include + +int read_file(const char *path, char *buf, size_t buflen); +int write_file(const char *path, const char *buf, size_t buflen); +int read_num(const char *path, unsigned long *num); +int write_num(const char *path, unsigned long num); +int write_num_ignore_einval(const char *path, unsigned long num); + +#endif diff --git a/tools/testing/selftests/mm/Makefile b/tools/testing/selftests/mm/Makefile index d3e9bd67904aa0..c36a9a7089bb26 100644 --- a/tools/testing/selftests/mm/Makefile +++ b/tools/testing/selftests/mm/Makefile @@ -37,7 +37,8 @@ endif # LDLIBS. MAKEFLAGS += --no-builtin-rules -CFLAGS = -Wall -O2 -I $(top_srcdir) $(EXTRA_CFLAGS) $(KHDR_INCLUDES) $(TOOLS_INCLUDES) +CFLAGS = -Wall -O2 -I $(top_srcdir) -I $(top_srcdir)/tools/lib +CFLAGS += $(EXTRA_CFLAGS) $(KHDR_INCLUDES) $(TOOLS_INCLUDES) CFLAGS += -Wunreachable-code LDLIBS = -lrt -lpthread -lm @@ -184,8 +185,8 @@ TEST_FILES += write_hugetlb_memory.sh include ../lib.mk -$(TEST_GEN_PROGS): vm_util.c hugepage_settings.c -$(TEST_GEN_FILES): vm_util.c hugepage_settings.c +$(TEST_GEN_PROGS): vm_util.c hugepage_settings.c $(top_srcdir)/tools/lib/mm/file_utils.c +$(TEST_GEN_FILES): vm_util.c hugepage_settings.c $(top_srcdir)/tools/lib/mm/file_utils.c $(OUTPUT)/uffd-stress: uffd-common.c $(OUTPUT)/uffd-unit-tests: uffd-common.c @@ -214,7 +215,7 @@ $(BINARIES_32): CFLAGS += -m32 -mxsave $(BINARIES_32): LDLIBS += -lrt -ldl -lm $(BINARIES_32): $(OUTPUT)/%_32: %.c $(call msg,CC,,$@) - $(Q)$(CC) $(CFLAGS) $(EXTRA_CFLAGS) $(notdir $^) $(LDLIBS) -o $@ + $(Q)$(CC) $(CFLAGS) $(EXTRA_CFLAGS) $^ $(LDLIBS) -o $@ $(foreach t,$(VMTARGETS),$(eval $(call gen-target-rule-32,$(t)))) endif @@ -223,7 +224,7 @@ $(BINARIES_64): CFLAGS += -m64 -mxsave $(BINARIES_64): LDLIBS += -lrt -ldl $(BINARIES_64): $(OUTPUT)/%_64: %.c $(call msg,CC,,$@) - $(Q)$(CC) $(CFLAGS) $(EXTRA_CFLAGS) $(notdir $^) $(LDLIBS) -o $@ + $(Q)$(CC) $(CFLAGS) $(EXTRA_CFLAGS) $^ $(LDLIBS) -o $@ $(foreach t,$(VMTARGETS),$(eval $(call gen-target-rule-64,$(t)))) endif diff --git a/tools/testing/selftests/mm/vm_util.c b/tools/testing/selftests/mm/vm_util.c index d04c904e76c35e..f8916b2abc2efb 100644 --- a/tools/testing/selftests/mm/vm_util.c +++ b/tools/testing/selftests/mm/vm_util.c @@ -885,103 +885,6 @@ int unpoison_memory(unsigned long pfn) return ret > 0 ? 0 : -errno; } -int read_file(const char *path, char *buf, size_t buflen) -{ - int fd, err; - ssize_t numread; - - fd = open(path, O_RDONLY); - if (fd == -1) - return -errno; - - numread = read(fd, buf, buflen - 1); - if (numread < 1) { - err = numread ? errno : ENODATA; - close(fd); - return -err; - } - - buf[numread] = '\0'; - close(fd); - - return 0; -} - -int write_file(const char *path, const char *buf, size_t buflen) -{ - int fd, saved_errno; - ssize_t numwritten; - - if (buflen < 2) - return -EINVAL; - - fd = open(path, O_WRONLY); - if (fd == -1) - return -errno; - - numwritten = write(fd, buf, buflen - 1); - saved_errno = errno; - close(fd); - - if (numwritten < 0) - return -saved_errno; - - if (numwritten != (ssize_t)(buflen - 1)) - return -EIO; - - return 0; -} - -int read_num(const char *path, unsigned long *num) -{ - unsigned long val; - int ret; - char buf[21]; - char *end; - - if (!num) - return -EINVAL; - - ret = read_file(path, buf, sizeof(buf)); - if (ret) - return ret; - - /* Reject signs and leading whitespace that are accepted by strtoul() */ - if (buf[0] < '0' || buf[0] > '9') - return -EINVAL; - - errno = 0; - val = strtoul(buf, &end, 10); - if (errno) - return -errno; - - /* Only allow a newline after the number */ - if (*end == '\n') - end++; - - if (*end != '\0') - return -EINVAL; - - *num = val; - return 0; -} - -int write_num(const char *path, unsigned long num) -{ - char buf[21]; - - sprintf(buf, "%lu", num); - return write_file(path, buf, strlen(buf) + 1); -} - -int write_num_ignore_einval(const char *path, unsigned long num) -{ - int ret; - - ret = write_num(path, num); - return ret == -EINVAL ? 0 : ret; -} - static unsigned long shmall, shmmax; void __shm_limits_restore(void) diff --git a/tools/testing/selftests/mm/vm_util.h b/tools/testing/selftests/mm/vm_util.h index 62f6f5b4264924..fe0475f2bdf288 100644 --- a/tools/testing/selftests/mm/vm_util.h +++ b/tools/testing/selftests/mm/vm_util.h @@ -8,6 +8,7 @@ #include /* _SC_PAGESIZE */ #include "kselftest.h" #include +#include #define BIT_ULL(nr) (1ULL << (nr)) #define PM_SOFT_DIRTY BIT_ULL(55) @@ -166,12 +167,6 @@ int unpoison_memory(unsigned long pfn); #define PAGEMAP_PRESENT(ent) (((ent) & (1ull << 63)) != 0) #define PAGEMAP_PFN(ent) ((ent) & ((1ull << 55) - 1)) -int write_file(const char *path, const char *buf, size_t buflen); -int read_file(const char *path, char *buf, size_t buflen); -int read_num(const char *path, unsigned long *num); -int write_num(const char *path, unsigned long num); -int write_num_ignore_einval(const char *path, unsigned long num); - void shm_limits_prepare(unsigned long length); void __shm_limits_restore(void); From 284b582109b375281d2982e387db2c06b008efcc Mon Sep 17 00:00:00 2001 From: Sarthak Sharma Date: Fri, 18 Sep 2026 16:52:31 +0530 Subject: [PATCH 1097/1352] tools/lib/mm: move hugepage_settings out of selftests Move hugepage_settings.[ch] from tools/testing/selftests/mm/ to tools/lib/mm/ so the THP and HugeTLB helpers can be shared more easily between selftests and other tools. Keep the helpers exposed to mm selftests through vm_util.h where possible, and use direct includes for files that do not include vm_util.h. Adjust the selftests/mm build to compile the moved implementation from its new location. Remove the remaining kselftest dependency by including file_utils.h directly and using EXIT_FAILURE in the signal handler. Link: https://lore.kernel.org/20260918112234.195857-4-sarthak.sharma@arm.com Signed-off-by: Sarthak Sharma Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Tested-by: Muhammad Usama Anjum Cc: Anshuman Khandual Cc: Baolin Wang Cc: Barry Song Cc: Dev Jain Cc: Jason Gunthorpe Cc: John Hubbard Cc: Jonathan Corbet Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Mark Brown Cc: Michal Hocko Cc: Nico Pache Cc: Peter Xu Cc: Ryan Roberts Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Zi Yan --- .../{testing/selftests => lib}/mm/hugepage_settings.c | 11 +++++++++-- .../{testing/selftests => lib}/mm/hugepage_settings.h | 0 tools/testing/selftests/mm/Makefile | 6 ++++-- tools/testing/selftests/mm/compaction_test.c | 2 +- tools/testing/selftests/mm/cow.c | 1 - tools/testing/selftests/mm/folio_split_race_test.c | 1 - tools/testing/selftests/mm/guard-regions.c | 1 - tools/testing/selftests/mm/gup_longterm.c | 1 - tools/testing/selftests/mm/gup_test.c | 1 - tools/testing/selftests/mm/hmm-tests.c | 6 +++--- tools/testing/selftests/mm/hugetlb-madvise.c | 1 - tools/testing/selftests/mm/hugetlb-mmap.c | 1 - tools/testing/selftests/mm/hugetlb-mremap.c | 1 - tools/testing/selftests/mm/hugetlb-shm.c | 1 - tools/testing/selftests/mm/hugetlb-soft-offline.c | 2 +- tools/testing/selftests/mm/hugetlb_dio.c | 1 - tools/testing/selftests/mm/hugetlb_fault_after_madv.c | 1 - tools/testing/selftests/mm/hugetlb_madv_vs_map.c | 1 - tools/testing/selftests/mm/khugepaged.c | 1 - tools/testing/selftests/mm/ksm_tests.c | 1 - tools/testing/selftests/mm/migration.c | 2 +- tools/testing/selftests/mm/pagemap_ioctl.c | 1 - tools/testing/selftests/mm/prctl_thp_disable.c | 1 - tools/testing/selftests/mm/protection_keys.c | 2 +- tools/testing/selftests/mm/soft-dirty.c | 1 - tools/testing/selftests/mm/split_huge_page_test.c | 1 - tools/testing/selftests/mm/thuge-gen.c | 1 - tools/testing/selftests/mm/transhuge-stress.c | 1 - tools/testing/selftests/mm/uffd-common.h | 1 - tools/testing/selftests/mm/uffd-wp-mremap.c | 2 +- tools/testing/selftests/mm/va_high_addr_switch.c | 1 - tools/testing/selftests/mm/vm_util.h | 1 + 32 files changed, 22 insertions(+), 34 deletions(-) rename tools/{testing/selftests => lib}/mm/hugepage_settings.c (99%) rename tools/{testing/selftests => lib}/mm/hugepage_settings.h (100%) diff --git a/tools/testing/selftests/mm/hugepage_settings.c b/tools/lib/mm/hugepage_settings.c similarity index 99% rename from tools/testing/selftests/mm/hugepage_settings.c rename to tools/lib/mm/hugepage_settings.c index 6f3abd35738592..656442c8d3954a 100644 --- a/tools/testing/selftests/mm/hugepage_settings.c +++ b/tools/lib/mm/hugepage_settings.c @@ -10,11 +10,16 @@ #include #include -#include "vm_util.h" +#include "file_utils.h" #include "hugepage_settings.h" #define THP_SYSFS "/sys/kernel/mm/transparent_hugepage/" #define MAX_SETTINGS_DEPTH 4 + +#ifndef ARRAY_SIZE +#define ARRAY_SIZE(arr) (sizeof(arr) / sizeof((arr)[0])) +#endif + static struct thp_settings settings_stack[MAX_SETTINGS_DEPTH]; static int settings_index; static struct thp_settings saved_settings; @@ -655,8 +660,10 @@ static void hugepage_restore_settings_atexit(void) static void hugepage_restore_settings_sighandler(int sig) { + (void)sig; + /* exit() will invoke the hugepage_restore_settings_atexit handler. */ - exit(KSFT_FAIL); + exit(EXIT_FAILURE); } void hugepage_save_settings(bool thp, bool hugetlb) diff --git a/tools/testing/selftests/mm/hugepage_settings.h b/tools/lib/mm/hugepage_settings.h similarity index 100% rename from tools/testing/selftests/mm/hugepage_settings.h rename to tools/lib/mm/hugepage_settings.h diff --git a/tools/testing/selftests/mm/Makefile b/tools/testing/selftests/mm/Makefile index c36a9a7089bb26..67882e52d4ff0a 100644 --- a/tools/testing/selftests/mm/Makefile +++ b/tools/testing/selftests/mm/Makefile @@ -185,8 +185,10 @@ TEST_FILES += write_hugetlb_memory.sh include ../lib.mk -$(TEST_GEN_PROGS): vm_util.c hugepage_settings.c $(top_srcdir)/tools/lib/mm/file_utils.c -$(TEST_GEN_FILES): vm_util.c hugepage_settings.c $(top_srcdir)/tools/lib/mm/file_utils.c +$(TEST_GEN_PROGS): vm_util.c $(top_srcdir)/tools/lib/mm/hugepage_settings.c \ + $(top_srcdir)/tools/lib/mm/file_utils.c +$(TEST_GEN_FILES): vm_util.c $(top_srcdir)/tools/lib/mm/hugepage_settings.c \ + $(top_srcdir)/tools/lib/mm/file_utils.c $(OUTPUT)/uffd-stress: uffd-common.c $(OUTPUT)/uffd-unit-tests: uffd-common.c diff --git a/tools/testing/selftests/mm/compaction_test.c b/tools/testing/selftests/mm/compaction_test.c index 30d4ace7155ae9..b3f5377119cb80 100644 --- a/tools/testing/selftests/mm/compaction_test.c +++ b/tools/testing/selftests/mm/compaction_test.c @@ -15,9 +15,9 @@ #include #include #include +#include #include "kselftest.h" -#include "hugepage_settings.h" #define MAP_SIZE_MB 100 #define MAP_SIZE (MAP_SIZE_MB * 1024 * 1024) diff --git a/tools/testing/selftests/mm/cow.c b/tools/testing/selftests/mm/cow.c index 8aa5249d9bef63..3264a828575bd4 100644 --- a/tools/testing/selftests/mm/cow.c +++ b/tools/testing/selftests/mm/cow.c @@ -29,7 +29,6 @@ #include "../../../../mm/gup_test.h" #include "kselftest.h" #include "vm_util.h" -#include "hugepage_settings.h" static size_t pagesize; static int pagemap_fd; diff --git a/tools/testing/selftests/mm/folio_split_race_test.c b/tools/testing/selftests/mm/folio_split_race_test.c index 1960635a953eb5..e4660bf89b624a 100644 --- a/tools/testing/selftests/mm/folio_split_race_test.c +++ b/tools/testing/selftests/mm/folio_split_race_test.c @@ -25,7 +25,6 @@ #include #include "vm_util.h" #include "kselftest.h" -#include "hugepage_settings.h" uint64_t page_size; uint64_t pmd_pagesize; diff --git a/tools/testing/selftests/mm/guard-regions.c b/tools/testing/selftests/mm/guard-regions.c index b724d62d2b7555..f7d53ea3c25270 100644 --- a/tools/testing/selftests/mm/guard-regions.c +++ b/tools/testing/selftests/mm/guard-regions.c @@ -21,7 +21,6 @@ #include #include #include "vm_util.h" -#include "hugepage_settings.h" #include "../pidfd/pidfd.h" diff --git a/tools/testing/selftests/mm/gup_longterm.c b/tools/testing/selftests/mm/gup_longterm.c index 510de93be6814f..c9d8b449126388 100644 --- a/tools/testing/selftests/mm/gup_longterm.c +++ b/tools/testing/selftests/mm/gup_longterm.c @@ -29,7 +29,6 @@ #include "../../../../mm/gup_test.h" #include "kselftest.h" #include "vm_util.h" -#include "hugepage_settings.h" static size_t pagesize; static int nr_hugetlbsizes; diff --git a/tools/testing/selftests/mm/gup_test.c b/tools/testing/selftests/mm/gup_test.c index 3f841a96f87068..5f44761dbec0be 100644 --- a/tools/testing/selftests/mm/gup_test.c +++ b/tools/testing/selftests/mm/gup_test.c @@ -14,7 +14,6 @@ #include #include "kselftest.h" #include "vm_util.h" -#include "hugepage_settings.h" #define MB (1UL << 20) diff --git a/tools/testing/selftests/mm/hmm-tests.c b/tools/testing/selftests/mm/hmm-tests.c index e2642eca0d02b4..fa1a651963fd01 100644 --- a/tools/testing/selftests/mm/hmm-tests.c +++ b/tools/testing/selftests/mm/hmm-tests.c @@ -10,9 +10,6 @@ * bugs. */ -#include "kselftest_harness.h" -#include "hugepage_settings.h" - #include #include #include @@ -33,6 +30,9 @@ #include #include #include +#include + +#include "kselftest_harness.h" /* * This is a private UAPI to the kernel test module so it isn't exported diff --git a/tools/testing/selftests/mm/hugetlb-madvise.c b/tools/testing/selftests/mm/hugetlb-madvise.c index 555b4b3d14307e..57cf790ca478d4 100644 --- a/tools/testing/selftests/mm/hugetlb-madvise.c +++ b/tools/testing/selftests/mm/hugetlb-madvise.c @@ -14,7 +14,6 @@ #include #include "vm_util.h" #include "kselftest.h" -#include "hugepage_settings.h" #define MIN_FREE_PAGES 20 #define NR_HUGE_PAGES 10 /* common number of pages to map/allocate */ diff --git a/tools/testing/selftests/mm/hugetlb-mmap.c b/tools/testing/selftests/mm/hugetlb-mmap.c index 2edbc992e6bcf0..8853271fc8259d 100644 --- a/tools/testing/selftests/mm/hugetlb-mmap.c +++ b/tools/testing/selftests/mm/hugetlb-mmap.c @@ -18,7 +18,6 @@ #include #include "vm_util.h" #include "kselftest.h" -#include "hugepage_settings.h" #define LENGTH (256UL*1024*1024) #define PROTECTION (PROT_READ | PROT_WRITE) diff --git a/tools/testing/selftests/mm/hugetlb-mremap.c b/tools/testing/selftests/mm/hugetlb-mremap.c index ed3d92e862d876..9b724af66e9388 100644 --- a/tools/testing/selftests/mm/hugetlb-mremap.c +++ b/tools/testing/selftests/mm/hugetlb-mremap.c @@ -26,7 +26,6 @@ #include #include "kselftest.h" #include "vm_util.h" -#include "hugepage_settings.h" #define DEFAULT_LENGTH_MB 10UL #define MB_TO_BYTES(x) (x * 1024 * 1024) diff --git a/tools/testing/selftests/mm/hugetlb-shm.c b/tools/testing/selftests/mm/hugetlb-shm.c index 3ff7f062b7eb45..f4514da49e1df7 100644 --- a/tools/testing/selftests/mm/hugetlb-shm.c +++ b/tools/testing/selftests/mm/hugetlb-shm.c @@ -29,7 +29,6 @@ #include #include "vm_util.h" -#include "hugepage_settings.h" #define LENGTH (256UL*1024*1024) diff --git a/tools/testing/selftests/mm/hugetlb-soft-offline.c b/tools/testing/selftests/mm/hugetlb-soft-offline.c index ffc85b958c6920..d9565219378aae 100644 --- a/tools/testing/selftests/mm/hugetlb-soft-offline.c +++ b/tools/testing/selftests/mm/hugetlb-soft-offline.c @@ -22,10 +22,10 @@ #include #include #include +#include #include "kselftest.h" #include "vm_util.h" -#include "hugepage_settings.h" #ifndef MADV_SOFT_OFFLINE #define MADV_SOFT_OFFLINE 101 diff --git a/tools/testing/selftests/mm/hugetlb_dio.c b/tools/testing/selftests/mm/hugetlb_dio.c index fb4600570e1319..9495974eccbea5 100644 --- a/tools/testing/selftests/mm/hugetlb_dio.c +++ b/tools/testing/selftests/mm/hugetlb_dio.c @@ -20,7 +20,6 @@ #include #include "vm_util.h" #include "kselftest.h" -#include "hugepage_settings.h" #ifndef STATX_DIOALIGN #define STATX_DIOALIGN 0x00002000U diff --git a/tools/testing/selftests/mm/hugetlb_fault_after_madv.c b/tools/testing/selftests/mm/hugetlb_fault_after_madv.c index 2dc158054f666b..56c5a8533e9d90 100644 --- a/tools/testing/selftests/mm/hugetlb_fault_after_madv.c +++ b/tools/testing/selftests/mm/hugetlb_fault_after_madv.c @@ -10,7 +10,6 @@ #include "vm_util.h" #include "kselftest.h" -#include "hugepage_settings.h" #define INLOOP_ITER 100 diff --git a/tools/testing/selftests/mm/hugetlb_madv_vs_map.c b/tools/testing/selftests/mm/hugetlb_madv_vs_map.c index 0f15eff1da0403..1d111f42dd5963 100644 --- a/tools/testing/selftests/mm/hugetlb_madv_vs_map.c +++ b/tools/testing/selftests/mm/hugetlb_madv_vs_map.c @@ -14,7 +14,6 @@ #include #include "vm_util.h" -#include "hugepage_settings.h" #define INLOOP_ITER 100 diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 6daa22f6da2f3f..525108cace54ab 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -22,7 +22,6 @@ #include "linux/magic.h" #include "vm_util.h" -#include "hugepage_settings.h" #define BASE_ADDR ((void *)(1UL << 30)) static unsigned long hpage_pmd_size; diff --git a/tools/testing/selftests/mm/ksm_tests.c b/tools/testing/selftests/mm/ksm_tests.c index 5fd7792a0d4794..6711f4c6137110 100644 --- a/tools/testing/selftests/mm/ksm_tests.c +++ b/tools/testing/selftests/mm/ksm_tests.c @@ -15,7 +15,6 @@ #include "kselftest.h" #include #include "vm_util.h" -#include "hugepage_settings.h" #define KSM_SYSFS_PATH "/sys/kernel/mm/ksm/" #define KSM_FP(s) (KSM_SYSFS_PATH s) diff --git a/tools/testing/selftests/mm/migration.c b/tools/testing/selftests/mm/migration.c index f19d53c6957644..a35e2b57e05b2d 100644 --- a/tools/testing/selftests/mm/migration.c +++ b/tools/testing/selftests/mm/migration.c @@ -5,7 +5,6 @@ */ #include "kselftest_harness.h" -#include "hugepage_settings.h" #include #include @@ -16,6 +15,7 @@ #include #include #include + #include "vm_util.h" #define TWOMEG (2<<20) diff --git a/tools/testing/selftests/mm/pagemap_ioctl.c b/tools/testing/selftests/mm/pagemap_ioctl.c index d9a4fb782ecfe7..03898b4f6cdab4 100644 --- a/tools/testing/selftests/mm/pagemap_ioctl.c +++ b/tools/testing/selftests/mm/pagemap_ioctl.c @@ -24,7 +24,6 @@ #include "vm_util.h" #include "kselftest.h" -#include "hugepage_settings.h" #define PAGEMAP_BITS_ALL (PAGE_IS_WPALLOWED | PAGE_IS_WRITTEN | \ PAGE_IS_FILE | PAGE_IS_PRESENT | \ diff --git a/tools/testing/selftests/mm/prctl_thp_disable.c b/tools/testing/selftests/mm/prctl_thp_disable.c index 82c6e96ea6eb37..f9ec1408a6e307 100644 --- a/tools/testing/selftests/mm/prctl_thp_disable.c +++ b/tools/testing/selftests/mm/prctl_thp_disable.c @@ -14,7 +14,6 @@ #include #include "kselftest_harness.h" -#include "hugepage_settings.h" #include "vm_util.h" #ifndef PR_THP_DISABLE_EXCEPT_ADVISED diff --git a/tools/testing/selftests/mm/protection_keys.c b/tools/testing/selftests/mm/protection_keys.c index ae6e1530b35484..b7882ab97683f2 100644 --- a/tools/testing/selftests/mm/protection_keys.c +++ b/tools/testing/selftests/mm/protection_keys.c @@ -45,8 +45,8 @@ #include #include #include +#include -#include "hugepage_settings.h" #include "pkey-helpers.h" u64 shadow_pkey_reg; diff --git a/tools/testing/selftests/mm/soft-dirty.c b/tools/testing/selftests/mm/soft-dirty.c index 5f278913c4d754..7f649b67335539 100644 --- a/tools/testing/selftests/mm/soft-dirty.c +++ b/tools/testing/selftests/mm/soft-dirty.c @@ -9,7 +9,6 @@ #include "kselftest.h" #include "vm_util.h" -#include "hugepage_settings.h" #define PAGEMAP_FILE_PATH "/proc/self/pagemap" #define TEST_ITERATIONS 10000 diff --git a/tools/testing/selftests/mm/split_huge_page_test.c b/tools/testing/selftests/mm/split_huge_page_test.c index a30927514b4f6c..68f508c9a355fc 100644 --- a/tools/testing/selftests/mm/split_huge_page_test.c +++ b/tools/testing/selftests/mm/split_huge_page_test.c @@ -21,7 +21,6 @@ #include #include "vm_util.h" #include "kselftest.h" -#include "hugepage_settings.h" uint64_t pagesize; unsigned int pageshift; diff --git a/tools/testing/selftests/mm/thuge-gen.c b/tools/testing/selftests/mm/thuge-gen.c index 50d0805b65db94..a04f588df780f1 100644 --- a/tools/testing/selftests/mm/thuge-gen.c +++ b/tools/testing/selftests/mm/thuge-gen.c @@ -14,7 +14,6 @@ #include #include "vm_util.h" #include "kselftest.h" -#include "hugepage_settings.h" #if !defined(MAP_HUGETLB) #define MAP_HUGETLB 0x40000 diff --git a/tools/testing/selftests/mm/transhuge-stress.c b/tools/testing/selftests/mm/transhuge-stress.c index 8eb0c5630e7e31..96f72898ebe0a5 100644 --- a/tools/testing/selftests/mm/transhuge-stress.c +++ b/tools/testing/selftests/mm/transhuge-stress.c @@ -17,7 +17,6 @@ #include #include "vm_util.h" #include "kselftest.h" -#include "hugepage_settings.h" int backing_fd = -1; int mmap_flags = MAP_ANONYMOUS | MAP_NORESERVE | MAP_PRIVATE; diff --git a/tools/testing/selftests/mm/uffd-common.h b/tools/testing/selftests/mm/uffd-common.h index 92a21b97f745af..0723843a7626b1 100644 --- a/tools/testing/selftests/mm/uffd-common.h +++ b/tools/testing/selftests/mm/uffd-common.h @@ -37,7 +37,6 @@ #include "kselftest.h" #include "vm_util.h" -#include "hugepage_settings.h" #define UFFD_FLAGS (O_CLOEXEC | O_NONBLOCK | UFFD_USER_MODE_ONLY) diff --git a/tools/testing/selftests/mm/uffd-wp-mremap.c b/tools/testing/selftests/mm/uffd-wp-mremap.c index 572c2516e874d7..c48eaab8e75cf9 100644 --- a/tools/testing/selftests/mm/uffd-wp-mremap.c +++ b/tools/testing/selftests/mm/uffd-wp-mremap.c @@ -7,8 +7,8 @@ #include #include #include +#include #include "kselftest.h" -#include "hugepage_settings.h" #include "uffd-common.h" static int pagemap_fd; diff --git a/tools/testing/selftests/mm/va_high_addr_switch.c b/tools/testing/selftests/mm/va_high_addr_switch.c index e24d7ba00b4417..5a354a664d1f7d 100644 --- a/tools/testing/selftests/mm/va_high_addr_switch.c +++ b/tools/testing/selftests/mm/va_high_addr_switch.c @@ -11,7 +11,6 @@ #include "vm_util.h" #include "kselftest.h" -#include "hugepage_settings.h" /* * The hint addr value is used to allocate addresses diff --git a/tools/testing/selftests/mm/vm_util.h b/tools/testing/selftests/mm/vm_util.h index fe0475f2bdf288..64a86e8a0c41bf 100644 --- a/tools/testing/selftests/mm/vm_util.h +++ b/tools/testing/selftests/mm/vm_util.h @@ -9,6 +9,7 @@ #include "kselftest.h" #include #include +#include #define BIT_ULL(nr) (1ULL << (nr)) #define PM_SOFT_DIRTY BIT_ULL(55) From db89df7d97a2ec7a80c6749052967eedec896c2a Mon Sep 17 00:00:00 2001 From: Sarthak Sharma Date: Fri, 18 Sep 2026 16:52:32 +0530 Subject: [PATCH 1098/1352] tools/mm: move gup_test from selftests/mm to tools/mm Move tools/testing/selftests/mm/gup_test.c to tools/mm/gup_bench.c. This is the first step in separating its benchmarking and functional testing components. Later patches will make this a purely benchmarking tool and introduce a new functional selftest under selftests/mm. Include hugepage_settings.h directly instead of vm_util.h and use getpagesize() instead of psize(). Adjust the Makefiles in both locations and add gup_bench to tools/mm/.gitignore. Remove the gup_test invocations from run_vmtests.sh and update MAINTAINERS. Also remove the gup_test reference from Documentation/core-api/pin_user_pages.rst. The selftest added later in the series is standalone and does not need per command documentation here. Link: https://lore.kernel.org/20260918112234.195857-5-sarthak.sharma@arm.com Signed-off-by: Sarthak Sharma Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Tested-by: Muhammad Usama Anjum Cc: Anshuman Khandual Cc: Baolin Wang Cc: Barry Song Cc: Dev Jain Cc: Jason Gunthorpe Cc: John Hubbard Cc: Jonathan Corbet Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Mark Brown Cc: Michal Hocko Cc: Nico Pache Cc: Peter Xu Cc: Ryan Roberts Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Zi Yan --- Documentation/core-api/pin_user_pages.rst | 9 ----- MAINTAINERS | 2 +- tools/mm/.gitignore | 1 + tools/mm/Makefile | 11 ++++-- .../mm/gup_test.c => mm/gup_bench.c} | 8 ++--- tools/testing/selftests/mm/Makefile | 1 - tools/testing/selftests/mm/run_vmtests.sh | 36 ------------------- 7 files changed, 14 insertions(+), 54 deletions(-) rename tools/{testing/selftests/mm/gup_test.c => mm/gup_bench.c} (97%) diff --git a/Documentation/core-api/pin_user_pages.rst b/Documentation/core-api/pin_user_pages.rst index c16ca163b55e3c..e0acedbd1d4860 100644 --- a/Documentation/core-api/pin_user_pages.rst +++ b/Documentation/core-api/pin_user_pages.rst @@ -226,15 +226,6 @@ will be pinned longterm, and whose data will be accessed. Unit testing ============ -This file:: - - tools/testing/selftests/mm/gup_test.c - -has the following new calls to exercise the new pin*() wrapper functions: - -* PIN_FAST_BENCHMARK (./gup_test -a) -* PIN_BASIC_TEST (./gup_test -b) - You can monitor how many total dma-pinned pages have been acquired and released since the system was booted, via two new /proc/vmstat entries: :: diff --git a/MAINTAINERS b/MAINTAINERS index dfd9f948390708..3a16011c3f7f75 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -17190,8 +17190,8 @@ T: git git://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm F: mm/gup.c F: mm/gup_test.c F: mm/gup_test.h +F: tools/mm/gup_bench.c F: tools/testing/selftests/mm/gup_longterm.c -F: tools/testing/selftests/mm/gup_test.c MEMORY MANAGEMENT - KSM (Kernel Samepage Merging) M: Andrew Morton diff --git a/tools/mm/.gitignore b/tools/mm/.gitignore index 1446a659e54088..154d740be02e82 100644 --- a/tools/mm/.gitignore +++ b/tools/mm/.gitignore @@ -3,3 +3,4 @@ slabinfo page-types page_owner_sort thp_swap_allocator_test +gup_bench diff --git a/tools/mm/Makefile b/tools/mm/Makefile index 858186a6eefdbd..f20a32d8cc22e2 100644 --- a/tools/mm/Makefile +++ b/tools/mm/Makefile @@ -3,13 +3,15 @@ # include ../scripts/Makefile.include -BUILD_TARGETS=page-types slabinfo page_owner_sort page_owner_filter thp_swap_allocator_test +BUILD_TARGETS=page-types slabinfo page_owner_sort page_owner_filter +BUILD_TARGETS += thp_swap_allocator_test gup_bench INSTALL_TARGETS = $(BUILD_TARGETS) thpmaps LIB_DIR = ../lib/api LIBS = $(LIB_DIR)/libapi.a +GUP_BENCH_OBJS = gup_bench.c ../lib/mm/hugepage_settings.c ../lib/mm/file_utils.c -CFLAGS += -Wall -Wextra -I../lib/ -pthread +CFLAGS += -Wall -Wextra -I../lib/ -I../.. -pthread LDFLAGS += $(LIBS) -pthread all: $(BUILD_TARGETS) @@ -22,8 +24,11 @@ $(LIBS): %: %.c $(CC) $(CFLAGS) -o $@ $< $(LDFLAGS) +gup_bench: $(GUP_BENCH_OBJS) $(LIBS) + $(CC) $(CFLAGS) -o $@ $(GUP_BENCH_OBJS) $(LDFLAGS) + clean: - $(RM) page-types slabinfo page_owner_sort page_owner_filter thp_swap_allocator_test + $(RM) page-types slabinfo page_owner_sort page_owner_filter thp_swap_allocator_test gup_bench make -C $(LIB_DIR) clean sbindir ?= /usr/sbin diff --git a/tools/testing/selftests/mm/gup_test.c b/tools/mm/gup_bench.c similarity index 97% rename from tools/testing/selftests/mm/gup_test.c rename to tools/mm/gup_bench.c index 5f44761dbec0be..da56aa5324d3fa 100644 --- a/tools/testing/selftests/mm/gup_test.c +++ b/tools/mm/gup_bench.c @@ -12,8 +12,8 @@ #include #include #include -#include "kselftest.h" -#include "vm_util.h" +#include +#include "../testing/selftests/kselftest.h" #define MB (1UL << 20) @@ -140,7 +140,7 @@ int main(int argc, char **argv) case 'n': nr_pages = atoi(optarg); if (nr_pages < 0) - nr_pages = size / psize(); + nr_pages = size / getpagesize(); break; case 't': thp = 1; @@ -254,7 +254,7 @@ int main(int argc, char **argv) madvise(p, size, MADV_NOHUGEPAGE); /* Fault them in here, from user space. */ - for (; (unsigned long)p < gup.addr + size; p += psize()) + for (; (unsigned long)p < gup.addr + size; p += getpagesize()) p[0] = 0; tid = malloc(sizeof(pthread_t) * nthreads); diff --git a/tools/testing/selftests/mm/Makefile b/tools/testing/selftests/mm/Makefile index 67882e52d4ff0a..d6337156111a4d 100644 --- a/tools/testing/selftests/mm/Makefile +++ b/tools/testing/selftests/mm/Makefile @@ -59,7 +59,6 @@ endif TEST_GEN_FILES = cow TEST_GEN_FILES += compaction_test TEST_GEN_FILES += gup_longterm -TEST_GEN_FILES += gup_test TEST_GEN_FILES += hmm-tests TEST_GEN_FILES += hugetlb-madvise TEST_GEN_FILES += hugetlb-mmap diff --git a/tools/testing/selftests/mm/run_vmtests.sh b/tools/testing/selftests/mm/run_vmtests.sh index a1db516557023f..4281bb6a859e03 100755 --- a/tools/testing/selftests/mm/run_vmtests.sh +++ b/tools/testing/selftests/mm/run_vmtests.sh @@ -148,30 +148,6 @@ test_selected() { fi } -run_gup_matrix() { - # -t: thp=on, -T: thp=off, -H: hugetlb=on - local hugetlb_mb=256 - - for huge in -t -T "-H -m $hugetlb_mb"; do - # -u: gup-fast, -U: gup-basic, -a: pin-fast, -b: pin-basic, -L: pin-longterm - for test_cmd in -u -U -a -b -L; do - # -w: write=1, -W: write=0 - for write in -w -W; do - # -S: shared - for share in -S " "; do - # -n: How many pages to fetch together? 512 is special - # because it's default thp size (or 2M on x86), 123 to - # just test partial gup when hit a huge in whatever form - for num in "-n 1" "-n 512" "-n 123" "-n -1"; do - CATEGORY="gup_test" run_test ./gup_test \ - $huge $test_cmd $write $share $num - done - done - done - done - done -} - # filter 64bit architectures ARCH64STR="arm64 mips64 parisc64 ppc64 ppc64le riscv64 s390x sparc64 x86_64" if [ -z "$ARCH" ]; then @@ -293,18 +269,6 @@ fi CATEGORY="mmap" run_test ./map_fixed_noreplace -if $RUN_ALL; then - run_gup_matrix -else - # get_user_pages_fast() benchmark - CATEGORY="gup_test" run_test ./gup_test -u -n 1 - CATEGORY="gup_test" run_test ./gup_test -u -n -1 - # pin_user_pages_fast() benchmark - CATEGORY="gup_test" run_test ./gup_test -a -n 1 - CATEGORY="gup_test" run_test ./gup_test -a -n -1 -fi -# Dump pages 0, 19, and 4096, using pin_user_pages: -CATEGORY="gup_test" run_test ./gup_test -ct -F 0x1 0 19 0x1000 CATEGORY="gup_test" run_test ./gup_longterm CATEGORY="userfaultfd" run_test ./uffd-unit-tests From 4c6b14578b3ac0a0d9e0fd9a0a2496b4d243b8c9 Mon Sep 17 00:00:00 2001 From: Sarthak Sharma Date: Fri, 18 Sep 2026 16:52:33 +0530 Subject: [PATCH 1099/1352] tools/mm: make gup_bench a benchmark only tool Remove the functional modes (GUP_BASIC_TEST, PIN_BASIC_TEST and DUMP_USER_PAGES_TEST) from gup_bench. Drop the kselftest dependency and use normal diagnostics and exit statuses. When no arguments are supplied, run a single GUP_FAST_BENCHMARK with the existing default values. Let users select other configurations through command-line options. Report ioctl failures and handle errors without relying on assert(). Link: https://lore.kernel.org/20260918112234.195857-6-sarthak.sharma@arm.com Signed-off-by: Sarthak Sharma Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Tested-by: Muhammad Usama Anjum Cc: Anshuman Khandual Cc: Baolin Wang Cc: Barry Song Cc: Dev Jain Cc: Jason Gunthorpe Cc: John Hubbard Cc: Jonathan Corbet Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Mark Brown Cc: Michal Hocko Cc: Nico Pache Cc: Peter Xu Cc: Ryan Roberts Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Zi Yan --- tools/mm/gup_bench.c | 183 ++++++++++++++++++------------------------- 1 file changed, 78 insertions(+), 105 deletions(-) diff --git a/tools/mm/gup_bench.c b/tools/mm/gup_bench.c index da56aa5324d3fa..ff6e466546813a 100644 --- a/tools/mm/gup_bench.c +++ b/tools/mm/gup_bench.c @@ -10,10 +10,10 @@ #include #include #include -#include +#include +#include #include #include -#include "../testing/selftests/kselftest.h" #define MB (1UL << 20) @@ -37,12 +37,6 @@ static char *cmd_to_str(unsigned long cmd) return "PIN_FAST_BENCHMARK"; case PIN_LONGTERM_BENCHMARK: return "PIN_LONGTERM_BENCHMARK"; - case GUP_BASIC_TEST: - return "GUP_BASIC_TEST"; - case PIN_BASIC_TEST: - return "PIN_BASIC_TEST"; - case DUMP_USER_PAGES_TEST: - return "DUMP_USER_PAGES_TEST"; } return "Unknown command"; } @@ -52,39 +46,29 @@ void *gup_thread(void *data) struct gup_test gup = *(struct gup_test *)data; int i, status; - /* Only report timing information on the *_BENCHMARK commands: */ - if ((cmd == PIN_FAST_BENCHMARK) || (cmd == GUP_FAST_BENCHMARK) || - (cmd == PIN_LONGTERM_BENCHMARK)) { - for (i = 0; i < repeats; i++) { - gup.size = size; - status = ioctl(gup_fd, cmd, &gup); - if (status) - break; + for (i = 0; i < repeats; i++) { + gup.size = size; + status = ioctl(gup_fd, cmd, &gup); + if (status) { + int err = errno; pthread_mutex_lock(&print_mutex); - ksft_print_msg("%s: Time: get:%lld put:%lld us", - cmd_to_str(cmd), gup.get_delta_usec, - gup.put_delta_usec); - if (gup.size != size) - ksft_print_msg(", truncated (size: %lld)", gup.size); - ksft_print_msg("\n"); + fprintf(stderr, "%s ioctl failed: %s\n", cmd_to_str(cmd), + strerror(err)); pthread_mutex_unlock(&print_mutex); + return data; } - } else { - gup.size = size; - status = ioctl(gup_fd, cmd, &gup); - if (status) - goto return_; pthread_mutex_lock(&print_mutex); - ksft_print_msg("%s: done\n", cmd_to_str(cmd)); + printf("%s: Time: get:%lld put:%lld us", + cmd_to_str(cmd), gup.get_delta_usec, + gup.put_delta_usec); if (gup.size != size) - ksft_print_msg("Truncated (size: %lld)\n", gup.size); + printf(", truncated (size: %lld)", gup.size); + printf("\n"); pthread_mutex_unlock(&print_mutex); } -return_: - ksft_test_result(!status, "ioctl status %d\n", status); return NULL; } @@ -92,38 +76,21 @@ int main(int argc, char **argv) { struct gup_test gup = { 0 }; int filed, i, opt, nr_pages = 1, thp = -1, write = 1, nthreads = 1, ret; - int flags = MAP_PRIVATE; + int flags = MAP_PRIVATE, started_threads = 0, exit_status = 1; char *file = "/dev/zero"; - bool hugetlb = false; + bool hugetlb = false, thread_error = false; + void *thread_result; pthread_t *tid; char *p; - while ((opt = getopt(argc, argv, "m:r:n:F:f:abcj:tTLUuwWSHpz")) != -1) { + while ((opt = getopt(argc, argv, "m:r:n:F:f:aj:tTLuwWSH")) != -1) { switch (opt) { case 'a': cmd = PIN_FAST_BENCHMARK; break; - case 'b': - cmd = PIN_BASIC_TEST; - break; case 'L': cmd = PIN_LONGTERM_BENCHMARK; break; - case 'c': - cmd = DUMP_USER_PAGES_TEST; - /* - * Dump page 0 (index 1). May be overridden later, by - * user's non-option arguments. - * - * .which_pages is zero-based, so that zero can mean "do - * nothing". - */ - gup.which_pages[0] = 1; - break; - case 'p': - /* works only with DUMP_USER_PAGES_TEST */ - gup.test_flags |= GUP_TEST_FLAG_DUMP_PAGES_USE_PIN; - break; case 'F': /* strtol, so you can pass flags in hex form */ gup.gup_flags = strtol(optarg, 0, 0); @@ -148,9 +115,6 @@ int main(int argc, char **argv) case 'T': thp = 0; break; - case 'U': - cmd = GUP_BASIC_TEST; - break; case 'u': cmd = GUP_FAST_BENCHMARK; break; @@ -172,52 +136,41 @@ int main(int argc, char **argv) hugetlb = true; break; default: - ksft_exit_fail_msg("Wrong argument\n"); + fprintf(stderr, "Wrong argument\n"); + exit(1); } } - if (optind < argc) { - int extra_arg_count = 0; - /* - * For example: - * - * ./gup_test -c 0 1 0x1001 - * - * ...to dump pages 0, 1, and 4097 - */ - - while ((optind < argc) && - (extra_arg_count < GUP_TEST_MAX_PAGES_TO_DUMP)) { - /* - * Do the 1-based indexing here, so that the user can - * use normal 0-based indexing on the command line. - */ - long page_index = strtol(argv[optind], 0, 0) + 1; - - gup.which_pages[extra_arg_count] = page_index; - extra_arg_count++; - optind++; - } + if (optind != argc) { + fprintf(stderr, "Unexpected argument '%s'\n", argv[optind]); + exit(1); } - ksft_print_header(); + if (geteuid()) { + fprintf(stderr, "Please run this test as root\n"); + exit(1); + } if (hugetlb) { unsigned long hp_size = default_huge_page_size(); - if (!hp_size) - ksft_exit_skip("HugeTLB is unavailable\n"); + if (!hp_size) { + fprintf(stderr, "Could not determine huge page size\n"); + return 1; + } size = (size + hp_size - 1) & ~(hp_size - 1); - if (!hugetlb_setup_default(size / hp_size)) - ksft_exit_skip("Not enough huge pages\n"); + if (!hugetlb_setup_default(size / hp_size)) { + fprintf(stderr, "Not enough huge pages\n"); + return 1; + } } - ksft_set_plan(nthreads); - filed = open(file, O_RDWR|O_CREAT, 0664); - if (filed < 0) - ksft_exit_fail_msg("Unable to open %s: %s\n", file, strerror(errno)); + if (filed < 0) { + fprintf(stderr, "Unable to open %s: %s\n", file, strerror(errno)); + return 1; + } gup.nr_pages_per_call = nr_pages; if (write) @@ -226,26 +179,24 @@ int main(int argc, char **argv) gup_fd = open(GUP_TEST_FILE, O_RDWR); if (gup_fd == -1) { switch (errno) { - case EACCES: - if (getuid()) - ksft_print_msg("Please run this test as root\n"); - break; case ENOENT: if (opendir("/sys/kernel/debug") == NULL) - ksft_print_msg("mount debugfs at /sys/kernel/debug\n"); - ksft_print_msg("check if CONFIG_GUP_TEST is enabled in kernel config\n"); + fprintf(stderr, "mount debugfs at /sys/kernel/debug\n"); + fprintf(stderr, "check if CONFIG_GUP_TEST is enabled in kernel config\n"); break; default: - ksft_print_msg("failed to open %s: %s\n", GUP_TEST_FILE, strerror(errno)); + fprintf(stderr, "failed to open %s: %s\n", GUP_TEST_FILE, + strerror(errno)); break; } - ksft_test_result_skip("Please run this test as root\n"); - ksft_exit_pass(); + goto err_close_filed; } p = mmap(NULL, size, PROT_READ | PROT_WRITE, flags, filed, 0); - if (p == MAP_FAILED) - ksft_exit_fail_msg("mmap: %s\n", strerror(errno)); + if (p == MAP_FAILED) { + fprintf(stderr, "mmap: %s\n", strerror(errno)); + goto err_close_gup_fd; + } gup.addr = (unsigned long)p; if (thp == 1) @@ -258,17 +209,39 @@ int main(int argc, char **argv) p[0] = 0; tid = malloc(sizeof(pthread_t) * nthreads); - assert(tid); + if (!tid) { + fprintf(stderr, "Failed to allocate %d threads: %s\n", + nthreads, strerror(errno)); + goto err_unmap; + } + for (i = 0; i < nthreads; i++) { ret = pthread_create(&tid[i], NULL, gup_thread, &gup); - assert(ret == 0); + if (ret) { + fprintf(stderr, "pthread_create failed: %s\n", strerror(ret)); + thread_error = true; + break; + } + started_threads++; } - for (i = 0; i < nthreads; i++) { - ret = pthread_join(tid[i], NULL); - assert(ret == 0); + for (i = 0; i < started_threads; i++) { + ret = pthread_join(tid[i], &thread_result); + if (ret) { + fprintf(stderr, "pthread_join failed: %s\n", strerror(ret)); + thread_error = true; + } else if (thread_result) + thread_error = true; } free(tid); - - ksft_exit_pass(); + if (!thread_error) + exit_status = 0; + +err_unmap: + munmap((void *)gup.addr, size); +err_close_gup_fd: + close(gup_fd); +err_close_filed: + close(filed); + return exit_status; } From f2681c358075c6d1054d19f704b525c3e6dffc24 Mon Sep 17 00:00:00 2001 From: Sarthak Sharma Date: Fri, 18 Sep 2026 16:52:34 +0530 Subject: [PATCH 1100/1352] selftests/mm: add a GUP selftest Add a new GUP selftest which uses kselftest_harness.h. Cover 12 mapping configurations: THP enabled, THP disabled and HugeTLB, each across private/shared mappings and with/without FOLL_WRITE. Run 5 test cases for every variant: get_user_pages, get_user_pages_fast, pin_user_pages, pin_user_pages_fast and pin_user_pages_longterm. Use two default hugeTLB pages and derive the mapping size from their size. This exercises GUP both within a single HugeTLB page and across a HugeTLB boundary, without reserving an excessive number of pages. Sweep four nr_pages_per_call values for each test: 1, 512, 123 and all pages. This preserves the coverage previously provided by run_gup_matrix(): 12 mapping combinations x 5 GUP/PUP operations x 4 batch sizes. In total the selftest reports 60 TAP cases and issues 240 ioctls. Do not carry DUMP_USER_PAGES_TEST into the new selftest because its output is written to the kernel log and the selftest does not verify that output. Add the new gup binary to the selftests/mm build, run_vmtests.sh and MAINTAINERS. Update mm/Kconfig to describe the benchmark and selftest split. Link: https://lore.kernel.org/20260918112234.195857-7-sarthak.sharma@arm.com Signed-off-by: Sarthak Sharma Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Tested-by: Muhammad Usama Anjum Cc: Anshuman Khandual Cc: Baolin Wang Cc: Barry Song Cc: Dev Jain Cc: Jason Gunthorpe Cc: John Hubbard Cc: Jonathan Corbet Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Mark Brown Cc: Michal Hocko Cc: Nico Pache Cc: Peter Xu Cc: Ryan Roberts Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Zi Yan --- MAINTAINERS | 1 + mm/Kconfig | 19 +- tools/testing/selftests/mm/Makefile | 1 + tools/testing/selftests/mm/gup.c | 262 ++++++++++++++++++++++ tools/testing/selftests/mm/run_vmtests.sh | 1 + 5 files changed, 272 insertions(+), 12 deletions(-) create mode 100644 tools/testing/selftests/mm/gup.c diff --git a/MAINTAINERS b/MAINTAINERS index 3a16011c3f7f75..596efa354ea235 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -17191,6 +17191,7 @@ F: mm/gup.c F: mm/gup_test.c F: mm/gup_test.h F: tools/mm/gup_bench.c +F: tools/testing/selftests/mm/gup.c F: tools/testing/selftests/mm/gup_longterm.c MEMORY MANAGEMENT - KSM (Kernel Samepage Merging) diff --git a/mm/Kconfig b/mm/Kconfig index 30170a936f1fc0..edb4a6c0a87021 100644 --- a/mm/Kconfig +++ b/mm/Kconfig @@ -1292,24 +1292,19 @@ config PERCPU_STATS be used to help understand percpu memory usage. config GUP_TEST - bool "Enable infrastructure for get_user_pages()-related unit tests" + bool "Enable infrastructure for get_user_pages()-related unit tests and benchmarks" depends on DEBUG_FS help Provides /sys/kernel/debug/gup_test, which in turn provides a way - to make ioctl calls that can launch kernel-based unit tests for - the get_user_pages*() and pin_user_pages*() family of API calls. + to make ioctl calls that can launch kernel-based unit tests and + benchmarks for the get_user_pages*() and pin_user_pages*() families + of API calls. - These tests include benchmark testing of the _fast variants of - get_user_pages*() and pin_user_pages*(), as well as smoke tests of + These include benchmark testing of the _fast variants of + get_user_pages*() and pin_user_pages*(), as well as tests of the non-_fast variants. - There is also a sub-test that allows running dump_page() on any - of up to eight pages (selected by command line args) within the - range of user-space addresses. These pages are either pinned via - pin_user_pages*(), or pinned via get_user_pages*(), as specified - by other command line arguments. - - See tools/testing/selftests/mm/gup_test.c + See tools/testing/selftests/mm/gup.c and tools/mm/gup_bench.c. comment "GUP_TEST needs to have DEBUG_FS enabled" depends on !GUP_TEST && !DEBUG_FS diff --git a/tools/testing/selftests/mm/Makefile b/tools/testing/selftests/mm/Makefile index d6337156111a4d..7d69baeb93f430 100644 --- a/tools/testing/selftests/mm/Makefile +++ b/tools/testing/selftests/mm/Makefile @@ -58,6 +58,7 @@ endif TEST_GEN_FILES = cow TEST_GEN_FILES += compaction_test +TEST_GEN_FILES += gup TEST_GEN_FILES += gup_longterm TEST_GEN_FILES += hmm-tests TEST_GEN_FILES += hugetlb-madvise diff --git a/tools/testing/selftests/mm/gup.c b/tools/testing/selftests/mm/gup.c new file mode 100644 index 00000000000000..a6a8ca47d12eb3 --- /dev/null +++ b/tools/testing/selftests/mm/gup.c @@ -0,0 +1,262 @@ +// SPDX-License-Identifier: GPL-2.0 +#define __SANE_USERSPACE_TYPES__ // Use ll64 +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include "vm_util.h" +#include "kselftest_harness.h" + +#define MB (1UL << 20) + +/* Just the flags we need, copied from the kernel internals. */ +#define FOLL_WRITE 0x01 /* check pte is writable */ + +/* Page counts exercising single, THP-batch, partial, and full-mapping GUP. */ +static const int nr_pages_list[] = { 1, 512, 123, -1 }; + +#define GUP_TEST_FILE "/sys/kernel/debug/gup_test" +#define NR_HUGETLB_PAGES 2 + +static unsigned long hp_size; + +FIXTURE(gup_test) +{ + int gup_fd; + char *addr; + unsigned long size; +}; + +FIXTURE_VARIANT(gup_test) +{ + bool thp; + bool hugetlb; + bool write; + bool shared; +}; + +FIXTURE_VARIANT_ADD(gup_test, private_write) +{ + .thp = false, + .hugetlb = false, + .write = true, + .shared = false, +}; + +FIXTURE_VARIANT_ADD(gup_test, private_read) +{ + .thp = false, + .hugetlb = false, + .write = false, + .shared = false, +}; + +FIXTURE_VARIANT_ADD(gup_test, private_write_thp) +{ + .thp = true, + .hugetlb = false, + .write = true, + .shared = false, +}; + +FIXTURE_VARIANT_ADD(gup_test, private_read_thp) +{ + .thp = true, + .hugetlb = false, + .write = false, + .shared = false, +}; + +FIXTURE_VARIANT_ADD(gup_test, private_write_hugetlb) +{ + .thp = false, + .hugetlb = true, + .write = true, + .shared = false, +}; + +FIXTURE_VARIANT_ADD(gup_test, private_read_hugetlb) +{ + .thp = false, + .hugetlb = true, + .write = false, + .shared = false, +}; + +FIXTURE_VARIANT_ADD(gup_test, shared_write) +{ + .thp = false, + .hugetlb = false, + .write = true, + .shared = true, +}; + +FIXTURE_VARIANT_ADD(gup_test, shared_read) +{ + .thp = false, + .hugetlb = false, + .write = false, + .shared = true, +}; + +FIXTURE_VARIANT_ADD(gup_test, shared_write_thp) +{ + .thp = true, + .hugetlb = false, + .write = true, + .shared = true, +}; + +FIXTURE_VARIANT_ADD(gup_test, shared_read_thp) +{ + .thp = true, + .hugetlb = false, + .write = false, + .shared = true, +}; + +FIXTURE_VARIANT_ADD(gup_test, shared_write_hugetlb) +{ + .thp = false, + .hugetlb = true, + .write = true, + .shared = true, +}; + +FIXTURE_VARIANT_ADD(gup_test, shared_read_hugetlb) +{ + .thp = false, + .hugetlb = true, + .write = false, + .shared = true, +}; + +FIXTURE_SETUP(gup_test) +{ + int mmap_flags = MAP_PRIVATE | MAP_ANONYMOUS; + char *p; + + self->size = 128 * MB; + + if (variant->hugetlb) { + if (!hp_size) + SKIP(return, "HugeTLB not available\n"); + + if (hugetlb_free_default_pages() < NR_HUGETLB_PAGES) + SKIP(return, "Not enough huge pages\n"); + + self->size = NR_HUGETLB_PAGES * hp_size; + mmap_flags |= MAP_HUGETLB; + } + + if (variant->shared) + mmap_flags = (mmap_flags & ~MAP_PRIVATE) | MAP_SHARED; + + /* gup_fd has to be >= 0. Already checked in main() */ + self->gup_fd = open(GUP_TEST_FILE, O_RDWR); + ASSERT_GE(self->gup_fd, 0); + + self->addr = mmap(NULL, self->size, PROT_READ | PROT_WRITE, + mmap_flags, -1, 0); + + ASSERT_NE(self->addr, MAP_FAILED) { + int err = errno; + + close(self->gup_fd); + TH_LOG("mmap failed: %s", strerror(err)); + } + + if (variant->thp) + madvise(self->addr, self->size, MADV_HUGEPAGE); + else if (!variant->hugetlb) + madvise(self->addr, self->size, MADV_NOHUGEPAGE); + + for (p = self->addr; (unsigned long)p < (unsigned long)self->addr + + self->size; p += psize()) + p[0] = 0; +} + +FIXTURE_TEARDOWN(gup_test) +{ + munmap(self->addr, self->size); + close(self->gup_fd); +} + +static void run_gup_cmd(struct __test_metadata *_metadata, + FIXTURE_DATA(gup_test) *self, + const FIXTURE_VARIANT(gup_test) *variant, + unsigned long command) +{ + int i; + + for (i = 0; i < (int)ARRAY_SIZE(nr_pages_list); i++) { + struct gup_test gup = { + .addr = (unsigned long)self->addr, + .size = self->size, + .nr_pages_per_call = nr_pages_list[i] < 0 ? + self->size / psize() : nr_pages_list[i], + .gup_flags = variant->write ? FOLL_WRITE : 0, + }; + + TH_LOG("nr_pages_per_call=%u", gup.nr_pages_per_call); + ASSERT_EQ(ioctl(self->gup_fd, command, &gup), 0); + ASSERT_EQ(gup.size, self->size); + } +} + +TEST_F(gup_test, get_user_pages) +{ + run_gup_cmd(_metadata, self, variant, GUP_BASIC_TEST); +} + +TEST_F(gup_test, pin_user_pages) +{ + run_gup_cmd(_metadata, self, variant, PIN_BASIC_TEST); +} + +TEST_F(gup_test, get_user_pages_fast) +{ + run_gup_cmd(_metadata, self, variant, GUP_FAST_BENCHMARK); +} + +TEST_F(gup_test, pin_user_pages_fast) +{ + run_gup_cmd(_metadata, self, variant, PIN_FAST_BENCHMARK); +} + +TEST_F(gup_test, pin_user_pages_longterm) +{ + run_gup_cmd(_metadata, self, variant, PIN_LONGTERM_BENCHMARK); +} + +int main(int argc, char **argv) +{ + const int fd = open(GUP_TEST_FILE, O_RDWR); + + if (fd == -1) { + ksft_print_header(); + if (errno == EACCES) + ksft_exit_skip("Please run this test as root\n"); + if (errno == ENOENT) { + DIR *debugfs = opendir("/sys/kernel/debug"); + + if (!debugfs) + ksft_exit_skip("Mount debugfs at /sys/kernel/debug\n"); + closedir(debugfs); + ksft_exit_skip("Check CONFIG_GUP_TEST in kernel config\n"); + } + ksft_exit_fail_msg("Failed to open %s: %s\n", GUP_TEST_FILE, strerror(errno)); + } + close(fd); + + hp_size = default_huge_page_size(); + if (hp_size) + hugetlb_setup_default(NR_HUGETLB_PAGES); + + return test_harness_run(argc, argv); +} diff --git a/tools/testing/selftests/mm/run_vmtests.sh b/tools/testing/selftests/mm/run_vmtests.sh index 4281bb6a859e03..0d8c93087d0529 100755 --- a/tools/testing/selftests/mm/run_vmtests.sh +++ b/tools/testing/selftests/mm/run_vmtests.sh @@ -269,6 +269,7 @@ fi CATEGORY="mmap" run_test ./map_fixed_noreplace +CATEGORY="gup_test" run_test ./gup CATEGORY="gup_test" run_test ./gup_longterm CATEGORY="userfaultfd" run_test ./uffd-unit-tests From ac2c8fe646e343b943a604e34d4315026845404c Mon Sep 17 00:00:00 2001 From: Breno Leitao Date: Fri, 18 Sep 2026 04:24:37 -0700 Subject: [PATCH 1101/1352] mm/shmem: report RCU-tasks quiescent states while undoing a range This shows up in the Meta fleet on ftruncate() of large tmpfs files: INFO: rcu_tasks detected stalls on tasks: 00000000752fd185: .. nvcsw: 59455/59455 holdout: 1 idle_cpu: -1/0 task:rocksdb:bottom state:R running task __folio_split find_get_entry find_get_entries truncate_inode_partial_folio shmem_undo_range shmem_setattr notify_change do_ftruncate __x64_sys_ftruncate do_syscall_64 shmem_undo_range() walks the whole of the requested range in folio_batch sized steps, twice, and its two cond_resched() calls are the only reschedule points in that walk. cond_resched() is not an RCU-tasks quiescent state. Use cond_resched_tasks_rcu_qs() at both points so the walk reports an RCU-tasks quiescent state as it proceeds. Link: https://lore.kernel.org/20260918-shmem-tasks-rcu-v1-1-79acf91a2569@debian.org Fixes: 8315f42295d2 ("rcu: Add call_rcu_tasks()") Signed-off-by: Breno Leitao Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Baolin Wang Cc: Hugh Dickins Cc: "Paul E . McKenney" Cc: --- mm/shmem.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/mm/shmem.c b/mm/shmem.c index f33dbf5af2cb79..f2a36a1b537506 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -1367,7 +1367,7 @@ static void shmem_undo_range(struct inode *inode, loff_t lstart, uoff_t lend, } folio_batch_remove_exceptionals(&fbatch); folio_batch_release(&fbatch); - cond_resched(); + cond_resched_tasks_rcu_qs(); } /* @@ -1408,7 +1408,7 @@ static void shmem_undo_range(struct inode *inode, loff_t lstart, uoff_t lend, index = start; while (index < end) { - cond_resched(); + cond_resched_tasks_rcu_qs(); if (!find_get_entries(mapping, &index, end - 1, &fbatch, indices)) { From 91de97de0e51b01e180e0dec0daa6b0a1cb3f9d6 Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Fri, 18 Sep 2026 12:19:28 +0530 Subject: [PATCH 1102/1352] mm: constify arguments in default pxdp_get() Generic MM default pxdp_get() helpers fetch the values contained in pgtable entries via READ_ONCE() without modifying them. Just make their arguments explicitly 'const' for some additional protection. Link: https://lore.kernel.org/20260918064928.793742-1-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand Acked-by: David Hildenbrand (Arm) Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: SJ Park Cc: Pedro Falcato --- include/linux/pgtable.h | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/include/linux/pgtable.h b/include/linux/pgtable.h index e3c8ab96941c5e..780fe849ff8b7e 100644 --- a/include/linux/pgtable.h +++ b/include/linux/pgtable.h @@ -490,35 +490,35 @@ static inline int pudp_set_access_flags(struct vm_area_struct *vma, #endif #ifndef ptep_get -static inline pte_t ptep_get(pte_t *ptep) +static inline pte_t ptep_get(const pte_t *ptep) { return READ_ONCE(*ptep); } #endif #ifndef pmdp_get -static inline pmd_t pmdp_get(pmd_t *pmdp) +static inline pmd_t pmdp_get(const pmd_t *pmdp) { return READ_ONCE(*pmdp); } #endif #ifndef pudp_get -static inline pud_t pudp_get(pud_t *pudp) +static inline pud_t pudp_get(const pud_t *pudp) { return READ_ONCE(*pudp); } #endif #ifndef p4dp_get -static inline p4d_t p4dp_get(p4d_t *p4dp) +static inline p4d_t p4dp_get(const p4d_t *p4dp) { return READ_ONCE(*p4dp); } #endif #ifndef pgdp_get -static inline pgd_t pgdp_get(pgd_t *pgdp) +static inline pgd_t pgdp_get(const pgd_t *pgdp) { return READ_ONCE(*pgdp); } From af1eece5f6ba42a1239bf49b78dd9358cc601af8 Mon Sep 17 00:00:00 2001 From: Hao Ge Date: Wed, 16 Sep 2026 15:55:57 +0800 Subject: [PATCH 1103/1352] mm/alloc_tag: account for reserved tag ids in the kernel tag check The tag ids stored in the page flags include two reserved markers. Id 0 means the page has no tag and id 1 means the tag was cleared, so real tags start at CODETAG_ID_FIRST. The kernel-side check in alloc_tag_sec_init() compared kernel_tags.count alone against the addressable limit, so with the count at or just under the limit the last tag ids wrapped into those markers. Pages allocated through them then look the same as untagged pages on free, nothing is ever subtracted from the real tag and /proc/allocinfo shows that memory as still allocated. Add the missing CODETAG_ID_FIRST, same as tags_addressable(). Link: https://lore.kernel.org/20260916075557.121316-1-hao.ge@linux.dev Fixes: 4835f747d3ed ("alloc_tag: support for page allocation tag compression") Signed-off-by: Hao Ge Signed-off-by: Andrew Morton Acked-by: Suren Baghdasaryan Cc: --- mm/alloc_tag.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/alloc_tag.c b/mm/alloc_tag.c index b3341031047796..82e2c3448dcf2c 100644 --- a/mm/alloc_tag.c +++ b/mm/alloc_tag.c @@ -621,7 +621,7 @@ void __init alloc_tag_sec_init(void) kernel_tags.count = last_codetag - kernel_tags.first_tag; /* Check if kernel tags fit into page flags */ - if (kernel_tags.count > (1UL << NR_UNUSED_PAGEFLAG_BITS)) { + if (CODETAG_ID_FIRST + kernel_tags.count > (1UL << NR_UNUSED_PAGEFLAG_BITS)) { shutdown_mem_profiling(false); /* allocinfo file does not exist yet */ pr_err("%lu allocation tags cannot be references using %d available page flag bits. Memory allocation profiling is disabled!\n", kernel_tags.count, NR_UNUSED_PAGEFLAG_BITS); From a73a1743371c90f294176a7040adce8855b0ac73 Mon Sep 17 00:00:00 2001 From: David Carlier Date: Sun, 20 Sep 2026 16:50:02 +0100 Subject: [PATCH 1104/1352] mm/shmem: don't release a swapin-error marker as a swap entry A failed shmem swapin (eg, EIO) frees the swap slot and leaves a PTE_MARKER_POISONED entry in the page cache. On truncate or eviction shmem_free_swap() passes that marker to swap_put_entries_direct(), which warns because it is not a swap entry. There is nothing left to release either way. Skip the release for non-swap entries, as every other caller already does. Link: https://lore.kernel.org/20260920155002.1030454-1-devnexen@gmail.com Fixes: ac2d3268284b ("mm/swapfile.c: remove the unneeded checking") Signed-off-by: David Carlier Signed-off-by: Andrew Morton Reported-by: syzbot+23b25ba3c6bf971f9c57@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=23b25ba3c6bf971f9c57 Reviewed-by: Baolin Wang Cc: Baoquan he Cc: Barry Song Cc: Chis Li (Google) Cc: Hugh Dickens Cc: Kairui Song Cc: Kemeng Shi Cc: Nhat Pham Cc: --- mm/shmem.c | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/mm/shmem.c b/mm/shmem.c index f2a36a1b537506..07b2855dfb7bd9 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -1184,6 +1184,7 @@ static long shmem_free_swap(struct address_space *mapping, pgoff_t index, pgoff_t end, void *radswap) { XA_STATE(xas, &mapping->i_pages, index); + const softleaf_t swp = radix_to_swp_entry(radswap); unsigned int nr_pages = 0; pgoff_t base; void *entry; @@ -1200,8 +1201,9 @@ static long shmem_free_swap(struct address_space *mapping, } xas_unlock_irq(&xas); - if (nr_pages) - swap_put_entries_direct(radix_to_swp_entry(radswap), nr_pages); + /* A swapin-error marker holds no swap slot, so just drop it. */ + if (nr_pages && softleaf_is_swap(swp)) + swap_put_entries_direct(swp, nr_pages); return nr_pages; } From ac871176a8ecf71190414984a146079ce297a856 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Sat, 19 Sep 2026 13:43:11 +0300 Subject: [PATCH 1105/1352] docs/mm: describe set_memory() and set_direct_map() APIs The set_memory() and set_direct_map() APIs change permissions of existing kernel mappings, but their semantics are only described by the code, and that code differs from architecture to architecture. Add Documentation/mm/kernel-page-tables.rst that briefly describes what the kernel page tables consist of, defines the semantics both APIs have in common, including the parts that are easy to get wrong, and lists the differences between the architecture implementations. Add kernel-doc comments for the generic set_memory() and set_direct_map() stubs and link them into Documentation/core-api/mm-api.rst. Link: https://lore.kernel.org/20260919-set-memory-docs-v2-1-a2a4b3657690@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Reviewed-by: Kevin Brodsky Assisted-by: copilot:claude-opus Cc: "David Hildenbrand (arm)" Cc: Jonathan Corbet Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Randy Dunlap Cc: Shuah khan Cc: Suren Baghdasaryan Cc: "Vlastimil Babka (SUSE)" --- Documentation/core-api/mm-api.rst | 9 + Documentation/mm/index.rst | 1 + Documentation/mm/kernel-page-tables.rst | 410 ++++++++++++++++++++++++ include/linux/set_memory.h | 110 +++++++ 4 files changed, 530 insertions(+) create mode 100644 Documentation/mm/kernel-page-tables.rst diff --git a/Documentation/core-api/mm-api.rst b/Documentation/core-api/mm-api.rst index c1d03a5a2a192e..e135b61bc87da8 100644 --- a/Documentation/core-api/mm-api.rst +++ b/Documentation/core-api/mm-api.rst @@ -52,6 +52,15 @@ Virtually Contiguous Mappings .. kernel-doc:: mm/vmalloc.c :export: +Kernel Page Table Permissions +============================= + +.. kernel-doc:: include/linux/set_memory.h + :doc: Kernel page table permissions + +.. kernel-doc:: include/linux/set_memory.h + :internal: + File Mapping and Page Cache =========================== diff --git a/Documentation/mm/index.rst b/Documentation/mm/index.rst index 13a79f5d092c0e..e9ae5cd82163af 100644 --- a/Documentation/mm/index.rst +++ b/Documentation/mm/index.rst @@ -25,6 +25,7 @@ see the :doc:`admin guide <../admin-guide/mm/index>`. physical_memory page_tables + kernel-page-tables process_addrs bootmem page_allocation diff --git a/Documentation/mm/kernel-page-tables.rst b/Documentation/mm/kernel-page-tables.rst new file mode 100644 index 00000000000000..b3148df07fc8b4 --- /dev/null +++ b/Documentation/mm/kernel-page-tables.rst @@ -0,0 +1,410 @@ +.. SPDX-License-Identifier: GPL-2.0 + +================== +Kernel Page Tables +================== + +Introduction +============ + +The kernel page tables are created early during boot and, unlike the page +tables of user processes, most of them remain static throughout the system +lifetime. + +Every architecture has a direct map (also called linear map) that maps the +physical memory at a fixed offset, so that a physical address can be +translated to a kernel virtual address with simple arithmetic. On most +architectures the direct map is a part of the kernel page tables, with a few +exceptions described in the `Direct map`_ section below. + +On 32-bit systems with high memory the direct map covers only a part of the +physical memory, see Documentation/mm/highmem.rst. + +The vmalloc area, present on every architecture with an MMU, is used for +allocations of virtually contiguous memory whose backing pages are not +necessarily physically contiguous, and for mapping of the device memory. Its +page tables are created and torn down at runtime, see +Documentation/mm/vmalloc.rst. + +Architectures that use the `SPARSEMEM_VMEMMAP` memory model reserve a range of +kernel address space for the memory map, so that `struct page` objects appear +as a virtually contiguous array indexed by the page frame number, see +Documentation/mm/memory-model.rst. + +Besides these, the kernel image may be mapped in a dedicated part of the +kernel address space rather than accessed through the direct map. In that +case its mapping is an alias of the direct map of the physical memory the +image occupies. + +The rest of the kernel address space is architecture specific. For instance, +x86 has a region for the EFI runtime services and s390 has a region for the +code that has to run in the 31-bit addressing mode. + +Direct map +========== + +On most architectures the direct map is an ordinary part of the kernel page +tables. It is created early during boot with the largest pages the hardware +and the kernel configuration allow. + +Several architectures are different. + +MIPS +---- + +MIPS does not map the physical memory with page tables at all. Instead, a +part of the kernel virtual address space is a window into the physical +address space: the hardware translates the addresses that fall into that +window by a fixed transformation of the address bits, without walking the +page tables and without using the TLB. The memory attributes, such as +cacheability and the privilege level required to access the memory, are a +property of the window rather than of an individual page. + +There are no page table entries describing the direct map, so its properties +cannot be changed for an individual page. On 32-bit systems the window covers +only 512 MiB of the physical memory, so everything above that is high memory. + +LoongArch +--------- + +Like MIPS, LoongArch maps the physical memory with a hardware window. The +window covers 256 TiB of the physical address space on 64-bit and 512 MiB on +32-bit systems, and everything above that is high memory. + +The window occupies the lower part of the kernel address space. The upper +part, which includes the vmalloc area, is mapped with kernel page tables. + +PowerPC with the hash MMU +------------------------- + +On 64-bit PowerPC systems with the hash MMU the direct map does not exist in +the Linux page tables. It is installed into the hardware hash page table early +during boot. + +Modifying such mappings requires updating the hash page table directly, and +the hash MMU code implements this only for the kernel image permissions, +`debug_pagealloc` and KFENCE. + +With the radix MMU the direct map is a part of the ordinary kernel page +tables. + +Modifying the kernel page tables +================================ + +Except for the vmalloc area, the kernel page tables are mostly static. Still, +there are cases when the permissions of existing kernel mappings have to be +updated, for instance when a module is loaded and its text becomes read-only +and executable, or when a page is temporarily removed from the direct map to +reduce its exposure. + +There are two families of functions for this, both declared in +`include/linux/set_memory.h`: + +* `set_memory_*()` change permissions of an arbitrary kernel mapping. They + take a kernel virtual address and the number of pages. + +* `set_direct_map_*()` change permissions of the direct mapping of the page + frame represented by a `struct page`. They take a `struct page` pointer and + the number of pages. + +Architectures that implement `set_memory()` select `CONFIG_ARCH_HAS_SET_MEMORY` + +Architectures that implement `set_direct_map()` select +`CONFIG_ARCH_HAS_SET_DIRECT_MAP`. + +Common semantics +---------------- + +Ranges +~~~~~~ + +The `set_memory()` functions expect a range described by a start address and a +number of pages. The address must be page aligned and the entire range must be +covered by page table entries the architecture knows how to update. + +The entries may be marked as not present, set_memory_p() and set_memory_valid() +exist exactly to bring such a mapping back. + +When a range does not qualify, an architecture will usually say so by returning +an error and sometimes by a WARN()ing as well, unless it prefers to keep it to +itself and return success, see `Architecture specific differences`_. + +The `set_direct_map()` functions expect a range described by the first +`struct page` and a number of pages, and they update the direct map starting +at that page. + +The pages that follow the first one are updated regardless of what they are, so +the caller has to make sure that the range does not extend beyond the memory it +owns. + +Some architectures cannot split a large mapping, and they reject a range that +is a part of one, see `Architecture specific differences`_. + +Calling `set_memory()` with the number of pages set to zero is a no-op that +returns success, except on arm64, where doing nothing to the wrong address is +still an error. + +Aliases +~~~~~~~ + +A physical page may be mapped several times, for instance in the direct map +and in the vmalloc area, and the permissions of these mappings may differ. + +Whether the direct map alias is updated by a `set_memory()` call, and +which permission bits make it there, is entirely up to the architecture, and +there is not much agreement between them, see +`Architecture specific differences`_. + +Relying on that is a gamble; code that needs the direct map alias to change +should say so with the `set_direct_map()` APIs. + +Failures and partial updates +~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +Both families of functions return 0 on success and a negative error code on +failure. The most common failures are `-EINVAL` for a range that cannot be +handled and `-ENOMEM` when splitting a large mapping fails to allocate a page +table. + +Some of the range checks happen upfront, so their failure leaves the page +tables unchanged. + +**The update is not atomic and there is no rollback.** + +The architecture implementations walk the range and update the page tables as +they go, and they stop at the first entry that cannot be updated. When an error +is returned, an arbitrary prefix of the range may have been updated already, +and the same is true for the direct map alias when the architecture updates it. + +For example, when a range spans two large mappings and splitting the second +one fails because there is no memory for a page table, the first one is +already split and updated by the time the error is returned. + +None of the APIs inform the caller where in the range they failed, so reverting +such a partial update is possible in principle but unreliable in practice. + +The caller may try to restore the original permissions over the entire range, +but that revert goes through the very code that has just failed, which does +not inspire much confidence. + +The callers should therefore be prepared to give up on the memory in question: +leak it or panic, but never return it to the allocator before the permissions +are restored and never assume that the requested permissions are in effect. +Neither option is appealing, but both beat handing out a page whose +permissions nobody knows. + +TLB flushing +~~~~~~~~~~~~ + +The `set_memory()` functions flush the TLB for the affected range before +they return, so that the new permissions are in effect for every CPU. + +The `set_direct_map()` functions have `_noflush` in their names because when +they were first introduced on x86, the intention was that the TLB flushing +could be optimized by letting the caller handle it. + +For example, vfree() batches the TLB flushes for the areas allocated with +`VM_FLUSH_RESET_PERMS`, folding the flush of the direct map into the flush it +has to do for the vmalloc mapping anyway. + +Some architectures flush the TLB in the `_noflush` functions anyway, so the +name is best read as a suggestion. It does not make the flush by the caller +unnecessary, it only makes it more expensive. + +A caller that changes the permissions to more restrictive ones must flush the +TLB itself. + +Context +~~~~~~~ + +Architectures use different locking mechanisms to synchronize kernel page table +updates, and both families of functions may sleep, for instance when they +allocate memory to split a large mapping. + +The caller cannot presume it is safe to call these APIs from an atomic context. + +The `set_direct_map()` functions must not be called for high memory pages, +which have no direct map alias to update. + +Unimplemented APIs silently succeed +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +When an architecture does not implement these APIs, the generic stubs in +`include/linux/set_memory.h` return 0, that is, they report success for work +they have no intention of doing. + +The same happens inside several architecture implementations. The arm64 +`set_direct_map()` functions return 0 when can_set_direct_map() is false, and +the LoongArch `set_memory()` functions return 0 for the addresses in its +windowed direct mapping, which is not backed by page tables at all. + +Returning success without actually updating the page tables is a deliberate +trade-off that keeps the callers free of `#ifdef`\ s, but they have to realize: + +**a return value of 0 does not imply that the permissions were actually +changed.** + +For best-effort hardening that is good enough. When correctness or security +depends on the permissions, the caller has to make sure the architecture +really implements what it needs. For instance, secretmem depends on +`CONFIG_ARCH_HAS_SET_DIRECT_MAP` and calls can_set_direct_map() at runtime. + +Architecture specific differences +================================= + +The APIs are implemented by seven architectures and, beyond the common +semantics described above, their behaviour differs in several respects. + +Which of the APIs are implemented: + +========= ===================== ========================= +Arch `ARCH_HAS_SET_MEMORY` `ARCH_HAS_SET_DIRECT_MAP` +========= ===================== ========================= +arm yes no +arm64 yes yes +loongarch yes yes +powerpc yes no +riscv yes (MMU only) yes (MMU only) +s390 yes yes +x86 yes yes +========= ===================== ========================= + +Only set_memory_ro(), set_memory_rw(), set_memory_x() and set_memory_nx() are +available everywhere, and even these are not universal: some architectures +restrict the ranges that can be modified, for instance arm64 rejects direct map +addresses and accepts only addresses in vmalloc space. The architectures that +may run on hardware without an execute permission bit, like x86 and s390, +silently skip the update of the executable bit. + +set_memory_rox() has a generic implementation that calls set_memory_ro() and +set_memory_x() in turn; PowerPC, s390 and x86 override it with a single-pass +version. + +Several architectures define additional APIs. Some of those may have identical +semantics but different names. For example, making a mapping present or not +present is spelled differently: set_memory_p() and set_memory_np() on x86 and +PowerPC, set_memory_valid() on arm and arm64. + +The direct map and the kernel image are normally mapped with the largest +possible pages, and changing the permissions of a single page inside such a +mapping requires splitting it, which not every architecture can do. + +arm +--- + +* Does not implement `set_direct_map()`. +* Provides set_memory_valid(). +* set_memory_ro(), set_memory_rw(), set_memory_x() and set_memory_nx() accept + only vmalloc and module addresses. +* set_memory_valid() accepts any address. +* Does not update mapping aliases. + +arm64 +----- + +* Provides set_memory_valid(). +* Provides the memory encryption helpers, which are effective only when the + kernel runs as a confidential guest. +* set_memory_ro(), set_memory_rw(), set_memory_x() and set_memory_nx() accept + only vmalloc and module addresses: + + - the range must fit in the VM area that contains its start + - the VM area must have `VM_ALLOC` set and `VM_ALLOW_HUGE_VMAP` clear + +* set_memory_valid() accepts any address. +* The encryption helpers accept only direct map addresses. +* The `set_memory()` functions propagate the read-only and the read-write + changes to the direct map alias when `rodata=full` is in effect. +* Splits leaf mappings before the update on the hardware that supports it. + The split itself may be partial when it fails midway, but the permissions are + left untouched in that case. + Without support for splitting large mappings, a range that covers a leaf + entry only partially fails with a WARN()ing and `-EINVAL`. + If a range spans one or more full leaf entries and a partial leaf entry, the + permissions of the full leaf entries are updated before the failure. +* The `set_direct_map()` functions return 0 without doing anything when the + direct map cannot be modified, see can_set_direct_map(). +* Skips the TLB flush in `set_memory()` when the update only turns an invalid + mapping into a valid one. +* Does not flush TLB in `set_direct_map()`. + +LoongArch +--------- + +* Accepts only the addresses above the hardware window and silently returns + success for the rest, see `Direct map`_. +* Does not update mapping aliases. +* Does not split anything: a leaf entry is updated as a whole, which changes + the permissions of the entire large mapping. +* Flushes the TLB in `set_direct_map()`. + +PowerPC +------- + +* Does not implement `set_direct_map()`. +* Provides set_memory_np() and set_memory_p(). +* Rejects huge vmalloc mappings. +* With the hash MMU on 64-bit systems accepts nothing but the vmalloc and the + I/O regions. +* With the radix MMU accepts direct map addresses, but still cannot split a + large mapping. +* Does not update mapping aliases. + +riscv +----- + +* Implements both APIs only when the MMU is enabled. +* Provides set_memory_rw_nx(). +* The `set_memory()` functions accept any mapped kernel address, including the + direct map, but a vmalloc range must have the `pages` array of its VM area + populated, which rules out vmap() and ioremap() mappings. +* On 64-bit systems the `set_memory()` functions update the direct map alias of + a vmalloc range, including the executable bit. +* Does not split vmalloc ranges: a leaf entry is updated as a whole, which + changes the permissions of the entire large mapping. +* Splits the direct map on 64-bit systems. +* Flushes the TLB in `set_direct_map()`. + +s390 +---- + +* Provides set_memory_4k(), set_memory_rwnx() and the + `__set_memory_*(start, end)` variants that take a range rather than a page + count. +* The `set_memory()` functions accept any mapped kernel address, including the + direct map. +* Skips the update of the executable bit when the hardware has no support for + it. +* The `set_memory()` functions propagate only the read-only and read-write + changes to the direct map alias of a `VM_ALLOC` area, and deliberately not + the executable bit. +* Splits leaf PUD and PMD entries when the range is not aligned to them or when + set_memory_4k() is requested. +* Updates the page table entries with instructions that invalidate the + corresponding TLB entries, so no separate flush is needed anywhere. + +x86 +--- + +* Provides the largest set of operations on top of the common ones: + + - the cache attribute helpers: set_memory_uc(), set_memory_wc(), + set_memory_wb() + - presence control: set_memory_np() and set_memory_p() + - set_memory_4k() + - set_memory_global() and set_memory_nonglobal() + - the array variants that operate on `struct page` arrays or arrays of + virtual addresses + - memory encryption: set_memory_encrypted() and set_memory_decrypted() + +* The `set_memory()` functions accept any mapped kernel address, including the + direct map, and silently succeed for the unmapped holes inside it. +* Does nothing in set_memory_x() and set_memory_nx() when the CPU has no + execute permission bit. +* The `set_memory()` functions apply the change to the direct map alias and, + for the kernel image, to the high kernel mapping. The NX bit is never + propagated, so that the direct map stays non-executable. +* Splits large mappings on demand and can collapse them back when the + permissions become uniform again. +* Does not flush the TLB in `set_direct_map()`, but splitting a large + mapping flushes it anyway. diff --git a/include/linux/set_memory.h b/include/linux/set_memory.h index 3fe293cfed8cc8..27c32fbabfec90 100644 --- a/include/linux/set_memory.h +++ b/include/linux/set_memory.h @@ -5,16 +5,92 @@ #ifndef _LINUX_SET_MEMORY_H_ #define _LINUX_SET_MEMORY_H_ +/** + * DOC: Kernel page table permissions + * + * The set_memory() and set_direct_map() APIs update permissions of existing + * kernel mappings. + * + * The set_memory() functions operate on a range of kernel virtual addresses, + * the set_direct_map() functions operate on the direct map. + * + * The updates are not atomic: when a call fails, an arbitrary prefix of the + * range may have been updated already and there is no automatic rollback. + * A caller must restore the required permissions before reusing or freeing + * the memory. + * + * When an architecture does not implement these APIs they succeed without + * doing anything, so a return value of 0 does not mean that the permissions + * were actually changed. + * + * Callers that depend on the permissions being applied must ensure that the + * architecture supports the required operation for the target addresses. The + * Kconfig symbols alone do not guarantee this. + * + * See Documentation/mm/kernel-page-tables.rst for the details and for the + * differences between the architecture implementations. + */ + #ifdef CONFIG_ARCH_HAS_SET_MEMORY #include #else +/** + * set_memory_ro - make a kernel mapping read-only + * @addr: page aligned start of the kernel virtual address range + * @numpages: number of pages in the range + * + * Flushes the TLB for the range. + * + * Return: 0 on success, negative error code on failure. + */ static inline int __must_check set_memory_ro(unsigned long addr, int numpages) { return 0; } + +/** + * set_memory_rw - make a kernel mapping writable + * @addr: page aligned start of the kernel virtual address range + * @numpages: number of pages in the range + * + * Flushes the TLB for the range. + * + * Return: 0 on success, negative error code on failure. + */ static inline int __must_check set_memory_rw(unsigned long addr, int numpages) { return 0; } + +/** + * set_memory_x - make a kernel mapping executable + * @addr: page aligned start of the kernel virtual address range + * @numpages: number of pages in the range + * + * Flushes the TLB for the range. + * + * Return: 0 on success, negative error code on failure. + */ static inline int __must_check set_memory_x(unsigned long addr, int numpages) { return 0; } + +/** + * set_memory_nx - make a kernel mapping non-executable + * @addr: page aligned start of the kernel virtual address range + * @numpages: number of pages in the range + * + * Flushes the TLB for the range. + * + * Return: 0 on success, negative error code on failure. + */ static inline int __must_check set_memory_nx(unsigned long addr, int numpages) { return 0; } #endif #ifndef set_memory_rox +/** + * set_memory_rox - make a kernel mapping read-only and executable + * @addr: page aligned start of the kernel virtual address range + * @numpages: number of pages in the range + * + * A failure may leave the range read-only but not executable. + * + * Flushes the TLB for the range. + * + * Return: 0 on success, negative error code on failure. + */ static inline int set_memory_rox(unsigned long addr, int numpages) { int ret = set_memory_ro(addr, numpages); @@ -25,11 +101,33 @@ static inline int set_memory_rox(unsigned long addr, int numpages) #endif #ifndef CONFIG_ARCH_HAS_SET_DIRECT_MAP +/** + * set_direct_map_invalid_noflush - remove pages from the direct map + * @page: first page to update + * @nr: number of pages to update + * + * Makes the direct mapping of @nr pages starting at @page not present. + * The caller is responsible for any required TLB flushing. + * + * Return: 0 on success, negative error code on failure. + */ static inline int set_direct_map_invalid_noflush(struct page *page, unsigned int nr) { return 0; } + +/** + * set_direct_map_default_noflush - restore the direct map of pages + * @page: first page to update + * @nr: number of pages to update + * + * Restores the default kernel permissions of the direct mapping of @nr + * pages starting at @page. + * The caller is responsible for any required TLB flushing. + * + * Return: 0 on success, negative error code on failure. + */ static inline int set_direct_map_default_noflush(struct page *page, unsigned int nr) { @@ -46,6 +144,18 @@ static inline bool kernel_page_present(struct page *page) * boot time. Let them overrive this query. */ #ifndef can_set_direct_map +/** + * can_set_direct_map - check if the direct map can be modified + * + * Available with CONFIG_ARCH_HAS_SET_DIRECT_MAP. Architectures may override + * this to report whether direct map updates are enabled at runtime. + * Even though the generic implementation returns true this does not guarantee + * that every address can be updated. + * + * See Documentation/mm/kernel-page-tables.rst for the details + * + * Return: true unless the architecture reports direct map updates disabled. + */ static inline bool can_set_direct_map(void) { return true; From 846940f034e3ee5d4b8fa2c0de73c1f463260157 Mon Sep 17 00:00:00 2001 From: Mike Rapoport Date: Tue, 22 Sep 2026 12:42:55 +0300 Subject: [PATCH 1106/1352] docs-mm-describe-set_memory-and-set_direct_map-apis-fix fix phrasing, per Kevin Link: https://lore.kernel.org/arJNn2QD_pY6e14R@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Cc: Kevin Brodsky Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/mm/kernel-page-tables.rst | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/Documentation/mm/kernel-page-tables.rst b/Documentation/mm/kernel-page-tables.rst index b3148df07fc8b4..c16de34b4f79d8 100644 --- a/Documentation/mm/kernel-page-tables.rst +++ b/Documentation/mm/kernel-page-tables.rst @@ -103,9 +103,9 @@ There are two families of functions for this, both declared in * `set_memory_*()` change permissions of an arbitrary kernel mapping. They take a kernel virtual address and the number of pages. -* `set_direct_map_*()` change permissions of the direct mapping of the page - frame represented by a `struct page`. They take a `struct page` pointer and - the number of pages. +* `set_direct_map_*()` change permissions of the direct mapping for the range + of page frames starting at the page represented by a `struct page`. They take + a `struct page` pointer and the number of pages. Architectures that implement `set_memory()` select `CONFIG_ARCH_HAS_SET_MEMORY` From 4ee2b2e789f15bd4ba229008ceddeec4608691ba Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:31 +0100 Subject: [PATCH 1107/1352] selftests/mm: raise the khugepaged test-case cap Patch series "selftests/mm: improve khugepaged coverage", v6. khugepaged collapses to mTHP orders since 7.2, and 7.3 added three selftest cases for it: the generic collapse cases run at one order named by -c, with the result detected by counting folios of that order. That leaves the collapse path largely untested. The suite does not run where a PMD is 512M. A folio count cannot say where a collapse landed. Fixed sleeps cannot tell "not collapsed" from "not scanned yet". And nothing exercises collapse under contention. Close those gaps in order: - Make the suite run at a 512M PMD: scale the collapse wait with the PMD size, skip what such a PMD cannot serve, make the swapout the swap cases depend on deterministic, and keep khugepaged out of the MADV_COLLAPSE cases. - Detect results per window rather than by count, with folio-order helpers in vm_util that are checked against the kernel before any collapse test trusts them. - Drive khugepaged deterministically: a completion barrier that wakes the daemon and waits for a full pass, and a check that one pass yields one attributed collapse. - Cover collapse at every supported order by default: which window collapses, occupancy at both limits, sources that are already large folios, and a fork-shared source under concurrent writes. - Race collapse against everything that can touch its sources, at both occupancy limits and over whole-table zaps, checked by content and by the kernel's own assertions. Everything passes on an unmodified kernel. This patch (of 19): TEST() ends the run with "MAX_TEST_CASES is too small" when the table fills, and the table holds 64. A full invocation already registers 63, so the next case added anywhere aborts the whole suite before a single test runs. Raise the cap to 256. The table is a static array of small structs, so the room costs nothing worth counting. Link: https://lore.kernel.org/20260919002451.496763-1-kirill@shutemov.name Link: https://lore.kernel.org/20260919002451.496763-2-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: Usama Arif Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Reviewed-by: Baolin Wang Assisted-by: LLM Cc: Alexander Gordeev Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Michal Hocko Cc: Muhammad Usama Anjum Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/khugepaged.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 525108cace54ab..c44bc18f753659 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -1290,7 +1290,7 @@ struct test_case { test_fn fn; }; -#define MAX_TEST_CASES 64 +#define MAX_TEST_CASES 256 static struct test_case test_cases[MAX_TEST_CASES]; static int nr_test_cases; From 93461abd0fd99ddd3b47509f7a151209eb43c43e Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:32 +0100 Subject: [PATCH 1108/1352] selftests/mm: skip collapse_compound_extreme() where the PMD is too large collapse_compound_extreme() builds a PTE table full of distinct PTE-mapped compound pages by cycling hpage_pmd_nr fault-time THPs through mremap. It therefore needs hpage_pmd_nr PMD-order allocations in a row. That is fine at a 2M PMD (4K base pages) or a 32M one (16K). A 512M PMD -- arm64 with 64K base pages -- makes each of those an order-13 allocation, which the allocator cannot reliably hand out even once, let alone 8192 times. The failure is not a quiet one: the case calls ksft_exit_fail_msg(), so the whole binary stops and every case after it is lost. Skip the case where the PMD is larger than 32M. The MADV_COLLAPSE cases still cover PMD-order collapse on those configurations, and 4K and 16K PMDs are unaffected. Link: https://lore.kernel.org/20260919002451.496763-3-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Reviewed-by: Baolin Wang Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Michal Hocko Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/khugepaged.c | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index c44bc18f753659..8a6d708026b725 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -940,6 +940,16 @@ static void collapse_compound_extreme(struct collapse_context *c, struct mem_ops void *p; int i; + /* + * This needs hpage_pmd_nr PMD-order allocations in a row, which the + * allocator will not supply if the PMD is very large. + */ + if (hpage_pmd_size > (32UL << 20)) { + ksft_test_result_skip("%s: PMD too large for fault-time THP construction\n", + __func__); + return; + } + p = ops->setup_area(1); ksft_print_msg("Construct PTE page table full of different PTE-mapped compound pages\n"); for (i = 0; i < hpage_pmd_nr; i++) { From ea5c69a50633a6fb5eace5f92c2d273ba4050abd Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:33 +0100 Subject: [PATCH 1109/1352] selftests/mm: scale khugepaged's collapse wait with the PMD size wait_for_scan() gives every case the same three seconds, whatever the huge page costs to build. collapse_full() asks for four of them: 8M at a 2M PMD, but 2G at a 512M PMD -- arm64 with 64K base pages. Three seconds is thin at that size, and the case has reported a failure for a collapse that was still going. The timeout is a ceiling on a poll loop, not a sleep: the loop stops as soon as ops->check_huge() sees the collapse, or as soon as full_scans has advanced by two. Raising it costs a passing case nothing. Across 80 runs of collapse_full() on arm64 with 64K pages the wait was half a second in 73 of them, with a tail to two seconds. Keep three seconds as the floor and add a second per 128M collapsed. A 2M PMD is unchanged, so x86-64 is too; a 512M PMD gets 19 seconds. On arm64 with 64K pages a passing ./khugepaged all:anon takes 49 seconds under TCG before and after this change. Link: https://lore.kernel.org/20260919002451.496763-4-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Reviewed-by: Baolin Wang Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Michal Hocko Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/khugepaged.c | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 8a6d708026b725..189cc4fee18ffe 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -561,8 +561,11 @@ static bool wait_for_scan(const char *msg, char *p, size_t len, int nr_hpages, int collap_order, struct mem_ops *ops) { unsigned long hpage_size = page_size << collap_order; - int full_scans; - int timeout = 6; /* 3 seconds */ + unsigned long bytes = (unsigned long)nr_hpages * hpage_size; + int timeout, full_scans; + + /* Half-second ticks: three seconds floor, plus a second per 128M */ + timeout = 6 + 2 * (bytes / (128UL << 20)); /* Sanity check */ if (!ops->check_huge(p, len, 0, hpage_size)) From 79b457b00a62f5393b8256d87f3248681ceccd4e Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:34 +0100 Subject: [PATCH 1110/1352] selftests/mm: skip khugepaged page cache cases without a PMD folio The page cache caps folio order at MAX_PAGECACHE_ORDER, which is below the PMD order on arm64 with 64K pages, where a PMD is 512M. A PMD-sized page cache folio is impossible there, so the kernel refuses these collapses: MADV_COLLAPSE answers -EINVAL and khugepaged passes over the range. Four shmem cases ask for a PMD-sized folio anyway, fail, and the run bails out in the middle. Skip the shmem and file mem types where the cap is below the PMD order. The cap is not shmem-specific: it applies to every file folio. Add thp_file_supported_orders() to read the orders the page cache allows. Anonymous collapse is unaffected: its orders are not capped this way. Link: https://lore.kernel.org/20260919002451.496763-5-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Reviewed-by: Mike Rapoport (Microsoft) Reviewed-by: Baolin Wang Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/lib/mm/hugepage_settings.h | 9 +++++++++ tools/testing/selftests/mm/khugepaged.c | 19 +++++++++++++++++++ 2 files changed, 28 insertions(+) diff --git a/tools/lib/mm/hugepage_settings.h b/tools/lib/mm/hugepage_settings.h index 548e9d288d1d16..94d9fc747f4979 100644 --- a/tools/lib/mm/hugepage_settings.h +++ b/tools/lib/mm/hugepage_settings.h @@ -87,6 +87,15 @@ void thp_set_read_ahead_path(char *path); unsigned long thp_supported_orders(void); unsigned long thp_shmem_supported_orders(void); +/* + * The per-order shmem_enabled attribute is created for the orders the page + * cache can hold, not just for shmem, so it answers for regular files too. + */ +static inline unsigned long thp_file_supported_orders(void) +{ + return thp_shmem_supported_orders(); +} + bool thp_available(void); bool thp_is_enabled(void); diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 189cc4fee18ffe..5318f3cfc0d044 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -1358,6 +1358,25 @@ int main(int argc, char **argv) setbuf(stdout, NULL); + /* + * Without a PMD-order page cache folio the kernel refuses these + * collapses, so there is nothing to test. + */ + if (!(thp_file_supported_orders() & (1UL << hpage_pmd_order))) { + if (shmem_ops) { + ksft_print_msg("no PMD-order page cache folio: skipping shmem\n"); + shmem_ops = NULL; + } + if (read_only_file_ops) { + ksft_print_msg("no PMD-order page cache folio: skipping file\n"); + read_only_file_ops = NULL; + read_write_file_read_ops = NULL; + read_write_file_write_ops = NULL; + } + if (!anon_ops && !shmem_ops && !read_only_file_ops) + ksft_exit_skip("No mem_type left to run\n"); + } + default_settings.khugepaged.max_ptes_none = hpage_pmd_nr - 1; default_settings.khugepaged.max_ptes_swap = hpage_pmd_nr / 8; default_settings.khugepaged.max_ptes_shared = hpage_pmd_nr / 2; From adb116526e52e818a8cbbd6e1c4ddbed45158c54 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:35 +0100 Subject: [PATCH 1111/1352] selftests/mm: make the swap cases' swapout reliable collapse_swapin_single_pte() and collapse_max_ptes_swap() swap a range out and then require smaps to report exactly the count they asked for. Two things keep that count from arriving. MADV_PAGEOUT is best effort, so the count often turns up a moment late. And wait_for_scan() leaves the range eligible for collapsing, so khugepaged is still working on it. Collapsing reads the swapped-out pages back in, so the daemon empties the swap as fast as the case fills it. On arm64 with 64K pages max_ptes_swap is 1024 pages, which is 64M a step, and the case loses the race: # Swapout 1024 of 8192 pages... Fail not ok 10 collapse_max_ptes_swap Retry for up to two seconds, holding the range out of khugepaged's reach meanwhile. The collapse each case runs next restores MADV_HUGEPAGE, so only the setup is affected. If the pages still won't swap out, skip: no swap, swap too small or full, a memcg cap or busy writeback. None of that is a kernel bug. Link: https://lore.kernel.org/20260919002451.496763-6-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Reviewed-by: Muhammad Usama Anjum Reviewed-by: Baolin Wang Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/khugepaged.c | 41 +++++++++++++++++-------- 1 file changed, 29 insertions(+), 12 deletions(-) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 5318f3cfc0d044..13a2a47ab1108e 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -221,6 +221,29 @@ static bool check_swap(void *addr, unsigned long size) return swap; } +static bool swapout_range(void *p, unsigned long size) +{ + int i; + + /* keep khugepaged from collapsing the range and swapping it back in */ + if (madvise(p, size, MADV_NOHUGEPAGE)) + ksft_exit_fail_perror("madvise(MADV_NOHUGEPAGE)"); + + /* + * Retry several times because MADV_PAGEOUT is best effort. Sleep + * between the retries to give outstanding writeback a chance to + * finish. + */ + for (i = 0; i < 40; i++) { + if (madvise(p, size, MADV_PAGEOUT)) + ksft_exit_fail_perror("madvise(MADV_PAGEOUT)"); + if (check_swap(p, size)) + return true; + usleep(50 * 1000); + } + return false; +} + static void *alloc_mapping(int nr) { void *p; @@ -828,12 +851,10 @@ static void collapse_swapin_single_pte(struct collapse_context *c, struct mem_op ops->fault(p, 0, hpage_pmd_size); ksft_print_msg("Swapout one page..."); - if (madvise(p, page_size, MADV_PAGEOUT)) - ksft_exit_fail_perror("madvise(MADV_PAGEOUT)"); - if (check_swap(p, page_size)) { + if (swapout_range(p, page_size)) { success("OK"); } else { - fail("Fail"); + skip("Could not swap out"); goto out; } @@ -854,12 +875,10 @@ static void collapse_max_ptes_swap(struct collapse_context *c, struct mem_ops *o ops->fault(p, 0, hpage_pmd_size); ksft_print_msg("Swapout %d of %d pages...", max_ptes_swap + 1, hpage_pmd_nr); - if (madvise(p, (max_ptes_swap + 1) * page_size, MADV_PAGEOUT)) - ksft_exit_fail_perror("madvise(MADV_PAGEOUT)"); - if (check_swap(p, (max_ptes_swap + 1) * page_size)) { + if (swapout_range(p, (max_ptes_swap + 1) * page_size)) { success("OK"); } else { - fail("Fail"); + skip("Could not swap out"); goto out; } @@ -871,12 +890,10 @@ static void collapse_max_ptes_swap(struct collapse_context *c, struct mem_ops *o ops->fault(p, 0, hpage_pmd_size); ksft_print_msg("Swapout %d of %d pages...", max_ptes_swap, hpage_pmd_nr); - if (madvise(p, max_ptes_swap * page_size, MADV_PAGEOUT)) - ksft_exit_fail_perror("madvise(MADV_PAGEOUT)"); - if (check_swap(p, max_ptes_swap * page_size)) { + if (swapout_range(p, max_ptes_swap * page_size)) { success("OK"); } else { - fail("Fail"); + skip("Could not swap out"); goto out; } From 8f3e3f71a8478af95bd8af71d5137402643f9c4c Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:36 +0100 Subject: [PATCH 1112/1352] selftests/mm: stop khugepaged during the MADV_COLLAPSE cases __madvise_collapse() turns THP off before each MADV_COLLAPSE, both to keep khugepaged out of the range and to prove MADV_COLLAPSE ignores the setting. It clears the global controls only, which is no longer enough. A per-order control overrides them, and -s, which makes the cases fault in folios of one order, leaves that order's control at "always". khugepaged then collapses the very range the case is working on, and the case fails on a collapse that was interfered with rather than refused. Clear the per-order controls too. MADV_COLLAPSE does not consult them: anon never did, and shmem stopped with "mm: shmem: ignore sysfs configs for shmem forced collapse". Link: https://lore.kernel.org/20260919002451.496763-7-kirill@shutemov.name Fixes: b7f16963efe7 ("mm/khugepaged: run khugepaged for all orders") Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Reviewed-by: Baolin Wang Assisted-by: LLM Cc: Alexander Gordeev Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Muhammad Usama Anjum Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/khugepaged.c | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 13a2a47ab1108e..e013eebc7136ed 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -538,8 +538,8 @@ static bool is_anon(struct mem_ops *ops) static void __madvise_collapse(const char *msg, char *p, int nr_hpages, struct mem_ops *ops, bool expect) { - int ret; struct thp_settings settings = *thp_current_settings(); + int ret, i; ksft_print_msg("%s...", msg); @@ -555,6 +555,10 @@ static void __madvise_collapse(const char *msg, char *p, int nr_hpages, */ settings.thp_enabled = THP_NEVER; settings.shmem_enabled = SHMEM_NEVER; + for (i = 0; i < NR_ORDERS; i++) { + settings.hugepages[i].enabled = THP_NEVER; + settings.shmem_hugepages[i].enabled = SHMEM_NEVER; + } thp_push_settings(&settings); /* Clear VM_NOHUGEPAGE */ From f7d823263b4c7fe3745b5edf0e5bfcf0c7bed03e Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:37 +0100 Subject: [PATCH 1113/1352] selftests/mm: move is_backed_by_folio() into vm_util Checking that an address range is backed by a folio of a given order is useful to any test that builds or collapses large folios. mTHP collapse coverage in the khugepaged selftest needs exactly that. split_huge_page_test.c already has the building block: is_backed_by_folio() reads the compound head and tail flags from /proc/kpageflags to classify the folio behind a page. Move it into vm_util so other tests can use it. No functional change. Link: https://lore.kernel.org/20260919002451.496763-8-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: Baolin Wang Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Michal Hocko Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- .../selftests/mm/split_huge_page_test.c | 61 ------------------- tools/testing/selftests/mm/vm_util.c | 61 +++++++++++++++++++ tools/testing/selftests/mm/vm_util.h | 2 + 3 files changed, 63 insertions(+), 61 deletions(-) diff --git a/tools/testing/selftests/mm/split_huge_page_test.c b/tools/testing/selftests/mm/split_huge_page_test.c index 68f508c9a355fc..c5d96a4b1db355 100644 --- a/tools/testing/selftests/mm/split_huge_page_test.c +++ b/tools/testing/selftests/mm/split_huge_page_test.c @@ -41,67 +41,6 @@ const char *kpageflags_proc = "/proc/kpageflags"; int pagemap_fd; int kpageflags_fd; -static bool is_backed_by_folio(char *vaddr, int order, int pagemap_fd, - int kpageflags_fd) -{ - const uint64_t folio_head_flags = KPF_THP | KPF_COMPOUND_HEAD; - const uint64_t folio_tail_flags = KPF_THP | KPF_COMPOUND_TAIL; - const unsigned long nr_pages = 1UL << order; - unsigned long pfn_head; - uint64_t pfn_flags; - unsigned long pfn; - unsigned long i; - - pfn = pagemap_get_pfn(pagemap_fd, vaddr); - - /* non present page */ - if (pfn == -1UL) - return false; - - if (pageflags_get(pfn, kpageflags_fd, &pfn_flags)) - goto fail; - - /* check for order-0 pages */ - if (!order) { - if (pfn_flags & (folio_head_flags | folio_tail_flags)) - return false; - return true; - } - - /* non THP folio */ - if (!(pfn_flags & KPF_THP)) - return false; - - pfn_head = pfn & ~(nr_pages - 1); - - if (pageflags_get(pfn_head, kpageflags_fd, &pfn_flags)) - goto fail; - - /* head PFN has no compound_head flag set */ - if ((pfn_flags & folio_head_flags) != folio_head_flags) - return false; - - /* check all tail PFN flags */ - for (i = 1; i < nr_pages; i++) { - if (pageflags_get(pfn_head + i, kpageflags_fd, &pfn_flags)) - goto fail; - if ((pfn_flags & folio_tail_flags) != folio_tail_flags) - return false; - } - - /* - * check the PFN after this folio, but if its flags cannot be obtained, - * assume this folio has the expected order - */ - if (pageflags_get(pfn_head + nr_pages, kpageflags_fd, &pfn_flags)) - return true; - - /* If we find another tail page, then the folio is larger. */ - return (pfn_flags & folio_tail_flags) != folio_tail_flags; -fail: - ksft_exit_fail_msg("Failed to get folio info\n"); -} - static int check_after_split_folio_orders(char *vaddr_start, size_t len, int pagemap_fd, int kpageflags_fd, int orders[], int nr_orders) { diff --git a/tools/testing/selftests/mm/vm_util.c b/tools/testing/selftests/mm/vm_util.c index f8916b2abc2efb..d2c5a20d724a29 100644 --- a/tools/testing/selftests/mm/vm_util.c +++ b/tools/testing/selftests/mm/vm_util.c @@ -490,6 +490,67 @@ int pageflags_get(unsigned long pfn, int kpageflags_fd, uint64_t *flags) return 0; } +bool is_backed_by_folio(char *vaddr, int order, int pagemap_fd, + int kpageflags_fd) +{ + const uint64_t folio_head_flags = KPF_THP | KPF_COMPOUND_HEAD; + const uint64_t folio_tail_flags = KPF_THP | KPF_COMPOUND_TAIL; + const unsigned long nr_pages = 1UL << order; + unsigned long pfn_head; + uint64_t pfn_flags; + unsigned long pfn; + unsigned long i; + + pfn = pagemap_get_pfn(pagemap_fd, vaddr); + + /* non present page */ + if (pfn == -1UL) + return false; + + if (pageflags_get(pfn, kpageflags_fd, &pfn_flags)) + goto fail; + + /* check for order-0 pages */ + if (!order) { + if (pfn_flags & (folio_head_flags | folio_tail_flags)) + return false; + return true; + } + + /* non THP folio */ + if (!(pfn_flags & KPF_THP)) + return false; + + pfn_head = pfn & ~(nr_pages - 1); + + if (pageflags_get(pfn_head, kpageflags_fd, &pfn_flags)) + goto fail; + + /* head PFN has no compound_head flag set */ + if ((pfn_flags & folio_head_flags) != folio_head_flags) + return false; + + /* check all tail PFN flags */ + for (i = 1; i < nr_pages; i++) { + if (pageflags_get(pfn_head + i, kpageflags_fd, &pfn_flags)) + goto fail; + if ((pfn_flags & folio_tail_flags) != folio_tail_flags) + return false; + } + + /* + * check the PFN after this folio, but if its flags cannot be obtained, + * assume this folio has the expected order + */ + if (pageflags_get(pfn_head + nr_pages, kpageflags_fd, &pfn_flags)) + return true; + + /* If we find another tail page, then the folio is larger. */ + return (pfn_flags & folio_tail_flags) != folio_tail_flags; +fail: + ksft_exit_fail_msg("Failed to get folio info\n"); +} + /* If `ioctls' non-NULL, the allowed ioctls will be returned into the var */ int uffd_register_with_ioctls(int uffd, void *addr, uint64_t len, bool miss, bool wp, bool minor, uint64_t *ioctls) diff --git a/tools/testing/selftests/mm/vm_util.h b/tools/testing/selftests/mm/vm_util.h index 64a86e8a0c41bf..f12979a70135c6 100644 --- a/tools/testing/selftests/mm/vm_util.h +++ b/tools/testing/selftests/mm/vm_util.h @@ -99,6 +99,8 @@ int64_t allocate_transhuge(void *ptr, int pagemap_fd); int pageflags_get(unsigned long pfn, int kpageflags_fd, uint64_t *flags); int gather_folio_orders(char *vaddr_start, size_t len, int pagemap_fd, int kpageflags_fd, int orders[], int nr_orders); +bool is_backed_by_folio(char *vaddr, int order, int pagemap_fd, + int kpageflags_fd); int uffd_register(int uffd, void *addr, uint64_t len, bool miss, bool wp, bool minor); From 621e78bbdcab55dfdeae8682feecd65503f5cef7 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:38 +0100 Subject: [PATCH 1114/1352] selftests/mm: add folio-order check for address ranges An mTHP collapse test needs to know that a range is backed by folios of the target order, and that they sit where a collapse would put them. Nothing answers both: is_backed_by_folio() classifies the folio behind a single page, and check_huge_anon() counts the folios of an order in a range without saying where they start. Add is_range_backed_by_order(). It requires every folio-sized, folio- aligned part of the range to map one folio of that order, head to tail, with the head at the start of the part. A part backed by two smaller folios fails, and so does a folio mapped off its natural alignment. The mTHP cases need both to tell a collapsed range from the one beside it. Link: https://lore.kernel.org/20260919002451.496763-9-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Reviewed-by: Mike Rapoport (Microsoft) Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Baolin Wang Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/vm_util.c | 47 ++++++++++++++++++++++++++++ tools/testing/selftests/mm/vm_util.h | 2 ++ 2 files changed, 49 insertions(+) diff --git a/tools/testing/selftests/mm/vm_util.c b/tools/testing/selftests/mm/vm_util.c index d2c5a20d724a29..1f88330fd0274a 100644 --- a/tools/testing/selftests/mm/vm_util.c +++ b/tools/testing/selftests/mm/vm_util.c @@ -551,6 +551,53 @@ bool is_backed_by_folio(char *vaddr, int order, int pagemap_fd, ksft_exit_fail_msg("Failed to get folio info\n"); } +/** + * is_range_backed_by_order() - check that a range is backed by @order folios + * @start: start of the range, a multiple of the folio size + * @len: length of the range in bytes, a multiple of the folio size + * @order: the folio order to check for + * @pagemap_fd: open /proc//pagemap of the range's owner + * @kpageflags_fd: open /proc/kpageflags + * + * Every folio-sized, folio-aligned part of the range must map one folio of + * @order, head to tail, with the head at the start of the part. A part + * backed by several smaller folios fails, and so does a folio mapped off + * its natural alignment. + * + * Returns: true if the whole range is backed that way, false otherwise. + */ +bool is_range_backed_by_order(char *start, size_t len, int order, + int pagemap_fd, int kpageflags_fd) +{ + const unsigned long nr_pages = 1UL << order; + const size_t folio_size = nr_pages * psize(); + char *vaddr; + + if ((uintptr_t)start % folio_size || len % folio_size) + return false; + + for (vaddr = start; vaddr < start + len; vaddr += folio_size) { + const unsigned long pfn = pagemap_get_pfn(pagemap_fd, vaddr); + unsigned long i; + + /* Not present, or a tail page */ + if (pfn == -1UL || pfn % nr_pages) + return false; + + for (i = 1; i < nr_pages; i++) { + char *page = vaddr + i * psize(); + + if (pagemap_get_pfn(pagemap_fd, page) != pfn + i) + return false; + } + + if (!is_backed_by_folio(vaddr, order, pagemap_fd, kpageflags_fd)) + return false; + } + + return true; +} + /* If `ioctls' non-NULL, the allowed ioctls will be returned into the var */ int uffd_register_with_ioctls(int uffd, void *addr, uint64_t len, bool miss, bool wp, bool minor, uint64_t *ioctls) diff --git a/tools/testing/selftests/mm/vm_util.h b/tools/testing/selftests/mm/vm_util.h index f12979a70135c6..0172003c16dcb9 100644 --- a/tools/testing/selftests/mm/vm_util.h +++ b/tools/testing/selftests/mm/vm_util.h @@ -101,6 +101,8 @@ int gather_folio_orders(char *vaddr_start, size_t len, int pagemap_fd, int kpageflags_fd, int orders[], int nr_orders); bool is_backed_by_folio(char *vaddr, int order, int pagemap_fd, int kpageflags_fd); +bool is_range_backed_by_order(char *start, size_t len, int order, + int pagemap_fd, int kpageflags_fd); int uffd_register(int uffd, void *addr, uint64_t len, bool miss, bool wp, bool minor); From edbad6bd134a8cb557cd8c365334bcdc074a3f62 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:39 +0100 Subject: [PATCH 1115/1352] selftests/mm: add folio-order detection self-check The khugepaged mTHP tests detect collapse results with the vm_util folio-order helpers rather than smaps AnonHugePages, which only sees PMD mappings. If those helpers are wrong, every case built on them is wrong the same way, and nothing says so. Check them directly. For every anon THP order the kernel supports, fault memory in with only that order enabled. Require the helpers to classify the backing as exactly that order: not the order below it, and base-page memory as order 0. Run it in the thp category, ahead of ./khugepaged, so a broken helper is reported as itself rather than as a collapse failure. Verified on x86-64 4K (orders 0, 2-9) and arm64 64K (orders 0, 2-13). The test needs ALIGN(), which hmm-tests.c and migration.c each defined privately. Move it to vm_util.h and drop both copies. Link: https://lore.kernel.org/20260919002451.496763-10-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Baolin Wang Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/Makefile | 1 + .../testing/selftests/mm/folio_order_check.c | 122 ++++++++++++++++++ tools/testing/selftests/mm/hmm-tests.c | 1 - tools/testing/selftests/mm/migration.c | 1 - tools/testing/selftests/mm/run_vmtests.sh | 2 + tools/testing/selftests/mm/vm_util.h | 2 + 6 files changed, 127 insertions(+), 2 deletions(-) create mode 100644 tools/testing/selftests/mm/folio_order_check.c diff --git a/tools/testing/selftests/mm/Makefile b/tools/testing/selftests/mm/Makefile index 7d69baeb93f430..cb32cf5d867e25 100644 --- a/tools/testing/selftests/mm/Makefile +++ b/tools/testing/selftests/mm/Makefile @@ -105,6 +105,7 @@ TEST_GEN_FILES += guard-regions TEST_GEN_FILES += merge TEST_GEN_FILES += rmap TEST_GEN_FILES += folio_split_race_test +TEST_GEN_FILES += folio_order_check TEST_GEN_FILES += soft-dirty ifeq ($(ARCH),x86_64) diff --git a/tools/testing/selftests/mm/folio_order_check.c b/tools/testing/selftests/mm/folio_order_check.c new file mode 100644 index 00000000000000..5eafbcc1b4f3c5 --- /dev/null +++ b/tools/testing/selftests/mm/folio_order_check.c @@ -0,0 +1,122 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Self-check for the vm_util folio-order helpers, is_backed_by_folio() and + * is_range_backed_by_order(), which the khugepaged mTHP cases use to detect + * collapse results. For every anon THP order the kernel supports, fault + * memory in with only that order enabled and require the helpers to report + * exactly that order. + */ +#define _GNU_SOURCE +#include +#include +#include +#include +#include + +#include "kselftest.h" +#include "vm_util.h" +#include + +static int pagemap_fd; +static int kpageflags_fd; + +static char *alloc_aligned(size_t size) +{ + size_t len = size * 2; + char *p, *aligned; + + p = mmap(NULL, len, PROT_READ | PROT_WRITE, + MAP_ANONYMOUS | MAP_PRIVATE, -1, 0); + if (p == MAP_FAILED) + ksft_exit_fail_perror("mmap()"); + + aligned = (char *)ALIGN((uintptr_t)p, size); + if (aligned != p) + munmap(p, aligned - p); + if (aligned + size != p + len) + munmap(aligned + size, p + len - aligned - size); + + return aligned; +} + +static void check_order(int order) +{ + struct thp_settings settings = *thp_current_settings(); + size_t size = psize() << order; + bool ok = true; + char *p; + int i; + + for (i = 0; i < NR_ORDERS; i++) + settings.hugepages[i].enabled = THP_NEVER; + if (order) + settings.hugepages[order].enabled = THP_ALWAYS; + thp_push_settings(&settings); + + p = alloc_aligned(size); + *p = 1; + + if (!is_range_backed_by_order(p, size, order, pagemap_fd, kpageflags_fd)) { + ksft_print_msg("order %d not detected after fault\n", order); + ok = false; + } + + /* A lower order must be rejected: the folio is larger */ + if (order && is_range_backed_by_order(p, size, order - 1, + pagemap_fd, kpageflags_fd)) { + ksft_print_msg("order %d also reported as order %d\n", + order, order - 1); + ok = false; + } + + /* A large folio must not pass as order 0 */ + if (order && is_range_backed_by_order(p, size, 0, + pagemap_fd, kpageflags_fd)) { + ksft_print_msg("order %d also reported as order 0\n", order); + ok = false; + } + + munmap(p, size); + thp_pop_settings(); + + ksft_test_result(ok, "order %d classified\n", order); +} + +int main(void) +{ + struct thp_settings settings; + unsigned long orders; + int order; + + ksft_print_header(); + + if (!thp_available()) + ksft_exit_skip("Transparent Hugepages not available\n"); + + pagemap_fd = open("/proc/self/pagemap", O_RDONLY); + if (pagemap_fd < 0) + ksft_exit_fail_perror("open(/proc/self/pagemap)"); + kpageflags_fd = open("/proc/kpageflags", O_RDONLY); + if (kpageflags_fd < 0) + ksft_exit_skip("open(/proc/kpageflags) requires root\n"); + + orders = thp_supported_orders(); + if (!orders) + ksft_exit_skip("No supported THP orders\n"); + + ksft_set_plan(__builtin_popcountl(orders) + 1); + + thp_save_settings(); + thp_read_settings(&settings); + /* Base of the settings stack; the bottom entry is never popped */ + thp_push_settings(&settings); + + check_order(0); + for (order = 1; order < NR_ORDERS; order++) { + if (!(orders & (1UL << order))) + continue; + check_order(order); + } + + ksft_finished(); +} diff --git a/tools/testing/selftests/mm/hmm-tests.c b/tools/testing/selftests/mm/hmm-tests.c index fa1a651963fd01..e5f273ca84c107 100644 --- a/tools/testing/selftests/mm/hmm-tests.c +++ b/tools/testing/selftests/mm/hmm-tests.c @@ -65,7 +65,6 @@ enum { #define HMM_PATH_MAX 64 #define NTIMES 10 -#define ALIGN(x, a) (((x) + (a - 1)) & (~((a) - 1))) /* Just the flags we need, copied from mm.h: */ #ifndef FOLL_WRITE diff --git a/tools/testing/selftests/mm/migration.c b/tools/testing/selftests/mm/migration.c index a35e2b57e05b2d..d1d0989ed2cada 100644 --- a/tools/testing/selftests/mm/migration.c +++ b/tools/testing/selftests/mm/migration.c @@ -20,7 +20,6 @@ #define TWOMEG (2<<20) #define RUNTIME (20) -#define ALIGN(x, a) (((x) + (a - 1)) & (~((a) - 1))) HUGETLB_SETUP_DEFAULT_PAGES(1) diff --git a/tools/testing/selftests/mm/run_vmtests.sh b/tools/testing/selftests/mm/run_vmtests.sh index 0d8c93087d0529..6b80cf2eab149c 100755 --- a/tools/testing/selftests/mm/run_vmtests.sh +++ b/tools/testing/selftests/mm/run_vmtests.sh @@ -382,6 +382,8 @@ CATEGORY="pfnmap" run_test ./pfnmap # COW tests CATEGORY="cow" run_test ./cow +CATEGORY="thp" run_test ./folio_order_check + CATEGORY="thp" run_test ./khugepaged CATEGORY="thp" run_test ./khugepaged -s 2 diff --git a/tools/testing/selftests/mm/vm_util.h b/tools/testing/selftests/mm/vm_util.h index 0172003c16dcb9..ea48e6a7527e13 100644 --- a/tools/testing/selftests/mm/vm_util.h +++ b/tools/testing/selftests/mm/vm_util.h @@ -12,6 +12,8 @@ #include #define BIT_ULL(nr) (1ULL << (nr)) +#define ALIGN(x, a) (((x) + (a) - 1) & ~((a) - 1)) + #define PM_SOFT_DIRTY BIT_ULL(55) #define PM_MMAP_EXCLUSIVE BIT_ULL(56) #define PM_UFFD_WP BIT_ULL(57) From 4f6b510d3c65aa4c88814f85da3c3f22de5dab0b Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:40 +0100 Subject: [PATCH 1116/1352] selftests/mm: add khugepaged completion barrier helper A khugepaged test has to tell "not collapsed" from "not scanned yet", and nothing in the selftests can. wait_for_scan() in khugepaged.c comes closest: it polls full_scans until the counter has advanced by two, since the pass in progress may already have passed the test's mm. But it only returns in time if scan_sleep_millisecs happens to be short, and it is private to that one test. Add khugepaged_full_pass() to hugepage_settings, built on the same advance-by-two wait but driven through sysfs: a store to scan_sleep_millisecs wakes the daemon, so the barrier completes whatever the scan cadence. A store made while the daemon is scanning rather than sleeping is lost, so the helper keeps storing until the pass lands. One wake completes one pass only if pages_to_scan covers every mm on the list, so callers need it large. Settings pushes must not start passes of their own. A store to either sleep knob wakes the daemon, so thp_write_settings() now writes a khugepaged knob only when its value changes. Link: https://lore.kernel.org/20260919002451.496763-11-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Reviewed-by: Baolin Wang Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/lib/mm/hugepage_settings.c | 60 +++++++++++++++++++++++++++----- tools/lib/mm/hugepage_settings.h | 2 ++ 2 files changed, 53 insertions(+), 9 deletions(-) diff --git a/tools/lib/mm/hugepage_settings.c b/tools/lib/mm/hugepage_settings.c index 656442c8d3954a..77918677a9cd02 100644 --- a/tools/lib/mm/hugepage_settings.c +++ b/tools/lib/mm/hugepage_settings.c @@ -217,6 +217,13 @@ void thp_read_settings(struct thp_settings *settings) } } +/* A store to either sleep knob wakes khugepaged, so write only on change */ +static void thp_update_num(const char *name, unsigned long num) +{ + if (thp_read_num(name) != num) + thp_write_num(name, num); +} + void thp_write_settings(struct thp_settings *settings) { struct khugepaged_settings *khugepaged = &settings->khugepaged; @@ -232,15 +239,15 @@ void thp_write_settings(struct thp_settings *settings) shmem_enabled_strings[settings->shmem_enabled]); thp_write_num("use_zero_page", settings->use_zero_page); - thp_write_num("khugepaged/defrag", khugepaged->defrag); - thp_write_num("khugepaged/alloc_sleep_millisecs", - khugepaged->alloc_sleep_millisecs); - thp_write_num("khugepaged/scan_sleep_millisecs", - khugepaged->scan_sleep_millisecs); - thp_write_num("khugepaged/max_ptes_none", khugepaged->max_ptes_none); - thp_write_num("khugepaged/max_ptes_swap", khugepaged->max_ptes_swap); - thp_write_num("khugepaged/max_ptes_shared", khugepaged->max_ptes_shared); - thp_write_num("khugepaged/pages_to_scan", khugepaged->pages_to_scan); + thp_update_num("khugepaged/defrag", khugepaged->defrag); + thp_update_num("khugepaged/alloc_sleep_millisecs", + khugepaged->alloc_sleep_millisecs); + thp_update_num("khugepaged/scan_sleep_millisecs", + khugepaged->scan_sleep_millisecs); + thp_update_num("khugepaged/max_ptes_none", khugepaged->max_ptes_none); + thp_update_num("khugepaged/max_ptes_swap", khugepaged->max_ptes_swap); + thp_update_num("khugepaged/max_ptes_shared", khugepaged->max_ptes_shared); + thp_update_num("khugepaged/pages_to_scan", khugepaged->pages_to_scan); if (dev_queue_read_ahead_path[0]) { int ret = write_num(dev_queue_read_ahead_path, @@ -271,6 +278,41 @@ void thp_write_settings(struct thp_settings *settings) } } +/* + * Wait for a full khugepaged scan pass that started after this call: the + * pass in progress may already have passed this mm, so full_scans has to + * advance twice. + * + * A store to scan_sleep_millisecs wakes the daemon, but one made while it + * is scanning rather than sleeping is lost, so keep storing until the pass + * lands. + * + * One wake is one pass only if pages_to_scan covers every mm on the list. + */ +bool khugepaged_full_pass(unsigned int timeout_s) +{ + unsigned long deadline_ms = timeout_s * 1000UL; + unsigned long elapsed_ms = 0, poll_ms = 10; + unsigned long sleep_ms; + int pass; + + sleep_ms = thp_read_num("khugepaged/scan_sleep_millisecs"); + for (pass = 0; pass < 2; pass++) { + unsigned long target = + thp_read_num("khugepaged/full_scans") + 1; + + while (thp_read_num("khugepaged/full_scans") < target) { + if (elapsed_ms >= deadline_ms) + return false; + thp_write_num("khugepaged/scan_sleep_millisecs", + sleep_ms); + usleep(poll_ms * 1000); + elapsed_ms += poll_ms; + } + } + return true; +} + struct thp_settings *thp_current_settings(void) { if (!settings_index) { diff --git a/tools/lib/mm/hugepage_settings.h b/tools/lib/mm/hugepage_settings.h index 94d9fc747f4979..8f4581099b7aba 100644 --- a/tools/lib/mm/hugepage_settings.h +++ b/tools/lib/mm/hugepage_settings.h @@ -83,6 +83,8 @@ static inline void thp_save_settings(void) hugepage_save_settings(/* thp = */ true, /* hugetlb = */ false); } +bool khugepaged_full_pass(unsigned int timeout_s); + void thp_set_read_ahead_path(char *path); unsigned long thp_supported_orders(void); unsigned long thp_shmem_supported_orders(void); From 7d01ec23e2447e002c523a58cb5906abfd7feb17 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:41 +0100 Subject: [PATCH 1117/1352] selftests/mm: add order-parameterized khugepaged collapse cases The mthp_khugepaged context runs the generic cases at a sub-PMD order, which answers how many folios of that order a range ends up with. It cannot say which order-sized window they landed in, so "the populated window collapsed" and "the empty window next to it collapsed instead" look alike. Add four cases that check each window on its own, with the folio-order helpers in vm_util: - collapse_order_single_window(): only the populated window collapses; - collapse_order_partial_window(): the default max_ptes_none lets a window with one present PTE collapse; - collapse_order_max_ptes_none(): with max_ptes_none=0 a full window collapses and one missing a page does not; - collapse_order_mixed_sources(): sources that are already large folios of a smaller order collapse to the target. Each case faults its region before MADV_HUGEPAGE with only the target order enabled, so the sources are order 0 and the result can only come from khugepaged. They wait for a full pass rather than for the result to appear: without a completed pass, "not collapsed" and "not scanned yet" are the same thing. Link: https://lore.kernel.org/20260919002451.496763-12-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Tested-by: Muhammad Usama Anjum Tested-by: Baolin Wang Assisted-by: LLM Cc: Alexander Gordeev Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/khugepaged.c | 212 ++++++++++++++++++++++++ 1 file changed, 212 insertions(+) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index e013eebc7136ed..51bda01446cd7e 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -30,6 +30,8 @@ static unsigned long page_size; static int hpage_pmd_nr; static int anon_order; static int collapse_order; +static int pagemap_fd = -1; +static int kpageflags_fd = -1; #define PID_SMAPS "/proc/self/smaps" #define TEST_FILE "collapse_test_file" @@ -1209,6 +1211,198 @@ static void madvise_retracted_page_tables(struct collapse_context *c, ksft_test_result_report(exit_status, "%s\n", __func__); } +/* Smallest order khugepaged will consider for mTHP collapse */ +#define MIN_MTHP_ORDER 2 + +/* Time budget for one khugepaged pass in the collapse_order_* cases */ +#define MTHP_PASS_TIMEOUT_S 30 + +static size_t mthp_window_size(void) +{ + return page_size << collapse_order; +} + +static void mthp_push_target_order(void) +{ + struct thp_settings settings = *thp_current_settings(); + int i; + + /* + * Only the target order, and only for madvise: the cases fault their + * region first, so the sources stay order 0 whatever -s asked for. + */ + settings.thp_enabled = THP_NEVER; + for (i = 0; i < NR_ORDERS; i++) + settings.hugepages[i].enabled = THP_NEVER; + settings.hugepages[collapse_order].enabled = THP_MADVISE; + thp_push_settings(&settings); +} + +static bool all_windows_at_order(void *p, size_t len) +{ + return is_range_backed_by_order(p, len, collapse_order, + pagemap_fd, kpageflags_fd); +} + +static bool any_window_at_order(void *p, size_t len) +{ + size_t window = mthp_window_size(); + char *addr = p; + + for (; len >= window; addr += window, len -= window) { + if (all_windows_at_order(addr, window)) + return true; + } + return false; +} + +static void collapse_order_single_window(struct collapse_context *c, + struct mem_ops *ops) +{ + size_t window = mthp_window_size(); + void *p; + + mthp_push_target_order(); + + p = ops->setup_area(1); + ops->fault(p, window, 2 * window); + if (any_window_at_order(p, hpage_pmd_size)) + ksft_exit_fail_msg("Unexpected large folio after fault\n"); + + if (madvise(p, hpage_pmd_size, MADV_HUGEPAGE)) + ksft_exit_fail_perror("madvise(MADV_HUGEPAGE)"); + ksft_print_msg("Collapse one fully populated window..."); + if (!khugepaged_full_pass(MTHP_PASS_TIMEOUT_S)) + fail("Timeout"); + else if (all_windows_at_order(p + window, window) && + !any_window_at_order(p, window) && + !any_window_at_order(p + 2 * window, + hpage_pmd_size - 2 * window)) + success("OK"); + else + fail("Fail"); + + validate_memory(p, window, 2 * window); + ops->cleanup_area(p, hpage_pmd_size); + thp_pop_settings(); + ksft_test_result_report(exit_status, "%s\n", __func__); +} + +static void collapse_order_partial_window(struct collapse_context *c, + struct mem_ops *ops) +{ + void *p; + + mthp_push_target_order(); + + p = ops->setup_area(1); + ops->fault(p, 0, page_size); + if (any_window_at_order(p, hpage_pmd_size)) + ksft_exit_fail_msg("Unexpected large folio after fault\n"); + + if (madvise(p, hpage_pmd_size, MADV_HUGEPAGE)) + ksft_exit_fail_perror("madvise(MADV_HUGEPAGE)"); + ksft_print_msg("Collapse window with single PTE entry present..."); + if (!khugepaged_full_pass(MTHP_PASS_TIMEOUT_S)) + fail("Timeout"); + else if (all_windows_at_order(p, mthp_window_size())) + success("OK"); + else + fail("Fail"); + + validate_memory(p, 0, page_size); + ops->cleanup_area(p, hpage_pmd_size); + thp_pop_settings(); + ksft_test_result_report(exit_status, "%s\n", __func__); +} + +static void collapse_order_max_ptes_none(struct collapse_context *c, + struct mem_ops *ops) +{ + struct thp_settings settings; + size_t window = mthp_window_size(); + void *p; + + mthp_push_target_order(); + settings = *thp_current_settings(); + settings.khugepaged.max_ptes_none = 0; + thp_push_settings(&settings); + + p = ops->setup_area(1); + ops->fault(p, 0, 2 * window - page_size); + if (any_window_at_order(p, hpage_pmd_size)) + ksft_exit_fail_msg("Unexpected large folio after fault\n"); + + if (madvise(p, hpage_pmd_size, MADV_HUGEPAGE)) + ksft_exit_fail_perror("madvise(MADV_HUGEPAGE)"); + ksft_print_msg("Collapse full window, not the one missing a page..."); + if (!khugepaged_full_pass(MTHP_PASS_TIMEOUT_S)) + fail("Timeout"); + else if (all_windows_at_order(p, window) && + !any_window_at_order(p + window, window)) + success("OK"); + else + fail("Fail"); + + validate_memory(p, 0, 2 * window - page_size); + ops->cleanup_area(p, hpage_pmd_size); + thp_pop_settings(); + thp_pop_settings(); + ksft_test_result_report(exit_status, "%s\n", __func__); +} + +static void collapse_order_mixed_sources(struct collapse_context *c, + struct mem_ops *ops) +{ + struct thp_settings settings; + void *p; + + if (collapse_order <= MIN_MTHP_ORDER) { + ksft_test_result_skip("%s: no source order below target\n", + __func__); + return; + } + + mthp_push_target_order(); + + settings = *thp_current_settings(); + settings.hugepages[MIN_MTHP_ORDER].enabled = THP_ALWAYS; + thp_push_settings(&settings); + p = ops->setup_area(1); + ops->fault(p, 0, hpage_pmd_size); + thp_pop_settings(); + + /* + * The allocator can fall back to smaller folios under fragmentation; + * having nothing to collapse from is not a failure. + */ + if (!is_range_backed_by_order(p, hpage_pmd_size, MIN_MTHP_ORDER, + pagemap_fd, kpageflags_fd)) { + ksft_print_msg("No order-%d sources to collapse...", + MIN_MTHP_ORDER); + skip("Skip"); + ops->cleanup_area(p, hpage_pmd_size); + thp_pop_settings(); + ksft_test_result_report(exit_status, "%s\n", __func__); + return; + } + + if (madvise(p, hpage_pmd_size, MADV_HUGEPAGE)) + ksft_exit_fail_perror("madvise(MADV_HUGEPAGE)"); + ksft_print_msg("Collapse region backed by smaller large folios..."); + if (!khugepaged_full_pass(MTHP_PASS_TIMEOUT_S)) + fail("Timeout"); + else if (all_windows_at_order(p, hpage_pmd_size)) + success("OK"); + else + fail("Fail"); + + validate_memory(p, 0, hpage_pmd_size); + ops->cleanup_area(p, hpage_pmd_size); + thp_pop_settings(); + ksft_test_result_report(exit_status, "%s\n", __func__); +} + static void usage(void) { fprintf(stderr, "\nUsage: ./khugepaged [OPTIONS] [dir]\n\n"); @@ -1377,6 +1571,20 @@ int main(int argc, char **argv) parse_test_type(argc, argv); + if (mthp_khugepaged_context && + !(thp_supported_orders() & (1UL << collapse_order))) + ksft_exit_skip("Order %d is not a supported anon THP order\n", + collapse_order); + + if (mthp_khugepaged_context) { + pagemap_fd = open("/proc/self/pagemap", O_RDONLY); + if (pagemap_fd < 0) + ksft_exit_fail_perror("open(/proc/self/pagemap)"); + kpageflags_fd = open("/proc/kpageflags", O_RDONLY); + if (kpageflags_fd < 0) + ksft_exit_fail_perror("open(/proc/kpageflags)"); + } + setbuf(stdout, NULL); /* @@ -1427,6 +1635,10 @@ int main(int argc, char **argv) TEST(collapse_empty, madvise_context, anon_ops); TEST(collapse_single_mthp, mthp_khugepaged_context, anon_ops); + TEST(collapse_order_single_window, mthp_khugepaged_context, anon_ops); + TEST(collapse_order_partial_window, mthp_khugepaged_context, anon_ops); + TEST(collapse_order_max_ptes_none, mthp_khugepaged_context, anon_ops); + TEST(collapse_order_mixed_sources, mthp_khugepaged_context, anon_ops); TEST(collapse_single_pte_entry, khugepaged_context, anon_ops); TEST(collapse_single_pte_entry, khugepaged_context, read_only_file_ops); From a609dc5474e3196eba3bb38449a244de90f27f72 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:42 +0100 Subject: [PATCH 1118/1352] selftests/mm: parameterize the mixed-source collapse case by source order collapse_order_mixed_sources() faults its region as order-2 folios and collapses them to the -c target. Order 2 is below the contpte size on every arm64 page size, so nothing in this suite collapses a contpte-mapped source on purpose. Let -s name the source order alongside -c. The case then faults at that order, keeping order 2 when -s is absent, and the source order has to be a supported mTHP order below the target. The other mTHP cases are unaffected: mthp_push_target_order() enables only the target order. A -c at or below -s is refused before any case runs: the sources would already be the size being asked for. Without the check the generic cases fail on that one by one instead of saying why. "-s 5 -c 7" on arm64/64K then collapses contpte-mapped sources into a larger mTHP. Link: https://lore.kernel.org/20260919002451.496763-13-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: Lorenzo Stoakes (ARM) Tested-by: Muhammad Usama Anjum Reviewed-by: Baolin Wang Tested-by: Baolin Wang Assisted-by: LLM Cc: Alexander Gordeev Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/khugepaged.c | 20 +++++++++++++------- 1 file changed, 13 insertions(+), 7 deletions(-) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 51bda01446cd7e..9d1c47bd501349 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -1354,11 +1354,13 @@ static void collapse_order_max_ptes_none(struct collapse_context *c, static void collapse_order_mixed_sources(struct collapse_context *c, struct mem_ops *ops) { + int source_order = anon_order ? anon_order : MIN_MTHP_ORDER; struct thp_settings settings; void *p; - if (collapse_order <= MIN_MTHP_ORDER) { - ksft_test_result_skip("%s: no source order below target\n", + if (source_order >= collapse_order || + !(thp_supported_orders() & (1UL << source_order))) { + ksft_test_result_skip("%s: no supported source order below target\n", __func__); return; } @@ -1366,7 +1368,7 @@ static void collapse_order_mixed_sources(struct collapse_context *c, mthp_push_target_order(); settings = *thp_current_settings(); - settings.hugepages[MIN_MTHP_ORDER].enabled = THP_ALWAYS; + settings.hugepages[source_order].enabled = THP_ALWAYS; thp_push_settings(&settings); p = ops->setup_area(1); ops->fault(p, 0, hpage_pmd_size); @@ -1376,10 +1378,9 @@ static void collapse_order_mixed_sources(struct collapse_context *c, * The allocator can fall back to smaller folios under fragmentation; * having nothing to collapse from is not a failure. */ - if (!is_range_backed_by_order(p, hpage_pmd_size, MIN_MTHP_ORDER, + if (!is_range_backed_by_order(p, hpage_pmd_size, source_order, pagemap_fd, kpageflags_fd)) { - ksft_print_msg("No order-%d sources to collapse...", - MIN_MTHP_ORDER); + ksft_print_msg("No order-%d sources to collapse...", source_order); skip("Skip"); ops->cleanup_area(p, hpage_pmd_size); thp_pop_settings(); @@ -1389,7 +1390,8 @@ static void collapse_order_mixed_sources(struct collapse_context *c, if (madvise(p, hpage_pmd_size, MADV_HUGEPAGE)) ksft_exit_fail_perror("madvise(MADV_HUGEPAGE)"); - ksft_print_msg("Collapse region backed by smaller large folios..."); + ksft_print_msg("Collapse region backed by order-%d sources...", + source_order); if (!khugepaged_full_pass(MTHP_PASS_TIMEOUT_S)) fail("Timeout"); else if (all_windows_at_order(p, hpage_pmd_size)) @@ -1420,6 +1422,7 @@ static void usage(void) fprintf(stderr, "\t\t-s: mTHP size, expressed as page order.\n"); fprintf(stderr, "\t\t Defaults to 0. Use this size for anon or shmem allocations.\n"); fprintf(stderr, "\t\t-c: collapse order for mTHP collapse, expressed as page order.\n"); + fprintf(stderr, "\t\t -s, if set, is the source order for the mixed-source case.\n"); exit(1); } @@ -1575,6 +1578,9 @@ int main(int argc, char **argv) !(thp_supported_orders() & (1UL << collapse_order))) ksft_exit_skip("Order %d is not a supported anon THP order\n", collapse_order); + if (mthp_khugepaged_context && collapse_order <= anon_order) + ksft_exit_skip("-c %d needs a source order below it, -s says %d\n", + collapse_order, anon_order); if (mthp_khugepaged_context) { pagemap_fd = open("/proc/self/pagemap", O_RDONLY); From 88b11f1232271c5a9aa451d34a1c79c44e09d1b4 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:43 +0100 Subject: [PATCH 1119/1352] selftests/mm: cover a shared-source collapse write race collapse_fork() checks that a fork-shared range collapses in the child while the parent keeps its own pages, but the parent sits still while that happens. Nothing checks that CoW isolation survives a collapse racing with writes to the shared source. Add a case where the parent writes to the shared range throughout the child's collapse. CoW has to keep the two apart: the child must see the content from before the fork, and the parent only its own writes. The parent unshares one page every 10ms, starting only once the child says it is about to collapse. Writing the range in a burst would break CoW on all of it before the collapse begins, leaving the child to collapse pages that are already exclusive to it. Preparation for changing how collapse handles fork-shared sources. Link: https://lore.kernel.org/20260919002451.496763-14-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Baolin Wang Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/khugepaged.c | 106 ++++++++++++++++++++++++ 1 file changed, 106 insertions(+) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 9d1c47bd501349..21ae258bd56eb7 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -1166,6 +1166,109 @@ static void collapse_max_ptes_shared(struct collapse_context *c, struct mem_ops ksft_test_result_report(exit_status, "%s\n", __func__); } +/* + * The parent writes to the fork-shared range throughout the child's + * collapse. CoW must keep the two apart: the child sees the pre-fork + * content, the parent only its own writes. + */ +static void collapse_fork_cow_race(struct collapse_context *c, struct mem_ops *ops) +{ + const int stride = page_size / sizeof(int); + int wstatus, child_status, i, n; + unsigned long shared; + volatile int *ip; + pid_t child; + int sync[2]; + char go = 1; + void *p; + + /* At a page per 10 ms, 64 pages spread the writes across the collapse */ + n = 64; + shared = n * page_size; + + p = ops->setup_area(1); + /* Shared prefix, with the pre-fork pattern */ + ops->fault(p, 0, shared); + if (pipe(sync)) + ksft_exit_fail_perror("pipe()"); + + /* A volatile pointer so the stores are not merged or dropped */ + ip = p; + + ksft_print_msg("Fork, collapse in the child while the parent rewrites..."); + child = fork(); + if (!child) { + int collapse_status; + + close(sync[0]); + /* Private remainder */ + ops->fault(p, shared, hpage_pmd_size); + /* Start the parent unsharing, and give it a head start */ + if (write(sync[1], &go, 1) != 1) + _exit(KSFT_FAIL); + usleep(5000); + c->collapse("Collapse a range the parent is writing to", + p, 1, ops, true); + collapse_status = exit_status; + for (i = 0; i < n; i++) + if (ip[i * stride] != i + 0xdead0000) + break; + if (i == n) + success("OK"); + else + fail("Fail: child content"); + /* The content check must not bury a failed collapse */ + if (exit_status != KSFT_FAIL) + exit_status = collapse_status; + ops->cleanup_area(p, hpage_pmd_size); + _exit(exit_status); + } + + close(sync[1]); + if (read(sync[0], &go, 1) != 1) + ksft_exit_fail_msg("child never reached the collapse\n"); + close(sync[0]); + + /* + * Unshare one page at a time: a burst would break CoW on the whole + * range before the collapse starts, leaving nothing shared to collapse. + */ + i = 0; + for (;;) { + pid_t ret; + + if (i < n) + ip[i * stride] = i + 0xbeef0000; + i++; + usleep(10 * 1000); + ret = waitpid(child, &wstatus, WNOHANG); + if (ret == child) + break; + if (ret < 0) + ksft_exit_fail_perror("waitpid()"); + } + + /* Finish whatever the paced sweep did not reach */ + for (; i < n; i++) + ip[i * stride] = i + 0xbeef0000; + /* A child that died reading the racing pages is a failure, not a zero */ + child_status = WIFEXITED(wstatus) ? WEXITSTATUS(wstatus) : KSFT_FAIL; + + ksft_print_msg("Check the parent sees only its own writes..."); + for (i = 0; i < n; i++) + if (ip[i * stride] != i + 0xbeef0000) + break; + if (i == n) + success("OK"); + else + fail("Fail: parent content"); + ops->cleanup_area(p, hpage_pmd_size); + /* The parent's check must not bury the child's verdict */ + if (exit_status != KSFT_FAIL) + exit_status = child_status; + ksft_test_result_report(exit_status, "%s\n", __func__); +} + static void madvise_collapse_existing_thps(struct collapse_context *c, struct mem_ops *ops) { @@ -1700,6 +1803,9 @@ int main(int argc, char **argv) TEST(collapse_max_ptes_shared, khugepaged_context, anon_ops); TEST(collapse_max_ptes_shared, madvise_context, anon_ops); + TEST(collapse_fork_cow_race, khugepaged_context, anon_ops); + TEST(collapse_fork_cow_race, madvise_context, anon_ops); + TEST(madvise_collapse_existing_thps, madvise_context, anon_ops); TEST(madvise_collapse_existing_thps, madvise_context, read_only_file_ops); TEST(madvise_collapse_existing_thps, madvise_context, read_write_file_read_ops); From 7b5e6b89e455c502020049e50a43c2b9f75a4223 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:44 +0100 Subject: [PATCH 1120/1352] selftests/mm: run every supported collapse order by default The mTHP collapse cases only run when the caller names both the context and an order, so a plain ./khugepaged covers the PMD contexts on anon and nothing else. run_vmtests.sh pinned order 4 and covered no other. Run the mTHP cases once per supported anon THP order below the PMD when -c is absent, and pull that context into both the no-argument invocation and "all". Also: - -c still pins one order, and now says what is wrong instead of printing the usage text. - Both orders end up as array indices and shift counts, so -s and -c are range-checked before they get there. - The mTHP context has only anon cases, so a run that names a different mem_type -- "all:shmem", say -- drops it again rather than refusing to start. Naming both explicitly still refuses. - A case carries the order it was registered at, so a result names it: # Run test: collapse_single_mthp (mthp_khugepaged:anon, order 6) On x86-64 with 4K pages that is orders 2 through 8, and ./khugepaged goes from 29 results in 17 seconds to 77 in 29, so run_vmtests.sh can drop its pinned order-4 line. Link: https://lore.kernel.org/20260919002451.496763-15-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Reviewed-by: Baolin Wang Tested-by: Muhammad Usama Anjum Tested-by: Baolin Wang Assisted-by: LLM Cc: Alexander Gordeev Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/khugepaged.c | 101 +++++++++++++++++----- tools/testing/selftests/mm/run_vmtests.sh | 2 - 2 files changed, 78 insertions(+), 25 deletions(-) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 21ae258bd56eb7..a2ac3b3ca5def3 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -30,6 +30,9 @@ static unsigned long page_size; static int hpage_pmd_nr; static int anon_order; static int collapse_order; +static bool collapse_order_set; +static int collapse_orders[NR_ORDERS]; +static int nr_collapse_orders; static int pagemap_fd = -1; static int kpageflags_fd = -1; @@ -1525,12 +1528,14 @@ static void usage(void) fprintf(stderr, "\t\t-s: mTHP size, expressed as page order.\n"); fprintf(stderr, "\t\t Defaults to 0. Use this size for anon or shmem allocations.\n"); fprintf(stderr, "\t\t-c: collapse order for mTHP collapse, expressed as page order.\n"); + fprintf(stderr, "\t\t Defaults to every supported order below the PMD.\n"); fprintf(stderr, "\t\t -s, if set, is the source order for the mixed-source case.\n"); exit(1); } static void parse_test_type(int argc, char **argv) { + bool mthp_context_implied = false; int opt; char *buf; const char *token; @@ -1542,6 +1547,7 @@ static void parse_test_type(int argc, char **argv) break; case 'c': collapse_order = atoi(optarg); + collapse_order_set = true; break; case 'h': default: @@ -1549,12 +1555,25 @@ static void parse_test_type(int argc, char **argv) } } + /* + * Both orders end up as array indices and shift counts, so neither + * can be negative, and a zero collapse order asks for base pages. + */ + if (anon_order < 0 || anon_order > hpage_pmd_order) + ksft_exit_fail_msg("-s takes an order in 0..%d, not %d\n", + hpage_pmd_order, anon_order); + if (collapse_order_set && + (collapse_order <= 0 || collapse_order >= hpage_pmd_order)) + ksft_exit_fail_msg("-c takes an order in 1..%d, not %d\n", + hpage_pmd_order - 1, collapse_order); + argv += optind; argc -= optind; if (argc == 0) { - /* Backwards compatibility */ + /* No arguments: anon under every context */ khugepaged_context = &__khugepaged_context; + mthp_khugepaged_context = &__mthp_khugepaged_context; madvise_context = &__madvise_context; anon_ops = &__anon_ops; return; @@ -1565,13 +1584,14 @@ static void parse_test_type(int argc, char **argv) if (!strcmp(token, "all")) { khugepaged_context = &__khugepaged_context; + mthp_khugepaged_context = &__mthp_khugepaged_context; madvise_context = &__madvise_context; + /* The mTHP context has only anon cases; let other mem_types drop it */ + mthp_context_implied = true; } else if (!strcmp(token, "khugepaged")) { khugepaged_context = &__khugepaged_context; } else if (!strcmp(token, "mthp_khugepaged")) { mthp_khugepaged_context = &__mthp_khugepaged_context; - if (collapse_order <= 0 || collapse_order >= hpage_pmd_order) - usage(); } else if (!strcmp(token, "madvise")) { madvise_context = &__madvise_context; } else { @@ -1587,20 +1607,20 @@ static void parse_test_type(int argc, char **argv) read_write_file_write_ops = &__read_write_file_write_ops; anon_ops = &__anon_ops; shmem_ops = &__shmem_ops; - if (mthp_khugepaged_context) - usage(); } else if (!strcmp(buf, "anon")) { anon_ops = &__anon_ops; } else if (!strcmp(buf, "file")) { read_only_file_ops = &__read_only_file_ops; read_write_file_read_ops = &__read_write_file_read_ops; read_write_file_write_ops = &__read_write_file_write_ops; - if (mthp_khugepaged_context) + if (mthp_khugepaged_context && !mthp_context_implied) usage(); + mthp_khugepaged_context = NULL; } else if (!strcmp(buf, "shmem")) { shmem_ops = &__shmem_ops; - if (mthp_khugepaged_context) + if (mthp_khugepaged_context && !mthp_context_implied) usage(); + mthp_khugepaged_context = NULL; } else { usage(); } @@ -1622,6 +1642,7 @@ struct test_case { struct mem_ops *ops; const char *desc; test_fn fn; + int order; /* mTHP contexts: the collapse order */ }; #define MAX_TEST_CASES 256 @@ -1637,6 +1658,7 @@ static int nr_test_cases; .ops = o, \ .desc = #t, \ .fn = t, \ + .order = collapse_order, \ }; \ } \ } while (0) @@ -1677,13 +1699,35 @@ int main(int argc, char **argv) parse_test_type(argc, argv); - if (mthp_khugepaged_context && - !(thp_supported_orders() & (1UL << collapse_order))) - ksft_exit_skip("Order %d is not a supported anon THP order\n", - collapse_order); - if (mthp_khugepaged_context && collapse_order <= anon_order) - ksft_exit_skip("-c %d needs a source order below it, -s says %d\n", - collapse_order, anon_order); + if (mthp_khugepaged_context) { + unsigned long orders = thp_supported_orders(); + + if (collapse_order_set) { + if (!(orders & (1UL << collapse_order))) + ksft_exit_skip("Order %d is not a supported anon THP order\n", + collapse_order); + if (collapse_order <= anon_order) + ksft_exit_skip("-c %d needs a source order below it, -s says %d\n", + collapse_order, anon_order); + collapse_orders[nr_collapse_orders++] = collapse_order; + } else { + /* + * Every supported order above the source: -s makes the + * fault path hand out folios of that order, so a target + * at or below it has nothing to collapse. + */ + int first = anon_order + 1; + + if (first < MIN_MTHP_ORDER) + first = MIN_MTHP_ORDER; + for (int i = first; i < hpage_pmd_order; i++) { + if (orders & (1UL << i)) + collapse_orders[nr_collapse_orders++] = i; + } + if (!nr_collapse_orders) + ksft_print_msg("mTHP cases skipped: no order above the source\n"); + } + } if (mthp_khugepaged_context) { pagemap_fd = open("/proc/self/pagemap", O_RDONLY); @@ -1732,7 +1776,17 @@ int main(int argc, char **argv) TEST(collapse_full, khugepaged_context, read_write_file_read_ops); TEST(collapse_full, khugepaged_context, read_write_file_write_ops); TEST(collapse_full, khugepaged_context, shmem_ops); - TEST(collapse_full, mthp_khugepaged_context, anon_ops); + for (int i = 0; i < nr_collapse_orders; i++) { + collapse_order = collapse_orders[i]; + TEST(collapse_full, mthp_khugepaged_context, anon_ops); + TEST(collapse_empty, mthp_khugepaged_context, anon_ops); + TEST(collapse_single_mthp, mthp_khugepaged_context, anon_ops); + TEST(collapse_order_single_window, mthp_khugepaged_context, anon_ops); + TEST(collapse_order_partial_window, mthp_khugepaged_context, anon_ops); + TEST(collapse_order_max_ptes_none, mthp_khugepaged_context, anon_ops); + TEST(collapse_order_mixed_sources, mthp_khugepaged_context, anon_ops); + } + TEST(collapse_full, madvise_context, anon_ops); TEST(collapse_full, madvise_context, read_only_file_ops); TEST(collapse_full, madvise_context, read_write_file_read_ops); @@ -1740,15 +1794,8 @@ int main(int argc, char **argv) TEST(collapse_full, madvise_context, shmem_ops); TEST(collapse_empty, khugepaged_context, anon_ops); - TEST(collapse_empty, mthp_khugepaged_context, anon_ops); TEST(collapse_empty, madvise_context, anon_ops); - TEST(collapse_single_mthp, mthp_khugepaged_context, anon_ops); - TEST(collapse_order_single_window, mthp_khugepaged_context, anon_ops); - TEST(collapse_order_partial_window, mthp_khugepaged_context, anon_ops); - TEST(collapse_order_max_ptes_none, mthp_khugepaged_context, anon_ops); - TEST(collapse_order_mixed_sources, mthp_khugepaged_context, anon_ops); - TEST(collapse_single_pte_entry, khugepaged_context, anon_ops); TEST(collapse_single_pte_entry, khugepaged_context, read_only_file_ops); TEST(collapse_single_pte_entry, khugepaged_context, read_write_file_read_ops); @@ -1821,7 +1868,15 @@ int main(int argc, char **argv) for (int i = 0; i < nr_test_cases; i++) { struct test_case *t = &test_cases[i]; - ksft_print_msg("\n# Run test: %s (%s:%s)\n", t->desc, t->ctx->name, t->ops->name); + if (t->ctx == &__mthp_khugepaged_context) { + collapse_order = t->order; + ksft_print_msg("\n# Run test: %s (%s:%s, order %d)\n", + t->desc, t->ctx->name, t->ops->name, + t->order); + } else { + ksft_print_msg("\n# Run test: %s (%s:%s)\n", t->desc, + t->ctx->name, t->ops->name); + } t->fn(t->ctx, t->ops); } diff --git a/tools/testing/selftests/mm/run_vmtests.sh b/tools/testing/selftests/mm/run_vmtests.sh index 6b80cf2eab149c..f1138412788574 100755 --- a/tools/testing/selftests/mm/run_vmtests.sh +++ b/tools/testing/selftests/mm/run_vmtests.sh @@ -392,8 +392,6 @@ CATEGORY="thp" run_test ./khugepaged all:shmem CATEGORY="thp" run_test ./khugepaged -s 4 all:shmem -CATEGORY="thp" run_test ./khugepaged -c 4 mthp_khugepaged:anon - # Try to create XFS if not provided if [ -z "${SPLIT_HUGE_PAGE_TEST_XFS_PATH}" ]; then if test_selected "thp"; then From 7574193ae943af663477efbba39b19ae9dfcad66 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:45 +0100 Subject: [PATCH 1121/1352] selftests/mm: check that one khugepaged pass collapses one window khugepaged_full_pass() drives the daemon through sysfs: a store to scan_sleep_millisecs wakes it, and full_scans advancing by two marks one pass that started after setup. Every mTHP collapse result in the suite rests on that pair, and nothing checks it. Add khugepaged_sync_check. Each step: - prepare one aligned window - record its source PFNs from pagemap - run one khugepaged_full_pass() barrier - require the window came out collapsed, with exactly one collapse attempt attributed to it The anon events carry no virtual address, so an attempt is matched by the source folio PFN and order that the mm_collapse_huge_page_isolate tracepoint reports. Reading the trace buffer takes four small helpers in vm_util: open an event subsystem's enable file, flip it, clear the buffer, and open it for reading. scan_sleep_millisecs is set to a minute, so a step that took a sleep instead of a wake would blow the budget. Passes 5/5 on x86-64 4K and arm64 64K. Link: https://lore.kernel.org/20260919002451.496763-16-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Baolin Wang Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/Makefile | 1 + .../selftests/mm/khugepaged_sync_check.c | 179 ++++++++++++++++++ tools/testing/selftests/mm/run_vmtests.sh | 2 + tools/testing/selftests/mm/vm_util.c | 38 ++++ tools/testing/selftests/mm/vm_util.h | 4 + 5 files changed, 224 insertions(+) create mode 100644 tools/testing/selftests/mm/khugepaged_sync_check.c diff --git a/tools/testing/selftests/mm/Makefile b/tools/testing/selftests/mm/Makefile index cb32cf5d867e25..2ed88132491164 100644 --- a/tools/testing/selftests/mm/Makefile +++ b/tools/testing/selftests/mm/Makefile @@ -106,6 +106,7 @@ TEST_GEN_FILES += merge TEST_GEN_FILES += rmap TEST_GEN_FILES += folio_split_race_test TEST_GEN_FILES += folio_order_check +TEST_GEN_FILES += khugepaged_sync_check TEST_GEN_FILES += soft-dirty ifeq ($(ARCH),x86_64) diff --git a/tools/testing/selftests/mm/khugepaged_sync_check.c b/tools/testing/selftests/mm/khugepaged_sync_check.c new file mode 100644 index 00000000000000..28a9b1ff5d4474 --- /dev/null +++ b/tools/testing/selftests/mm/khugepaged_sync_check.c @@ -0,0 +1,179 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Check that khugepaged_full_pass() drives khugepaged in step: one barrier + * over one prepared window must collapse it with exactly one collapse + * attempt attributed to its source pages, step after step. + * + * scan_sleep_millisecs is a minute so that a step which slept instead of + * being woken blows the budget. + */ +#define _GNU_SOURCE +#include +#include +#include +#include +#include +#include + +#include "kselftest.h" +#include "vm_util.h" +#include + +#define BASE_ADDR ((void *)(1UL << 30)) +/* Smallest order khugepaged considers */ +#define TARGET_ORDER 2 +#define NR_ITERATIONS 5 +#define PASS_TIMEOUT_S 30 + +static int pagemap_fd; +static int kpageflags_fd; +static int trace_events_fd = -1; +static unsigned long hpage_pmd_size; + +/* + * The events are system-wide: switch them off however the test ends, + * including from inside a helper that gives up. + */ +static void trace_events_off(void) +{ + if (trace_events_fd >= 0) + tracing_events_enable(trace_events_fd, false); +} + +/* Count the isolate events whose scan_pfn is one of the window's source PFNs */ +static int count_attributed(unsigned long *pfns, int nr_pfns, + unsigned int order) +{ + char line[1024]; + int count = 0; + FILE *fp; + + fp = tracing_open_trace(); + if (!fp) + ksft_exit_fail_msg("Cannot open trace buffer\n"); + + while (fgets(line, sizeof(line), fp)) { + unsigned long val; + unsigned int ord; + char *s, *o; + int i; + + s = strstr(line, "mm_collapse_huge_page_isolate:"); + if (!s) + continue; + if (sscanf(s, "mm_collapse_huge_page_isolate: scan_pfn=0x%lx", + &val) != 1) + continue; + o = strstr(s, "order="); + if (!o || sscanf(o, "order=%u", &ord) != 1 || ord != order) + continue; + for (i = 0; i < nr_pfns; i++) { + if (val == pfns[i]) { + count++; + break; + } + } + } + fclose(fp); + return count; +} + +static void one_step(int iteration) +{ + const size_t window = getpagesize() << TARGET_ORDER; + const int nr_pages = 1 << TARGET_ORDER; + unsigned long pfns[1 << TARGET_ORDER]; + bool collapsed, passed; + int attributed; + char *p; + int i; + + p = mmap(BASE_ADDR, hpage_pmd_size, PROT_READ | PROT_WRITE, + MAP_ANONYMOUS | MAP_PRIVATE | MAP_FIXED_NOREPLACE, -1, 0); + if (p != BASE_ADDR) + ksft_exit_fail_perror("mmap() window"); + + for (i = 0; i < nr_pages; i++) { + p[i * getpagesize()] = i + 1; + pfns[i] = pagemap_get_pfn(pagemap_fd, p + i * getpagesize()); + if (pfns[i] == -1UL) + ksft_exit_fail_msg("Source page not present\n"); + } + + /* Clear before enabling so the buffer holds only this step's events */ + if (tracing_clear_trace()) + ksft_exit_fail_msg("Cannot clear the trace buffer\n"); + if (tracing_events_enable(trace_events_fd, true)) + ksft_exit_fail_msg("Cannot enable huge_memory events\n"); + + if (madvise(p, hpage_pmd_size, MADV_HUGEPAGE)) + ksft_exit_fail_perror("madvise(MADV_HUGEPAGE)"); + passed = khugepaged_full_pass(PASS_TIMEOUT_S); + + /* Off before anything that can give up: the events are system-wide */ + if (tracing_events_enable(trace_events_fd, false)) + ksft_exit_fail_msg("Cannot disable huge_memory events\n"); + if (!passed) + ksft_exit_fail_msg("khugepaged did not complete a full pass\n"); + + collapsed = is_range_backed_by_order(p, window, TARGET_ORDER, + pagemap_fd, kpageflags_fd); + attributed = count_attributed(pfns, nr_pages, TARGET_ORDER); + + ksft_test_result(collapsed && attributed == 1, + "step %d: window collapsed, %d attributed result(s)\n", + iteration, attributed); + + munmap(p, hpage_pmd_size); +} + +int main(void) +{ + struct thp_settings settings; + int i; + + ksft_print_header(); + + if (!thp_available()) + ksft_exit_skip("Transparent Hugepages not available\n"); + if (!(thp_supported_orders() & (1UL << TARGET_ORDER))) + ksft_exit_skip("Order %d is not a supported anon THP order\n", + TARGET_ORDER); + + hpage_pmd_size = read_pmd_pagesize(); + if (!hpage_pmd_size) + ksft_exit_fail_msg("Reading PMD pagesize failed\n"); + pagemap_fd = open("/proc/self/pagemap", O_RDONLY); + if (pagemap_fd < 0) + ksft_exit_fail_perror("open(/proc/self/pagemap)"); + kpageflags_fd = open("/proc/kpageflags", O_RDONLY); + if (kpageflags_fd < 0) + ksft_exit_skip("open(/proc/kpageflags) requires root\n"); + trace_events_fd = tracing_events_open("huge_memory"); + if (trace_events_fd < 0) + ksft_exit_skip("huge_memory events require tracefs and root\n"); + atexit(trace_events_off); + + ksft_set_plan(NR_ITERATIONS); + + thp_save_settings(); + thp_read_settings(&settings); + settings.thp_enabled = THP_MADVISE; + settings.thp_defrag = THP_DEFRAG_ALWAYS; + settings.khugepaged.defrag = 1; + settings.khugepaged.scan_sleep_millisecs = 60 * 1000; + settings.khugepaged.alloc_sleep_millisecs = 60 * 1000; + settings.khugepaged.max_ptes_none = (hpage_pmd_size / getpagesize()) - 1; + /* One wake must complete one full pass; see khugepaged_full_pass() */ + settings.khugepaged.pages_to_scan = 1UL << 24; + for (i = 0; i < NR_ORDERS; i++) + settings.hugepages[i].enabled = THP_NEVER; + settings.hugepages[TARGET_ORDER].enabled = THP_INHERIT; + /* Base of the settings stack; the bottom entry is never popped */ + thp_push_settings(&settings); + + for (i = 0; i < NR_ITERATIONS; i++) + one_step(i); + + ksft_finished(); +} diff --git a/tools/testing/selftests/mm/run_vmtests.sh b/tools/testing/selftests/mm/run_vmtests.sh index f1138412788574..3e8c76cf4948b0 100755 --- a/tools/testing/selftests/mm/run_vmtests.sh +++ b/tools/testing/selftests/mm/run_vmtests.sh @@ -384,6 +384,8 @@ CATEGORY="cow" run_test ./cow CATEGORY="thp" run_test ./folio_order_check +CATEGORY="thp" run_test ./khugepaged_sync_check + CATEGORY="thp" run_test ./khugepaged CATEGORY="thp" run_test ./khugepaged -s 2 diff --git a/tools/testing/selftests/mm/vm_util.c b/tools/testing/selftests/mm/vm_util.c index 1f88330fd0274a..a0ab78ceedb130 100644 --- a/tools/testing/selftests/mm/vm_util.c +++ b/tools/testing/selftests/mm/vm_util.c @@ -598,6 +598,44 @@ bool is_range_backed_by_order(char *start, size_t len, int order, return true; } +#define TRACEFS_ROOT "/sys/kernel/tracing" + +/* + * Returns -1 without tracefs or the subsystem. The events are system-wide: + * whoever switches them on has to switch them off again, on every exit path. + */ +int tracing_events_open(const char *subsys) +{ + char path[256]; + + snprintf(path, sizeof(path), TRACEFS_ROOT "/events/%s/enable", + subsys); + return open(path, O_WRONLY); +} + +int tracing_events_enable(int fd, bool enable) +{ + if (pwrite(fd, enable ? "1" : "0", 1, 0) != 1) + return -1; + return 0; +} + +/* Drop what the trace buffer holds so far */ +int tracing_clear_trace(void) +{ + int fd = open(TRACEFS_ROOT "/trace", O_WRONLY | O_TRUNC); + + if (fd < 0) + return -1; + close(fd); + return 0; +} + +FILE *tracing_open_trace(void) +{ + return fopen(TRACEFS_ROOT "/trace", "r"); +} + /* If `ioctls' non-NULL, the allowed ioctls will be returned into the var */ int uffd_register_with_ioctls(int uffd, void *addr, uint64_t len, bool miss, bool wp, bool minor, uint64_t *ioctls) diff --git a/tools/testing/selftests/mm/vm_util.h b/tools/testing/selftests/mm/vm_util.h index ea48e6a7527e13..072a6c756c5170 100644 --- a/tools/testing/selftests/mm/vm_util.h +++ b/tools/testing/selftests/mm/vm_util.h @@ -121,6 +121,10 @@ int close_procmap(struct procmap_fd *procmap); int write_sysfs(const char *file_path, unsigned long val); int read_sysfs(const char *file_path, unsigned long *val); bool softdirty_supported(void); +int tracing_events_open(const char *subsys); +int tracing_events_enable(int fd, bool enable); +int tracing_clear_trace(void); +FILE *tracing_open_trace(void); static inline int open_self_procmap(struct procmap_fd *procmap_out) { From c278e6c81a83330682b1584bbdcd68aa932e77f7 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:46 +0100 Subject: [PATCH 1122/1352] selftests/mm: add khugepaged race harness Collapse serialises against faults, GUP, fork, mremap and zapping through a protocol of locks, TLB flushes and refcount checks. No khugepaged selftest exercises any of it under contention. Add khugepaged_race. Six racing threads work the same address space: - two faulters - an MADV_DONTNEED thread - a transient FOLL_PIN thread (gup_test) - a forker - an mremap thread One of three drivers collapses under them: stepped khugepaged, one full pass at a time via khugepaged_full_pass(), so each step covers a known extent; free khugepaged left to run (scan_sleep_millisecs=0), for soak; madvise an MADV_COLLAPSE and MADV_DONTNEED loop. Every mode runs in turn unless -m names one, five seconds each. Every supported anon THP order is set to inherit and max_ptes_none is 0, so a window collapses only once fully populated and the racing MADV_DONTNEED steers selection across orders. The rule is that a racing page reads as its pattern or as zero, never anything else. The faulters and fork children check it throughout, and a final sweep checks it again. The other half of the check is the kernel's own assertions, so read dmesg too. The pin thread goes through gup_test, so the harness skips without CONFIG_GUP_TEST or root. The default playground is three shared PMD-sized areas plus the mremap thread's, over two gigabytes at a 512M PMD; -a shrinks it. Link: https://lore.kernel.org/20260919002451.496763-17-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Baolin Wang Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/Makefile | 1 + tools/testing/selftests/mm/khugepaged_race.c | 415 +++++++++++++++++++ tools/testing/selftests/mm/run_vmtests.sh | 2 + 3 files changed, 418 insertions(+) create mode 100644 tools/testing/selftests/mm/khugepaged_race.c diff --git a/tools/testing/selftests/mm/Makefile b/tools/testing/selftests/mm/Makefile index 2ed88132491164..beacc0f873049c 100644 --- a/tools/testing/selftests/mm/Makefile +++ b/tools/testing/selftests/mm/Makefile @@ -107,6 +107,7 @@ TEST_GEN_FILES += rmap TEST_GEN_FILES += folio_split_race_test TEST_GEN_FILES += folio_order_check TEST_GEN_FILES += khugepaged_sync_check +TEST_GEN_FILES += khugepaged_race TEST_GEN_FILES += soft-dirty ifeq ($(ARCH),x86_64) diff --git a/tools/testing/selftests/mm/khugepaged_race.c b/tools/testing/selftests/mm/khugepaged_race.c new file mode 100644 index 00000000000000..66c9e2c10f6601 --- /dev/null +++ b/tools/testing/selftests/mm/khugepaged_race.c @@ -0,0 +1,415 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Race collapse against faults, GUP pins, fork, mremap and MADV_DONTNEED + * over the same ranges. A racing page must read as its pattern or as + * zero, never anything else; the kernel's own assertions in dmesg are the + * other half of the check. + */ +#define _GNU_SOURCE +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "kselftest.h" +#include "vm_util.h" +#include +#include "../../../../mm/gup_test.h" + +#ifndef FOLL_WRITE +#define FOLL_WRITE 0x01 +#endif + +#define BASE_ADDR ((void *)(1UL << 30)) +#define PASS_TIMEOUT_S 30 + +/* + * PMD-sized areas the racing threads share, plus one for the mremap + * thread. -a shrinks it where a PMD is 512M. + */ +#define DEFAULT_SHARED_AREAS 3 +static int nr_shared_areas; +static int nr_areas; + +static unsigned long hpage_pmd_size; +static unsigned long page_size; +/* nr_areas PMD-sized areas; the last one belongs to the mremap thread */ +static char *region; +static char *mremap_area; +static char *mremap_scratch; +static int gup_fd = -1; +static volatile int stop; +static volatile int corrupted; + +static unsigned int pattern(unsigned long page_idx) +{ + unsigned int val = (unsigned int)page_idx * 2654435761U; + + return val ? val : 1; /* never collides with the zero-fill */ +} + +/* Zero means never written; anything else must be this page's pattern */ +static bool page_is_corrupt(unsigned long page_idx, unsigned int *val) +{ + *val = *(unsigned int *)(region + page_idx * page_size); + + return *val && *val != pattern(page_idx); +} + +static void check_page(unsigned long page_idx) +{ + unsigned int val; + + if (page_is_corrupt(page_idx, &val)) { + corrupted = 1; + ksft_print_msg("Corruption at page %lu: %#x != %#x\n", + page_idx, val, pattern(page_idx)); + } +} + +static unsigned long shared_pages(void) +{ + return nr_shared_areas * hpage_pmd_size / page_size; +} + +static unsigned long rand_page(unsigned int *seed) +{ + return (unsigned long)rand_r(seed) % shared_pages(); +} + +/* Clamp so a range never reaches the mremap thread's area */ +static unsigned long room_from(unsigned long page_idx, unsigned long want) +{ + unsigned long left = shared_pages() - page_idx; + + return want < left ? want : left; +} + +static void *faulter_fn(void *arg) +{ + unsigned int seed = (unsigned long)arg; + + while (!stop) { + unsigned long page_idx = rand_page(&seed); + + if (rand_r(&seed) & 1) + *(unsigned int *)(region + page_idx * page_size) = + pattern(page_idx); + else + check_page(page_idx); + } + return NULL; +} + +static void *dontneed_fn(void *arg) +{ + unsigned int seed = (unsigned long)arg; + + while (!stop) { + unsigned long page_idx = rand_page(&seed); + unsigned long nr = 1UL << (rand_r(&seed) % 6); /* 1..32 pages */ + + madvise(region + page_idx * page_size, + room_from(page_idx, nr) * page_size, MADV_DONTNEED); + usleep(rand_r(&seed) % 500); + } + return NULL; +} + +static void *pinner_fn(void *arg) +{ + unsigned int seed = (unsigned long)arg; + + while (!stop) { + struct gup_test gup = {}; + unsigned long page_idx = rand_page(&seed); + unsigned long nr = room_from(page_idx, 16); + + gup.addr = (unsigned long)(region + page_idx * page_size); + gup.size = nr * page_size; + gup.nr_pages_per_call = nr; + gup.gup_flags = FOLL_WRITE; + /* Racing MADV_DONTNEED makes transient failures expected */ + ioctl(gup_fd, PIN_FAST_BENCHMARK, &gup); + usleep(rand_r(&seed) % 200); + } + return NULL; +} + +static void *forker_fn(void *arg) +{ + unsigned int seed = (unsigned long)arg; + + while (!stop) { + pid_t pid = fork(); + + if (pid == 0) { + unsigned int val; + int bad = 0; + + /* + * No stdio in the child: a thread may hold stdout's + * lock across the fork, and printing under it hangs. + */ + for (int i = 0; i < 16; i++) + bad |= page_is_corrupt(rand_page(&seed), &val); + _exit(bad); + } + if (pid > 0) { + int wstatus; + + if (waitpid(pid, &wstatus, 0) < 0) + ksft_exit_fail_perror("waitpid()"); + /* A child killed on the read counts too */ + if (!WIFEXITED(wstatus) || WEXITSTATUS(wstatus)) + corrupted = 1; + } + usleep(rand_r(&seed) % 2000); + } + return NULL; +} + +static void *mremapper_fn(void *arg) +{ + unsigned int seed = (unsigned long)arg; + + while (!stop) { + void *p; + + p = mremap(mremap_area, hpage_pmd_size, hpage_pmd_size, + MREMAP_MAYMOVE | MREMAP_FIXED, mremap_scratch); + if (p == MAP_FAILED) + ksft_exit_fail_perror("mremap() away"); + for (int i = 0; i < 8; i++) + mremap_scratch[(rand_r(&seed) % + (hpage_pmd_size / page_size)) * page_size] = 1; + p = mremap(mremap_scratch, hpage_pmd_size, hpage_pmd_size, + MREMAP_MAYMOVE | MREMAP_FIXED, mremap_area); + if (p == MAP_FAILED) + ksft_exit_fail_perror("mremap() back"); + /* The move back unmapped the scratch address: claim it again */ + if (mmap(mremap_scratch, hpage_pmd_size, PROT_NONE, + MAP_ANONYMOUS | MAP_PRIVATE | MAP_FIXED_NOREPLACE, + -1, 0) != (void *)mremap_scratch) + ksft_exit_fail_perror("mmap() mremap scratch"); + usleep(rand_r(&seed) % 2000); + } + return NULL; +} + +static unsigned long now_ms(void) +{ + struct timeval tv; + + gettimeofday(&tv, NULL); + return tv.tv_sec * 1000UL + tv.tv_usec / 1000; +} + +static void usage(void) +{ + fprintf(stderr, + "Usage: khugepaged_race [-d seconds] [-m stepped|free|madvise] [-a areas] [-t mask]\n" + "\tWithout -m, every mode runs in turn.\n" + "\t-d: seconds per mode (default 5)\n" + "\t-a: number of shared PMD-sized playground areas (default 3)\n" + "\t-t: bitmask of racing threads to start, for bisecting a failure\n"); + exit(1); +} + +int main(int argc, char **argv) +{ + static const char * const thread_names[] = { + "faulter", "faulter2", "dontneed", "pinner", "forker", + "mremapper", + }; + void *(*const thread_fns[])(void *) = { + faulter_fn, faulter_fn, dontneed_fn, pinner_fn, forker_fn, + mremapper_fn, + }; + const int nr_threads = ARRAY_SIZE(thread_names); + pthread_t threads[ARRAY_SIZE(thread_names)]; + static const char * const all_modes[] = { "stepped", "free", "madvise" }; + const char *one_mode[1]; + const char * const *modes = all_modes; + int nr_modes = ARRAY_SIZE(all_modes); + const char *mode_arg = NULL; + struct thp_settings settings; + unsigned long end_ms; + int duration_s = 5; + unsigned long thread_mask = ~0UL; + int nr_areas_arg = 0; + unsigned long i; + int steps = 0; + int opt; + + while ((opt = getopt(argc, argv, "a:d:m:t:h")) != -1) { + switch (opt) { + case 'a': + nr_areas_arg = atoi(optarg); + break; + case 'd': + duration_s = atoi(optarg); + break; + case 'm': + mode_arg = optarg; + break; + case 't': + thread_mask = strtoul(optarg, NULL, 0); + break; + default: + usage(); + } + } + + if (mode_arg) { + if (strcmp(mode_arg, "stepped") && strcmp(mode_arg, "free") && + strcmp(mode_arg, "madvise")) + usage(); + one_mode[0] = mode_arg; + modes = one_mode; + nr_modes = 1; + } + + ksft_print_header(); + if (!thp_available()) + ksft_exit_skip("Transparent Hugepages not available\n"); + + page_size = getpagesize(); + hpage_pmd_size = read_pmd_pagesize(); + if (!hpage_pmd_size) + ksft_exit_fail_msg("Reading PMD pagesize failed\n"); + + gup_fd = open("/sys/kernel/debug/gup_test", O_RDWR); + if (gup_fd < 0) + ksft_exit_skip("/sys/kernel/debug/gup_test requires CONFIG_GUP_TEST and root\n"); + + nr_shared_areas = nr_areas_arg > 0 ? nr_areas_arg : DEFAULT_SHARED_AREAS; + nr_areas = nr_shared_areas + 1; + + /* + * MREMAP_FIXED unmaps whatever is in the way without saying so, so + * claim the mremap thread's scratch address up front. + */ + mremap_scratch = (char *)BASE_ADDR + 2 * nr_areas * hpage_pmd_size; + if (mmap(mremap_scratch, hpage_pmd_size, PROT_NONE, + MAP_ANONYMOUS | MAP_PRIVATE | MAP_FIXED_NOREPLACE, + -1, 0) != (void *)mremap_scratch) + ksft_exit_fail_perror("mmap() mremap scratch"); + + ksft_set_plan(nr_modes); + + thp_save_settings(); + thp_read_settings(&settings); + + /* Base of the settings stack; the bottom entry is never popped */ + thp_push_settings(&settings); + + for (int m = 0; m < nr_modes; m++) { + const char *mode = modes[m]; + + thp_read_settings(&settings); + settings.thp_enabled = THP_MADVISE; + settings.thp_defrag = THP_DEFRAG_ALWAYS; + settings.shmem_enabled = SHMEM_NEVER; + settings.khugepaged.defrag = 1; + settings.khugepaged.scan_sleep_millisecs = + strcmp(mode, "free") ? 1000 : 0; + settings.khugepaged.alloc_sleep_millisecs = 10; + /* + * mTHP collapse honours only 0 or HPAGE_PMD_NR - 1 here, and 0 + * keeps a step from being spent on PMD allocations that racing + * MADV_DONTNEED will not let succeed. + */ + settings.khugepaged.max_ptes_none = 0; + /* One wake, one pass: the playground plus the forked children's copies */ + settings.khugepaged.pages_to_scan = + nr_areas * (hpage_pmd_size / page_size) * 8; + for (i = 0; i < NR_ORDERS; i++) { + if (thp_supported_orders() & (1UL << i)) + settings.hugepages[i].enabled = THP_INHERIT; + } + thp_push_settings(&settings); + + region = mmap(BASE_ADDR, nr_areas * hpage_pmd_size, + PROT_READ | PROT_WRITE, MAP_ANONYMOUS | + MAP_PRIVATE | MAP_FIXED_NOREPLACE, -1, 0); + if (region != BASE_ADDR) + ksft_exit_fail_perror("mmap() playground"); + mremap_area = region + nr_shared_areas * hpage_pmd_size; + + /* Populate so the first pass has something to collapse */ + for (i = 0; i < nr_shared_areas * hpage_pmd_size / page_size; i++) + *(unsigned int *)(region + i * page_size) = pattern(i); + memset(mremap_area, 1, hpage_pmd_size); + if (madvise(region, nr_areas * hpage_pmd_size, MADV_HUGEPAGE)) + ksft_exit_fail_perror("madvise(MADV_HUGEPAGE)"); + + for (i = 0; i < nr_threads; i++) { + if (!(thread_mask & (1UL << i))) { + threads[i] = 0; + continue; + } + if (pthread_create(&threads[i], NULL, thread_fns[i], + (void *)(i + 1))) + ksft_exit_fail_perror(thread_names[i]); + } + + end_ms = now_ms() + duration_s * 1000UL; + if (!strcmp(mode, "stepped")) { + while (now_ms() < end_ms && !corrupted) { + if (!khugepaged_full_pass(PASS_TIMEOUT_S)) + ksft_exit_fail_msg("khugepaged pass timed out\n"); + steps++; + } + } else if (!strcmp(mode, "free")) { + while (now_ms() < end_ms && !corrupted) + usleep(100 * 1000); + } else { /* madvise */ + while (now_ms() < end_ms && !corrupted) { + for (i = 0; i < nr_shared_areas; i++) { + madvise(region + i * hpage_pmd_size, + hpage_pmd_size, MADV_COLLAPSE); + } + madvise(region, nr_shared_areas * hpage_pmd_size, + MADV_DONTNEED); + steps++; + } + } + + stop = 1; + for (i = 0; i < nr_threads; i++) { + if (threads[i]) + pthread_join(threads[i], NULL); + } + + for (i = 0; i < nr_shared_areas * hpage_pmd_size / page_size; i++) + check_page(i); + + ksft_test_result(!corrupted, + "%s: %ds, %d steps, no corruption\n", + mode, duration_s, steps); + + /* The next mode maps the same fixed address with its own settings */ + munmap(region, nr_areas * hpage_pmd_size); + thp_pop_settings(); + stop = 0; + steps = 0; + + if (corrupted) { + /* Memory is suspect; the rest would prove nothing */ + while (++m < nr_modes) + ksft_test_result_skip("%s: skipped after corruption\n", + modes[m]); + break; + } + } + + ksft_finished(); +} diff --git a/tools/testing/selftests/mm/run_vmtests.sh b/tools/testing/selftests/mm/run_vmtests.sh index 3e8c76cf4948b0..a1b45a3dedaeea 100755 --- a/tools/testing/selftests/mm/run_vmtests.sh +++ b/tools/testing/selftests/mm/run_vmtests.sh @@ -386,6 +386,8 @@ CATEGORY="thp" run_test ./folio_order_check CATEGORY="thp" run_test ./khugepaged_sync_check +CATEGORY="thp" run_test ./khugepaged_race + CATEGORY="thp" run_test ./khugepaged CATEGORY="thp" run_test ./khugepaged -s 2 From aff511dd72e87e7e89febee8b17497b5191e5642 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:47 +0100 Subject: [PATCH 1123/1352] selftests/mm: race the collapse of windows with holes The harness pins max_ptes_none to 0, so khugepaged only collapses a window once every PTE in it is present. A window with holes takes a different route, and never gets raced. A hole is zero-filled in the new folio rather than copied. Which slots count as holes keeps moving under the racing MADV_DONTNEED, right up to the moment the PMD is detached. Run both ends of the occupancy scale for every driver mode, one after the other. mTHP collapse supports only those two, 0 and HPAGE_PMD_NR - 1, and coerces anything between them to 0. Each result says which end it ran: ok 1 stepped/strict: 5s, 231 steps, no corruption ok 2 stepped/holes: 5s, 194 steps, no corruption Link: https://lore.kernel.org/20260919002451.496763-18-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Baolin Wang Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/khugepaged_race.c | 32 ++++++++++++-------- 1 file changed, 20 insertions(+), 12 deletions(-) diff --git a/tools/testing/selftests/mm/khugepaged_race.c b/tools/testing/selftests/mm/khugepaged_race.c index 66c9e2c10f6601..6575045c048262 100644 --- a/tools/testing/selftests/mm/khugepaged_race.c +++ b/tools/testing/selftests/mm/khugepaged_race.c @@ -236,6 +236,8 @@ int main(int argc, char **argv) const int nr_threads = ARRAY_SIZE(thread_names); pthread_t threads[ARRAY_SIZE(thread_names)]; static const char * const all_modes[] = { "stepped", "free", "madvise" }; + static const bool occupancies[] = { false, true }; /* strict, holes */ + const int nr_occupancies = ARRAY_SIZE(occupancies); const char *one_mode[1]; const char * const *modes = all_modes; int nr_modes = ARRAY_SIZE(all_modes); @@ -303,7 +305,7 @@ int main(int argc, char **argv) -1, 0) != (void *)mremap_scratch) ksft_exit_fail_perror("mmap() mremap scratch"); - ksft_set_plan(nr_modes); + ksft_set_plan(nr_modes * nr_occupancies); thp_save_settings(); thp_read_settings(&settings); @@ -311,8 +313,9 @@ int main(int argc, char **argv) /* Base of the settings stack; the bottom entry is never popped */ thp_push_settings(&settings); - for (int m = 0; m < nr_modes; m++) { - const char *mode = modes[m]; + for (int run = 0; run < nr_modes * nr_occupancies; run++) { + const char *mode = modes[run / nr_occupancies]; + bool holes = occupancies[run % nr_occupancies]; thp_read_settings(&settings); settings.thp_enabled = THP_MADVISE; @@ -322,12 +325,14 @@ int main(int argc, char **argv) settings.khugepaged.scan_sleep_millisecs = strcmp(mode, "free") ? 1000 : 0; settings.khugepaged.alloc_sleep_millisecs = 10; + /* - * mTHP collapse honours only 0 or HPAGE_PMD_NR - 1 here, and 0 - * keeps a step from being spent on PMD allocations that racing - * MADV_DONTNEED will not let succeed. + * mTHP collapse honours only 0 or HPAGE_PMD_NR - 1 here. The two + * ends race different paths: a strict window has every PTE + * present, a hole-heavy one is mostly zero-filled. */ - settings.khugepaged.max_ptes_none = 0; + settings.khugepaged.max_ptes_none = holes ? + (hpage_pmd_size / page_size) - 1 : 0; /* One wake, one pass: the playground plus the forked children's copies */ settings.khugepaged.pages_to_scan = nr_areas * (hpage_pmd_size / page_size) * 8; @@ -393,8 +398,9 @@ int main(int argc, char **argv) check_page(i); ksft_test_result(!corrupted, - "%s: %ds, %d steps, no corruption\n", - mode, duration_s, steps); + "%s/%s: %ds, %d steps, no corruption\n", + mode, holes ? "holes" : "strict", + duration_s, steps); /* The next mode maps the same fixed address with its own settings */ munmap(region, nr_areas * hpage_pmd_size); @@ -404,9 +410,11 @@ int main(int argc, char **argv) if (corrupted) { /* Memory is suspect; the rest would prove nothing */ - while (++m < nr_modes) - ksft_test_result_skip("%s: skipped after corruption\n", - modes[m]); + while (++run < nr_modes * nr_occupancies) + ksft_test_result_skip("%s/%s: skipped after corruption\n", + modes[run / nr_occupancies], + occupancies[run % nr_occupancies] ? + "holes" : "strict"); break; } } From e8d4c0870eea6b65a3fec6806f2abd008980fe48 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:48 +0100 Subject: [PATCH 1124/1352] selftests/mm: add memory-pressure threads to the khugepaged race harness The harness races collapse against faults, pins, fork, mremap and MADV_DONTNEED, but nothing in it runs reclaim or compaction against the collapse. Add two more threads, and run every mode and occupancy limit both with and without them: - pageout: cycles MADV_PAGEOUT over a region of its own, faults it back in and checks the content each round, since a page's pattern must survive the trip through swap. Left out when the host has no swap, because then there is no anon reclaim to drive. - compactor: writes /proc/sys/vm/compact_memory in a loop. Compaction isolates and migrates folios, so it competes with a collapse for the pages it is gathering, with refcount elevations and migration entries of its own. Each result says whether it ran under pressure: ok 2 stepped/strict/pressure: 5s, 88 steps, no corruption A full run is now twelve combinations; -m picks one mode, -d shortens each run. Link: https://lore.kernel.org/20260919002451.496763-19-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Baolin Wang Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/khugepaged_race.c | 142 ++++++++++++++++--- 1 file changed, 122 insertions(+), 20 deletions(-) diff --git a/tools/testing/selftests/mm/khugepaged_race.c b/tools/testing/selftests/mm/khugepaged_race.c index 6575045c048262..68046f0bf6d318 100644 --- a/tools/testing/selftests/mm/khugepaged_race.c +++ b/tools/testing/selftests/mm/khugepaged_race.c @@ -44,6 +44,8 @@ static unsigned long page_size; static char *region; static char *mremap_area; static char *mremap_scratch; +static char *pageout_area; +static size_t pageout_size; static int gup_fd = -1; static volatile int stop; static volatile int corrupted; @@ -204,6 +206,69 @@ static void *mremapper_fn(void *arg) return NULL; } +/* + * Swap traffic and LRU churn on a region nothing else writes, so a page's + * pattern must survive the trip through swap exactly. + */ +static void *pageout_fn(void *arg) +{ + unsigned int seed = (unsigned long)arg; + unsigned long nr = pageout_size / page_size; + unsigned long i; + + for (i = 0; i < nr; i++) + *(unsigned int *)(pageout_area + i * page_size) = pattern(i); + + while (!stop) { + madvise(pageout_area, pageout_size, MADV_PAGEOUT); + for (i = 0; i < nr && !stop; i++) { + unsigned int val = *(unsigned int *)(pageout_area + + i * page_size); + + if (val != pattern(i)) { + corrupted = 1; + ksft_print_msg("Pageout corruption at page %lu: %#x != %#x\n", + i, val, pattern(i)); + } + } + usleep(rand_r(&seed) % 2000); + } + return NULL; +} + +/* Compaction migrates the collapse sources while they are being gathered */ +static void *compactor_fn(void *arg) +{ + unsigned int seed = (unsigned long)arg; + int fd = open("/proc/sys/vm/compact_memory", O_WRONLY); + + if (fd < 0) { + ksft_print_msg("No compact_memory; compactor idle\n"); + return NULL; + } + while (!stop) { + if (write(fd, "1", 1) < 0) + break; + usleep(10000 + rand_r(&seed) % 100000); + } + close(fd); + return NULL; +} + +static bool swap_available(void) +{ + char line[256]; + int lines = 0; + FILE *fp = fopen("/proc/swaps", "r"); + + if (!fp) + return false; + while (fgets(line, sizeof(line), fp)) + lines++; + fclose(fp); + return lines > 1; +} + static unsigned long now_ms(void) { struct timeval tv; @@ -227,17 +292,23 @@ int main(int argc, char **argv) { static const char * const thread_names[] = { "faulter", "faulter2", "dontneed", "pinner", "forker", - "mremapper", + "mremapper", "pageout", "compactor", }; void *(*const thread_fns[])(void *) = { faulter_fn, faulter_fn, dontneed_fn, pinner_fn, forker_fn, - mremapper_fn, + mremapper_fn, pageout_fn, compactor_fn, }; + enum { T_FAULTER, T_FAULTER2, T_DONTNEED, T_PINNER, T_FORKER, + T_MREMAPPER, T_PAGEOUT, T_COMPACTOR }; + const unsigned long pageout_bit = 1UL << T_PAGEOUT; + const unsigned long compactor_bit = 1UL << T_COMPACTOR; const int nr_threads = ARRAY_SIZE(thread_names); pthread_t threads[ARRAY_SIZE(thread_names)]; static const char * const all_modes[] = { "stepped", "free", "madvise" }; static const bool occupancies[] = { false, true }; /* strict, holes */ + static const bool pressures[] = { false, true }; /* quiet, under pressure */ const int nr_occupancies = ARRAY_SIZE(occupancies); + const int nr_pressures = ARRAY_SIZE(pressures); const char *one_mode[1]; const char * const *modes = all_modes; int nr_modes = ARRAY_SIZE(all_modes); @@ -246,6 +317,9 @@ int main(int argc, char **argv) unsigned long end_ms; int duration_s = 5; unsigned long thread_mask = ~0UL; + unsigned long base_mask; + bool have_swap; + char label[64]; int nr_areas_arg = 0; unsigned long i; int steps = 0; @@ -305,7 +379,13 @@ int main(int argc, char **argv) -1, 0) != (void *)mremap_scratch) ksft_exit_fail_perror("mmap() mremap scratch"); - ksft_set_plan(nr_modes * nr_occupancies); + base_mask = thread_mask; + have_swap = swap_available(); + if (!have_swap) + /* No swap, no anon reclaim: compaction-only pressure */ + ksft_print_msg("no swap: the pageout thread is not started\n"); + + ksft_set_plan(nr_modes * nr_occupancies * nr_pressures); thp_save_settings(); thp_read_settings(&settings); @@ -313,9 +393,25 @@ int main(int argc, char **argv) /* Base of the settings stack; the bottom entry is never popped */ thp_push_settings(&settings); - for (int run = 0; run < nr_modes * nr_occupancies; run++) { - const char *mode = modes[run / nr_occupancies]; - bool holes = occupancies[run % nr_occupancies]; + for (int run = 0; run < nr_modes * nr_occupancies * nr_pressures; run++) { + int rem = run % (nr_occupancies * nr_pressures); + const char *mode = modes[run / (nr_occupancies * nr_pressures)]; + bool holes = occupancies[rem / nr_pressures]; + bool pressure = pressures[rem % nr_pressures]; + + snprintf(label, sizeof(label), "%s/%s%s", mode, + holes ? "holes" : "strict", pressure ? "/pressure" : ""); + if (corrupted) { + /* Memory is suspect; the rest would prove nothing */ + ksft_test_result_skip("%s: skipped after corruption\n", label); + continue; + } + + thread_mask = base_mask; + if (!pressure) + thread_mask &= ~(pageout_bit | compactor_bit); + else if (!have_swap) + thread_mask &= ~pageout_bit; thp_read_settings(&settings); settings.thp_enabled = THP_MADVISE; @@ -349,6 +445,20 @@ int main(int argc, char **argv) ksft_exit_fail_perror("mmap() playground"); mremap_area = region + nr_shared_areas * hpage_pmd_size; + if (thread_mask & pageout_bit) { + /* Enough to drive real reclaim without swamping a small guest */ + pageout_size = 4 * hpage_pmd_size; + if (pageout_size < 16UL << 20) + pageout_size = 16UL << 20; + if (pageout_size > 64UL << 20) + pageout_size = 64UL << 20; + pageout_area = mmap(NULL, pageout_size, + PROT_READ | PROT_WRITE, + MAP_ANONYMOUS | MAP_PRIVATE, -1, 0); + if (pageout_area == MAP_FAILED) + ksft_exit_fail_perror("mmap() pageout area"); + } + /* Populate so the first pass has something to collapse */ for (i = 0; i < nr_shared_areas * hpage_pmd_size / page_size; i++) *(unsigned int *)(region + i * page_size) = pattern(i); @@ -397,26 +507,18 @@ int main(int argc, char **argv) for (i = 0; i < nr_shared_areas * hpage_pmd_size / page_size; i++) check_page(i); - ksft_test_result(!corrupted, - "%s/%s: %ds, %d steps, no corruption\n", - mode, holes ? "holes" : "strict", - duration_s, steps); + ksft_test_result(!corrupted, "%s: %ds, %d steps, no corruption\n", + label, duration_s, steps); /* The next mode maps the same fixed address with its own settings */ munmap(region, nr_areas * hpage_pmd_size); + if (pageout_area) { + munmap(pageout_area, pageout_size); + pageout_area = NULL; + } thp_pop_settings(); stop = 0; steps = 0; - - if (corrupted) { - /* Memory is suspect; the rest would prove nothing */ - while (++run < nr_modes * nr_occupancies) - ksft_test_result_skip("%s/%s: skipped after corruption\n", - modes[run / nr_occupancies], - occupancies[run % nr_occupancies] ? - "holes" : "strict"); - break; - } } ksft_finished(); From e597ac0ed6753a7ae48cde37afdad837bdfbb58d Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:49 +0100 Subject: [PATCH 1125/1352] selftests/mm: zap whole PTE tables in the khugepaged race harness The harness's MADV_DONTNEED thread zaps 1 to 32 pages at a time, never a whole PMD-aligned area, and only a zap that covers a full table frees the table itself (CONFIG_PT_RECLAIM). Make the thread zap a whole PMD-aligned area about one iteration in 64, and keep the fine-grained zaps as the common case. The new case frees page tables, racing that against a collapse walking the same table. Link: https://lore.kernel.org/20260919002451.496763-20-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Baolin Wang Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/khugepaged_race.c | 17 +++++++++++++++-- 1 file changed, 15 insertions(+), 2 deletions(-) diff --git a/tools/testing/selftests/mm/khugepaged_race.c b/tools/testing/selftests/mm/khugepaged_race.c index 68046f0bf6d318..abb0678522cd9f 100644 --- a/tools/testing/selftests/mm/khugepaged_race.c +++ b/tools/testing/selftests/mm/khugepaged_race.c @@ -118,8 +118,21 @@ static void *dontneed_fn(void *arg) unsigned long page_idx = rand_page(&seed); unsigned long nr = 1UL << (rand_r(&seed) % 6); /* 1..32 pages */ - madvise(region + page_idx * page_size, - room_from(page_idx, nr) * page_size, MADV_DONTNEED); + /* + * Now and then zap a whole PMD-aligned area: only a zap that + * covers the full table frees the table itself (CONFIG_PT_RECLAIM). + */ + if (!(rand_r(&seed) % 64)) { + unsigned long area = page_idx / + (hpage_pmd_size / page_size); + + madvise(region + area * hpage_pmd_size, + hpage_pmd_size, MADV_DONTNEED); + } else { + madvise(region + page_idx * page_size, + room_from(page_idx, nr) * page_size, + MADV_DONTNEED); + } usleep(rand_r(&seed) % 500); } return NULL; From deedbd6c6c77dd785aa5fe03ad370dbf6f04c497 Mon Sep 17 00:00:00 2001 From: Lance Yang Date: Mon, 21 Sep 2026 13:42:25 +0800 Subject: [PATCH 1126/1352] mm: disallow raw PFN mappings of huge/shared zeropage Handling the huge/shared zeropage correctly in vmf_insert_pfn_pmd() and vmf_insert_pfn_prot() is more involved. We would need to check whether the VMA allows it and keep the mapping read-only, similar to the checks in vm_mixed_ok(). No in-tree user needs that support, so reject these mappings with VM_FAULT_SIGBUS rather than complicate the code for now. Link: https://lore.kernel.org/20260921054225.28537-1-lance.yang@linux.dev Link: https://lore.kernel.org/all/20260917121010.60966-1-lance.yang@linux.dev/ Signed-off-by: Lance Yang Signed-off-by: Andrew Morton Suggested-by: Kiryl Shutsemau (Meta) Suggested-by: David Hildenbrand (Arm) Reviewed-by: Kiryl Shutsemau (Meta) Acked-by: David Hildenbrand (Arm) Acked-by: Zi Yan Reviewed-by: Lorenzo Stoakes (ARM) Cc: Baolin Wang Cc: Barry Song Cc: Dev Jain Cc: Liam Howlett Cc: Michal Hocko Cc: "Mike Rapoport (IBM)" Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" --- mm/huge_memory.c | 3 +++ mm/memory.c | 3 +++ 2 files changed, 6 insertions(+) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index ba5e20bbfd3bc3..c5c210b3ebc646 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -1722,6 +1722,9 @@ vm_fault_t vmf_insert_pfn_pmd(struct vm_fault *vmf, unsigned long pfn, (VM_PFNMAP|VM_MIXEDMAP)); BUG_ON((vma->vm_flags & VM_PFNMAP) && vma_is_cow_mapping(vma)); + if (unlikely(is_huge_zero_pfn(pfn))) + return VM_FAULT_SIGBUS; + pfnmap_setup_cachemode_pfn(pfn, &pgprot); return insert_pmd(vma, addr, vmf->pmd, fop, pgprot, write); diff --git a/mm/memory.c b/mm/memory.c index 6349ef676549a8..79fa57a381ce00 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -2956,6 +2956,9 @@ vm_fault_t vmf_insert_pfn_prot(struct vm_area_struct *vma, unsigned long addr, BUG_ON((vma->vm_flags & VM_PFNMAP) && vma_is_cow_mapping(vma)); BUG_ON((vma->vm_flags & VM_MIXEDMAP) && pfn_valid(pfn)); + if (unlikely(is_zero_pfn(pfn))) + return VM_FAULT_SIGBUS; + if (addr < vma->vm_start || addr >= vma->vm_end) return VM_FAULT_SIGBUS; From 0412bc23798994ade94a74818170875e4ac94dc0 Mon Sep 17 00:00:00 2001 From: Donggeun Yoo Date: Mon, 21 Sep 2026 08:24:41 -0700 Subject: [PATCH 1127/1352] mm/damon/core: charge only the part of a region the filter left Patch series "mm/damon/core: fix the size charged for a filter-trimmed region", v2. An address range DAMOS filter that partially overlaps a monitoring region splits it, so the scheme action reaches only one side of the boundary. damos_apply_scheme() reads the region size before that split and never re-reads it, so the quota charge and schemes//stats/sz_tried account for the whole original region. Patch 1 re-reads the size after the core filters have run. Patch 2 adds KUnit coverage for what the function charges and reports, for the trimmed cases and for the cases that must not change. This patch (of 2): An address range DAMOS filter that partially overlaps a monitoring region splits the region at the filter boundary, so the scheme action is applied to only one side of it. damos_apply_scheme() reads the region size once on entry, before damos_core_filter_out() performs that split. The quota charge and the statistics update therefore account for the region as it was before the trim. schemes//stats/sz_tried counts memory the filter excluded, which Documentation/mm/damon/design.rst says is not counted as tried, and with quotas/bytes set the excluded part is charged against the budget, throttling the scheme to a fraction of what was configured. User impact is that DAMOS could unexpectedly slowly run, due to over-charged quota. DAMOS stat could also be confusing. No critical events such as crashes or leask happen. Re-read the region size after the core filters have run. Link: https://lore.kernel.org/20260921152443.80132-1-sj@kernel.org Link: https://lore.kernel.org/20260921152443.80132-2-sj@kernel.org Fixes: ab9bda001b68 ("mm/damon/core: introduce address range type damos filter") Signed-off-by: Donggeun Yoo Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Brendan Higgins Cc: David Gow Cc: --- mm/damon/core.c | 1 + 1 file changed, 1 insertion(+) diff --git a/mm/damon/core.c b/mm/damon/core.c index add1b7afb957ac..8d8b0cba1a56c9 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2784,6 +2784,7 @@ static void damos_apply_scheme(struct damon_ctx *c, struct damon_target *t, } if (damos_core_filter_out(c, t, r, s)) return; + sz = damon_sz_region(r); ktime_get_coarse_ts64(&begin); trace_damos_before_apply(cidx, sidx, tidx, r, nr_accesses, damon_nr_regions(t), do_trace); From 216c895d998dab79deee375e8ac243ae5eeffc24 Mon Sep 17 00:00:00 2001 From: Donggeun Yoo Date: Mon, 21 Sep 2026 08:24:42 -0700 Subject: [PATCH 1128/1352] mm/damon/tests/core-kunit: test the size charged for a filter-trimmed region damos_test_filter_out() checks that an address range filter splits a region at the filter boundary, and stops there. Nothing checks what damos_apply_scheme() then charges to the quota and reports as tried. damos_test_apply_scheme_filtered_sz() covers the two ways such a filter trims a region: one that starts before the filter's range, and one that starts inside it. damos_test_apply_scheme_filter_sz_unchanged() and damos_test_apply_scheme_quota_sz() cover the cases whose accounting must not change: a region wholly inside a reject range, one wholly inside an allow range, two filters trimming in a single call, a filter type that never splits, no filter at all, a region the quota trims, and a quota remainder too small for one region. Link: https://lore.kernel.org/20260921152443.80132-3-sj@kernel.org Signed-off-by: Donggeun Yoo Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Brendan Higgins Cc: David Gow Cc: --- mm/damon/tests/core-kunit.h | 316 ++++++++++++++++++++++++++++++++++++ 1 file changed, 316 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index 5ff0436c58441f..df84d9cc7d204a 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -1706,6 +1706,319 @@ static void damos_test_filter_out(struct kunit *test) damos_free_filter(f); } +static unsigned long damos_test_apply_scheme_stub(struct damon_ctx *c, + struct damon_target *t, struct damon_region *r, + struct damos *s, unsigned long *sz_filter_passed) +{ + return 0; +} + +static void damos_test_apply_scheme_filtered_sz(struct kunit *test) +{ + struct damos_access_pattern pattern = { + .min_sz_region = 0, + .max_sz_region = ULONG_MAX, + .min_nr_accesses = 0, + .max_nr_accesses = UINT_MAX, + .min_age_region = 0, + .max_age_region = UINT_MAX, + }; + unsigned long min_sz = DAMON_MIN_REGION_SZ; + struct damos_watermarks wmarks = {}; + struct damos_quota quota = {}; + struct damon_ctx *c; + struct damon_target *t; + struct damon_region *r; + struct damos_filter *f; + struct damos *s; + + c = damon_new_ctx(); + if (!c) + kunit_skip(test, "ctx alloc fail"); + c->ops.apply_scheme = damos_test_apply_scheme_stub; + + s = damon_new_scheme(&pattern, DAMOS_STAT, 0, "a, &wmarks, + NUMA_NO_NODE); + if (!s) { + damon_destroy_ctx(c); + kunit_skip(test, "scheme alloc fail"); + } + damon_add_scheme(c, s); + + t = damon_new_target(); + if (!t) { + damon_destroy_ctx(c); + kunit_skip(test, "target alloc fail"); + } + damon_add_target(c, t); + + f = damos_new_filter(DAMOS_FILTER_TYPE_ADDR, true, false); + if (!f) { + damon_destroy_ctx(c); + kunit_skip(test, "filter alloc fail"); + } + f->addr_range = (struct damon_addr_range){ + .start = 2 * min_sz, + .end = 8 * min_sz + }; + damos_add_filter(s, f); + damos_set_filters_default_reject(s); + + /* reject filter, region starting before the range */ + r = damon_new_region(0, 4 * min_sz); + if (!r) { + damon_destroy_ctx(c); + kunit_skip(test, "region alloc fail"); + } + damon_add_region(r, t); + + damos_apply_scheme(c, t, r, s); + KUNIT_EXPECT_EQ(test, damon_nr_regions(t), 2); + KUNIT_EXPECT_EQ(test, r->ar.end, 2 * min_sz); + KUNIT_EXPECT_EQ(test, s->stat.sz_tried, 2 * min_sz); + + if (damon_nr_regions(t) != 2) + goto out; + damon_destroy_region(damon_next_region(r), t); + damon_destroy_region(r, t); + s->stat = (struct damos_stat){}; + + /* allow filter, region starting inside the range */ + f->allow = true; + damos_set_filters_default_reject(s); + r = damon_new_region(2 * min_sz, 10 * min_sz); + if (!r) { + damon_destroy_ctx(c); + kunit_skip(test, "region alloc fail"); + } + damon_add_region(r, t); + + damos_apply_scheme(c, t, r, s); + KUNIT_EXPECT_EQ(test, damon_nr_regions(t), 2); + KUNIT_EXPECT_EQ(test, r->ar.end, 8 * min_sz); + KUNIT_EXPECT_EQ(test, s->stat.sz_tried, 6 * min_sz); + +out: + damon_destroy_ctx(c); +} + +static void damos_test_apply_scheme_filter_sz_unchanged(struct kunit *test) +{ + struct damos_access_pattern pattern = { + .min_sz_region = 0, + .max_sz_region = ULONG_MAX, + .min_nr_accesses = 0, + .max_nr_accesses = UINT_MAX, + .min_age_region = 0, + .max_age_region = UINT_MAX, + }; + unsigned long min_sz = DAMON_MIN_REGION_SZ; + struct damos_watermarks wmarks = {}; + struct damos_quota quota = {}; + struct damos_filter *f, *f2; + struct damon_ctx *c; + struct damon_target *t; + struct damon_region *r; + struct damos *s; + + c = damon_new_ctx(); + if (!c) + kunit_skip(test, "ctx alloc fail"); + c->ops.apply_scheme = damos_test_apply_scheme_stub; + + s = damon_new_scheme(&pattern, DAMOS_STAT, 0, "a, &wmarks, + NUMA_NO_NODE); + if (!s) { + damon_destroy_ctx(c); + kunit_skip(test, "scheme alloc fail"); + } + damon_add_scheme(c, s); + + t = damon_new_target(); + if (!t) { + damon_destroy_ctx(c); + kunit_skip(test, "target alloc fail"); + } + damon_add_target(c, t); + + f = damos_new_filter(DAMOS_FILTER_TYPE_ADDR, true, false); + if (!f) { + damon_destroy_ctx(c); + kunit_skip(test, "filter alloc fail"); + } + f->addr_range = (struct damon_addr_range){ + .start = 2 * min_sz, .end = 8 * min_sz}; + damos_add_filter(s, f); + damos_set_filters_default_reject(s); + + /* wholly inside a reject range: not counted at all */ + r = damon_new_region(4 * min_sz, 6 * min_sz); + if (!r) { + damon_destroy_ctx(c); + kunit_skip(test, "region alloc fail"); + } + damon_add_region(r, t); + + damos_apply_scheme(c, t, r, s); + KUNIT_EXPECT_EQ(test, s->stat.nr_tried, 0); + KUNIT_EXPECT_EQ(test, s->stat.sz_tried, 0); + KUNIT_EXPECT_EQ(test, damon_nr_regions(t), 1); + + /* wholly inside an allow range: counted whole, not split */ + f->allow = true; + damos_set_filters_default_reject(s); + + damos_apply_scheme(c, t, r, s); + KUNIT_EXPECT_EQ(test, s->stat.sz_tried, 2 * min_sz); + KUNIT_EXPECT_EQ(test, damon_nr_regions(t), 1); + + /* two filters, each trimming: the size left after both is counted */ + damon_destroy_region(r, t); + s->stat = (struct damos_stat){}; + f->allow = false; + f->addr_range = (struct damon_addr_range){ + .start = 4 * min_sz, .end = 12 * min_sz}; + f2 = damos_new_filter(DAMOS_FILTER_TYPE_ADDR, true, false); + if (!f2) { + damon_destroy_ctx(c); + kunit_skip(test, "filter alloc fail"); + } + f2->addr_range = (struct damon_addr_range){ + .start = 2 * min_sz, .end = 3 * min_sz}; + damos_add_filter(s, f2); + damos_set_filters_default_reject(s); + + r = damon_new_region(0, 8 * min_sz); + if (!r) { + damon_destroy_ctx(c); + kunit_skip(test, "region alloc fail"); + } + damon_add_region(r, t); + + damos_apply_scheme(c, t, r, s); + KUNIT_EXPECT_EQ(test, damon_nr_regions(t), 3); + KUNIT_EXPECT_EQ(test, r->ar.end, 2 * min_sz); + KUNIT_EXPECT_EQ(test, s->stat.sz_tried, 2 * min_sz); + + /* a core filter that never splits: counted whole */ + damos_destroy_filter(f2); + damos_destroy_filter(f); + f = damos_new_filter(DAMOS_FILTER_TYPE_TARGET, true, true); + if (!f) { + damon_destroy_ctx(c); + kunit_skip(test, "filter alloc fail"); + } + f->target_idx = 0; + damos_add_filter(s, f); + damos_set_filters_default_reject(s); + s->stat = (struct damos_stat){}; + + damos_apply_scheme(c, t, r, s); + KUNIT_EXPECT_EQ(test, damon_nr_regions(t), 3); + KUNIT_EXPECT_EQ(test, r->ar.end, 2 * min_sz); + KUNIT_EXPECT_EQ(test, s->stat.sz_tried, 2 * min_sz); + + damon_destroy_ctx(c); +} + +static void damos_test_apply_scheme_quota_sz(struct kunit *test) +{ + struct damos_access_pattern pattern = { + .min_sz_region = 0, + .max_sz_region = ULONG_MAX, + .min_nr_accesses = 0, + .max_nr_accesses = UINT_MAX, + .min_age_region = 0, + .max_age_region = UINT_MAX, + }; + unsigned long min_sz = DAMON_MIN_REGION_SZ; + struct damos_watermarks wmarks = {}; + struct damos_quota quota = {}; + struct damon_region *r, *next; + struct damon_ctx *c; + struct damon_target *t; + struct damos *s; + + c = damon_new_ctx(); + if (!c) + kunit_skip(test, "ctx alloc fail"); + + s = damon_new_scheme(&pattern, DAMOS_STAT, 0, "a, &wmarks, + NUMA_NO_NODE); + if (!s) { + damon_destroy_ctx(c); + kunit_skip(test, "scheme alloc fail"); + } + damon_add_scheme(c, s); + damos_set_filters_default_reject(s); + + t = damon_new_target(); + if (!t) { + damon_destroy_ctx(c); + kunit_skip(test, "target alloc fail"); + } + damon_add_target(c, t); + + /* no apply_scheme operation: the whole region is counted */ + r = damon_new_region(0, 4 * min_sz); + if (!r) { + damon_destroy_ctx(c); + kunit_skip(test, "region alloc fail"); + } + damon_add_region(r, t); + + damos_apply_scheme(c, t, r, s); + KUNIT_EXPECT_EQ(test, s->stat.sz_tried, 4 * min_sz); + KUNIT_EXPECT_EQ(test, damon_nr_regions(t), 1); + + /* no filter and no quota: the whole region is counted */ + c->ops.apply_scheme = damos_test_apply_scheme_stub; + s->stat = (struct damos_stat){}; + + damos_apply_scheme(c, t, r, s); + KUNIT_EXPECT_EQ(test, s->stat.sz_tried, 4 * min_sz); + KUNIT_EXPECT_EQ(test, damon_nr_regions(t), 1); + + /* the quota trims the region: the trimmed size is counted */ + damon_for_each_region_safe(r, next, t) + damon_destroy_region(r, t); + s->stat = (struct damos_stat){}; + s->quota.esz = 2 * min_sz; + s->quota.charged_sz = 0; + + r = damon_new_region(0, 8 * min_sz); + if (!r) { + damon_destroy_ctx(c); + kunit_skip(test, "region alloc fail"); + } + damon_add_region(r, t); + + damos_apply_scheme(c, t, r, s); + KUNIT_EXPECT_EQ(test, r->ar.end, 2 * min_sz); + KUNIT_EXPECT_EQ(test, s->stat.sz_tried, 2 * min_sz); + + /* quota remainder below one region: tried, but nothing counted */ + damon_for_each_region_safe(r, next, t) + damon_destroy_region(r, t); + s->stat = (struct damos_stat){}; + s->quota.esz = 1; + s->quota.charged_sz = 0; + + r = damon_new_region(0, 4 * min_sz); + if (!r) { + damon_destroy_ctx(c); + kunit_skip(test, "region alloc fail"); + } + damon_add_region(r, t); + + damos_apply_scheme(c, t, r, s); + KUNIT_EXPECT_EQ(test, s->stat.nr_tried, 1); + KUNIT_EXPECT_EQ(test, s->stat.sz_tried, 0); + KUNIT_EXPECT_EQ(test, r->ar.end, 4 * min_sz); + + damon_destroy_ctx(c); +} + static void damon_test_feed_loop_next_input(struct kunit *test) { unsigned long last_input = 900000, current_score = 200; @@ -1959,6 +2272,9 @@ static struct kunit_case damon_test_cases[] = { KUNIT_CASE(damon_test_commit_ctx), KUNIT_CASE(damon_test_valid_probe_params), KUNIT_CASE(damos_test_filter_out), + KUNIT_CASE(damos_test_apply_scheme_filtered_sz), + KUNIT_CASE(damos_test_apply_scheme_filter_sz_unchanged), + KUNIT_CASE(damos_test_apply_scheme_quota_sz), KUNIT_CASE(damon_test_feed_loop_next_input), KUNIT_CASE(damon_test_set_filters_default_reject), KUNIT_CASE(damon_test_apply_min_nr_regions), From f8ee3dd1a0897bec99f3ccaace2457525e2235de Mon Sep 17 00:00:00 2001 From: Liew Rui Yan Date: Mon, 21 Sep 2026 08:15:43 -0700 Subject: [PATCH 1129/1352] mm/damon/core: skip quota score setup when the quota is full Patch series "mm/damon: improvements in efficiency, error handling, documents". Four various improvements. Patch 1 from Liew Rui Yan skips unnecessary quota setup work when relevant. Patch 2 from Xuewen Wang propagates ignored damon_call() error to sysfs users. Patch 3 from Adrian Huang (Lenovo) fixes typos in comments. Patch 4 from Karthikeyan KS clarifies zero sample interval acceptance on kernel-doc comment. This patch (of 4): In damos_adjust_quota(), the quota could already be full. In this situation, damos_adjust_quota() will still calculates quota->min_score. However, this min_score will not be used in this window, because in damon_do_apply_schemes(), damos_quota_is_full() will always returns true, preventing the scheme from being applied to any region. Therefore, add a short circuit for 'esz < min_region_sz' schemes to early return from damos_adjust_quota() before calculating min_score. Link: https://lore.kernel.org/20260921151547.78472-1-sj@kernel.org Link: https://lore.kernel.org/20260921151547.78472-2-sj@kernel.org Signed-off-by: Liew Rui Yan Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Adrian Huang Cc: Karthikeyan KS Cc: Xuewen Wang Cc: Zenghui Yu (Huawei) --- mm/damon/core.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/mm/damon/core.c b/mm/damon/core.c index 8d8b0cba1a56c9..749897846b4c2b 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -3385,6 +3385,8 @@ static void damos_adjust_quota(struct damon_ctx *c, struct damos *s) damos_trace_esz(c, s, quota); } + if (damos_quota_is_full(quota, c->min_region_sz)) + return; if (!c->ops.get_scheme_score) return; From 91758fb31e4f7f6c3fb2c8f48bc4ddcc32d638d1 Mon Sep 17 00:00:00 2001 From: Xuewen Wang Date: Mon, 21 Sep 2026 08:15:44 -0700 Subject: [PATCH 1130/1352] mm/damon/sysfs: propagate damon_call() error in turn_damon_on damon_sysfs_turn_damon_on() ignored the return value of damon_call() for the repeat call control registration and always returned the stale result of damon_start() (0 at that point). When damon_call() fails, e.g., the kdamond is already exiting, the user still gets success from the state file write while monitoring is not actually on. Save and return the damon_call() result instead. No rollback of damon_start() is needed since a failed damon_call() guarantees the context is stopped. Link: https://lore.kernel.org/20260921151547.78472-3-sj@kernel.org Signed-off-by: Xuewen Wang Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Adrian Huang Cc: Karthikeyan KS Cc: Liew Rui Yan Cc: Zenghui Yu (Huawei) --- mm/damon/sysfs.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index 43519afb9eb7f3..70a4b64fb8ea9f 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -2607,7 +2607,8 @@ static int damon_sysfs_turn_damon_on(struct damon_sysfs_kdamond *kdamond) repeat_call_control->data = kdamond; repeat_call_control->repeat = true; repeat_call_control->dealloc_on_cancel = true; - if (damon_call(ctx, repeat_call_control)) + err = damon_call(ctx, repeat_call_control); + if (err) kfree(repeat_call_control); return err; } From 0ef60aabe6c0374c81492f7efdf76c3cbfed46b7 Mon Sep 17 00:00:00 2001 From: "Adrian Huang (Lenovo)" Date: Mon, 21 Sep 2026 08:15:45 -0700 Subject: [PATCH 1131/1352] mm/damon: fix typos in comments Correct spelling mistakes. No functional changes. Link: https://lore.kernel.org/20260921151547.78472-4-sj@kernel.org Signed-off-by: Adrian Huang (Lenovo) Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Zenghui Yu (Huawei) Reviewed-by: SJ Park Cc: Karthikeyan KS Cc: Liew Rui Yan Cc: Xuewen Wang --- mm/damon/core.c | 6 +++--- mm/damon/lru_sort.c | 2 +- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 749897846b4c2b..733025b367457b 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2486,7 +2486,7 @@ static bool damos_valid_target(struct damon_ctx *c, struct damon_region *r, * This function checks if a given region should be skipped or not for the * reason. If only the starting part of the region has previously charged, * this function splits the region into two so that the second one covers the - * area that not charged in the previous charge widnow, and return true. The + * area that not charged in the previous charge window, and return true. The * caller can see the second one on the next iteration of the region walk. * Note that this means the caller should use damon_for_each_region() instead * of damon_for_each_region_safe(). If damon_for_each_region_safe() is used, @@ -2652,7 +2652,7 @@ static void damos_walk_call_walk(struct damon_ctx *ctx, struct damon_target *t, * This function is called when kdamond finished applying the action of a DAMOS * scheme to all regions that eligible for the given &damos->apply_interval_us. * If every scheme of @ctx including @s now finished walking for at least one - * &damos->apply_interval_us, this function makrs the handling of the given + * &damos->apply_interval_us, this function marks the handling of the given * DAMOS walk request is done, so that damos_walk() can wake up and return. */ static void damos_walk_complete(struct damon_ctx *ctx, struct damos *s) @@ -4194,7 +4194,7 @@ static bool damon_find_system_rams_range(unsigned long *start, * This function sets the region of @t as requested by @start and @end. If the * values of @start and @end are zero, however, this function finds 'System * RAM' resources and sets the region to cover all the resource. In the latter - * case, this function saves the start and the end addresseses of the first and + * case, this function saves the start and the end addresses of the first and * the last resources in @start and @end, respectively. * * Return: 0 on success, negative error code otherwise. diff --git a/mm/damon/lru_sort.c b/mm/damon/lru_sort.c index ad8e86dd3a93e1..273efa3c913ed4 100644 --- a/mm/damon/lru_sort.c +++ b/mm/damon/lru_sort.c @@ -260,7 +260,7 @@ static int damon_lru_sort_add_filters(struct damos *hot_scheme, return -ENOMEM; damos_add_filter(hot_scheme, filter); - /* disabllow de-prioritizing young pages */ + /* disallow de-prioritizing young pages */ filter = damos_new_filter(DAMOS_FILTER_TYPE_YOUNG, true, false); if (!filter) return -ENOMEM; From d2a1e854b944dfb175f53a1c9bc3b3683067bf8c Mon Sep 17 00:00:00 2001 From: Karthikeyan KS Date: Mon, 21 Sep 2026 08:15:46 -0700 Subject: [PATCH 1132/1352] mm/damon: document that a zero sample_interval is accepted damon_set_attrs() accepts sample_interval == 0. This was reported as a bug in v1 of this patch (rejecting it in damon_set_attrs()). A similar patch was already declined for the same reason: a zero interval is intentionally supported [1]. Document the behavior instead of changing it. Link: https://lore.kernel.org/20260921151547.78472-5-sj@kernel.org Link: https://lore.kernel.org/all/20260722094304.3132750-1-dayou5941@163.com/ [1] Signed-off-by: Karthikeyan KS Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Adrian Huang Cc: Liew Rui Yan Cc: Xuewen Wang Cc: Zenghui Yu (Huawei) --- include/linux/damon.h | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index 836353c4ab9aab..844b175120f09b 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -836,7 +836,8 @@ struct damon_probe { /** * struct damon_attrs - Monitoring attributes for accuracy/overhead control. * - * @sample_interval: The time between access samplings. + * @sample_interval: The time between access samplings. Zero is + * accepted. * @aggr_interval: The time between monitor results aggregations. * @ops_update_interval: The time between monitoring operations updates. * @intervals_goal: Intervals auto-tuning goal. From 803f266addd0f0f651cd47ebaed838341c46eeaf Mon Sep 17 00:00:00 2001 From: Breno Leitao Date: Mon, 21 Sep 2026 05:30:26 -0700 Subject: [PATCH 1133/1352] mm: kmemleak: move the struct page scan into a helper Patch series "mm: kmemleak: batch the struct page scan". kmemleak walks the struct page array as a scan root and hands it to scan_block() one page at a time. scan_block() takes kmemleak_lock with interrupts disabled for the duration of the call, so the scanner acquires the lock once per online PFN in order to look at a single 64-byte struct page. Batching adjacent pages up to that size keeps exactly the same pages under the same maximum lock hold time, with MAX_SCAN_SIZE / pagesize fewer acquisitions. Patch 1 lifts the loop out of __kmemleak_scan() into scan_zone_pages() with no functional change. Patch 2 does the batching, which is then contained in that one function. This improves the performance due to less atomic operations, which is not a big deal on a regular machine, but, given kmemleak usually comes with extra debug options, such as PROVE_LOCKING, DEBUG_SPINLOCK, etc. For instance, measuring Meta's "debug kernel flavor" on an arm64 hosts, this improve the scan time by 20%. It is safe to get more work into scan_block(), given it has the protections, added by commit eb11f56eeca560 ("mm/kmemleak: stop the task stack scan early when interrupted") MAX_SCAN_SIZE is also not a new maximum for a single scan_block() call. kmemleak_scan_task_stacks() already hands it a whole task stack in one go. On arm64 and x86_64 THREAD_SIZE is never below 16 KiB, four times MAX_SCAN_SIZE, and it is 64 KiB on arm64 with 64K pages. This patch (of 2): The struct page scanning loop sits inline in __kmemleak_scan(), nested three levels deep and sharing the caller's "stop" variable with the zone walk around it. Move it into scan_zone_pages(), which scans one zone and returns 1 if scanning should stop. No functional change; this only makes room for changing how the pages are handed to scan_block(). Link: https://lore.kernel.org/20260921-b4-kmemleak-page-scan-v1-1-fb97d4801b3a@debian.org Signed-off-by: Breno Leitao Signed-off-by: Andrew Morton Reviewed-by: Catalin Marinas --- mm/kmemleak.c | 57 ++++++++++++++++++++++++++++++--------------------- 1 file changed, 34 insertions(+), 23 deletions(-) diff --git a/mm/kmemleak.c b/mm/kmemleak.c index 5d0daea93c471f..fe34c5650f42fc 100644 --- a/mm/kmemleak.c +++ b/mm/kmemleak.c @@ -1855,6 +1855,39 @@ static void dedup_flush(struct xarray *dedup) } } +/* + * Scan the struct pages of a zone, skipping memory holes, pages that belong to + * another zone and pages that are not in use. Returns 1 if the scan should be + * stopped. + */ +static int scan_zone_pages(struct zone *zone) +{ + unsigned long start_pfn = zone->zone_start_pfn; + unsigned long end_pfn = zone_end_pfn(zone); + unsigned long pfn; + + for (pfn = start_pfn; pfn < end_pfn; pfn++) { + struct page *page = pfn_to_online_page(pfn); + + if (!(pfn & 63)) + cond_resched_tasks_rcu_qs(); + + if (!page) + continue; + + /* only scan pages belonging to this zone */ + if (page_zone(page) != zone) + continue; + /* only scan if page is in use */ + if (page_count(page) == 0) + continue; + if (scan_block(page, page + 1, NULL)) + return 1; + } + + return 0; +} + /* * Scan data sections and all the referenced memory blocks allocated via the * kernel's standard allocators. This function must be called with the @@ -1928,29 +1961,7 @@ static int __kmemleak_scan(bool full) */ get_online_mems(); for_each_populated_zone(zone) { - unsigned long start_pfn = zone->zone_start_pfn; - unsigned long end_pfn = zone_end_pfn(zone); - unsigned long pfn; - - for (pfn = start_pfn; pfn < end_pfn; pfn++) { - struct page *page = pfn_to_online_page(pfn); - - if (!(pfn & 63)) - cond_resched_tasks_rcu_qs(); - - if (!page) - continue; - - /* only scan pages belonging to this zone */ - if (page_zone(page) != zone) - continue; - /* only scan if page is in use */ - if (page_count(page) == 0) - continue; - stop = scan_block(page, page + 1, NULL); - if (stop) - break; - } + stop = scan_zone_pages(zone); if (stop) break; } From 81a4818b8953e8361a6c461c372e72a1197bde61 Mon Sep 17 00:00:00 2001 From: Breno Leitao Date: Mon, 21 Sep 2026 05:30:27 -0700 Subject: [PATCH 1134/1352] mm: kmemleak: scan the struct page array in MAX_SCAN_SIZE batches scan_zone_pages() scans the struct page array one page per scan_block() call: if (scan_block(page, page + 1, NULL)) scan_block() takes kmemleak_lock with interrupts disabled for the duration of the call, so this acquires the lock once per online PFN to scan a single struct page. Gather runs of adjacent eligible struct pages and pass each run to scan_block() in one call, capped at MAX_SCAN_SIZE. The longest kmemleak_lock hold duration is now MAX_SCAN_SIZE worth of bytes. The same bound scan_large_block() already applies to the data sections and the per-CPU areas. This can make the scan faster. On an arm64 VM with 24 GiB and ~17 GiB in use, the struct page phase goes from 4390k scan_block() calls down to 69k for the same 4390k pages scanned, and the whole scan about 20% faster (on debug kernel). The win is in the per-acquisition cost, so on a kernel built without the lock debugging options the scan time is unchanged. I don't think it will be a problem doing more on scan_block(), given we have the scan_should_stop() protection. Coverage is unchanged: a temporary assertion comparing the number of eligible pages against the number actually passed to scan_block() matched on every zone of every scan, including while memory was being freed underneath the scan. Link: https://lore.kernel.org/20260921-b4-kmemleak-page-scan-v1-2-fb97d4801b3a@debian.org Signed-off-by: Breno Leitao Signed-off-by: Andrew Morton Reviewed-by: Catalin Marinas --- mm/kmemleak.c | 29 +++++++++++++++++++++-------- 1 file changed, 21 insertions(+), 8 deletions(-) diff --git a/mm/kmemleak.c b/mm/kmemleak.c index fe34c5650f42fc..20c0eb41c8730b 100644 --- a/mm/kmemleak.c +++ b/mm/kmemleak.c @@ -1862,8 +1862,11 @@ static void dedup_flush(struct xarray *dedup) */ static int scan_zone_pages(struct zone *zone) { + const unsigned int max_batch = MAX_SCAN_SIZE / sizeof(struct page); unsigned long start_pfn = zone->zone_start_pfn; unsigned long end_pfn = zone_end_pfn(zone); + struct page *first = NULL, *last = NULL; + unsigned int batch = 0; unsigned long pfn; for (pfn = start_pfn; pfn < end_pfn; pfn++) { @@ -1872,19 +1875,29 @@ static int scan_zone_pages(struct zone *zone) if (!(pfn & 63)) cond_resched_tasks_rcu_qs(); - if (!page) - continue; + /* only scan in-use pages belonging to this zone */ + if (page && (page_zone(page) != zone || + page_count(page) == 0)) + page = NULL; - /* only scan pages belonging to this zone */ - if (page_zone(page) != zone) - continue; - /* only scan if page is in use */ - if (page_count(page) == 0) + if (page && first && page == last + 1 && + batch < max_batch) { + last = page; + batch++; continue; - if (scan_block(page, page + 1, NULL)) + } + + if (first && scan_block(first, last + 1, NULL)) return 1; + + first = page; + last = page; + batch = page ? 1 : 0; } + if (first && scan_block(first, last + 1, NULL)) + return 1; + return 0; } From 96fb7b79bc9077118ac81ac18921539f26f66d38 Mon Sep 17 00:00:00 2001 From: Ridong Chen Date: Sun, 20 Sep 2026 21:25:19 +0800 Subject: [PATCH 1135/1352] mm: vmscan: put rotation-missed folios at the LRU tail The page reclaim isolates a batch of folios from the tail of an LRU list and works on them one by one. For a suitable swap-backed folio on an async swap device, it queues the folio for writeback and, after finishing the batch, puts the folio back to the head of the original LRU list. Meanwhile the page writeback flushes the queued folios in its own, independent batches. For each folio it writes back it calls folio_rotate_reclaimable(), which tries to rotate the folio to the LRU tail. But folio_rotate_reclaimable() only takes effect once the folio has been put back by reclaim. If the async swap device is fast enough, the writeback can complete a folio while reclaim is still working on the rest of the batch that contains it. In that case the folio stays near the head and reclaim will not revisit it before wrapping around, causing a cold/hot inversion: a clean, written-back folio that should be a prime reclaim candidate is kept ahead of hotter folios. commit 359a5e1416ca ("mm: multi-gen LRU: retry folios written back while isolated") addressed this for MGLRU only. The traditional active/inactive LRU has the same problem, reported at [1]. A reproducer is available at [2]. Rather than re-reclaiming those folios (which would drop the swap cache that may still be useful for a future hit [4]), restore the rotation that was missed: when move_folios_to_lru() puts a folio back, add it to the LRU tail if it looks like it missed folio_rotate_reclaimable() (inactive, not mapped, not dirty and not under writeback). A referenced folio is left at the head so it still gets a second chance, and a folio with an unexpected reference (e.g. a GUP or speculative pin) is left at the head because it cannot be reclaimed yet anyway. A new do_rotate parameter gates this so it only applies on the reclaim put-back path (shrink_inactive_list()), not on shrink_active_list() where the list order is already deliberate. This approach was suggested by Barry Song [3]. Only the traditional LRU is handled here. MGLRU already retries such folios via its own clean-list retry pass in evict_folios(), so it is left unchanged. The same do_rotate scheme could later replace that retry pass to unify both LRUs, which is left for a follow-up. Test result with [2]: Without patch: cat memory.usage_in_bytes 1073700864 cat memory.memsw.usage_in_bytes 1413124096 free -h total used free Mem: 1.6Gi 1.2Gi 299Mi Swap: 1.0Gi 678Mi 346Mi With patch: cat memory.usage_in_bytes 1071140864 cat memory.memsw.usage_in_bytes 1413423104 free -h total used free Mem: 1.6Gi 1.2Gi 322Mi Swap: 1.0Gi 328Mi 695Mi After applying the patch, the difference between memory.memsw.usage_in_bytes and memory.usage_in_bytes is close to the swap "used" value reported by 'free -h'. Link: https://lore.kernel.org/20260920132519.3369946-1-ridong.chen@linux.dev Link: https://lore.kernel.org/linux-kernel/20241010081802.290893-1-chenridong@huaweicloud.com/ [1] Link: https://lore.kernel.org/lkml/46037a37-4cf6-448e-a94b-30a4d16e8814@linux.dev/ [2] Link: https://lore.kernel.org/linux-mm/20260911121341.178028-1-alex@ghiti.fr/ [4] Signed-off-by: Ridong Chen Signed-off-by: Andrew Morton Suggested-by: Barry Song Link: https://lore.kernel.org/lkml/CAGsJ_4zwP3_+EYY5Ug9EJ+yD1UdxsBSGr25u8s1K3u_i7LH3Zg@mail.gmail.com/ [3] Reviewed-by: Barry Song Reviewed-by: Baolin Wang Cc: Johannes Weiner Cc: David Hildenbrand Cc: Michal Hocko Cc: Qi Zheng Cc: Shakeel Butt Cc: Lorenzo Stoakes Cc: Kairui Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He --- mm/vmscan.c | 24 ++++++++++++++++++------ 1 file changed, 18 insertions(+), 6 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index e200ce3eb056b1..91295070ca336c 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -1971,7 +1971,7 @@ static bool too_many_isolated(struct pglist_data *pgdat, int file, * * Note: The caller must not hold any lruvec lock. */ -static unsigned int move_folios_to_lru(struct list_head *list) +static unsigned int move_folios_to_lru(struct list_head *list, bool do_rotate) { int nr_pages, nr_moved = 0; struct lruvec *lruvec = NULL; @@ -2018,7 +2018,19 @@ static unsigned int move_folios_to_lru(struct list_head *list) continue; } - lruvec_add_folio(lruvec, folio); + /* + * Put clean, unreferenced and unpinned folios that may have + * missed folio_rotate_reclaimable() at the tail to avoid + * cold/hot inversion. + */ + if (do_rotate && !folio_test_active(folio) && !folio_mapped(folio) && + !folio_test_dirty(folio) && !folio_test_writeback(folio) && + !folio_test_referenced(folio) && + folio_ref_count(folio) == folio_expected_ref_count(folio)) + lruvec_add_folio_tail(lruvec, folio); + else + lruvec_add_folio(lruvec, folio); + nr_pages = folio_nr_pages(folio); nr_moved += nr_pages; if (folio_test_active(folio)) @@ -2135,7 +2147,7 @@ static unsigned long shrink_inactive_list(unsigned long nr_to_scan, nr_reclaimed = shrink_folio_list(&folio_list, pgdat, sc, &stat, false, lruvec_memcg(lruvec)); - move_folios_to_lru(&folio_list); + move_folios_to_lru(&folio_list, true); mod_lruvec_state(lruvec, PGDEMOTE_KSWAPD + reclaimer_offset(sc), stat.nr_demoted); @@ -2246,8 +2258,8 @@ static void shrink_active_list(unsigned long nr_to_scan, /* * Move folios back to the lru list. */ - nr_activate = move_folios_to_lru(&l_active); - nr_deactivate = move_folios_to_lru(&l_inactive); + nr_activate = move_folios_to_lru(&l_active, false); + nr_deactivate = move_folios_to_lru(&l_inactive, false); count_vm_events(PGDEACTIVATE, nr_deactivate); count_memcg_events(lruvec_memcg(lruvec), PGDEACTIVATE, nr_deactivate); @@ -5115,7 +5127,7 @@ static int evict_folios(unsigned long nr_to_scan, struct lruvec *lruvec, folio_set_active(folio); } - move_folios_to_lru(&list); + move_folios_to_lru(&list, false); walk = current->reclaim_state->mm_walk; if (walk && walk->batched) { From ae0b995aeba27fde75bc5c46184c2ba61f293d5d Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Wed, 23 Sep 2026 10:05:15 +0800 Subject: [PATCH 1136/1352] mm/hugetlb: account migration target folio in per-node NR_HUGETLB vmstat Patch series "mm: fix hugetlb NR_HUGETLB accounting on folio migration", v2. The hugeTLB counters added by 05d4532b60e3 ("memcg/hugetlb: add hugeTLB counters to memcg") are maintained in two per-node places: - the per-node vmstat counter NR_HUGETLB, exposed as nr_hugetlb in /proc/vmstat; - the per-node memcg lruvec stat, exposed via memory.numa_stat. Both are accounted against the folio's node, and both drift when a hugetlb folio is migrated, though in different ways. A migration target folio is allocated by alloc_hugetlb_folio_nodemask() and inherits the old folio's state without ever being accounted, while the old folio is freed right after and its free is accounted. That alone loses vmstat accounting: the target node has no matching increment for the decrement on the old node, so /proc/vmstat's nr_hugetlb shrinks by nr_pages per migration. Patch 1 accounts the folio where it is obtained, so the increment pairs with the free in free_huge_folio() on the successful as well as the failed migration path. The memfd page cache preallocation helper has the same asymmetry and is fixed in the same patch. The per-node lruvec stat breaks differently. mem_cgroup_migrate() moves the charge to the new folio and drops the old folio's memcg data, so the old folio's free right after migration skips the memcg per-node decrement; the count stays attributed to the old node for the rest of the charge's life, and the target folio never gets an increment on its new node. Patch 2 moves that per-node accounting along with the charge, in move_hugetlb_state(). This patch (of 2): The NR_HUGETLB vmstat counter is maintained per folio's node: incremented when a huge page is handed to a user via hugetlb_alloc_folio() and decremented when it is returned to the pool via free_huge_folio(). A folio obtained by alloc_hugetlb_folio_nodemask() never goes through hugetlb_alloc_folio(), so it is never accounted, while its free always is. For a migration target this means the target node gets no matching increment for the decrement on the old node, so the global nr_hugetlb in /proc/vmstat drops by nr_pages for each migration. The same asymmetry affects the failed migration path, which frees the target again right away, and the temporary folio hugetlb_mfill_atomic_pte() takes from the same helper. alloc_hugetlb_folio_reserve(), used to preallocate the memfd page cache folios, has the same asymmetry: the folio is handed to a user without being accounted, while its free is accounted through free_huge_folio(). Account the folio where it is obtained, in alloc_hugetlb_folio_nodemask() and alloc_hugetlb_folio_reserve(), so that the increment pairs with the decrement in free_huge_folio(): a successful migration hands the folio to a user, a failed one frees it again. Link: https://lore.kernel.org/20260923-for-hugetlb_state3-v2-0-e8a36245bfab@kylinos.cn Link: https://lore.kernel.org/20260923-for-hugetlb_state3-v2-1-e8a36245bfab@kylinos.cn Fixes: 05d4532b60e3 ("memcg/hugetlb: add hugeTLB counters to memcg") Signed-off-by: Hongfu Li Signed-off-by: Andrew Morton Tested-by: Joshua Hahn Reviewed-by: Joshua Hahn Acked-by: Muchun Song Acked-by: Oscar Salvador Cc: David Hildenbrand Cc: Shakeel Butt Cc: Michal Hocko Cc: Johannes Weiner Cc: Nhat Pham Cc: Michal Hocko Cc: Roman Gushchin Cc: Chris Down Cc: --- mm/hugetlb.c | 35 +++++++++++++++++++++++------------ 1 file changed, 23 insertions(+), 12 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index da980377d35339..519c30b338a8b0 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -2206,6 +2206,11 @@ struct folio *alloc_hugetlb_folio_reserve(struct hstate *h, int preferred_nid, } spin_unlock_irq(&hugetlb_lock); + + if (folio) + lruvec_stat_mod_folio(folio, NR_HUGETLB, + folio_nr_pages(folio)); + return folio; } @@ -2213,24 +2218,30 @@ struct folio *alloc_hugetlb_folio_reserve(struct hstate *h, int preferred_nid, struct folio *alloc_hugetlb_folio_nodemask(struct hstate *h, int preferred_nid, nodemask_t *nmask, gfp_t gfp_mask, bool allow_alloc_fallback) { - spin_lock_irq(&hugetlb_lock); - if (available_huge_pages(h)) { - struct folio *folio; + struct folio *folio = NULL; + spin_lock_irq(&hugetlb_lock); + if (available_huge_pages(h)) folio = dequeue_hugetlb_folio_nodemask(h, gfp_mask, preferred_nid, nmask); - if (folio) { - spin_unlock_irq(&hugetlb_lock); - return folio; - } - } spin_unlock_irq(&hugetlb_lock); - /* We cannot fallback to other nodes, as we could break the per-node pool. */ - if (!allow_alloc_fallback) - gfp_mask |= __GFP_THISNODE; + if (!folio) { + /* + * We cannot fallback to other nodes, as we could break the + * per-node pool. + */ + if (!allow_alloc_fallback) + gfp_mask |= __GFP_THISNODE; - return alloc_migrate_hugetlb_folio(h, gfp_mask, preferred_nid, nmask); + folio = alloc_migrate_hugetlb_folio(h, gfp_mask, preferred_nid, + nmask); + } + + if (folio) + lruvec_stat_mod_folio(folio, NR_HUGETLB, folio_nr_pages(folio)); + + return folio; } static nodemask_t *policy_mbind_nodemask(gfp_t gfp) From 9327ce41b657c2c0b24e677c67f95ed205a00125 Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Wed, 23 Sep 2026 10:05:16 +0800 Subject: [PATCH 1137/1352] mm/memcg: migrate per-node hugetlb lruvec stat together with hugetlb folio memory.numa_stat exposes per-node hugetlb counters from per-node lruvec stats. These stats are accounted against folio_nid(): incremented on the folio's node when handed to a user, decremented when the folio is returned to the pool. During hugetlb folio migration, mem_cgroup_migrate() moves the charge to the new folio and drops the memcg data of the old one, so the free of the old folio right after the migration skips the memcg per-node lruvec decrement. The hugetlb count stays attributed to the old node for the rest of the life of the charge, while the target folio gets no increment on the new node; its later free decrements a counter that was never incremented. Migrate the per-node lruvec accounting alongside migration. Global memcg totals remain balanced because they track resource consumption, not node placement. Link: https://lore.kernel.org/20260923-for-hugetlb_state3-v2-2-e8a36245bfab@kylinos.cn Fixes: 05d4532b60e3 ("memcg/hugetlb: add hugeTLB counters to memcg") Signed-off-by: Hongfu Li Signed-off-by: Andrew Morton Tested-by: Joshua Hahn Reviewed-by: Joshua Hahn Reviewed-by: Oscar Salvador Acked-by: Muchun Song Cc: David Hildenbrand Cc: Shakeel Butt Cc: Michal Hocko Cc: Johannes Weiner Cc: Nhat Pham Cc: Michal Hocko Cc: Roman Gushchin Cc: Chris Down Cc: --- include/linux/memcontrol.h | 8 ++++++++ mm/hugetlb.c | 25 +++++++++++++++++++++++++ mm/memcontrol.c | 5 ++--- 3 files changed, 35 insertions(+), 3 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 64c183be8cbfe7..cd223f60b3a245 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -983,6 +983,9 @@ unsigned long lruvec_page_state_monotonic(const struct lruvec *lruvec, unsigned long lruvec_page_state_local(const struct lruvec *lruvec, enum node_stat_item idx); +void mod_memcg_lruvec_state(struct lruvec *lruvec, + enum node_stat_item idx, int val); + void mem_cgroup_flush_stats(struct mem_cgroup *memcg); void mem_cgroup_flush_stats_ratelimited(struct mem_cgroup *memcg); @@ -1451,6 +1454,11 @@ static inline unsigned long lruvec_page_state_local(const struct lruvec *lruvec, return node_page_state(lruvec_pgdat(lruvec), idx); } +static inline void mod_memcg_lruvec_state(struct lruvec *lruvec, + enum node_stat_item idx, int val) +{ +} + static inline void mem_cgroup_flush_stats(struct mem_cgroup *memcg) { } diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 519c30b338a8b0..76d019594b39c3 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -23,6 +23,7 @@ #include #include #include +#include #include #include #include @@ -7378,12 +7379,36 @@ void folio_putback_hugetlb(struct folio *folio) folio_put(folio); } +static void move_hugetlb_lruvec_stat(struct folio *old_folio, + struct folio *new_folio) +{ + struct mem_cgroup *memcg; + long nr_pages = folio_nr_pages(old_folio); + int old_nid = folio_nid(old_folio); + int new_nid = folio_nid(new_folio); + + if (old_nid == new_nid) + return; + + guard(rcu)(); + + memcg = folio_memcg(new_folio); + if (!memcg) + return; + + mod_memcg_lruvec_state(mem_cgroup_lruvec(memcg, NODE_DATA(old_nid)), + NR_HUGETLB, -nr_pages); + mod_memcg_lruvec_state(mem_cgroup_lruvec(memcg, NODE_DATA(new_nid)), + NR_HUGETLB, nr_pages); +} + void move_hugetlb_state(struct folio *old_folio, struct folio *new_folio, enum migrate_reason reason) { struct hstate *h = folio_hstate(old_folio); hugetlb_cgroup_migrate(old_folio, new_folio); + move_hugetlb_lruvec_stat(old_folio, new_folio); folio_set_owner_migrate_reason(new_folio, reason); /* diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 4d00748c8a5b87..f2ad4de8d80569 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -1015,9 +1015,8 @@ static void __mod_memcg_lruvec_state(struct mem_cgroup_per_node *pn, put_cpu(); } -static void mod_memcg_lruvec_state(struct lruvec *lruvec, - enum node_stat_item idx, - int val) +void mod_memcg_lruvec_state(struct lruvec *lruvec, + enum node_stat_item idx, int val) { struct pglist_data *pgdat = lruvec_pgdat(lruvec); struct mem_cgroup_per_node *pn; From 9e5c50a1bb0333e2fa0d06f29b7081e6e6249599 Mon Sep 17 00:00:00 2001 From: Zhenghui Hao Date: Wed, 23 Sep 2026 10:28:58 +0800 Subject: [PATCH 1138/1352] hugetlbfs: fix stale comment in hugetlbfs_file_mmap() The comment above the VMA flag setup in hugetlbfs_file_mmap() says the VMA address alignment has already been checked by prepare_hugepage_range(). That function no longer exists: it was removed by commit eff41389d824 ("mm/hugetlb: remove prepare_hugepage_range()"), and the check now lives in hugetlb_get_unmapped_area(). The comment also mentions ia64, which was removed from the tree by commit cf8e8658100d ("arch: Remove Itanium (IA-64) architecture"). Point the comment at the current function and drop the ia64 reference. The rest of the comment, which tells future edits to add any error return only after VM_HUGETLB has been set, is still accurate and is left untouched. No functional change. Link: https://lore.kernel.org/tencent_AA61551D1F50E61F46C8542F5C8874512C05@qq.com Signed-off-by: Zhenghui Hao Signed-off-by: Andrew Morton Acked-by: Muchun Song Cc: Oscar Salvador Cc: David Hildenbrand --- fs/hugetlbfs/inode.c | 9 ++++----- 1 file changed, 4 insertions(+), 5 deletions(-) diff --git a/fs/hugetlbfs/inode.c b/fs/hugetlbfs/inode.c index ba7097d5720c07..f643855d2acc70 100644 --- a/fs/hugetlbfs/inode.c +++ b/fs/hugetlbfs/inode.c @@ -106,11 +106,10 @@ static int hugetlbfs_file_mmap(struct file *file, struct vm_area_struct *vma) /* * vma address alignment (but not the pgoff alignment) has - * already been checked by prepare_hugepage_range. If you add - * any error returns here, do so after setting VM_HUGETLB, so - * vma_is_hugetlb tests below unmap_region go the right - * way when do_mmap unwinds (may be important on powerpc - * and ia64). + * already been checked by hugetlb_get_unmapped_area(). If you + * add any error returns here, do so after setting VM_HUGETLB, + * so vma_is_hugetlb tests below unmap_region go the right + * way when do_mmap unwinds (may be important on powerpc). */ vma_set_flags(vma, VMA_HUGETLB_BIT, VMA_DONTEXPAND_BIT); vma->vm_ops = &hugetlb_vm_ops; From b3bcacefb88731f445ce3aed3860c4991f3625b3 Mon Sep 17 00:00:00 2001 From: Yeoreum Yun Date: Tue, 29 Sep 2026 11:33:18 +0100 Subject: [PATCH 1139/1352] kselftest: mm: return fail when child test result is fail in khugepaged Patch series "kselftest: mm: fix intermittent failure khugepaged test", v4. There are intermittent failures in collapse_max_ptes_swap() and collapse_max_ptes_shared() when using the khugepaged_context: # Run test: collapse_max_ptes_shared (khugepaged:anon) # Allocate huge page... OK # Share huge page over fork()... OK # Trigger CoW on page 1023 of 2048... OK # Maybe collapse with max_ptes_shared exceeded.... OK # Trigger CoW on page 1024 of 2048... Fail Bail out! Unexpected huge page # Planned tests != run tests (26 != 23) # Totals: pass:23 fail:0 xfail:0 xpass:0 skip:0 error:0 # Run test: collapse_max_ptes_swap (khugepaged:anon) # Swapout 257 of 2048 pages... OK # Maybe collapse with max_ptes_swap exceeded.... OK # Swapout 256 of 2048 pages... OK Bail out! Unexpected huge page # Planned tests != run tests (26 != 17) # Totals: pass:17 fail:0 xfail:0 xpass:0 skip:0 error:0 This happens because khugepaged may collapse the pages before wait_for_scan() is called, causing a sanity check that expects uncollapsed pages to fail. For example, in collapse_max_ptes_swap(), after faulting the pages back in and paging out up to max_ptes_swap pages, khugepaged may collapse them again before c->collapse() is called. To prevent this, mark the VMA with MADV_NOHUGEPAGE after it has been collapsed by wait_for_scan() for anon. This prevents khugepaged from collapsing it again before c->collapse() is called. Also, fix false-positive results when a child process fails in tests such as collapse_fork*() or collapse_max_ptes_shared(): # ------------------------- # running ./khugepaged -s 2 # ------------------------- # # Run test: collapse_max_ptes_shared (khugepaged:anon) # Allocate huge page... OK # Share huge page over fork()... OK # Trigger CoW on page 1023 of 2048... OK # Maybe collapse with max_ptes_shared exceeded.... OK # Trigger CoW on page 1024 of 2048... Fail Bail out! Unexpected huge page # Planned tests != run tests (26 != 23) # Totals: pass:23 fail:0 xfail:0 xpass:0 skip:0 error:0 // child failed. # Check if parent still has huge page... OK // parent hpage success ok 24 collapse_max_ptes_shared // considered as success ... # Totals: pass:26 fail:0 xfail:0 xpass:0 skip:0 error:0 This failure was observed on NVIDIA Spark with 16KB page. This patch (of 2): Although the child process in collapse_fork*() or collapse_max_ptes_shared() reports `KSFT_FAIL`, the result is ignored because the test only checks whether the parent's page was collapsed into a huge page. As a result, the test is considered successful whenever the parent's page is a huge page, even if the child test fails, as shown below: # # Run test: collapse_max_ptes_shared (khugepaged:anon) # Allocate huge page... OK # Share huge page over fork()... OK # Trigger CoW on page 1023 of 2048... OK # Maybe collapse with max_ptes_shared exceeded.... OK # Trigger CoW on page 1024 of 2048... Fail Bail out! Unexpected huge page # Planned tests != run tests (26 != 23) # Totals: pass:23 fail:0 xfail:0 xpass:0 skip:0 error:0 // child failed. # Check if parent still has huge page... OK // parent hpage success ok 24 collapse_max_ptes_shared // considered as success ... # Totals: pass:26 fail:0 xfail:0 xpass:0 skip:0 error:0 To address this, propagate the child's failure and skip the subsequent check in the parent. Link: https://lore.kernel.org/20260929-fix_khugepagd_fail-v4-0-2169c18f2576@arm.com Link: https://lore.kernel.org/20260929-fix_khugepagd_fail-v4-1-2169c18f2576@arm.com Signed-off-by: Yeoreum Yun Signed-off-by: Andrew Morton Reviewed-by: Baolin Wang Reviewed-by: Gregory Price (Meta) Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) Acked-by: Zi Yan Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Kiryl Shutsemau Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Shuah Khan --- tools/testing/selftests/mm/khugepaged.c | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index a2ac3b3ca5def3..2aa7c919715848 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -1053,6 +1053,8 @@ static void collapse_fork(struct collapse_context *c, struct mem_ops *ops) wait(&wstatus); exit_status = WEXITSTATUS(wstatus); + if (exit_status == KSFT_FAIL) + goto out; ksft_print_msg("Check if parent still has small page..."); if (ops->check_huge(p, hpage_pmd_size, 0, hpage_pmd_size)) @@ -1060,6 +1062,7 @@ static void collapse_fork(struct collapse_context *c, struct mem_ops *ops) else fail("Fail"); validate_memory(p, 0, page_size); +out: ops->cleanup_area(p, hpage_pmd_size); ksft_test_result_report(exit_status, "%s\n", __func__); } @@ -1100,6 +1103,8 @@ static void collapse_fork_compound(struct collapse_context *c, struct mem_ops *o wait(&wstatus); exit_status = WEXITSTATUS(wstatus); + if (exit_status == KSFT_FAIL) + goto out; ksft_print_msg("Check if parent still has huge page..."); if (ops->check_huge(p, hpage_pmd_size, 1, hpage_pmd_size)) @@ -1107,6 +1112,7 @@ static void collapse_fork_compound(struct collapse_context *c, struct mem_ops *o else fail("Fail"); validate_memory(p, 0, hpage_pmd_size); +out: ops->cleanup_area(p, hpage_pmd_size); ksft_test_result_report(exit_status, "%s\n", __func__); } @@ -1158,6 +1164,8 @@ static void collapse_max_ptes_shared(struct collapse_context *c, struct mem_ops wait(&wstatus); exit_status = WEXITSTATUS(wstatus); + if (exit_status == KSFT_FAIL) + goto out; ksft_print_msg("Check if parent still has huge page..."); if (ops->check_huge(p, hpage_pmd_size, 1, hpage_pmd_size)) @@ -1165,6 +1173,7 @@ static void collapse_max_ptes_shared(struct collapse_context *c, struct mem_ops else fail("Fail"); validate_memory(p, 0, hpage_pmd_size); +out: ops->cleanup_area(p, hpage_pmd_size); ksft_test_result_report(exit_status, "%s\n", __func__); } From 961054966ffb7a2653bd78cc40eec01ae25ddb31 Mon Sep 17 00:00:00 2001 From: Yeoreum Yun Date: Tue, 29 Sep 2026 11:33:19 +0100 Subject: [PATCH 1140/1352] kselftest: mm: fix intermittent failure khugepaged test There are intermittent failures in collapse_max_ptes_swap() and collapse_max_ptes_shared() when using the khugepaged_context: // while running ./khugepaged -s 2 # Run test: collapse_max_ptes_shared (khugepaged:anon) # Allocate huge page... OK # Share huge page over fork()... OK # Trigger CoW on page 1023 of 2048... OK # Maybe collapse with max_ptes_shared exceeded.... OK # Trigger CoW on page 1024 of 2048... Fail Bail out! Unexpected huge page # Planned tests != run tests (26 != 23) # Totals: pass:23 fail:0 xfail:0 xpass:0 skip:0 error:0 # Run test: collapse_max_ptes_swap (khugepaged:anon) # Swapout 257 of 2048 pages... OK # Maybe collapse with max_ptes_swap exceeded.... OK # Swapout 256 of 2048 pages... OK Bail out! Unexpected huge page # Planned tests != run tests (26 != 17) # Totals: pass:17 fail:0 xfail:0 xpass:0 skip:0 error:0 This happens because khugepaged may collapse the pages before wait_for_scan() is called, causing a sanity check that expects uncollapsed pages to fail. For example, in collapse_max_ptes_swap(), after faulting the pages back in and paging out up to max_ptes_swap pages, khugepaged may collapse them again before c->collapse() is called. To prevent this, mark the VMA with MADV_NOHUGEPAGE after it has been collapsed by wait_for_scan() for anon. This prevents khugepaged from collapsing it again before c->collapse() is called. This failure was observed on NVIDIA Spark with 16KB page. Link: https://lore.kernel.org/20260929-fix_khugepagd_fail-v4-2-2169c18f2576@arm.com Signed-off-by: Yeoreum Yun Signed-off-by: Andrew Morton Reviewed-by: Gregory Price (Meta) Reviewed-by: Baolin Wang Tested-by: Baolin Wang Acked-by: Lorenzo Stoakes (ARM) Cc: Zi Yan Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Kiryl Shutsemau Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: David Hildenbrand Cc: Shuah Khan --- tools/testing/selftests/mm/khugepaged.c | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 2aa7c919715848..4e888b7bf31007 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -618,6 +618,16 @@ static bool wait_for_scan(const char *msg, char *p, size_t len, usleep(TICK); } + /* + * The file and shmem tests rely on refaults to install PMD mappings + * after collapse. MADV_NOHUGEPAGE would prevent those mappings. + * + * Apply MADV_NOHUGEPAGE only to anonymous VMAs to prevent khugepaged + * from unexpectedly collapsing pages during the test. + */ + if (is_anon(ops)) + madvise(p, len, MADV_NOHUGEPAGE); + return timeout == -1; } From ea5bb88072221ef68de1f8303dad37ebf6423085 Mon Sep 17 00:00:00 2001 From: Bingfang Guo Date: Mon, 21 Sep 2026 14:16:24 +0800 Subject: [PATCH 1141/1352] memcg: keep swap charging under RCU protection Patch series "memcg: move memcgid refcount to objcg to unpin dying memcgs", v2. Although the dying memcg problem caused by LRU pages is fixed, I can still see many dying memcgs on some workloads that use shmem and those pages are swapped out. For example, programs populating logs to tmpfs or containers sharing data using shmem. This series binds the memcgid refcount to objcgs so dying memcgs can be freed normally in this case. The memcg private ID identifies memcgs for objects that can outlive the cgroup itself: swap entries and workingset shadows. Today the ID's refcount is embedded in the css, and every outstanding ID reference (mostly swap entries) pins the css, keeping the entire memcg alive. This causes a problem: A swapped-out page holds a memcgid reference that pins the css, so the memcg cannot be freed until the page is swapped back in and charged back to its online parent. The work done by Muchun Song and Qi Zheng already charges folios to the objcg, which is reparented to its parent when the memcg offlines. This series applies similar idea to the memcg private ID: the ID's refcount moves from the css into the objcg, and the memcgid xarray holds a reference to an objcg instead of pinning the css. When the memcg offlines, the objcg is reparented and any remaining memcgid references resolve to the ancestor, so swapped-out pages no longer pin the dying memcg and get the online parent naturally on swapin. Unbinding the ID from the memcg has three consequences the series has to deal with: 1. The ID stops pinning the memcg, so the paths that relied on the ID reference to keep the memcg alive have to hold the RCU read lock instead. (Patch 1) 2. Charge and uncharge no longer necessarily happen on the same memcg: swapout charges the folio's memcg, while the slot free resolves the nearest live ancestor. The counters are hierarchical, and the MEMCG_SWAP stat is either reparented at offline (v1) or not visible (v2), so nothing leaks. But "does this entry carry a counter charge at all" can no longer be answered from the resolved memcg: root is skipped only because root's swap is not accounted, and a non-root ID whose memcg was reparented into root still carries a charge that must be released. Patch 2 adds mem_cgroup_private_id_is_root() and makes all three swap paths decide on the ID's root status. 3. An ID can now outlive the memcg it was allocated to, so the memcg resolved from an ID is not necessarily the memcg the ID was handed out for. Callers that need exactly that memcg (list_lru, the workingset and MGLRU shadow tests) now get NULL and skip the entry, while the swap paths, which only need something to account to, get the nearest live ancestor. (Patch 4) The series is four patches. The first three are preparation that keeps today's semantics while the ID is still bound to the css. Only the last patch changes behavior. This patch (of 4): This is a preparatory work for unbinding memcgid from memcg. No functional change. The swap charging path currently drops its RCU read lock after acquiring a private ID reference. This is safe because the ID reference pins the memcg's CSS. After ID references are moved to pin objcgs, that lifetime guarantee will no longer hold. Keep the RCU read lock held while accessing the memcg for counter charging, statistics and failure handling. (This matches what __memcg1_swapout() already does). Save the private ID for swap memcg recording before dropping the RCU read lock. So the swap cluster locking remains outside the RCU read-side critical section. Link: https://lore.kernel.org/20260921-bingfangguo-memcgid-rework-v2-0-6c0637dc0edb@tencent.com Link: https://lore.kernel.org/20260921-bingfangguo-memcgid-rework-v2-1-6c0637dc0edb@tencent.com Signed-off-by: Bingfang Guo Signed-off-by: Andrew Morton Acked-by: Muchun Song Cc: Johannes Weiner Cc: Michal Hocko Cc: Roman Gushchin Cc: Shakeel Butt Cc: Dave Chinner Cc: Qi Zheng Cc: Kairui Song Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Bingfang Guo --- mm/memcontrol.c | 38 +++++++++++++++++++------------------- 1 file changed, 19 insertions(+), 19 deletions(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index f2ad4de8d80569..059eaca173bb20 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -5954,6 +5954,7 @@ int __mem_cgroup_try_charge_swap(struct folio *folio) struct page_counter *counter; struct mem_cgroup *memcg; struct obj_cgroup *objcg; + unsigned short private_id; if (do_memsw_account()) return 0; @@ -5963,30 +5964,29 @@ int __mem_cgroup_try_charge_swap(struct folio *folio) if (!objcg) return 0; - rcu_read_lock(); - memcg = obj_cgroup_memcg(objcg); - if (!folio_test_swapcache(folio)) { - memcg_memory_event(memcg, MEMCG_SWAP_FAIL); - rcu_read_unlock(); - return 0; - } + scoped_guard(rcu) { + memcg = obj_cgroup_memcg(objcg); + if (!folio_test_swapcache(folio)) { + memcg_memory_event(memcg, MEMCG_SWAP_FAIL); + return 0; + } - memcg = mem_cgroup_private_id_get_online(memcg, nr_pages); - /* memcg is pined by memcg ID. */ - rcu_read_unlock(); + memcg = mem_cgroup_private_id_get_online(memcg, nr_pages); + /* memcg is pined by memcg ID. */ + private_id = mem_cgroup_private_id(memcg); - if (!mem_cgroup_is_root(memcg) && - !page_counter_try_charge(&memcg->swap, nr_pages, &counter)) { - memcg_memory_event(memcg, MEMCG_SWAP_MAX); - memcg_memory_event(memcg, MEMCG_SWAP_FAIL); - mem_cgroup_private_id_put(memcg, nr_pages); - return -ENOMEM; + if (!mem_cgroup_is_root(memcg) && + !page_counter_try_charge(&memcg->swap, nr_pages, &counter)) { + memcg_memory_event(memcg, MEMCG_SWAP_MAX); + memcg_memory_event(memcg, MEMCG_SWAP_FAIL); + mem_cgroup_private_id_put(memcg, nr_pages); + return -ENOMEM; + } + mod_memcg_state(memcg, MEMCG_SWAP, nr_pages); } - mod_memcg_state(memcg, MEMCG_SWAP, nr_pages); ci = swap_cluster_get_and_lock(folio); - __swap_cgroup_set(ci, swp_cluster_offset(folio->swap), nr_pages, - mem_cgroup_private_id(memcg)); + __swap_cgroup_set(ci, swp_cluster_offset(folio->swap), nr_pages, private_id); swap_cluster_unlock(ci); return 0; From a819dca6ea69ec3fbe25bcb10f076c2184c0f367 Mon Sep 17 00:00:00 2001 From: Bingfang Guo Date: Mon, 21 Sep 2026 14:16:25 +0800 Subject: [PATCH 1142/1352] memcg: base swap charge accounting on memcgid root status A private ID currently pins its original memcg, so testing whether the ID belongs to root is equivalent to testing whether the resolved memcg is root. That equivalence will no longer hold when IDs refer to objcgs. A non-root ID may resolve to root after reparenting, but its swap entries still carry counter charges inherited by root. Skipping their uncharge based on the resolved memcg would leave those charges behind. Use the private ID's root status to decide whether a swap entry carries a counter charge. A root-ID entry carries none, while a non-root-ID entry must release its charge even if its current accounting memcg has become root. This patch adds a new helper to check if the memcgid equals to that of the root memcg. For swap uncharging and v2 swap charging, simply decide whether to charge/uncharge memsw or swap counter based on the swap memcgid is root or not. For v1 swapout, don't recharge the memsw counter, just cancel the charge if the swap memcg ID refers to the root memcg. This makes memory and swap charging have similar semantics: one relys on the objcg's root status, and the other relys on the memcgid's root status. Link: https://lore.kernel.org/20260921-bingfangguo-memcgid-rework-v2-2-6c0637dc0edb@tencent.com Signed-off-by: Bingfang Guo Signed-off-by: Andrew Morton Acked-by: Muchun Song Cc: Johannes Weiner Cc: Michal Hocko Cc: Roman Gushchin Cc: Shakeel Butt Cc: Dave Chinner Cc: Qi Zheng Cc: Kairui Song Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Bingfang Guo --- mm/memcontrol-v1.c | 16 ++++++++-------- mm/memcontrol-v1.h | 5 +++++ mm/memcontrol.c | 4 ++-- 3 files changed, 15 insertions(+), 10 deletions(-) diff --git a/mm/memcontrol-v1.c b/mm/memcontrol-v1.c index bf2c7d53b01b1c..ed015fdd951233 100644 --- a/mm/memcontrol-v1.c +++ b/mm/memcontrol-v1.c @@ -271,6 +271,7 @@ void __memcg1_swapout(struct folio *folio, struct swap_cluster_info *ci) struct mem_cgroup *memcg, *swap_memcg; struct obj_cgroup *objcg; unsigned int nr_entries; + unsigned short private_id; VM_WARN_ON_ONCE_FOLIO(!folio_test_swapcache(folio), folio); VM_WARN_ON_ONCE_FOLIO(!folio_test_locked(folio), folio); @@ -293,25 +294,24 @@ void __memcg1_swapout(struct folio *folio, struct swap_cluster_info *ci) /* * In case the memcg owning these pages has been offlined and doesn't * have an ID allocated to it anymore, charge the closest online - * ancestor for the swap instead and transfer the memory+swap charge. + * ancestor for the swap instead and cancel the memory+swap charge + * if the ID refers to the root memcg. */ nr_entries = folio_nr_pages(folio); swap_memcg = mem_cgroup_private_id_get_online(memcg, nr_entries); + private_id = mem_cgroup_private_id(swap_memcg); mod_memcg_state(swap_memcg, MEMCG_SWAP, nr_entries); - __swap_cgroup_set(ci, swp_cluster_offset(folio->swap), nr_entries, - mem_cgroup_private_id(swap_memcg)); + __swap_cgroup_set(ci, swp_cluster_offset(folio->swap), nr_entries, private_id); folio_unqueue_deferred_split(folio); folio->memcg_data = 0; - if (!obj_cgroup_is_root(objcg)) + if (!obj_cgroup_is_root(objcg)) { page_counter_uncharge(&memcg->memory, nr_entries); - if (memcg != swap_memcg) { - if (!mem_cgroup_is_root(swap_memcg)) - page_counter_charge(&swap_memcg->memsw, nr_entries); - page_counter_uncharge(&memcg->memsw, nr_entries); + if (mem_cgroup_private_id_is_root(private_id)) + page_counter_uncharge(&memcg->memsw, nr_entries); } /* diff --git a/mm/memcontrol-v1.h b/mm/memcontrol-v1.h index 0952b2a783e524..7286456125dbef 100644 --- a/mm/memcontrol-v1.h +++ b/mm/memcontrol-v1.h @@ -22,6 +22,11 @@ void drain_all_stock(struct mem_cgroup *root_memcg); int memory_stat_show(struct seq_file *m, void *v); +static inline bool mem_cgroup_private_id_is_root(unsigned short id) +{ + return id == mem_cgroup_private_id(root_mem_cgroup); +} + struct mem_cgroup *mem_cgroup_private_id_get_online(struct mem_cgroup *memcg, unsigned int n); diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 059eaca173bb20..c43b0917851248 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -5975,7 +5975,7 @@ int __mem_cgroup_try_charge_swap(struct folio *folio) /* memcg is pined by memcg ID. */ private_id = mem_cgroup_private_id(memcg); - if (!mem_cgroup_is_root(memcg) && + if (!mem_cgroup_private_id_is_root(private_id) && !page_counter_try_charge(&memcg->swap, nr_pages, &counter)) { memcg_memory_event(memcg, MEMCG_SWAP_MAX); memcg_memory_event(memcg, MEMCG_SWAP_FAIL); @@ -6004,7 +6004,7 @@ void __mem_cgroup_uncharge_swap(unsigned short id, unsigned int nr_pages) rcu_read_lock(); memcg = mem_cgroup_from_private_id(id); if (memcg) { - if (!mem_cgroup_is_root(memcg)) { + if (!mem_cgroup_private_id_is_root(id)) { if (do_memsw_account()) page_counter_uncharge(&memcg->memsw, nr_pages); else From 9f8924c0f280abceb89e830f29c77f7a11ea6daf Mon Sep 17 00:00:00 2001 From: Bingfang Guo Date: Mon, 21 Sep 2026 14:16:26 +0800 Subject: [PATCH 1143/1352] memcg: manipulate memcg private ID references by ID This is a preparatory work for moving memcgid from memcg to objcg. Swap entries retain a private ID rather than a memcg pointer. Once private ID references are moved to objcgs, the ID can also outlive the memcg to which it was originally assigned. So it's better to make the get and put functions accept the ID itself instead of the memcg. Rename mem_cgroup_private_id_get_online() to mem_cgroup_private_id_get(), and make it return the ID only. If the memcg is already dying, the dying memcg will still be used for charging and stats accounting in v2 swap charging path. But they are hierarchical and will be reparented after offlining so it doesn't matter. Make mem_cgroup_private_id_put() take the ID and resolve the reference holder internally. Convert swap uncharge and charge rollback to release the reference using that ID. This introduces an extra xarray lookup for now, which will be removed in the final patch. Separate the online-state reference release into mem_cgroup_private_id_kill(). The offline path already has the memcg pointer and can call the underlying put helper directly. Link: https://lore.kernel.org/20260921-bingfangguo-memcgid-rework-v2-3-6c0637dc0edb@tencent.com Signed-off-by: Bingfang Guo Signed-off-by: Andrew Morton Acked-by: Muchun Song Cc: Johannes Weiner Cc: Michal Hocko Cc: Roman Gushchin Cc: Shakeel Butt Cc: Dave Chinner Cc: Qi Zheng Cc: Kairui Song Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Bingfang Guo --- mm/memcontrol-v1.c | 7 +++---- mm/memcontrol-v1.h | 3 +-- mm/memcontrol.c | 32 +++++++++++++++++++++++--------- 3 files changed, 27 insertions(+), 15 deletions(-) diff --git a/mm/memcontrol-v1.c b/mm/memcontrol-v1.c index ed015fdd951233..b7f28688850715 100644 --- a/mm/memcontrol-v1.c +++ b/mm/memcontrol-v1.c @@ -268,7 +268,7 @@ void memcg1_commit_charge(struct folio *folio, struct mem_cgroup *memcg) */ void __memcg1_swapout(struct folio *folio, struct swap_cluster_info *ci) { - struct mem_cgroup *memcg, *swap_memcg; + struct mem_cgroup *memcg; struct obj_cgroup *objcg; unsigned int nr_entries; unsigned short private_id; @@ -298,9 +298,8 @@ void __memcg1_swapout(struct folio *folio, struct swap_cluster_info *ci) * if the ID refers to the root memcg. */ nr_entries = folio_nr_pages(folio); - swap_memcg = mem_cgroup_private_id_get_online(memcg, nr_entries); - private_id = mem_cgroup_private_id(swap_memcg); - mod_memcg_state(swap_memcg, MEMCG_SWAP, nr_entries); + private_id = mem_cgroup_private_id_get(memcg, nr_entries); + mod_memcg_state(memcg, MEMCG_SWAP, nr_entries); __swap_cgroup_set(ci, swp_cluster_offset(folio->swap), nr_entries, private_id); diff --git a/mm/memcontrol-v1.h b/mm/memcontrol-v1.h index 7286456125dbef..8201396571842a 100644 --- a/mm/memcontrol-v1.h +++ b/mm/memcontrol-v1.h @@ -27,8 +27,7 @@ static inline bool mem_cgroup_private_id_is_root(unsigned short id) return id == mem_cgroup_private_id(root_mem_cgroup); } -struct mem_cgroup *mem_cgroup_private_id_get_online(struct mem_cgroup *memcg, - unsigned int n); +unsigned short mem_cgroup_private_id_get(struct mem_cgroup *memcg, unsigned int n); void reparent_memcg_lruvec_state_local(struct mem_cgroup *memcg, struct mem_cgroup *parent, int idx); diff --git a/mm/memcontrol.c b/mm/memcontrol.c index c43b0917851248..53c09a0e500e99 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -4082,7 +4082,7 @@ static void mem_cgroup_private_id_remove(struct mem_cgroup *memcg) } } -static inline void mem_cgroup_private_id_put(struct mem_cgroup *memcg, unsigned int n) +static void __mem_cgroup_private_id_put(struct mem_cgroup *memcg, unsigned int n) { if (refcount_sub_and_test(n, &memcg->private_id_ref)) { mem_cgroup_private_id_remove(memcg); @@ -4092,7 +4092,22 @@ static inline void mem_cgroup_private_id_put(struct mem_cgroup *memcg, unsigned } } -struct mem_cgroup *mem_cgroup_private_id_get_online(struct mem_cgroup *memcg, unsigned int n) +static inline void mem_cgroup_private_id_put(unsigned short id, unsigned int n) +{ + struct mem_cgroup *memcg; + + lockdep_assert_in_rcu_read_lock(); + + memcg = mem_cgroup_from_private_id(id); + __mem_cgroup_private_id_put(memcg, n); +} + +static void mem_cgroup_private_id_kill(struct mem_cgroup *memcg) +{ + __mem_cgroup_private_id_put(memcg, 1); +} + +unsigned short mem_cgroup_private_id_get(struct mem_cgroup *memcg, unsigned int n) { while (!refcount_add_not_zero(n, &memcg->private_id_ref)) { /* @@ -4105,7 +4120,8 @@ struct mem_cgroup *mem_cgroup_private_id_get_online(struct mem_cgroup *memcg, un } memcg = parent_mem_cgroup(memcg); } - return memcg; + + return mem_cgroup_private_id(memcg); } /** @@ -4430,7 +4446,7 @@ static void mem_cgroup_css_offline(struct cgroup_subsys_state *css) drain_all_stock(memcg); - mem_cgroup_private_id_put(memcg, 1); + mem_cgroup_private_id_kill(memcg); } static void mem_cgroup_css_released(struct cgroup_subsys_state *css) @@ -5971,15 +5987,13 @@ int __mem_cgroup_try_charge_swap(struct folio *folio) return 0; } - memcg = mem_cgroup_private_id_get_online(memcg, nr_pages); - /* memcg is pined by memcg ID. */ - private_id = mem_cgroup_private_id(memcg); + private_id = mem_cgroup_private_id_get(memcg, nr_pages); if (!mem_cgroup_private_id_is_root(private_id) && !page_counter_try_charge(&memcg->swap, nr_pages, &counter)) { memcg_memory_event(memcg, MEMCG_SWAP_MAX); memcg_memory_event(memcg, MEMCG_SWAP_FAIL); - mem_cgroup_private_id_put(memcg, nr_pages); + mem_cgroup_private_id_put(private_id, nr_pages); return -ENOMEM; } mod_memcg_state(memcg, MEMCG_SWAP, nr_pages); @@ -6011,7 +6025,7 @@ void __mem_cgroup_uncharge_swap(unsigned short id, unsigned int nr_pages) page_counter_uncharge(&memcg->swap, nr_pages); } mod_memcg_state(memcg, MEMCG_SWAP, -nr_pages); - mem_cgroup_private_id_put(memcg, nr_pages); + mem_cgroup_private_id_put(id, nr_pages); } rcu_read_unlock(); } From 4a3537af0cd175b3b4d538b73d89f29b04c2d599 Mon Sep 17 00:00:00 2001 From: Bingfang Guo Date: Mon, 21 Sep 2026 14:16:27 +0800 Subject: [PATCH 1144/1352] memcg: move memcg private ID refcount to objcg The memcg private ID is used by objects that can't afford storing a whole pointer and can outlive memcgs to track the memcg (notably swap entries). The current design holds a refcount to the css, preventing the memcg from being freed. This patch unbinds the lifetime of memcgid from the memcg so it can be freed. The idea is to move the refcount of memcgid to one of the memcg's objcg. The objcg is stored in the global memcgid xarray instead and used for retrieving the online memcg from it. So swapped out pages no longer pin the dying memcg. After the change, a memcgid can refer to a non present memcg. To handle this situation, when trying to get the original memcg from the id, compare the memcgid passed in with that of the memcg, and return NULL to indicate its death if they differ. NULL checks are added for list_lru_walk_node(), workingset_test_recent() to skip dead memcgs. For MGLRU recency test, mem_cgroup_lruvec() will substitute NULL with root_mem_cgroup. In the earlier patch, an extra xarray lookup was introduced in swap uncharging path. Now that we have the objcg pointer in the function, the extra overhead can be removed by using it for putting directly. Link: https://lore.kernel.org/20260921-bingfangguo-memcgid-rework-v2-4-6c0637dc0edb@tencent.com Signed-off-by: Bingfang Guo Signed-off-by: Andrew Morton Acked-by: Muchun Song Cc: Johannes Weiner Cc: Michal Hocko Cc: Roman Gushchin Cc: Shakeel Butt Cc: Dave Chinner Cc: Qi Zheng Cc: Kairui Song Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Bingfang Guo --- include/linux/memcontrol.h | 7 ++-- mm/list_lru.c | 2 +- mm/memcontrol.c | 73 ++++++++++++++++++++++++++++---------- mm/workingset.c | 2 +- 4 files changed, 60 insertions(+), 24 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index cd223f60b3a245..74110a324f9e22 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -180,6 +180,7 @@ struct obj_cgroup { struct percpu_ref refcnt; struct mem_cgroup *memcg; atomic_t nr_charged_bytes; + refcount_t private_id_ref; union { struct list_head list; /* protected by objcg_lock */ struct rcu_head rcu; @@ -225,9 +226,6 @@ struct mem_cgroup { /* vmpressure notifications. Written on every reclaim iteration. */ struct vmpressure vmpressure; - /* Written on every swap charge and uncharge. */ - refcount_t private_id_ref; - #ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC /* MEMCG_KMEM for nmi context */ atomic_t kmem_stat; @@ -324,6 +322,9 @@ struct mem_cgroup { unsigned long zswap_max; #endif + /* The objcg holding private memcg ID. */ + struct obj_cgroup *private_id_objcg; + /* Private memcg ID. Used to ID objects that outlive the cgroup */ int private_id; diff --git a/mm/list_lru.c b/mm/list_lru.c index 8a6dd0a489e12c..7edd79113cc56d 100644 --- a/mm/list_lru.c +++ b/mm/list_lru.c @@ -428,7 +428,7 @@ unsigned long list_lru_walk_node(struct list_lru *lru, int nid, xa_for_each(&lru->xa, index, mlru) { rcu_read_lock(); memcg = mem_cgroup_from_private_id(index); - if (!mem_cgroup_tryget(memcg)) { + if (!memcg || !mem_cgroup_tryget(memcg)) { rcu_read_unlock(); continue; } diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 53c09a0e500e99..a5335da5d4257a 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -4074,6 +4074,13 @@ static void memcg_wb_domain_size_changed(struct mem_cgroup *memcg) #define MEM_CGROUP_ID_MAX ((1UL << MEM_CGROUP_ID_SHIFT) - 1) static DEFINE_XARRAY_ALLOC1(mem_cgroup_private_ids); +static inline struct obj_cgroup *obj_cgroup_from_private_id(unsigned short id) +{ + lockdep_assert_once(rcu_read_lock_held()); + + return xa_load(&mem_cgroup_private_ids, id); +} + static void mem_cgroup_private_id_remove(struct mem_cgroup *memcg) { if (memcg->private_id > 0) { @@ -4082,34 +4089,44 @@ static void mem_cgroup_private_id_remove(struct mem_cgroup *memcg) } } -static void __mem_cgroup_private_id_put(struct mem_cgroup *memcg, unsigned int n) +static void __mem_cgroup_private_id_put(struct obj_cgroup *objcg, + unsigned short id, unsigned int n) { - if (refcount_sub_and_test(n, &memcg->private_id_ref)) { - mem_cgroup_private_id_remove(memcg); + struct obj_cgroup *objcg_free; - /* Memcg ID pins CSS */ - css_put(&memcg->css); + if (refcount_sub_and_test(n, &objcg->private_id_ref)) { + objcg_free = xa_erase(&mem_cgroup_private_ids, id); + VM_WARN_ON(objcg_free != objcg); + + /* Memcg ID pins the objcg */ + obj_cgroup_put(objcg); } } static inline void mem_cgroup_private_id_put(unsigned short id, unsigned int n) { - struct mem_cgroup *memcg; + struct obj_cgroup *objcg; lockdep_assert_in_rcu_read_lock(); - memcg = mem_cgroup_from_private_id(id); - __mem_cgroup_private_id_put(memcg, n); + objcg = obj_cgroup_from_private_id(id); + __mem_cgroup_private_id_put(objcg, id, n); } static void mem_cgroup_private_id_kill(struct mem_cgroup *memcg) { - __mem_cgroup_private_id_put(memcg, 1); + __mem_cgroup_private_id_put(memcg->private_id_objcg, memcg->private_id, 1); } unsigned short mem_cgroup_private_id_get(struct mem_cgroup *memcg, unsigned int n) { - while (!refcount_add_not_zero(n, &memcg->private_id_ref)) { + struct obj_cgroup *objcg; + + lockdep_assert_once(rcu_read_lock_held()); + + objcg = memcg->private_id_objcg; + + while (!refcount_add_not_zero(n, &objcg->private_id_ref)) { /* * The root cgroup cannot be destroyed, so it's refcount must * always be >= 1. @@ -4119,6 +4136,7 @@ unsigned short mem_cgroup_private_id_get(struct mem_cgroup *memcg, unsigned int break; } memcg = parent_mem_cgroup(memcg); + objcg = memcg->private_id_objcg; } return mem_cgroup_private_id(memcg); @@ -4129,11 +4147,25 @@ unsigned short mem_cgroup_private_id_get(struct mem_cgroup *memcg, unsigned int * @id: the memcg id to look up * * Caller must hold rcu_read_lock(). + * + * @return: the memcg, or NULL if the memcg referred to is already dead. */ struct mem_cgroup *mem_cgroup_from_private_id(unsigned short id) { + struct obj_cgroup *objcg; + struct mem_cgroup *memcg; + WARN_ON_ONCE(!rcu_read_lock_held()); - return xa_load(&mem_cgroup_private_ids, id); + + objcg = obj_cgroup_from_private_id(id); + if (!objcg) + return NULL; + + memcg = obj_cgroup_memcg(objcg); + if (mem_cgroup_private_id(memcg) != id) + return NULL; + + return memcg; } struct mem_cgroup *mem_cgroup_get_from_id(u64 id) @@ -4380,9 +4412,10 @@ static int mem_cgroup_css_online(struct cgroup_subsys_state *css) FLUSH_TIME); lru_gen_online_memcg(memcg); - /* Online state pins memcg ID, memcg ID pins CSS */ - refcount_set(&memcg->private_id_ref, 1); - css_get(css); + /* CSS pins memcg ID, memcg ID pins obj cgroup */ + memcg->private_id_objcg = objcg; + refcount_set(&memcg->private_id_objcg->private_id_ref, 1); + obj_cgroup_get(memcg->private_id_objcg); /* * Ensure mem_cgroup_from_private_id() works once we're fully online. @@ -4394,7 +4427,7 @@ static int mem_cgroup_css_online(struct cgroup_subsys_state *css) * publish it here at the end of onlining. This matches the * regular ID destruction during offlining. */ - xa_store(&mem_cgroup_private_ids, memcg->private_id, memcg, GFP_KERNEL); + xa_store(&mem_cgroup_private_ids, memcg->private_id, memcg->private_id_objcg, GFP_KERNEL); return 0; free_objcg: @@ -5832,8 +5865,6 @@ static void __init memcg_struct_check(void) memory_events_local); CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, vmpressure); - CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, - private_id_ref); #ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, kmem_stat); @@ -5878,6 +5909,8 @@ static void __init memcg_struct_check(void) CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_read_mostly, zswap_writeback); #endif + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_read_mostly, + private_id_objcg); CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_read_mostly, private_id); CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_read_mostly, @@ -6013,10 +6046,12 @@ int __mem_cgroup_try_charge_swap(struct folio *folio) */ void __mem_cgroup_uncharge_swap(unsigned short id, unsigned int nr_pages) { + struct obj_cgroup *objcg; struct mem_cgroup *memcg; rcu_read_lock(); - memcg = mem_cgroup_from_private_id(id); + objcg = obj_cgroup_from_private_id(id); + memcg = obj_cgroup_memcg(objcg); if (memcg) { if (!mem_cgroup_private_id_is_root(id)) { if (do_memsw_account()) @@ -6025,7 +6060,7 @@ void __mem_cgroup_uncharge_swap(unsigned short id, unsigned int nr_pages) page_counter_uncharge(&memcg->swap, nr_pages); } mod_memcg_state(memcg, MEMCG_SWAP, -nr_pages); - mem_cgroup_private_id_put(id, nr_pages); + __mem_cgroup_private_id_put(objcg, id, nr_pages); } rcu_read_unlock(); } diff --git a/mm/workingset.c b/mm/workingset.c index 8412f4840ae35c..1504f91cdca5f6 100644 --- a/mm/workingset.c +++ b/mm/workingset.c @@ -470,7 +470,7 @@ bool workingset_test_recent(void *shadow, bool file, bool *workingset, * configurations instead. */ eviction_memcg = mem_cgroup_from_private_id(memcgid); - if (!mem_cgroup_tryget(eviction_memcg)) + if (eviction_memcg && !mem_cgroup_tryget(eviction_memcg)) eviction_memcg = NULL; rcu_read_unlock(); From ae9788a424d597c463c774c7d4459ac73ae4247c Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:58:53 +0200 Subject: [PATCH 1145/1352] mm/sparse: move mem_section init to sparse_extreme_init() Patch series "mm/sparse: remove SECTION_MARKED_PRESENT and further cleanups", v2. SECTION_MARKED_PRESENT is really only needed during boot, where we can instead just rely on SECTION_IS_EARLY_BIT by setting that flag earlier. Some preparations for doing that conversion and some cleanups in the same (sparse) area. This patch (of 13): Let's just avoid another pair of ifdef inside a function. While at it, switch to INTERNODE_CACHE_BYTES by just defining a fallback in cache.h as well, given that the x86 variant already provides one. Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-0-54d81d65e125@kernel.org Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-1-54d81d65e125@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Acked-by: Oscar Salvador Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Reviewed-by: Zi Yan Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Jan Kiszka Cc: Kieran Bingham Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- include/linux/cache.h | 1 + mm/sparse.c | 19 ++++++++++++------- 2 files changed, 13 insertions(+), 7 deletions(-) diff --git a/include/linux/cache.h b/include/linux/cache.h index e69768f50d5327..b6e857b985bcaa 100644 --- a/include/linux/cache.h +++ b/include/linux/cache.h @@ -89,6 +89,7 @@ */ #ifndef INTERNODE_CACHE_SHIFT #define INTERNODE_CACHE_SHIFT L1_CACHE_SHIFT +#define INTERNODE_CACHE_BYTES (1 << INTERNODE_CACHE_SHIFT) #endif #if !defined(____cacheline_internodealigned_in_smp) diff --git a/mm/sparse.c b/mm/sparse.c index b75921c622edef..8cd2b06b231ac0 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -103,11 +103,22 @@ int __meminit sparse_index_init(unsigned long section_nr, int nid) return 0; } + +static void __init sparse_extreme_init(void) +{ + const unsigned long size = sizeof(struct mem_section *) * NR_SECTION_ROOTS; + + mem_section = memblock_alloc_or_panic(size, INTERNODE_CACHE_BYTES); +} #else /* !SPARSEMEM_EXTREME */ int __meminit sparse_index_init(unsigned long section_nr, int nid) { return 0; } + +static void __init sparse_extreme_init(void) +{ +} #endif /* @@ -196,13 +207,7 @@ void __init sparse_sections_init(void) unsigned long start, end; int i, nid; -#ifdef CONFIG_SPARSEMEM_EXTREME - unsigned long size, align; - - size = sizeof(struct mem_section *) * NR_SECTION_ROOTS; - align = 1 << (INTERNODE_CACHE_SHIFT); - mem_section = memblock_alloc_or_panic(size, align); -#endif + sparse_extreme_init(); for_each_mem_pfn_range(i, MAX_NUMNODES, &start, &end, &nid) memory_present(nid, start, end); From 11b62a4d36244d8b4c89550f6daa5e5c37f13a95 Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:58:54 +0200 Subject: [PATCH 1146/1352] mm/sparse: refactor sparse_sections_init() memory_present() really identifies+prepares all early sections so the initialization in sparse_init() can properly iterating them to initialize metadata. Let's just inline memory_present() into sparse_sections_init() and cleaning up the code a bit while at it: make it clear that we are operating on pfns. Note that we call set_section_nid() now only if the section was not already created earlier. Now, there is no more inconsistency between what we (temporarily) store in ms->section_mem_map and what we store in our section->nid array. Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-2-54d81d65e125@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Acked-by: Oscar Salvador Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Zi Yan Cc: Jan Kiszka Cc: Kieran Bingham Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- mm/sparse.c | 41 +++++++++++++++++------------------------ 1 file changed, 17 insertions(+), 24 deletions(-) diff --git a/mm/sparse.c b/mm/sparse.c index 8cd2b06b231ac0..2b285653abf1e8 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -179,40 +179,33 @@ static inline unsigned long first_present_section_nr(void) return next_present_section_nr(-1); } -/* Record a memory area against a node. */ -static void __init memory_present(int nid, unsigned long start, unsigned long end) +void __init sparse_sections_init(void) { - unsigned long pfn; + unsigned long pfn, start_pfn, end_pfn; + int i, nid; + + sparse_extreme_init(); - start &= PAGE_SECTION_MASK; - mminit_validate_memmodel_limits(&start, &end); - for (pfn = start; pfn < end; pfn += PAGES_PER_SECTION) { - unsigned long section_nr = pfn_to_section_nr(pfn); - struct mem_section *ms; + for_each_mem_pfn_range(i, MAX_NUMNODES, &start_pfn, &end_pfn, &nid) { + start_pfn &= PAGE_SECTION_MASK; + mminit_validate_memmodel_limits(&start_pfn, &end_pfn); - sparse_index_init(section_nr, nid); - set_section_nid(section_nr, nid); + for (pfn = start_pfn; pfn < end_pfn; pfn += PAGES_PER_SECTION) { + unsigned long section_nr = pfn_to_section_nr(pfn); + struct mem_section *ms; - ms = __nr_to_section(section_nr); - if (!ms->section_mem_map) { + sparse_index_init(section_nr, nid); + ms = __nr_to_section(section_nr); + if (ms->section_mem_map) + continue; + + set_section_nid(section_nr, nid); ms->section_mem_map = sparse_encode_early_nid(nid) | SECTION_IS_ONLINE; __section_mark_present(ms, section_nr); } } } - -void __init sparse_sections_init(void) -{ - unsigned long start, end; - int i, nid; - - sparse_extreme_init(); - - for_each_mem_pfn_range(i, MAX_NUMNODES, &start, &end, &nid) - memory_present(nid, start, end); -} - #ifndef CONFIG_SPARSEMEM_VMEMMAP struct page __init *__populate_section_memmap(unsigned long pfn, unsigned long nr_pages, int nid, struct vmem_altmap *altmap, From e8266fec2b3e83d5b7583b7a106fd6842ba5e939 Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:58:55 +0200 Subject: [PATCH 1147/1352] mm/sparse: move initialization of section metadata to sparse_metadata_init() Let's move the code responsible for initializing sparse metadata (usemap, memmap) into a helper. Cleanup the variable while at it (e.g., "map_count"). Drop the rather obvious code comments. No functional change intended. Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-3-54d81d65e125@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Acked-by: Oscar Salvador Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Zi Yan Cc: Jan Kiszka Cc: Kieran Bingham Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- mm/sparse.c | 43 ++++++++++++++++++++++--------------------- 1 file changed, 22 insertions(+), 21 deletions(-) diff --git a/mm/sparse.c b/mm/sparse.c index 2b285653abf1e8..71288daeb36653 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -255,38 +255,39 @@ static void __init sparse_init_nid(int nid, unsigned long pnum_begin, } } +static void __init sparse_metadata_init(void) +{ + unsigned long start_section_nr = first_present_section_nr(); + int nid_begin = sparse_early_nid(__nr_to_section(start_section_nr)); + unsigned long section_nr, nr_sections = 1; + + for_each_present_section_nr(start_section_nr + 1, section_nr) { + const int nid = sparse_early_nid(__nr_to_section(section_nr)); + + if (nid == nid_begin) { + nr_sections++; + continue; + } + sparse_init_nid(nid_begin, start_section_nr, section_nr, nr_sections); + nid_begin = nid; + start_section_nr = section_nr; + nr_sections = 1; + } + sparse_init_nid(nid_begin, start_section_nr, section_nr, nr_sections); +} + /* * Allocate the accumulated non-linear sections, allocate a mem_map * for each and record the physical to section mapping. */ void __init sparse_init(void) { - unsigned long pnum_end, pnum_begin, map_count = 1; - int nid_begin; - if (compound_info_has_mask()) { VM_WARN_ON_ONCE(!IS_ALIGNED((unsigned long) pfn_to_page(0), MAX_FOLIO_VMEMMAP_ALIGN)); } - pnum_begin = first_present_section_nr(); - nid_begin = sparse_early_nid(__nr_to_section(pnum_begin)); - - for_each_present_section_nr(pnum_begin + 1, pnum_end) { - int nid = sparse_early_nid(__nr_to_section(pnum_end)); - - if (nid == nid_begin) { - map_count++; - continue; - } - /* Init node with sections in range [pnum_begin, pnum_end) */ - sparse_init_nid(nid_begin, pnum_begin, pnum_end, map_count); - nid_begin = nid; - pnum_begin = pnum_end; - map_count = 1; - } - /* cover the last node */ - sparse_init_nid(nid_begin, pnum_begin, pnum_end, map_count); + sparse_metadata_init(); sparse_init_subsection_map(); vmemmap_populate_print_last(); } From b561a8e31d1c1578c1487d82ee894c7197cc6cc9 Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:58:56 +0200 Subject: [PATCH 1148/1352] mm/sparse: rename and cleanup sparse_init_nid() Let's rename it to "sparse_metadata_init_nid", avoid the "pnum" terminology and drop the function comment. Further, rename the "map" variable to "mem_map" for consistency with other functions. Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-4-54d81d65e125@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Acked-by: Oscar Salvador Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Zi Yan Cc: Jan Kiszka Cc: Kieran Bingham Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- mm/sparse.c | 41 ++++++++++++++++++++--------------------- 1 file changed, 20 insertions(+), 21 deletions(-) diff --git a/mm/sparse.c b/mm/sparse.c index 71288daeb36653..3c19d475590906 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -221,36 +221,33 @@ void __weak __meminit vmemmap_populate_print_last(void) { } -/* - * Initialize sparse on a specific node. The node spans [pnum_begin, pnum_end) - * And number of present sections in this node is map_count. - */ -static void __init sparse_init_nid(int nid, unsigned long pnum_begin, - unsigned long pnum_end, - unsigned long map_count) +static void __init sparse_metadata_init_nid(int nid, + unsigned long start_section_nr, unsigned long end_section_nr, + unsigned long nr_sections) { - unsigned long pnum; struct mem_section_usage *usage; + unsigned long section_nr; - usage = memblock_alloc_node(map_count * mem_section_usage_size(), + usage = memblock_alloc_node(nr_sections * mem_section_usage_size(), SMP_CACHE_BYTES, nid); if (!usage) panic("Failed to allocate usemap for node %d\n", nid); - for_each_present_section_nr(pnum_begin, pnum) { - unsigned long pfn = section_nr_to_pfn(pnum); - struct page *map; + for_each_present_section_nr(start_section_nr, section_nr) { + const unsigned long pfn = section_nr_to_pfn(section_nr); + struct page *mem_map; - if (pnum >= pnum_end) + if (section_nr >= end_section_nr) break; - map = __populate_section_memmap(pfn, PAGES_PER_SECTION, - nid, NULL, NULL); - if (!map) - panic("Failed to allocate memmap for section %lu\n", pnum); + mem_map = __populate_section_memmap(pfn, PAGES_PER_SECTION, nid, + NULL, NULL); + if (!mem_map) + panic("Failed to allocate memmap for section %lu\n", + section_nr); memmap_boot_pages_add(section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION)); - sparse_init_one_section(__nr_to_section(pnum), pnum, map, usage, - SECTION_IS_EARLY); + sparse_init_one_section(__nr_to_section(section_nr), section_nr, + mem_map, usage, SECTION_IS_EARLY); usage = (void *)usage + mem_section_usage_size(); } } @@ -268,12 +265,14 @@ static void __init sparse_metadata_init(void) nr_sections++; continue; } - sparse_init_nid(nid_begin, start_section_nr, section_nr, nr_sections); + sparse_metadata_init_nid(nid_begin, start_section_nr, + section_nr, nr_sections); nid_begin = nid; start_section_nr = section_nr; nr_sections = 1; } - sparse_init_nid(nid_begin, start_section_nr, section_nr, nr_sections); + sparse_metadata_init_nid(nid_begin, start_section_nr, section_nr, + nr_sections); } /* From c67abb1b635abaa50cdef1836161423b7a4b74f1 Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:58:57 +0200 Subject: [PATCH 1149/1352] mm/sparse: cleanup sparse_init_one_section() Let's avoid the "pnum" terminology. Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-5-54d81d65e125@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Acked-by: Oscar Salvador Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Zi Yan Cc: Jan Kiszka Cc: Kieran Bingham Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- mm/sparse.h | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/mm/sparse.h b/mm/sparse.h index 530692cdd516fe..4ee507c34b98d5 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -20,7 +20,7 @@ void sparse_sections_init(void); int sparse_index_init(unsigned long section_nr, int nid); static inline void sparse_init_one_section(struct mem_section *ms, - unsigned long pnum, struct page *mem_map, + unsigned long section_nr, struct page *mem_map, struct mem_section_usage *usage, unsigned long flags) { unsigned long coded_mem_map; @@ -32,7 +32,7 @@ static inline void sparse_init_one_section(struct mem_section *ms, * page_to_pfn() on !CONFIG_SPARSEMEM_VMEMMAP can simply subtract it * from the page pointer to obtain the PFN. */ - coded_mem_map = (unsigned long)(mem_map - section_nr_to_pfn(pnum)); + coded_mem_map = (unsigned long)(mem_map - section_nr_to_pfn(section_nr)); VM_WARN_ON_ONCE(coded_mem_map & ~SECTION_MAP_MASK); ms->section_mem_map &= ~SECTION_MAP_MASK; From 5f71451ae7f6f8fb8b05438f9f80774bbd3357f4 Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:58:58 +0200 Subject: [PATCH 1150/1352] mm/sparse: rename __highest_present_section_nr to __highest_used_section_nr In preparation for getting rid of SECTION_MARKED_PRESENT, rename __highest_present_section_nr and clarify the comment. Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-6-54d81d65e125@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Acked-by: Oscar Salvador Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Zi Yan Cc: Jan Kiszka Cc: Kieran Bingham Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- include/linux/mmzone.h | 6 +++--- mm/compaction.c | 2 +- mm/sparse.c | 12 ++++-------- mm/sparse.h | 4 ++-- 4 files changed, 10 insertions(+), 14 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 8e4e0bda3b586b..0555936f654db8 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -2158,7 +2158,7 @@ static inline struct mem_section *__pfn_to_section(unsigned long pfn) return __nr_to_section(pfn_to_section_nr(pfn)); } -extern unsigned long __highest_present_section_nr; +extern unsigned long __highest_used_section_nr; static inline int subsection_map_index(unsigned long pfn) { @@ -2257,7 +2257,7 @@ static inline unsigned long first_valid_pfn(unsigned long pfn, unsigned long end rcu_read_lock_sched(); - while (nr <= __highest_present_section_nr && pfn < end_pfn) { + while (nr <= __highest_used_section_nr && pfn < end_pfn) { struct mem_section *ms = __pfn_to_section(pfn); if (valid_section(ms) && @@ -2312,7 +2312,7 @@ static inline int pfn_in_present_section(unsigned long pfn) static inline unsigned long next_present_section_nr(unsigned long section_nr) { - while (++section_nr <= __highest_present_section_nr) { + while (++section_nr <= __highest_used_section_nr) { if (present_section_nr(section_nr)) return section_nr; } diff --git a/mm/compaction.c b/mm/compaction.c index 4994e200bbecd7..f1b2060eb20168 100644 --- a/mm/compaction.c +++ b/mm/compaction.c @@ -216,7 +216,7 @@ static unsigned long skip_offline_sections(unsigned long start_pfn) if (online_section_nr(start_nr)) return 0; - while (++start_nr <= __highest_present_section_nr) { + while (++start_nr <= __highest_used_section_nr) { if (online_section_nr(start_nr)) return section_nr_to_pfn(start_nr); } diff --git a/mm/sparse.c b/mm/sparse.c index 3c19d475590906..bb89017254f4da 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -164,15 +164,11 @@ static void __init mminit_validate_memmodel_limits(unsigned long *start_pfn, } /* - * There are a number of times that we loop over NR_MEM_SECTIONS, - * looking for section_present() on each. But, when we have very - * large physical address spaces, NR_MEM_SECTIONS can also be - * very large which makes the loops quite long. - * - * Keeping track of this gives us an easy way to break out of - * those loops early. + * Looping over all possible memory sections is expensive, especially if + * NR_MEM_SECTIONS is large but only a fraction is actually used. Keep track of + * the highest section number we ever used. */ -unsigned long __highest_present_section_nr; +unsigned long __highest_used_section_nr; static inline unsigned long first_present_section_nr(void) { diff --git a/mm/sparse.h b/mm/sparse.h index 4ee507c34b98d5..242b7bab0013c6 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -44,8 +44,8 @@ static inline void sparse_init_one_section(struct mem_section *ms, static inline void __section_mark_present(struct mem_section *ms, unsigned long section_nr) { - if (section_nr > __highest_present_section_nr) - __highest_present_section_nr = section_nr; + if (section_nr > __highest_used_section_nr) + __highest_used_section_nr = section_nr; ms->section_mem_map |= SECTION_MARKED_PRESENT; } From abcd556472e40715ab7182b178feec333cefb400 Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:58:59 +0200 Subject: [PATCH 1151/1352] mm/sparse: remove pfn_in_present_section() Unused, let's remove it. Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-7-54d81d65e125@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Acked-by: Oscar Salvador Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Zi Yan Cc: Jan Kiszka Cc: Kieran Bingham Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- include/linux/mmzone.h | 10 ---------- 1 file changed, 10 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 0555936f654db8..7e7aa57d2a0a6c 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -2303,13 +2303,6 @@ static inline unsigned long next_valid_pfn(unsigned long pfn, unsigned long end_ #endif -static inline int pfn_in_present_section(unsigned long pfn) -{ - if (pfn_to_section_nr(pfn) >= NR_MEM_SECTIONS) - return 0; - return present_section(__pfn_to_section(pfn)); -} - static inline unsigned long next_present_section_nr(unsigned long section_nr) { while (++section_nr <= __highest_used_section_nr) { @@ -2339,9 +2332,6 @@ static inline unsigned long next_present_section_nr(unsigned long section_nr) #else #define pfn_to_nid(pfn) (0) #endif - -#else -#define pfn_in_present_section pfn_valid #endif /* CONFIG_SPARSEMEM */ /* From 064b7ae8ac3e330722a3e1c5d53e1721a5cf4d9b Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:59:00 +0200 Subject: [PATCH 1152/1352] mm/sparse: move __highest_used_section_nr handling In preparation for removing __section_mark_present(), let's move __highest_used_section_nr handling into its callers. Verify in sparse_init_one_section() that it was properly updated. In sparse_sections_init() we can just set it to the last processed section_nr. No functional change intended. Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-8-54d81d65e125@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Reviewed-by: Mike Rapoport (Microsoft) Acked-by: Oscar Salvador Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Zi Yan Cc: Jan Kiszka Cc: Kieran Bingham Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- mm/sparse-vmemmap.c | 1 + mm/sparse.c | 5 +++-- mm/sparse.h | 4 +--- 3 files changed, 5 insertions(+), 5 deletions(-) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 23b31bc4bf94ab..d5fd61c83bac0c 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -904,6 +904,7 @@ int __meminit sparse_add_section(int nid, unsigned long start_pfn, page_init_poison(memmap, sizeof(struct page) * nr_pages); __section_mark_present(ms, section_nr); + __highest_used_section_nr = max(section_nr, __highest_used_section_nr); /* Align memmap to section boundary in the subsection case */ if (section_nr_to_pfn(section_nr) != start_pfn) diff --git a/mm/sparse.c b/mm/sparse.c index bb89017254f4da..a0f50ca5acf5a5 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -177,7 +177,7 @@ static inline unsigned long first_present_section_nr(void) void __init sparse_sections_init(void) { - unsigned long pfn, start_pfn, end_pfn; + unsigned long pfn, start_pfn, end_pfn, section_nr; int i, nid; sparse_extreme_init(); @@ -187,9 +187,9 @@ void __init sparse_sections_init(void) mminit_validate_memmodel_limits(&start_pfn, &end_pfn); for (pfn = start_pfn; pfn < end_pfn; pfn += PAGES_PER_SECTION) { - unsigned long section_nr = pfn_to_section_nr(pfn); struct mem_section *ms; + section_nr = pfn_to_section_nr(pfn); sparse_index_init(section_nr, nid); ms = __nr_to_section(section_nr); if (ms->section_mem_map) @@ -201,6 +201,7 @@ void __init sparse_sections_init(void) __section_mark_present(ms, section_nr); } } + __highest_used_section_nr = section_nr; } #ifndef CONFIG_SPARSEMEM_VMEMMAP struct page __init *__populate_section_memmap(unsigned long pfn, diff --git a/mm/sparse.h b/mm/sparse.h index 242b7bab0013c6..c396d5c05cfc8f 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -26,6 +26,7 @@ static inline void sparse_init_one_section(struct mem_section *ms, unsigned long coded_mem_map; BUILD_BUG_ON(SECTION_MAP_LAST_BIT > PFN_SECTION_SHIFT); + VM_WARN_ON_ONCE(section_nr > __highest_used_section_nr); /* * We encode the start PFN of the section into the mem_map such that @@ -44,9 +45,6 @@ static inline void sparse_init_one_section(struct mem_section *ms, static inline void __section_mark_present(struct mem_section *ms, unsigned long section_nr) { - if (section_nr > __highest_used_section_nr) - __highest_used_section_nr = section_nr; - ms->section_mem_map |= SECTION_MARKED_PRESENT; } From 7c9108fe8b7d9c8ba5332bd36ba04f8696a31c44 Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:59:01 +0200 Subject: [PATCH 1153/1352] scripts/gdb: mm.py: remove fallbacks for SECTION_HAS_MEM_MAP and SECTION_IS_EARLY There is no reason to handle the absence of the corresponding BIT values. So let's remove the fallbacks. Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-9-54d81d65e125@kernel.org Link: https://lore.kernel.org/r/vpz5zqg5nwbpolxbkip45vxxdt3lni7j2haal7q42skc3b57tu@uh4cec4x32dq Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Cc: Seongjun Hong Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Zi Yan Cc: Jan Kiszka Cc: Kieran Bingham Cc: Oscar Salvador Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- scripts/gdb/linux/mm.py | 8 ++------ 1 file changed, 2 insertions(+), 6 deletions(-) diff --git a/scripts/gdb/linux/mm.py b/scripts/gdb/linux/mm.py index 193a88d763abf7..7398105ff3f677 100644 --- a/scripts/gdb/linux/mm.py +++ b/scripts/gdb/linux/mm.py @@ -71,12 +71,8 @@ def __init__(self): self.NR_SECTION_ROOTS = DIV_ROUND_UP(self.NR_MEM_SECTIONS, self.SECTIONS_PER_ROOT) - try: - self.SECTION_HAS_MEM_MAP = 1 << int(gdb.parse_and_eval('SECTION_HAS_MEM_MAP_BIT')) - self.SECTION_IS_EARLY = 1 << int(gdb.parse_and_eval('SECTION_IS_EARLY_BIT')) - except: - self.SECTION_HAS_MEM_MAP = 1 << 0 - self.SECTION_IS_EARLY = 1 << 3 + self.SECTION_HAS_MEM_MAP = 1 << int(gdb.parse_and_eval('SECTION_HAS_MEM_MAP_BIT')) + self.SECTION_IS_EARLY = 1 << int(gdb.parse_and_eval('SECTION_IS_EARLY_BIT')) self.SUBSECTION_SHIFT = 21 self.PAGES_PER_SUBSECTION = 1 << (self.SUBSECTION_SHIFT - self.PAGE_SHIFT) From 6dde1239b76b4e32e3b583d739ff398a08c0de39 Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:59:02 +0200 Subject: [PATCH 1154/1352] mm/sparse: remove SECTION_MARKED_PRESENT All present section iterators run before memory hotplug added any further memory sections, Therefore, we can simply use the SECTION_IS_EARLY flag by setting that flag earlier in sparse_sections_init(). Get rid of SECTION_MARKED_PRESENT entirely and rename for_each_present_section_nr() to for_each_early_section_nr(). Take care of the .clang-format for_each_present_section_nr() handling. No functional change intended. Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-10-54d81d65e125@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Acked-by: Oscar Salvador Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Zi Yan Cc: Jan Kiszka Cc: Kieran Bingham Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- .clang-format | 2 +- drivers/base/memory.c | 2 +- include/linux/mmzone.h | 27 ++++++++++----------------- mm/sparse-vmemmap.c | 1 - mm/sparse.c | 17 ++++++++--------- mm/sparse.h | 6 ------ 6 files changed, 20 insertions(+), 35 deletions(-) diff --git a/.clang-format b/.clang-format index 5ef5743b77c95f..395292ab2ac634 100644 --- a/.clang-format +++ b/.clang-format @@ -282,6 +282,7 @@ ForEachMacros: - 'for_each_dpcm_fe' - 'for_each_drhd_unit' - 'for_each_dss_dev' + - 'for_each_early_section_nr' - 'for_each_efi_memory_desc' - 'for_each_efi_memory_desc_in_map' - 'for_each_element' @@ -403,7 +404,6 @@ ForEachMacros: - 'for_each_possible_cpu_wrap' - 'for_each_present_blessed_reg' - 'for_each_present_cpu' - - 'for_each_present_section_nr' - 'for_each_prime_number' - 'for_each_prime_number_from' - 'for_each_probe_cache_entry' diff --git a/drivers/base/memory.c b/drivers/base/memory.c index 5eead3346f1e32..b0338de2f1d82c 100644 --- a/drivers/base/memory.c +++ b/drivers/base/memory.c @@ -972,7 +972,7 @@ void __init memory_dev_init(void) * block so that it can be covered. */ block_id = ULONG_MAX; - for_each_present_section_nr(0, nr) { + for_each_early_section_nr(0, nr) { if (block_id != ULONG_MAX && memory_block_id(nr) == block_id) continue; diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 7e7aa57d2a0a6c..851b14c91cb1a5 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -2062,7 +2062,6 @@ static inline struct mem_section *__nr_to_section(unsigned long nr) * accommodate SECTION_MAP_LAST_BIT. We use BUILD_BUG_ON() to ensure this. */ enum { - SECTION_MARKED_PRESENT_BIT, SECTION_HAS_MEM_MAP_BIT, SECTION_IS_ONLINE_BIT, SECTION_IS_EARLY_BIT, @@ -2072,7 +2071,6 @@ enum { SECTION_MAP_LAST_BIT, }; -#define SECTION_MARKED_PRESENT BIT(SECTION_MARKED_PRESENT_BIT) #define SECTION_HAS_MEM_MAP BIT(SECTION_HAS_MEM_MAP_BIT) #define SECTION_IS_ONLINE BIT(SECTION_IS_ONLINE_BIT) #define SECTION_IS_EARLY BIT(SECTION_IS_EARLY_BIT) @@ -2089,16 +2087,6 @@ static inline struct page *__section_mem_map_addr(struct mem_section *section) return (struct page *)map; } -static inline int present_section(const struct mem_section *section) -{ - return (section && (section->section_mem_map & SECTION_MARKED_PRESENT)); -} - -static inline int present_section_nr(unsigned long nr) -{ - return present_section(__nr_to_section(nr)); -} - static inline int valid_section(const struct mem_section *section) { return (section && (section->section_mem_map & SECTION_HAS_MEM_MAP)); @@ -2114,6 +2102,11 @@ static inline int valid_section_nr(unsigned long nr) return valid_section(__nr_to_section(nr)); } +static inline int early_section_nr(unsigned long nr) +{ + return early_section(__nr_to_section(nr)); +} + static inline int online_section(const struct mem_section *section) { return (section && (section->section_mem_map & SECTION_IS_ONLINE)); @@ -2303,20 +2296,20 @@ static inline unsigned long next_valid_pfn(unsigned long pfn, unsigned long end_ #endif -static inline unsigned long next_present_section_nr(unsigned long section_nr) +static inline unsigned long next_early_section_nr(unsigned long section_nr) { while (++section_nr <= __highest_used_section_nr) { - if (present_section_nr(section_nr)) + if (early_section_nr(section_nr)) return section_nr; } return -1; } -#define for_each_present_section_nr(start, section_nr) \ - for (section_nr = next_present_section_nr(start - 1); \ +#define for_each_early_section_nr(start, section_nr) \ + for (section_nr = next_early_section_nr(start - 1); \ section_nr != -1; \ - section_nr = next_present_section_nr(section_nr)) + section_nr = next_early_section_nr(section_nr)) /* * These are _only_ used during initialisation, therefore they diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index d5fd61c83bac0c..09376f5b1d921c 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -903,7 +903,6 @@ int __meminit sparse_add_section(int nid, unsigned long start_pfn, if (!section_vmemmap_optimizable(ms)) page_init_poison(memmap, sizeof(struct page) * nr_pages); - __section_mark_present(ms, section_nr); __highest_used_section_nr = max(section_nr, __highest_used_section_nr); /* Align memmap to section boundary in the subsection case */ diff --git a/mm/sparse.c b/mm/sparse.c index a0f50ca5acf5a5..5d2d9f95d4024a 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -170,13 +170,14 @@ static void __init mminit_validate_memmodel_limits(unsigned long *start_pfn, */ unsigned long __highest_used_section_nr; -static inline unsigned long first_present_section_nr(void) +static inline unsigned long first_early_section_nr(void) { - return next_present_section_nr(-1); + return next_early_section_nr(-1); } void __init sparse_sections_init(void) { + const unsigned long flags = SECTION_IS_EARLY | SECTION_IS_ONLINE; unsigned long pfn, start_pfn, end_pfn, section_nr; int i, nid; @@ -196,9 +197,7 @@ void __init sparse_sections_init(void) continue; set_section_nid(section_nr, nid); - ms->section_mem_map = sparse_encode_early_nid(nid) | - SECTION_IS_ONLINE; - __section_mark_present(ms, section_nr); + ms->section_mem_map = sparse_encode_early_nid(nid) | flags; } } __highest_used_section_nr = section_nr; @@ -230,7 +229,7 @@ static void __init sparse_metadata_init_nid(int nid, if (!usage) panic("Failed to allocate usemap for node %d\n", nid); - for_each_present_section_nr(start_section_nr, section_nr) { + for_each_early_section_nr(start_section_nr, section_nr) { const unsigned long pfn = section_nr_to_pfn(section_nr); struct page *mem_map; @@ -244,18 +243,18 @@ static void __init sparse_metadata_init_nid(int nid, section_nr); memmap_boot_pages_add(section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION)); sparse_init_one_section(__nr_to_section(section_nr), section_nr, - mem_map, usage, SECTION_IS_EARLY); + mem_map, usage, 0); usage = (void *)usage + mem_section_usage_size(); } } static void __init sparse_metadata_init(void) { - unsigned long start_section_nr = first_present_section_nr(); + unsigned long start_section_nr = first_early_section_nr(); int nid_begin = sparse_early_nid(__nr_to_section(start_section_nr)); unsigned long section_nr, nr_sections = 1; - for_each_present_section_nr(start_section_nr + 1, section_nr) { + for_each_early_section_nr(start_section_nr + 1, section_nr) { const int nid = sparse_early_nid(__nr_to_section(section_nr)); if (nid == nid_begin) { diff --git a/mm/sparse.h b/mm/sparse.h index c396d5c05cfc8f..19664d702af844 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -42,12 +42,6 @@ static inline void sparse_init_one_section(struct mem_section *ms, ms->usage = usage; } -static inline void __section_mark_present(struct mem_section *ms, - unsigned long section_nr) -{ - ms->section_mem_map |= SECTION_MARKED_PRESENT; -} - static inline size_t mem_section_usage_size(void) { return struct_size_t(struct mem_section_usage, pageblock_flags, From a42b50a6d313bc0105a3aad5ce48faaa3d23288e Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:59:03 +0200 Subject: [PATCH 1155/1352] mm/sparse: remove flags parameter from sparse_init_one_section() The last user of the flags parameter that used to pass SECTION_IS_EARLY was removed, so let's remove the parameter. Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-11-54d81d65e125@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Acked-by: Oscar Salvador Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Zi Yan Cc: Jan Kiszka Cc: Kieran Bingham Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- mm/sparse-vmemmap.c | 2 +- mm/sparse.c | 2 +- mm/sparse.h | 5 ++--- 3 files changed, 4 insertions(+), 5 deletions(-) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 09376f5b1d921c..09897586152386 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -908,7 +908,7 @@ int __meminit sparse_add_section(int nid, unsigned long start_pfn, /* Align memmap to section boundary in the subsection case */ if (section_nr_to_pfn(section_nr) != start_pfn) memmap = pfn_to_page(section_nr_to_pfn(section_nr)); - sparse_init_one_section(ms, section_nr, memmap, ms->usage, 0); + sparse_init_one_section(ms, section_nr, memmap, ms->usage); return 0; } diff --git a/mm/sparse.c b/mm/sparse.c index 5d2d9f95d4024a..702904b4170063 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -243,7 +243,7 @@ static void __init sparse_metadata_init_nid(int nid, section_nr); memmap_boot_pages_add(section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION)); sparse_init_one_section(__nr_to_section(section_nr), section_nr, - mem_map, usage, 0); + mem_map, usage); usage = (void *)usage + mem_section_usage_size(); } } diff --git a/mm/sparse.h b/mm/sparse.h index 19664d702af844..fdbc4cacf91d17 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -21,7 +21,7 @@ int sparse_index_init(unsigned long section_nr, int nid); static inline void sparse_init_one_section(struct mem_section *ms, unsigned long section_nr, struct page *mem_map, - struct mem_section_usage *usage, unsigned long flags) + struct mem_section_usage *usage) { unsigned long coded_mem_map; @@ -37,8 +37,7 @@ static inline void sparse_init_one_section(struct mem_section *ms, VM_WARN_ON_ONCE(coded_mem_map & ~SECTION_MAP_MASK); ms->section_mem_map &= ~SECTION_MAP_MASK; - ms->section_mem_map |= coded_mem_map; - ms->section_mem_map |= flags | SECTION_HAS_MEM_MAP; + ms->section_mem_map |= coded_mem_map | SECTION_HAS_MEM_MAP; ms->usage = usage; } From fd9bd0afa84b9c9b7e37f3891e1d8dd338fa7850 Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:59:04 +0200 Subject: [PATCH 1156/1352] fs/proc/page: clarify comment in get_max_dump_pfn() pfn_to_online_page() will only succeed on some PFNs within the same section, not necessarily all. Let's make that clearer. While at it, rephrase it to "Allow inspection of". Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-12-54d81d65e125@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Acked-by: Oscar Salvador Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Zi Yan Cc: Jan Kiszka Cc: Kieran Bingham Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- fs/proc/page.c | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/fs/proc/page.c b/fs/proc/page.c index f90e1030825e94..6aa99e42ec2221 100644 --- a/fs/proc/page.c +++ b/fs/proc/page.c @@ -31,10 +31,9 @@ static inline unsigned long get_max_dump_pfn(void) { #ifdef CONFIG_SPARSEMEM /* - * The memmap of early sections is completely populated and marked - * online even if max_pfn does not fall on a section boundary - - * pfn_to_online_page() will succeed on all pages. Allow inspecting - * these memmaps. + * If max_pfn does not fall on a section boundary, pfn_to_online_page() + * can succeed on PFNs beyond max_pfn within the same section. Allow + * inspection of these memmaps. */ return round_up(max_pfn, PAGES_PER_SECTION); #else From e2a703c3492ffb165c282e28b9119be4f965d8b2 Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:59:05 +0200 Subject: [PATCH 1157/1352] mm/memory_hotplug: drop CONFIG_HAVE_ARCH_PFN_VALID handling from pfn_to_online_page() Drop CONFIG_HAVE_ARCH_PFN_VALID handling, as CONFIG_HAVE_ARCH_PFN_VALID is never used with CONFIG_MEMORY_HOTPLUG, as the latter depends on CONFIG_SPARSEMEM_VMEMMAP. Make sure it stays that way. Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-13-54d81d65e125@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Acked-by: Oscar Salvador Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Zi Yan Cc: Jan Kiszka Cc: Kieran Bingham Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- mm/memory_hotplug.c | 9 ++------- 1 file changed, 2 insertions(+), 7 deletions(-) diff --git a/mm/memory_hotplug.c b/mm/memory_hotplug.c index d7a59167bec43c..796af1028ee239 100644 --- a/mm/memory_hotplug.c +++ b/mm/memory_hotplug.c @@ -345,6 +345,8 @@ struct page *pfn_to_online_page(unsigned long pfn) struct dev_pagemap *pgmap; struct mem_section *ms; + BUILD_BUG_ON(IS_ENABLED(CONFIG_HAVE_ARCH_PFN_VALID)); + if (nr >= NR_MEM_SECTIONS) return NULL; @@ -352,13 +354,6 @@ struct page *pfn_to_online_page(unsigned long pfn) if (!online_section(ms)) return NULL; - /* - * Save some code text when online_section() + - * pfn_section_valid() are sufficient. - */ - if (IS_ENABLED(CONFIG_HAVE_ARCH_PFN_VALID) && !pfn_valid(pfn)) - return NULL; - if (!pfn_section_valid(ms, pfn)) return NULL; From 16c881a9c95c152fec6f748c75809fd163ce6fa3 Mon Sep 17 00:00:00 2001 From: Zhijian Han Date: Tue, 22 Sep 2026 11:18:43 +0800 Subject: [PATCH 1158/1352] mm: fix typos in various comments Fix spelling errors found by codespell in several mm source files: - "aray" -> "array" in include/linux/mm.h - "indiciate" -> "indicate" in include/linux/mm_types.h - "aggresive" -> "aggressive" in include/linux/mm_types.h - "progated" -> "propagated" in mm/huge_memory.c - "Pressumably" -> "Presumably" in mm/hugetlb.c - "fime" -> "time" in mm/hugetlb.c - "farest" -> "farthest" in mm/list_lru.c - "trivally" -> "trivially" in mm/madvise.c - "atmost" -> "at most" in mm/memcontrol.c - "splited" -> "split" in mm/memory-failure.c - "appliable" -> "applicable" in mm/memory.c - "incase" -> "in case" in mm/page_io.c - "contigous" -> "contiguous" in mm/vmalloc.c - "probablity" -> "probability" in mm/kfence/core.c - "possesss" -> "possesses" in mm/util.c No functional changes. Link: https://lore.kernel.org/20260922031843.2857104-1-hanzhijian1991@gmail.com Signed-off-by: Zhijian Han Signed-off-by: Andrew Morton --- include/linux/mm.h | 2 +- include/linux/mm_types.h | 4 ++-- mm/huge_memory.c | 2 +- mm/hugetlb.c | 4 ++-- mm/kfence/core.c | 2 +- mm/list_lru.c | 2 +- mm/madvise.c | 2 +- mm/memcontrol.c | 2 +- mm/memory-failure.c | 2 +- mm/memory.c | 2 +- mm/page_io.c | 2 +- mm/util.c | 2 +- mm/vmalloc.c | 2 +- 13 files changed, 15 insertions(+), 15 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index 4fd47cc796a6c1..5e35864eb731c5 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -4795,7 +4795,7 @@ static inline void mmap_action_simple_ioremap(struct vm_area_desc *desc, * @desc: The VMA descriptor for the VMA requiring kernel pags to be mapped. * @start: The virtual address from which to map them. * @pages: An array of struct page pointers describing the memory to map. - * @nr_pages: The number of entries in the @pages aray. + * @nr_pages: The number of entries in the @pages array. */ static inline void mmap_action_map_kernel_pages(struct vm_area_desc *desc, unsigned long start, struct page **pages, diff --git a/include/linux/mm_types.h b/include/linux/mm_types.h index 95a768fac987a5..6141160ec6526b 100644 --- a/include/linux/mm_types.h +++ b/include/linux/mm_types.h @@ -759,7 +759,7 @@ static inline struct anon_vma_name *anon_vma_name_alloc(const char *name) /* * While __vma_enter_locked() is working to ensure are no read-locks held on a * VMA (either while acquiring a VMA write lock or marking a VMA detached) we - * set the VM_REFCNT_EXCLUDE_READERS_FLAG in vma->vm_refcnt to indiciate to + * set the VM_REFCNT_EXCLUDE_READERS_FLAG in vma->vm_refcnt to indicate to * vma_start_read() that the reference count should be left alone. * * See the comment describing vm_refcnt in vm_area_struct for details as to @@ -2011,7 +2011,7 @@ enum { /* * MMF_HAS_PINNED: Whether this mm has pinned any pages. This can be either * replaced in the future by mm.pinned_vm when it becomes stable, or grow into - * a counter on its own. We're aggresive on this bit for now: even if the + * a counter on its own. We're aggressive on this bit for now: even if the * pinned pages were unpinned later on, we'll still keep this bit set for the * lifecycle of this mm, just for simplicity. */ diff --git a/mm/huge_memory.c b/mm/huge_memory.c index c5c210b3ebc646..ddc631a388b93e 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -3415,7 +3415,7 @@ static void __split_huge_pmd_locked(struct vm_area_struct *vma, pmd_t *pmd, swp_entry = make_readable_device_private_entry( page_to_pfn(page + i)); /* - * Young and dirty bits are not progated via swp_entry + * Young and dirty bits are not propagated via swp_entry */ entry = swp_entry_to_pte(swp_entry); if (soft_dirty) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 76d019594b39c3..37f8272f0f2a73 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -5408,7 +5408,7 @@ void __unmap_hugepage_range(struct mmu_gather *tlb, struct vm_area_struct *vma, int rc = vma_needs_reservation(h, vma, address); if (rc < 0) - /* Pressumably allocate_file_region_entries failed + /* Presumably allocate_file_region_entries failed * to allocate a file_region struct. Clear * hugetlb_restore_reserve so that global reserve * count will not be incremented by free_huge_folio. @@ -5621,7 +5621,7 @@ static vm_fault_t hugetlb_wp(struct vm_fault *vmf) * In order to determine where this is a COW on a MAP_PRIVATE mapping it * is enough to check whether the old_folio is anonymous. This means that * the reserve for this address was consumed. If reserves were used, a - * partial faulted mapping at the fime of fork() could consume its reserves + * partial faulted mapping at the time of fork() could consume its reserves * on COW instead of the full address range. */ if (is_vma_resv_set(vma, HPAGE_RESV_OWNER) && diff --git a/mm/kfence/core.c b/mm/kfence/core.c index 90925c646c4c24..55d5c114eadf6c 100644 --- a/mm/kfence/core.c +++ b/mm/kfence/core.c @@ -156,7 +156,7 @@ atomic_t kfence_allocation_gate = ATOMIC_INIT(1); * allocations of the same source filling up the pool. * * Assuming a range of 15%-85% unique allocations in the pool at any point in - * time, the below parameters provide a probablity of 0.02-0.33 for false + * time, the below parameters provide a probability of 0.02-0.33 for false * positive hits respectively: * * P(alloc_traces) = (1 - e^(-HNUM * (alloc_traces / SIZE)) ^ HNUM diff --git a/mm/list_lru.c b/mm/list_lru.c index 7edd79113cc56d..d1d256822ccc73 100644 --- a/mm/list_lru.c +++ b/mm/list_lru.c @@ -587,7 +587,7 @@ static int __memcg_list_lru_alloc(struct mem_cgroup *memcg, */ do { /* - * Keep finding the farest parent that wasn't populated + * Keep finding the farthest parent that wasn't populated * until found memcg itself. */ pos = memcg; diff --git a/mm/madvise.c b/mm/madvise.c index 20135275cb5550..d83ce6abf8c323 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -1471,7 +1471,7 @@ static bool is_discard(int behavior) * We are restricted from madvise()'ing mseal()'d VMAs only in very particular * circumstances - discarding of data from read-only anonymous SEALED mappings. * - * This is because users cannot trivally discard data from these VMAs, and may + * This is because users cannot trivially discard data from these VMAs, and may * only do so via an appropriate madvise() call. */ static bool can_madvise_modify(struct madvise_behavior *madv_behavior) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index a5335da5d4257a..aad0498a7bd658 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -741,7 +741,7 @@ static unsigned long *memcg_events_local_array(struct mem_cgroup *memcg) * * 2) Flush the stats synchronously on reader side only when there are more than * (MEMCG_CHARGE_BATCH * nr_cpus) update events. Though this optimization - * will let stats be out of sync by atmost (MEMCG_CHARGE_BATCH * nr_cpus) but + * will let stats be out of sync by at most (MEMCG_CHARGE_BATCH * nr_cpus) but * only for 2 seconds due to (1). */ static void flush_memcg_stats_dwork(struct work_struct *w); diff --git a/mm/memory-failure.c b/mm/memory-failure.c index d237f556b3a09e..0c96eb5119976e 100644 --- a/mm/memory-failure.c +++ b/mm/memory-failure.c @@ -2560,7 +2560,7 @@ int memory_failure(unsigned long pfn, int flags) /* * We're only intended to deal with the non-Compound page here. * The page cannot become compound pages again as folio has been - * splited and extra refcnt is held. + * split and extra refcnt is held. */ WARN_ON(folio_test_large(folio)); diff --git a/mm/memory.c b/mm/memory.c index 79fa57a381ce00..330cde31bf8b40 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -5165,7 +5165,7 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) * have changed, so sub pages might got charged to the wrong cgroup, * or even should be shmem. So we have to free it and fallback. * Nothing should have touched it, both anon and shmem checks if a - * large folio is fully appliable before use. + * large folio is fully applicable before use. * * This will be removed once we unify folio allocation in the swap cache * layer, where allocation of a folio stabilizes the swap entries. diff --git a/mm/page_io.c b/mm/page_io.c index 0808808f0309ce..c6824fcd483e0a 100644 --- a/mm/page_io.c +++ b/mm/page_io.c @@ -135,7 +135,7 @@ static bool is_folio_zero_filled(struct folio *folio) for (i = 0; i < folio_nr_pages(folio); i++) { data = kmap_local_folio(folio, i * PAGE_SIZE); /* - * Check last word first, incase the page is zero-filled at + * Check last word first, in case the page is zero-filled at * the start and has non-zero data at the end, which is common * in real-world workloads. */ diff --git a/mm/util.c b/mm/util.c index c5ee52aede1e41..ab67cfc7357165 100644 --- a/mm/util.c +++ b/mm/util.c @@ -1252,7 +1252,7 @@ EXPORT_SYMBOL(__compat_vma_mmap); /** * compat_vma_mmap() - Apply the file's .mmap_prepare() hook to an * existing VMA and execute any requested actions. - * @file: The file which possesss an f_op->mmap_prepare() hook. + * @file: The file which possesses an f_op->mmap_prepare() hook. * @vma: The VMA to apply the .mmap_prepare() hook to. * * Ordinarily, .mmap_prepare() is invoked directly upon mmap(). However, certain diff --git a/mm/vmalloc.c b/mm/vmalloc.c index fad918765f054e..b24896aedac212 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -3698,7 +3698,7 @@ vm_area_alloc_pages(gfp_t gfp, int nid, /* * Initially, attempt to have the page allocator give us large order * pages. Do not attempt allocating smaller than order chunks since - * __vmap_pages_range() expects physically contigous pages of exactly + * __vmap_pages_range() expects physically contiguous pages of exactly * order long chunks. */ while (large_order > order && nr_remaining) { From 347aabd2dfedc92528180045c2d4e12b4c7ea078 Mon Sep 17 00:00:00 2001 From: Li Zhe Date: Mon, 28 Sep 2026 10:47:23 +0800 Subject: [PATCH 1159/1352] mm/hugetlb: fix overbroad MMU notifiers for unshared PMDs Hugetlb currently expands MMU notifier ranges to PUD boundaries whenever PMD sharing is possible. That is only needed when huge_pmd_unshare() actually detaches a shared PMD page table, because clearing the PUD invalidates the whole PUD-sized virtual address range. For hugetlbfs hole punch, and similarly for other hugetlb unmap paths, a shared mapping can pass the "PMD sharing is possible" range test in adjust_range_if_pmd_sharing_possible() even when the hugetlbfs file does not currently have any shared PMD page tables. KVM then receives a 1G invalidation for a 2M operation and zaps unrelated secondary mappings, so the guest has to fault them back in. Avoid this by tracking active PMD-sharing attachments per hugetlbfs inode. The count is incremented only after huge_pmd_share() successfully installs a shared PMD table, and decremented when __huge_pmd_unshare() actually detaches one. Since huge_pmd_share() can run concurrently under i_mmap_lock_read(), use a 64-bit atomic counter. A zero count is used to skip the conservative notifier range expansion only after excluding concurrent PMD sharing with the mapping write lock. On a Redis-in-VM workload that punches cold 2M hugetlb pages, this patch improves P99 QPS stability while punching pages, reducing the QPS degradation ratio from 7.09% to 1.45%. Link: https://lore.kernel.org/20260928024723.87708-1-lizhe.67@bytedance.com Signed-off-by: Li Zhe Signed-off-by: Andrew Morton Reported-by: aiqi.i7 Acked-by: Muchun Song Cc: David Hildenbrand Cc: Oscar Salvador --- fs/hugetlbfs/inode.c | 1 + include/linux/hugetlb.h | 37 +++++++++++++++++++++++++++++++++++++ mm/hugetlb.c | 13 ++++++++++--- 3 files changed, 48 insertions(+), 3 deletions(-) diff --git a/fs/hugetlbfs/inode.c b/fs/hugetlbfs/inode.c index f643855d2acc70..ab1e4dec3f77b8 100644 --- a/fs/hugetlbfs/inode.c +++ b/fs/hugetlbfs/inode.c @@ -920,6 +920,7 @@ static struct inode *hugetlbfs_get_inode(struct super_block *sb, simple_inode_init_ts(inode); info->resv_map = resv_map; info->seals = F_SEAL_SEAL; + hugetlbfs_pmd_sharing_init(inode); switch (mode & S_IFMT) { default: init_special_inode(inode, mode, dev); diff --git a/include/linux/hugetlb.h b/include/linux/hugetlb.h index 24727ece20fe52..5029c72418631b 100644 --- a/include/linux/hugetlb.h +++ b/include/linux/hugetlb.h @@ -11,6 +11,7 @@ #include #include #include +#include #include #include #include @@ -507,6 +508,9 @@ struct hugetlbfs_inode_info { struct inode vfs_inode; struct resv_map *resv_map; unsigned int seals; +#ifdef CONFIG_HUGETLB_PMD_PAGE_TABLE_SHARING + atomic64_t pmd_sharing_count; +#endif }; static inline struct hugetlbfs_inode_info *HUGETLBFS_I(struct inode *inode) @@ -514,6 +518,39 @@ static inline struct hugetlbfs_inode_info *HUGETLBFS_I(struct inode *inode) return container_of(inode, struct hugetlbfs_inode_info, vfs_inode); } +#ifdef CONFIG_HUGETLB_PMD_PAGE_TABLE_SHARING +static inline void hugetlbfs_pmd_sharing_init(struct inode *inode) +{ + atomic64_set(&HUGETLBFS_I(inode)->pmd_sharing_count, 0); +} + +static inline void hugetlbfs_pmd_sharing_inc(struct inode *inode) +{ + atomic64_inc(&HUGETLBFS_I(inode)->pmd_sharing_count); +} + +static inline void hugetlbfs_pmd_sharing_dec(struct inode *inode) +{ + atomic64_dec(&HUGETLBFS_I(inode)->pmd_sharing_count); +} + +static inline bool hugetlbfs_pmd_sharing_active(struct inode *inode) +{ + return atomic64_read(&HUGETLBFS_I(inode)->pmd_sharing_count) != 0; +} +#else +static inline void hugetlbfs_pmd_sharing_init(struct inode *inode) {} + +static inline void hugetlbfs_pmd_sharing_inc(struct inode *inode) {} + +static inline void hugetlbfs_pmd_sharing_dec(struct inode *inode) {} + +static inline bool hugetlbfs_pmd_sharing_active(struct inode *inode) +{ + return false; +} +#endif + extern const struct vm_operations_struct hugetlb_vm_ops; struct file *hugetlb_file_setup(const char *name, size_t size, vma_flags_t acct, int creat_flags, int page_size_log); diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 37f8272f0f2a73..a1b51251103b36 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -5438,10 +5438,12 @@ void __hugetlb_zap_begin(struct vm_area_struct *vma, if (!vma->vm_file) /* hugetlbfs_file_mmap error */ return; - adjust_range_if_pmd_sharing_possible(vma, start, end); hugetlb_vma_lock_write(vma); - if (vma->vm_file) + if (vma->vm_file) { i_mmap_lock_write(vma->vm_file->f_mapping); + if (hugetlbfs_pmd_sharing_active(file_inode(vma->vm_file))) + adjust_range_if_pmd_sharing_possible(vma, start, end); + } } void __hugetlb_zap_end(struct vm_area_struct *vma, @@ -5480,7 +5482,10 @@ void unmap_hugepage_range(struct vm_area_struct *vma, unsigned long start, mmu_notifier_range_init(&range, MMU_NOTIFY_CLEAR, 0, vma->vm_mm, start, end); - adjust_range_if_pmd_sharing_possible(vma, &range.start, &range.end); + i_mmap_assert_write_locked(vma->vm_file->f_mapping); + if (hugetlbfs_pmd_sharing_active(file_inode(vma->vm_file))) + adjust_range_if_pmd_sharing_possible(vma, &range.start, + &range.end); mmu_notifier_invalidate_range_start(&range); tlb_gather_mmu(&tlb, vma->vm_mm); @@ -7083,6 +7088,7 @@ pte_t *huge_pmd_share(struct mm_struct *mm, struct vm_area_struct *vma, if (pud_none(*pud)) { pud_populate(mm, pud, (pmd_t *)((unsigned long)spte & PAGE_MASK)); + hugetlbfs_pmd_sharing_inc(file_inode(vma->vm_file)); mm_inc_nr_pmds(mm); } else { ptdesc_pmd_pts_dec(virt_to_ptdesc(spte)); @@ -7114,6 +7120,7 @@ static int __huge_pmd_unshare(struct mmu_gather *tlb, pud_clear(pud); tlb_unshare_pmd_ptdesc(tlb, virt_to_ptdesc(ptep), addr); + hugetlbfs_pmd_sharing_dec(file_inode(vma->vm_file)); mm_dec_nr_pmds(mm); return 1; From 95066d13d9b980009e53c646aa31fe577ef7bf85 Mon Sep 17 00:00:00 2001 From: Park Tae-sun Date: Wed, 23 Sep 2026 18:26:48 +0900 Subject: [PATCH 1160/1352] selftests/mm: fix mlock2 errno handling and false PASS on ENOSYS While inspecting selftests/mm syscall wrappers, I noticed that mlock2_() in mlock2.h handles the syscall return value differently from other wrappers: int ret = syscall(__NR_mlock2, start, len, flags); if (ret) { errno = ret; return -1; } Commit 1ddae9d67ee1 ("selftests/mm/mlock: print error on failure") introduced this intending to make mlock2_() behave like libc by setting errno and returning -1. However, glibc syscall(2) already returns -1 on failure and sets positive errno. Assigning "errno = ret;" overwrites errno with -1. To verify this, mlock2 was disabled in the kernel (via sys_ni_syscall) to return -ENOSYS. Testing revealed two interrelated defects: 1. In the unmodified test, mlock2_() clobbered errno to -1. The check "if (ret && errno == ENOSYS)" in main() was bypassed, resulting in an immediate crash in the first test: ~ # ./mlock2-tests TAP version 13 1..15 Bail out! mlock2(0): Unknown error -1 # Planned tests != run tests (15 != 0) # Totals: pass:0 fail:0 xfail:0 xpass:0 skip:0 error:0 (exit code: 1 - FAIL) 2. After restoring mlock2_() to directly return syscall(), errno correctly retained ENOSYS (38), entering the ENOSYS check in main(). However, it then called ksft_finished(): ~ # ./mlock2-tests TAP version 13 # Totals: pass:0 fail:0 xfail:0 xpass:0 skip:0 error:0 ~ # echo $? 0 Because ksft_set_plan() had not been called yet (ksft_plan == 0) and zero tests ran (ksft_pass == 0), ksft_finished() evaluated 0 == 0 as success and exited with KSFT_PASS (code 0) without any TAP skip header. Fix both issues by: 1. Returning the syscall() result directly in mlock2_() so that errno is preserved. 2. Calling ksft_exit_skip() on ENOSYS so unsupported kernels report a TAP skip ("1..0 # SKIP ...") and exit with KSFT_SKIP (code 4). Verification on the mlock2-disabled kernel: ~ # ./mlock2-tests TAP version 13 1..0 # SKIP mlock2() syscall is not supported ~ # echo $? 4 Re-enabling mlock2 in the kernel confirmed all 15 tests pass cleanly: ~ # ./mlock2-tests TAP version 13 1..15 ok 1 test_mlock_lock: Locked ... ok 15 test_mlockall_future_droppable: droppable memory not locked # Totals: pass:15 fail:0 xfail:0 xpass:0 skip:0 error:0 ~ # echo $? 0 Link: https://lore.kernel.org/20260923-selftests-mm-mlock2-fix-v1-1-750b627854c6@dgu.ac.kr Fixes: 1ddae9d67ee1 ("selftests/mm/mlock: print error on failure") Fixes: 65c89684896d ("selftests/mm: mlock2-tests: conform test to TAP format output") Signed-off-by: Park Tae-sun Signed-off-by: Andrew Morton Reviewed-by: Gregory Price Acked-by: David Hildenbrand (Arm) Reviewed-by: Muhammad Usama Anjum Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Shuah Khan Cc: Brendan Jackman --- tools/testing/selftests/mm/mlock2-tests.c | 2 +- tools/testing/selftests/mm/mlock2.h | 8 +------- 2 files changed, 2 insertions(+), 8 deletions(-) diff --git a/tools/testing/selftests/mm/mlock2-tests.c b/tools/testing/selftests/mm/mlock2-tests.c index e16e288cc7c1f2..144b550813a6e1 100644 --- a/tools/testing/selftests/mm/mlock2-tests.c +++ b/tools/testing/selftests/mm/mlock2-tests.c @@ -502,7 +502,7 @@ int main(int argc, char **argv) ret = mlock2_(map, size, MLOCK_ONFAULT); if (ret && errno == ENOSYS) - ksft_finished(); + ksft_exit_skip("mlock2() syscall is not supported\n"); munmap(map, size); diff --git a/tools/testing/selftests/mm/mlock2.h b/tools/testing/selftests/mm/mlock2.h index 81e77fa41901a0..4417eaa5cfb78b 100644 --- a/tools/testing/selftests/mm/mlock2.h +++ b/tools/testing/selftests/mm/mlock2.h @@ -6,13 +6,7 @@ static int mlock2_(void *start, size_t len, int flags) { - int ret = syscall(__NR_mlock2, start, len, flags); - - if (ret) { - errno = ret; - return -1; - } - return 0; + return syscall(__NR_mlock2, start, len, flags); } static FILE *seek_to_smaps_entry(unsigned long addr) From 89b3d17048820e2353c4b461a94a412cf35c725a Mon Sep 17 00:00:00 2001 From: Yeoreum Yun Date: Thu, 1 Oct 2026 22:17:51 +0100 Subject: [PATCH 1161/1352] kselftest: mm: prevent random failure of huge page split for khugepaged MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Patch series "kselftest: mm: fix some failure of split_huge_page_test", v9. split_huge_page_test can fail for the following reasons: 1. During the test, khugepaged may collapse previously split pages again, causing intermittent failures. 2. Since glibc commit 321e1fc73f (“malloc: Enable 2MB THP by default on AArch64”), glibc may call madvise(MADV_HUGEPAGE) for sufficiently large allocations made by memalign(). The underlying VMA may start at a different address from the aligned address returned by memalign(). Moreover, a subsequent madvise(MADV_HUGEPAGE) call does not split the VMA because it already has the same advice. This causes the test to fail because the check_huge_xxx() helpers incorrectly require the address returned by memalign() to match the VMA start address reported in /proc/self/smaps. Address these issues by applying MADV_NOHUGEPAGE after faulting in the huge page, preventing khugepaged from collapsing it again, and instead of relying on /proc/self/smaps, use /proc/self/pagemap and /proc/kpageflags to detect huge-page mappings and large folios: 1. If hpage_size == pmd_pagesize, check PAGE_IS_HUGE instead of using check_large_folios(), since only the mapping type matters. This identifies PMD-mapped huge pages. 2. Otherwise, use check_large_folios() to detect large folios. This covers mTHP cases. 3. Check the folio flags according to the type of huge page. Since check_huge_shmem() was required to distinguish shmem huge pages because /proc/self/smaps reports them using a dedicated “ShmemPmdMapped” entry, as opposed to “FilePmdMapped” for file-backed huge pages. Now that /proc/self/smaps is no longer used to detect huge pages and /proc/kpageflags is used instead, it is sufficient to distinguish between file-backed and anonymous pages since the ShmemPmdMapped is also kind of file-backed. Therefore, remove check_huge_shmem() and use check_huge_file() instead. This patch (of 4): There're some random failure for split_huge_page_test when khugepaged collapses pages into pmd again which had split by the test. Prevent the khugepaged's collapses for split page by setting the mapped pmd-huge-page with MADV_NOHUGEPAGE before split. Link: https://lore.kernel.org/20261001-fix_split-v9-0-0f4ba8bbdbdf@arm.com Link: https://lore.kernel.org/20261001-fix_split-v9-1-0f4ba8bbdbdf@arm.com Signed-off-by: Yeoreum Yun Signed-off-by: Andrew Morton Suggested-by: Kevin Brodsky Suggested-by: Lorenzo Stoakes (ARM) Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Zi Yan Reviewed-by: Sarthak Sharma Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Shuah Khan --- .../testing/selftests/mm/split_huge_page_test.c | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/tools/testing/selftests/mm/split_huge_page_test.c b/tools/testing/selftests/mm/split_huge_page_test.c index c5d96a4b1db355..36a6719ff09a88 100644 --- a/tools/testing/selftests/mm/split_huge_page_test.c +++ b/tools/testing/selftests/mm/split_huge_page_test.c @@ -108,6 +108,15 @@ static char *allocate_zero_filled_hugepage(size_t len) return result; } +static void disable_khugepaged(void *addr, size_t len) +{ + /* Disables khugepaged from collapsing pages in range into THPs */ + if (!madvise(addr, len, MADV_NOHUGEPAGE)) + return; + + ksft_exit_fail_perror("MADV_NOHUGEPAGE failed"); +} + static void verify_rss_anon_split_huge_page_all_zeroes(char *one_page, int nr_hpages, size_t len) { unsigned long rss_anon_before, rss_anon_after; @@ -120,6 +129,8 @@ static void verify_rss_anon_split_huge_page_all_zeroes(char *one_page, int nr_hp if (!rss_anon_before) ksft_exit_fail_msg("No RssAnon is allocated before split\n"); + disable_khugepaged(one_page, len); + /* split all THPs */ write_debugfs(PID_FMT, getpid(), (uint64_t)one_page, (uint64_t)one_page + len, 0); @@ -167,6 +178,8 @@ static void split_pmd_thp_to_order(int order) if (!check_huge_anon(one_page, 4 * pmd_pagesize, 4, pmd_pagesize)) ksft_exit_fail_msg("No THP is allocated\n"); + disable_khugepaged(one_page, len); + /* split all THPs */ write_debugfs(PID_FMT, getpid(), (uint64_t)one_page, (uint64_t)one_page + len, order); @@ -215,6 +228,8 @@ static void split_pte_mapped_thp(void) goto out; } + disable_khugepaged(thp_area, thp_area_size); + /* * To challenge spitting code, we will mremap a single page of each * THP (page[i] of thp[i]) in the thp_area into page_area. This will @@ -482,6 +497,7 @@ static int create_pagecache_thp_and_fd(const char *testfile, size_t fd_size, ksft_test_result_skip("Pagecache folio split skipped\n"); return -2; } + disable_khugepaged(*addr, fd_size); return 0; err_out_close: close(*fd); From fbcc0d3dfee84ff1cae8c0d0bf2d5270a0fcfb55 Mon Sep 17 00:00:00 2001 From: Yeoreum Yun Date: Thu, 1 Oct 2026 22:17:52 +0100 Subject: [PATCH 1162/1352] kselftest: mm: replace usage of /proc/self/smaps for __check_pmd_huge() MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Since glibc commit 321e1fc73f (“malloc: Enable 2MB THP by default on AArch64”), glibc may call madvise(MADV_HUGEPAGE) for sufficiently large allocations made by memalign(). The underlying VMA may start at a different address from the aligned address returned by memalign(). Furthermore, a subsequent madvise(MADV_HUGEPAGE) call does not split the VMA because the flag is already set. This causes split_huge_page_test to fail because the check_pmd_huge() helpers incorrectly require the address returned by memalign() to match the VMA start address reported in /proc/self/smaps. Instead of relying on /proc/self/smaps, use /proc/self/pagemap and /proc/kpageflags to detect huge-page mappings checking PAGE_IS_HUGE and PAGE_IS_FILE according to type of huge page. Since shmem pages are also file-backed, simply check whether the page is file-backed. Link: https://lore.kernel.org/20261001-fix_split-v9-2-0f4ba8bbdbdf@arm.com Fixes: 642bc52aed9c ("selftests: vm: bring common functions to a new file") Signed-off-by: Yeoreum Yun Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand (Arm) Reviewed-by: Sarthak Sharma Reviewed-by: Baolin Wang Acked-by: David Hildenbrand (Arm) Tested-by: Baolin Wang Cc: Lorenzo Stoakes Cc: Zi Yan Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Shuah Khan Cc: Kevin Brodsky --- tools/testing/selftests/mm/vm_util.c | 78 +++++++++++++++++----------- 1 file changed, 49 insertions(+), 29 deletions(-) diff --git a/tools/testing/selftests/mm/vm_util.c b/tools/testing/selftests/mm/vm_util.c index a0ab78ceedb130..a3c3a2706c222e 100644 --- a/tools/testing/selftests/mm/vm_util.c +++ b/tools/testing/selftests/mm/vm_util.c @@ -351,24 +351,6 @@ char *__get_smap_entry(void *addr, const char *pattern, char *buf, size_t len) return entry; } -static bool __check_pmd_huge(void *addr, char *pattern, int nr_hpages, - uint64_t hpage_size) -{ - char buffer[MAX_LINE_LENGTH]; - uint64_t thp = -1; - char *entry; - - entry = __get_smap_entry(addr, pattern, buffer, sizeof(buffer)); - if (!entry) - goto err_out; - - if (sscanf(entry, "%9" SCNu64 " kB", &thp) != 1) - ksft_exit_fail_msg("Reading smap error\n"); - -err_out: - return thp == (nr_hpages * (hpage_size >> 10)); -} - static bool check_large_folios(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) { @@ -410,20 +392,51 @@ static bool check_large_folios(void *addr, size_t len, int nr_hpages, return ret; } -bool check_huge_anon(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) +enum check_huge_type { + CHECK_HUGE_ANON, + CHECK_HUGE_FILE, +}; + +static bool check_huge_type(uint64_t categories, enum check_huge_type type) { - uint64_t pmd_pagesize = read_pmd_pagesize(); + const bool file = categories & PAGE_IS_FILE; - if (!pmd_pagesize) - ksft_exit_fail_msg("reading PMD pagesize failed\n"); + switch (type) { + case CHECK_HUGE_ANON: + return !file; + case CHECK_HUGE_FILE: + return file; + } - if (hpage_size == pmd_pagesize) - return __check_pmd_huge(addr, "AnonHugePages: ", nr_hpages, hpage_size); + return false; +} - return check_large_folios(addr, len, nr_hpages, hpage_size); +static bool __check_pmd_huge(void *addr, size_t len, int nr_hpages, + uint64_t hpage_size, enum check_huge_type type) +{ + int pagemap_fd; + int nr_pmd_mappings = 0; + uint64_t categories; + char *start = addr; + char *end = start + len; + + pagemap_fd = open(PAGEMAP_PATH, O_RDONLY); + if (pagemap_fd < 0) + ksft_exit_fail_perror("open pagemap"); + + for (; start < end; start += hpage_size) { + categories = pagemap_scan_get_categories(pagemap_fd, start); + if (!(categories & PAGE_IS_HUGE)) + continue; + if (check_huge_type(categories, type)) + nr_pmd_mappings++; + } + close(pagemap_fd); + + return nr_hpages == nr_pmd_mappings; } -bool check_huge_file(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) +bool check_huge_anon(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) { uint64_t pmd_pagesize = read_pmd_pagesize(); @@ -431,12 +444,13 @@ bool check_huge_file(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) ksft_exit_fail_msg("reading PMD pagesize failed\n"); if (hpage_size == pmd_pagesize) - return __check_pmd_huge(addr, "FilePmdMapped:", nr_hpages, hpage_size); + return __check_pmd_huge(addr, len, nr_hpages, hpage_size, + CHECK_HUGE_ANON); return check_large_folios(addr, len, nr_hpages, hpage_size); } -bool check_huge_shmem(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) +bool check_huge_file(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) { uint64_t pmd_pagesize = read_pmd_pagesize(); @@ -444,11 +458,17 @@ bool check_huge_shmem(void *addr, size_t len, int nr_hpages, uint64_t hpage_size ksft_exit_fail_msg("reading PMD pagesize failed\n"); if (hpage_size == pmd_pagesize) - return __check_pmd_huge(addr, "ShmemPmdMapped:", nr_hpages, hpage_size); + return __check_pmd_huge(addr, len, nr_hpages, hpage_size, + CHECK_HUGE_FILE); return check_large_folios(addr, len, nr_hpages, hpage_size); } +bool check_huge_shmem(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) +{ + return check_huge_file(addr, len, nr_hpages, hpage_size); +} + int64_t allocate_transhuge(void *ptr, int pagemap_fd) { uint64_t ent[2]; From 5d8c4bfd6dd9432f3d4c89a7fe832974221be397 Mon Sep 17 00:00:00 2001 From: Yeoreum Yun Date: Thu, 1 Oct 2026 22:17:53 +0100 Subject: [PATCH 1163/1352] kselftest: mm: integrate huge page checks check_large_folios() only checks for large folios without distinguishing between anonymous and file-backed huge pages. To add huge page type checking, integrate the huge page checks into __check_huge() and __check_type(): 1. If hpage_size == pmd_pagesize, check PAGE_IS_HUGE instead of using check_large_folios(), since only the mapping type matters. This identifies PMD-mapped huge pages. 2. Otherwise, use check_large_folios() to detect large folios. This covers mTHP cases. 3. Check the folio flags according to the huge page type via __check_type() Link: https://lore.kernel.org/20261001-fix_split-v9-3-0f4ba8bbdbdf@arm.com Signed-off-by: Yeoreum Yun Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand (Arm) Suggested-by: Zi Yan Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Shuah Khan Cc: Kevin Brodsky --- tools/testing/selftests/mm/vm_util.c | 90 +++++++++++++++++----------- 1 file changed, 56 insertions(+), 34 deletions(-) diff --git a/tools/testing/selftests/mm/vm_util.c b/tools/testing/selftests/mm/vm_util.c index a3c3a2706c222e..65c06e17c326f6 100644 --- a/tools/testing/selftests/mm/vm_util.c +++ b/tools/testing/selftests/mm/vm_util.c @@ -392,76 +392,98 @@ static bool check_large_folios(void *addr, size_t len, int nr_hpages, return ret; } -enum check_huge_type { - CHECK_HUGE_ANON, - CHECK_HUGE_FILE, +enum check_type { + CHECK_TYPE_ANON, + CHECK_TYPE_FILE, }; -static bool check_huge_type(uint64_t categories, enum check_huge_type type) +static bool __check_type(void *addr, size_t len, uint64_t page_size, + enum check_type type) { - const bool file = categories & PAGE_IS_FILE; + bool ret = false; + int pagemap_fd; + char *start = addr; + char *end = start + len; + uint64_t categories; - switch (type) { - case CHECK_HUGE_ANON: - return !file; - case CHECK_HUGE_FILE: - return file; + pagemap_fd = open(PAGEMAP_PATH, O_RDONLY); + if (pagemap_fd < 0) + ksft_exit_fail_perror("open pagemap"); + + for (; start < end; start += page_size) { + categories = pagemap_scan_get_categories(pagemap_fd, start); + if ((categories & PAGE_IS_PRESENT) != PAGE_IS_PRESENT) + continue; + + if ((type == CHECK_TYPE_FILE) != !!(categories & PAGE_IS_FILE)) + goto out; } - return false; + ret = true; + +out: + close(pagemap_fd); + return ret; } -static bool __check_pmd_huge(void *addr, size_t len, int nr_hpages, - uint64_t hpage_size, enum check_huge_type type) +static bool __check_huge(void *addr, size_t len, int nr_hpages, + uint64_t hpage_size) { + bool ret = false; int pagemap_fd; int nr_pmd_mappings = 0; + uint64_t pmd_pagesize; uint64_t categories; char *start = addr; char *end = start + len; + pmd_pagesize = read_pmd_pagesize(); + if (!pmd_pagesize) + ksft_exit_fail_msg("reading PMD pagesize failed\n"); + pagemap_fd = open(PAGEMAP_PATH, O_RDONLY); if (pagemap_fd < 0) ksft_exit_fail_perror("open pagemap"); + if (hpage_size != pmd_pagesize) { + ret = check_large_folios(addr, len, nr_hpages, hpage_size); + goto out; + } + for (; start < end; start += hpage_size) { categories = pagemap_scan_get_categories(pagemap_fd, start); - if (!(categories & PAGE_IS_HUGE)) - continue; - if (check_huge_type(categories, type)) + if (categories & PAGE_IS_HUGE) nr_pmd_mappings++; } - close(pagemap_fd); - return nr_hpages == nr_pmd_mappings; + if (nr_pmd_mappings != nr_hpages) + goto out; + + ret = true; + +out: + close(pagemap_fd); + return ret; } bool check_huge_anon(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) { - uint64_t pmd_pagesize = read_pmd_pagesize(); - - if (!pmd_pagesize) - ksft_exit_fail_msg("reading PMD pagesize failed\n"); + const uint64_t scan_mapping_size = (nr_hpages > 0) ? hpage_size : psize(); - if (hpage_size == pmd_pagesize) - return __check_pmd_huge(addr, len, nr_hpages, hpage_size, - CHECK_HUGE_ANON); + if (!__check_huge(addr, len, nr_hpages, hpage_size)) + return false; - return check_large_folios(addr, len, nr_hpages, hpage_size); + return __check_type(addr, len, scan_mapping_size, CHECK_TYPE_ANON); } bool check_huge_file(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) { - uint64_t pmd_pagesize = read_pmd_pagesize(); + const uint64_t scan_mapping_size = (nr_hpages > 0) ? hpage_size : psize(); - if (!pmd_pagesize) - ksft_exit_fail_msg("reading PMD pagesize failed\n"); - - if (hpage_size == pmd_pagesize) - return __check_pmd_huge(addr, len, nr_hpages, hpage_size, - CHECK_HUGE_FILE); + if (!__check_huge(addr, len, nr_hpages, hpage_size)) + return false; - return check_large_folios(addr, len, nr_hpages, hpage_size); + return __check_type(addr, len, scan_mapping_size, CHECK_TYPE_FILE); } bool check_huge_shmem(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) From 608b401b37d2e32fdd621b163b80468280a16c47 Mon Sep 17 00:00:00 2001 From: Yeoreum Yun Date: Thu, 1 Oct 2026 22:17:54 +0100 Subject: [PATCH 1164/1352] kselftest: mm: remove check_huge_shmem() MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit check_huge_shmem() was required to distinguish shmem huge pages because /proc/self/smaps reports them using a dedicated “ShmemPmdMapped” entry, as opposed to “FilePmdMapped” for file-backed huge pages. Now that /proc/self/smaps is no longer used to detect huge pages and /proc/kpageflags is used instead, it is sufficient to distinguish between file-backed and anonymous pages since the ShmemPmdMapped is also kind of file-backed. Therefore, remove check_huge_shmem() and use check_huge_file() instead and cleanup khugepaged's check_huge operation in mem_ops. Link: https://lore.kernel.org/20261001-fix_split-v9-4-0f4ba8bbdbdf@arm.com Signed-off-by: Yeoreum Yun Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand (Arm) Reviewed-by: Baolin Wang Reviewed-by: Sarthak Sharma Acked-by: Zi Yan Acked-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Shuah Khan Cc: Kevin Brodsky --- .../selftests/mm/folio_split_race_test.c | 2 +- tools/testing/selftests/mm/khugepaged.c | 38 +++---------------- tools/testing/selftests/mm/uffd-common.c | 4 +- tools/testing/selftests/mm/vm_util.c | 5 --- tools/testing/selftests/mm/vm_util.h | 1 - 5 files changed, 9 insertions(+), 41 deletions(-) diff --git a/tools/testing/selftests/mm/folio_split_race_test.c b/tools/testing/selftests/mm/folio_split_race_test.c index e4660bf89b624a..1a9840b73e6189 100644 --- a/tools/testing/selftests/mm/folio_split_race_test.c +++ b/tools/testing/selftests/mm/folio_split_race_test.c @@ -181,7 +181,7 @@ static uint64_t run_iteration(void) for (i = 0; i < TOTAL_PAGES; i++) fill_page(mmap_base, i); - if (!check_huge_shmem(mmap_base, FILE_SIZE, NR_PMD_PAGE, pmd_pagesize)) + if (!check_huge_file(mmap_base, FILE_SIZE, NR_PMD_PAGE, pmd_pagesize)) ksft_exit_fail_msg("No shmem THP is allocated\n"); if (pthread_barrier_init(&ctl.barrier, NULL, NUM_READER_THREADS + 1) != 0) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 4e888b7bf31007..1bc7acc66a4697 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -57,7 +57,7 @@ struct mem_ops { void *(*setup_area)(int nr_hpages); void (*cleanup_area)(void *p, unsigned long size); void (*fault)(void *p, unsigned long start, unsigned long end); - bool (*check_huge)(void *addr, size_t len, int nr_hpages, unsigned long hpage_size); + bool (*check_huge)(void *addr, size_t len, int nr_hpages, uint64_t hpage_size); const char *name; }; @@ -343,12 +343,6 @@ static void anon_fault(void *p, unsigned long start, unsigned long end) fill_memory(p, start, end); } -static bool anon_check_huge(void *addr, size_t len, int nr_hpages, - unsigned long hpage_size) -{ - return check_huge_anon(addr, len, nr_hpages, hpage_size); -} - static void *file_setup_area_common(int nr_hpages, enum file_setup_ops setup) { const int open_opt = setup == FILE_SETUP_READ_ONLY_FS ? O_RDONLY : O_RDWR; @@ -444,20 +438,6 @@ static void file_fault_write(void *p, unsigned long start, unsigned long end) ksft_exit_fail_perror("madvise(MADV_POPULATE_WRITE)"); } -static bool file_check_huge(void *addr, size_t len, int nr_hpages, - unsigned long hpage_size) -{ - switch (finfo.type) { - case VMA_FILE: - return check_huge_file(addr, len, nr_hpages, hpage_size); - case VMA_SHMEM: - return check_huge_shmem(addr, len, nr_hpages, hpage_size); - default: - ksft_exit_fail_msg("Unknown VMA type\n"); - return false; - } -} - static void *shmem_setup_area(int nr_hpages) { void *p; @@ -481,17 +461,11 @@ static void shmem_cleanup_area(void *p, unsigned long size) close(finfo.fd); } -static bool shmem_check_huge(void *addr, size_t len, int nr_hpages, - unsigned long hpage_size) -{ - return check_huge_shmem(addr, len, nr_hpages, hpage_size); -} - static struct mem_ops __anon_ops = { .setup_area = &anon_setup_area, .cleanup_area = &anon_cleanup_area, .fault = &anon_fault, - .check_huge = &anon_check_huge, + .check_huge = &check_huge_anon, .name = "anon", }; @@ -499,7 +473,7 @@ static struct mem_ops __read_only_file_ops = { .setup_area = &file_setup_read_only_area, .cleanup_area = &file_cleanup_area, .fault = &file_fault_read, - .check_huge = &file_check_huge, + .check_huge = &check_huge_file, .name = "file", }; @@ -507,7 +481,7 @@ static struct mem_ops __read_write_file_read_ops = { .setup_area = &file_setup_read_write_fs_read_area, .cleanup_area = &file_cleanup_area, .fault = &file_fault_read_and_flush, - .check_huge = &file_check_huge, + .check_huge = &check_huge_file, .name = "file", }; @@ -515,7 +489,7 @@ static struct mem_ops __read_write_file_write_ops = { .setup_area = &file_setup_read_write_fs_write_area, .cleanup_area = &file_cleanup_area, .fault = &file_fault_write, - .check_huge = &file_check_huge, + .check_huge = &check_huge_file, .name = "file", }; @@ -523,7 +497,7 @@ static struct mem_ops __shmem_ops = { .setup_area = &shmem_setup_area, .cleanup_area = &shmem_cleanup_area, .fault = &anon_fault, - .check_huge = &shmem_check_huge, + .check_huge = &check_huge_file, .name = "shmem", }; diff --git a/tools/testing/selftests/mm/uffd-common.c b/tools/testing/selftests/mm/uffd-common.c index 1fb967ef498540..a76851aced4d0b 100644 --- a/tools/testing/selftests/mm/uffd-common.c +++ b/tools/testing/selftests/mm/uffd-common.c @@ -196,8 +196,8 @@ static void shmem_check_pmd_mapping(uffd_global_test_opts_t *gopts, void *p, int { size_t len = expect_nr_hpages * read_pmd_pagesize(); - if (!check_huge_shmem(gopts->area_dst_alias, len, expect_nr_hpages, - read_pmd_pagesize())) + if (!check_huge_file(gopts->area_dst_alias, len, expect_nr_hpages, + read_pmd_pagesize())) err("Did not find expected %d number of hugepages", expect_nr_hpages); } diff --git a/tools/testing/selftests/mm/vm_util.c b/tools/testing/selftests/mm/vm_util.c index 65c06e17c326f6..515031b8283dac 100644 --- a/tools/testing/selftests/mm/vm_util.c +++ b/tools/testing/selftests/mm/vm_util.c @@ -486,11 +486,6 @@ bool check_huge_file(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) return __check_type(addr, len, scan_mapping_size, CHECK_TYPE_FILE); } -bool check_huge_shmem(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) -{ - return check_huge_file(addr, len, nr_hpages, hpage_size); -} - int64_t allocate_transhuge(void *ptr, int pagemap_fd) { uint64_t ent[2]; diff --git a/tools/testing/selftests/mm/vm_util.h b/tools/testing/selftests/mm/vm_util.h index 072a6c756c5170..b5d59729d43275 100644 --- a/tools/testing/selftests/mm/vm_util.h +++ b/tools/testing/selftests/mm/vm_util.h @@ -96,7 +96,6 @@ uint64_t read_pmd_pagesize(void); unsigned long rss_anon(void); bool check_huge_anon(void *addr, size_t len, int nr_hpages, uint64_t hpage_size); bool check_huge_file(void *addr, size_t len, int nr_hpages, uint64_t hpage_size); -bool check_huge_shmem(void *addr, size_t len, int nr_hpages, uint64_t hpage_size); int64_t allocate_transhuge(void *ptr, int pagemap_fd); int pageflags_get(unsigned long pfn, int kpageflags_fd, uint64_t *flags); int gather_folio_orders(char *vaddr_start, size_t len, From 6e1d927cffe64f1cb4728feabfde7618db785fae Mon Sep 17 00:00:00 2001 From: Usama Arif Date: Thu, 24 Sep 2026 12:11:03 -0700 Subject: [PATCH 1165/1352] mm: remove the unused zone->unaccepted_cleanup Commit fefc07518227 ("mm/page_alloc: fix race condition in unaccepted memory handling") removed the zones_with_unaccepted_pages static key and the work that decremented it, but left the work_struct behind in struct zone. Nothing uses it anymore, so let's remove it. Link: https://lore.kernel.org/20260924191103.3475117-1-usama.arif@linux.dev Signed-off-by: Usama Arif Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Shakeel Butt Reviewed-by: Johannes Weiner Reviewed-by: Barry Song Reviewed-by: Kiryl Shutsemau (Meta) Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Kairui Song Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Qi Zheng Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie --- include/linux/mmzone.h | 3 --- 1 file changed, 3 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 851b14c91cb1a5..3b96d6c7123b45 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -1077,9 +1077,6 @@ struct zone { #ifdef CONFIG_UNACCEPTED_MEMORY /* Pages to be accepted. All pages on the list are MAX_PAGE_ORDER */ struct list_head unaccepted_pages; - - /* To be called once the last page in the zone is accepted */ - struct work_struct unaccepted_cleanup; #endif /* zone flags, see below */ From c3222a5915e8393e08c1f80e09b4aaae8a4e24bf Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Wed, 16 Sep 2026 10:19:32 +0530 Subject: [PATCH 1166/1352] arm64/mm: move __check_safe_pte_update() Patch series "arm64/mm: Standardize printing for pgtable entries", v3. Standardize printing for pgtable entries using recently introduced generic helper ptval_bytes_to_hex_str() in core MM which automatically enables 128 bits entries when added later. But first move __check_safe_pte_update() outside to avoid a cyclic dependency while accessing these afore mentioned core MM helpers defined in . This replaces the original page table entry print standardisation proposal which was part of the D128 series [1]. This patch (of 2): The page table entry print helpers and related macros which are defined in will not be accessible in platform which is basically caused by cycling dependency. Move __check_safe_pte_update() inside arch/arm64/mm/mmu.c as a preparation for subsequent usage of the afore mentioned generic MM helpers. While here drop IS_ENABLED(CONFIG_DEBUG_VM), although wrap __check_safe_pte_update() inside #ifdef CONFIG_DEBUG_VM that preserves the current code optimization which is achieved via the static inline functions. This does not cause any functional change. Link: https://lore.kernel.org/20260916044933.2689426-2-anshuman.khandual@arm.com Link: https://lore.kernel.org/linux-mm/20260729122452.3797443-11-anshuman.khandual@arm.com/ [1] Signed-off-by: Anshuman Khandual Signed-off-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) Reviewed-by: Catalin Marinas Cc: Will Deacon Cc: Mike Rapoport Cc: Lorenzo Stoakes --- arch/arm64/include/asm/pgtable.h | 49 ++++---------------------------- arch/arm64/mm/mmu.c | 44 ++++++++++++++++++++++++++++ 2 files changed, 49 insertions(+), 44 deletions(-) diff --git a/arch/arm64/include/asm/pgtable.h b/arch/arm64/include/asm/pgtable.h index e89ec5f4787b49..763c5a411d64e3 100644 --- a/arch/arm64/include/asm/pgtable.h +++ b/arch/arm64/include/asm/pgtable.h @@ -387,52 +387,13 @@ static inline pte_t __ptep_get(pte_t *ptep) extern void __sync_icache_dcache(pte_t pteval); bool pgattr_change_is_safe(pteval_t old, pteval_t new); -/* - * PTE bits configuration in the presence of hardware Dirty Bit Management - * (PTE_WRITE == PTE_DBM): - * - * Dirty Writable | PTE_RDONLY PTE_WRITE PTE_DIRTY (sw) - * 0 0 | 1 0 0 - * 0 1 | 1 1 0 - * 1 0 | 1 0 1 - * 1 1 | 0 1 x - * - * When hardware DBM is not present, the software PTE_DIRTY bit is updated via - * the page fault mechanism. Checking the dirty status of a pte becomes: - * - * PTE_DIRTY || (PTE_WRITE && !PTE_RDONLY) - */ - -static inline void __check_safe_pte_update(struct mm_struct *mm, pte_t *ptep, - pte_t pte) +#ifdef CONFIG_DEBUG_VM +void __check_safe_pte_update(struct mm_struct *mm, pte_t *ptep, pte_t pte); +#else +static inline void __check_safe_pte_update(struct mm_struct *mm, pte_t *ptep, pte_t pte) { - pte_t old_pte; - - if (!IS_ENABLED(CONFIG_DEBUG_VM)) - return; - - old_pte = __ptep_get(ptep); - - if (!pte_valid(old_pte) || !pte_valid(pte)) - return; - if (mm != current->active_mm && atomic_read(&mm->mm_users) <= 1) - return; - - /* - * Check for potential race with hardware updates of the pte - * (__ptep_set_access_flags safely changes valid ptes without going - * through an invalid entry). - */ - VM_WARN_ONCE(!pte_young(pte), - "%s: racy access flag clearing: 0x%016llx -> 0x%016llx", - __func__, pte_val(old_pte), pte_val(pte)); - VM_WARN_ONCE(pte_write(old_pte) && !pte_dirty(pte), - "%s: racy dirty state clearing: 0x%016llx -> 0x%016llx", - __func__, pte_val(old_pte), pte_val(pte)); - VM_WARN_ONCE(!pgattr_change_is_safe(pte_val(old_pte), pte_val(pte)), - "%s: unsafe attribute change: 0x%016llx -> 0x%016llx", - __func__, pte_val(old_pte), pte_val(pte)); } +#endif /* CONFIG_DEBUG_VM */ static inline void __sync_cache_and_tags(pte_t pte, unsigned int nr_pages) { diff --git a/arch/arm64/mm/mmu.c b/arch/arm64/mm/mmu.c index 79d90226fd5dc9..cb49469707a877 100644 --- a/arch/arm64/mm/mmu.c +++ b/arch/arm64/mm/mmu.c @@ -2392,4 +2392,48 @@ int arch_set_user_pkey_access(int pkey, unsigned long init_val) return 0; } + +/* + * PTE bits configuration in the presence of hardware Dirty Bit Management + * (PTE_WRITE == PTE_DBM): + * + * Dirty Writable | PTE_RDONLY PTE_WRITE PTE_DIRTY (sw) + * 0 0 | 1 0 0 + * 0 1 | 1 1 0 + * 1 0 | 1 0 1 + * 1 1 | 0 1 x + * + * When hardware DBM is not present, the software PTE_DIRTY bit is updated via + * the page fault mechanism. Checking the dirty status of a pte becomes: + * + * PTE_DIRTY || (PTE_WRITE && !PTE_RDONLY) + */ +#ifdef CONFIG_DEBUG_VM +void __check_safe_pte_update(struct mm_struct *mm, pte_t *ptep, pte_t pte) +{ + pte_t old_pte; + + old_pte = __ptep_get(ptep); + + if (!pte_valid(old_pte) || !pte_valid(pte)) + return; + if (mm != current->active_mm && atomic_read(&mm->mm_users) <= 1) + return; + + /* + * Check for potential race with hardware updates of the pte + * (__ptep_set_access_flags safely changes valid ptes without going + * through an invalid entry). + */ + VM_WARN_ONCE(!pte_young(pte), + "%s: racy access flag clearing: 0x%016llx -> 0x%016llx", + __func__, pte_val(old_pte), pte_val(pte)); + VM_WARN_ONCE(pte_write(old_pte) && !pte_dirty(pte), + "%s: racy dirty state clearing: 0x%016llx -> 0x%016llx", + __func__, pte_val(old_pte), pte_val(pte)); + VM_WARN_ONCE(!pgattr_change_is_safe(pte_val(old_pte), pte_val(pte)), + "%s: unsafe attribute change: 0x%016llx -> 0x%016llx", + __func__, pte_val(old_pte), pte_val(pte)); +} +#endif /* CONFIG_DEBUG_VM */ #endif From 2d01b73e513b3f0da6e953aaa3d2d2288fcec66d Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 28 Sep 2026 09:42:06 +0200 Subject: [PATCH 1167/1352] arm64-mm-move-__check_safe_pte_update-fix fix this: >> aarch64-linux-ld: Unexpected GOT/PLT entries detected! >> aarch64-linux-ld: Unexpected run-time procedure linkages detected! aarch64-linux-ld: arch/arm64/mm/mmu.o: in function `__set_ptes_anysz.isra.40.constprop.47': >> mmu.c:(.text+0x9a8): undefined reference to `__check_safe_pte_update' aarch64-linux-ld: arch/arm64/mm/hugetlbpage.o: in function `__set_ptes_anysz.isra.32': >> hugetlbpage.c:(.text+0x1a8): undefined reference to `__check_safe_pte_update' aarch64-linux-ld: kernel/bpf/arena.o: in function `apply_range_set_cb': >> arena.c:(.text+0x11fc): undefined reference to `__check_safe_pte_update' aarch64-linux-ld: mm/gup.o: in function `follow_page_pte': >> gup.c:(.text+0x2b48): undefined reference to `__check_safe_pte_update' aarch64-linux-ld: mm/memory.o: in function `__set_ptes.isra.207': >> memory.c:(.text+0x3418): undefined reference to `__check_safe_pte_update' aarch64-linux-ld: mm/mprotect.o:mprotect.c:(.text+0x1080): more undefined references to `__check_safe_pte_update' follow Link: https://lore.kernel.org/99023678-fd8a-4eef-8e43-a07b504e3e96@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Reported-by: kernel test robot Closes: https://lore.kernel.org/oe-kbuild-all/202609270030.Y0QVBPmK-lkp@intel.com/ Cc: Anshuman Khandual Cc: Catalin Marinas --- arch/arm64/mm/mmu.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/arch/arm64/mm/mmu.c b/arch/arm64/mm/mmu.c index cb49469707a877..05b7b52d761cdf 100644 --- a/arch/arm64/mm/mmu.c +++ b/arch/arm64/mm/mmu.c @@ -2392,6 +2392,7 @@ int arch_set_user_pkey_access(int pkey, unsigned long init_val) return 0; } +#endif /* * PTE bits configuration in the presence of hardware Dirty Bit Management @@ -2436,4 +2437,3 @@ void __check_safe_pte_update(struct mm_struct *mm, pte_t *ptep, pte_t pte) __func__, pte_val(old_pte), pte_val(pte)); } #endif /* CONFIG_DEBUG_VM */ -#endif From 5309564718949792966dc8c26333dafd1a45e5c0 Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Wed, 16 Sep 2026 10:19:33 +0530 Subject: [PATCH 1168/1352] arm64/mm: standardize printing for pgtable entries Standardize printing for pgtable entries using recently introduced generic helper ptval_bytes_to_hex_str() in core MM which automatically enables 128 bits entries when added later. Link: https://lore.kernel.org/20260916044933.2689426-3-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Signed-off-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) Reviewed-by: Catalin Marinas Cc: Will Deacon Cc: Mike Rapoport Cc: Lorenzo Stoakes --- arch/arm64/mm/fault.c | 16 +++++++++++----- arch/arm64/mm/mmu.c | 16 ++++++++++------ 2 files changed, 21 insertions(+), 11 deletions(-) diff --git a/arch/arm64/mm/fault.c b/arch/arm64/mm/fault.c index 75c3e463df2ef2..2cecf6ba6df7cf 100644 --- a/arch/arm64/mm/fault.c +++ b/arch/arm64/mm/fault.c @@ -131,6 +131,7 @@ static inline unsigned long mm_to_pgd_phys(struct mm_struct *mm) */ static void show_pte(unsigned long addr) { + char pxd_str[PTVAL_STR_MAX]; struct mm_struct *mm; pgd_t *pgdp; pgd_t pgd; @@ -160,7 +161,8 @@ static void show_pte(unsigned long addr) pgdp = pgd_offset(mm, addr); pgd = READ_ONCE(*pgdp); - pr_alert("[%016lx] pgd=%016llx", addr, pgd_val(pgd)); + ptval_to_str(pxd_str, pgd_val(pgd)); + pr_alert("[%016lx] pgd=%s", addr, pxd_str); do { p4d_t *p4dp, p4d; @@ -173,19 +175,22 @@ static void show_pte(unsigned long addr) p4dp = p4d_offset_lockless(pgdp, pgd, addr); p4d = READ_ONCE(*p4dp); - pr_cont(", p4d=%016llx", p4d_val(p4d)); + ptval_to_str(pxd_str, p4d_val(p4d)); + pr_cont(", p4d=%s", pxd_str); if (p4d_none(p4d) || p4d_bad(p4d)) break; pudp = pud_offset_lockless(p4dp, p4d, addr); pud = READ_ONCE(*pudp); - pr_cont(", pud=%016llx", pud_val(pud)); + ptval_to_str(pxd_str, pud_val(pud)); + pr_cont(", pud=%s", pxd_str); if (pud_none(pud) || pud_bad(pud)) break; pmdp = pmd_offset_lockless(pudp, pud, addr); pmd = READ_ONCE(*pmdp); - pr_cont(", pmd=%016llx", pmd_val(pmd)); + ptval_to_str(pxd_str, pmd_val(pmd)); + pr_cont(", pmd=%s", pxd_str); if (pmd_none(pmd) || pmd_bad(pmd)) break; @@ -194,7 +199,8 @@ static void show_pte(unsigned long addr) break; pte = __ptep_get(ptep); - pr_cont(", pte=%016llx", pte_val(pte)); + ptval_to_str(pxd_str, pte_val(pte)); + pr_cont(", pte=%s", pxd_str); pte_unmap(ptep); } while(0); diff --git a/arch/arm64/mm/mmu.c b/arch/arm64/mm/mmu.c index 05b7b52d761cdf..7343ac9294f8d6 100644 --- a/arch/arm64/mm/mmu.c +++ b/arch/arm64/mm/mmu.c @@ -2412,6 +2412,8 @@ int arch_set_user_pkey_access(int pkey, unsigned long init_val) #ifdef CONFIG_DEBUG_VM void __check_safe_pte_update(struct mm_struct *mm, pte_t *ptep, pte_t pte) { + char pte_str_old[PTVAL_STR_MAX]; + char pte_str[PTVAL_STR_MAX]; pte_t old_pte; old_pte = __ptep_get(ptep); @@ -2426,14 +2428,16 @@ void __check_safe_pte_update(struct mm_struct *mm, pte_t *ptep, pte_t pte) * (__ptep_set_access_flags safely changes valid ptes without going * through an invalid entry). */ + ptval_to_str(pte_str, pte_val(pte)); + ptval_to_str(pte_str_old, pte_val(old_pte)); VM_WARN_ONCE(!pte_young(pte), - "%s: racy access flag clearing: 0x%016llx -> 0x%016llx", - __func__, pte_val(old_pte), pte_val(pte)); + "%s: racy access flag clearing: %s -> %s", + __func__, pte_str_old, pte_str); VM_WARN_ONCE(pte_write(old_pte) && !pte_dirty(pte), - "%s: racy dirty state clearing: 0x%016llx -> 0x%016llx", - __func__, pte_val(old_pte), pte_val(pte)); + "%s: racy dirty state clearing: %s -> %s", + __func__, pte_str_old, pte_str); VM_WARN_ONCE(!pgattr_change_is_safe(pte_val(old_pte), pte_val(pte)), - "%s: unsafe attribute change: 0x%016llx -> 0x%016llx", - __func__, pte_val(old_pte), pte_val(pte)); + "%s: unsafe attribute change: %s -> %s", + __func__, pte_str_old, pte_str); } #endif /* CONFIG_DEBUG_VM */ From 465d19caee9f895b52b54708c05e7695f6c2c665 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Sat, 26 Sep 2026 12:29:29 +0300 Subject: [PATCH 1169/1352] arch, mm: promote DEBUG_WX to CHECK_WX Verification that the kernel does not have writable + executable mappings is about detecting security risks rather than a pure debug feature. Major distribution configurations enable it in their kernels as well as defconfigs of most architectures that have ARCH_HAS_DEBUG_WX. Rename relevant generic configuration options to use CHECK_WX and move their definitions from mm/Kconfig.debug to mm/Kconfig. Rename *debug_checkwx() funcitons and macros to *pgtable_checkwx(). For arm that does not widely enable it, only rename its variants of the config options. Enabling CHECK_WX adds a few kilobytes to the kernel binary and while the added size can be slightly reduced with churny updates of architecture implementations of ptdump, the core functionality takes most of the added size. It cannot be moved to .init.text because the verification has to happen after init sections are freed. With this, make generic CHECK_WX default to STRICT_KERNEL_RWX while still leaving users targeting small kernels the possibility to opt-out. Link: https://lore.kernel.org/20260926-direct-map-verify-wx-v2-1-efcd64a6b74a@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Suggested-by: Dave Hansen Acked-by: Lorenzo Stoakes (ARM) Acked-by: Dave Hansen Acked-by: David Hildenbrand (Arm) Reviewed-by: Anshuman Khandual Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Borislav Petkov Cc: Catalin Marinas Cc: Christophe Leroy Cc: Christian Borntraeger Cc: Gerald Schaefer Cc: Heiko Carstens Cc: Ingo Molnar Cc: Liam R. Howlett Cc: Madhavan Srinivasan Cc: Mark Rutland Cc: Michael Ellerman Cc: Michal Hocko Cc: Nicholas Piggin Cc: Palmer Dabbelt Cc: Paul Walmsley Cc: H. Peter Anvin Cc: Ritesh Harjani (IBM) Cc: Russell King Cc: Shrikanth Hegde Cc: Suren Baghdasaryan Cc: Sven Schnelle Cc: Thomas Gleixner Cc: Vasily Gorbik Cc: Vlastimil Babka Cc: Will Deacon --- arch/arm/Kconfig.debug | 2 +- arch/arm/configs/aspeed_g4_defconfig | 2 +- arch/arm/configs/aspeed_g5_defconfig | 2 +- arch/arm/configs/shmobile_defconfig | 2 +- arch/arm/include/asm/ptdump.h | 6 ++-- arch/arm/mm/init.c | 2 +- arch/arm64/Kconfig | 2 +- arch/powerpc/Kconfig | 2 +- arch/powerpc/configs/ppc64_defconfig | 2 +- arch/powerpc/mm/ptdump/ptdump.c | 2 +- arch/riscv/Kconfig | 2 +- arch/s390/Kconfig | 2 +- arch/s390/configs/debug_defconfig | 2 +- arch/s390/configs/defconfig | 2 +- arch/s390/mm/dump_pagetables.c | 2 +- arch/x86/Kconfig | 2 +- arch/x86/configs/x86_64_defconfig | 2 +- arch/x86/include/asm/pgtable.h | 6 ++-- arch/x86/mm/pti.c | 2 +- include/linux/ptdump.h | 4 +-- init/main.c | 2 +- kernel/configs/debug.config | 2 +- mm/Kconfig | 41 ++++++++++++++++++++++++++++ mm/Kconfig.debug | 39 -------------------------- 24 files changed, 68 insertions(+), 66 deletions(-) diff --git a/arch/arm/Kconfig.debug b/arch/arm/Kconfig.debug index 366f162e147d11..abcf14f10276be 100644 --- a/arch/arm/Kconfig.debug +++ b/arch/arm/Kconfig.debug @@ -17,7 +17,7 @@ config ARM_PTDUMP_DEBUGFS kernel. If in doubt, say "N" -config ARM_DEBUG_WX +config ARM_CHECK_WX bool "Warn on W+X mappings at boot" depends on MMU select ARM_PTDUMP_CORE diff --git a/arch/arm/configs/aspeed_g4_defconfig b/arch/arm/configs/aspeed_g4_defconfig index f86dd4ce7d0def..2c9d5a644ae965 100644 --- a/arch/arm/configs/aspeed_g4_defconfig +++ b/arch/arm/configs/aspeed_g4_defconfig @@ -249,7 +249,7 @@ CONFIG_DEBUG_INFO_REDUCED=y CONFIG_GDB_SCRIPTS=y CONFIG_STRIP_ASM_SYMS=y CONFIG_DEBUG_FS=y -CONFIG_ARM_DEBUG_WX=y +CONFIG_ARM_CHECK_WX=y CONFIG_SCHED_STACK_END_CHECK=y CONFIG_PANIC_ON_OOPS=y CONFIG_PANIC_TIMEOUT=-1 diff --git a/arch/arm/configs/aspeed_g5_defconfig b/arch/arm/configs/aspeed_g5_defconfig index 45b937419dbdaa..1327a09e163ab8 100644 --- a/arch/arm/configs/aspeed_g5_defconfig +++ b/arch/arm/configs/aspeed_g5_defconfig @@ -300,7 +300,7 @@ CONFIG_DEBUG_INFO_REDUCED=y CONFIG_GDB_SCRIPTS=y CONFIG_STRIP_ASM_SYMS=y CONFIG_DEBUG_FS=y -CONFIG_ARM_DEBUG_WX=y +CONFIG_ARM_CHECK_WX=y CONFIG_SCHED_STACK_END_CHECK=y CONFIG_PANIC_ON_OOPS=y CONFIG_PANIC_TIMEOUT=-1 diff --git a/arch/arm/configs/shmobile_defconfig b/arch/arm/configs/shmobile_defconfig index 6f9696e9fe17dd..cc22e22b989e97 100644 --- a/arch/arm/configs/shmobile_defconfig +++ b/arch/arm/configs/shmobile_defconfig @@ -225,4 +225,4 @@ CONFIG_CMA_SIZE_MBYTES=64 CONFIG_PRINTK_TIME=y CONFIG_DEBUG_KERNEL=y CONFIG_DEBUG_FS=y -CONFIG_ARM_DEBUG_WX=y +CONFIG_ARM_CHECK_WX=y diff --git a/arch/arm/include/asm/ptdump.h b/arch/arm/include/asm/ptdump.h index 46a4575146ee85..3c5245220ecfee 100644 --- a/arch/arm/include/asm/ptdump.h +++ b/arch/arm/include/asm/ptdump.h @@ -32,10 +32,10 @@ void ptdump_check_wx(void); #endif /* CONFIG_ARM_PTDUMP_CORE */ -#ifdef CONFIG_ARM_DEBUG_WX -#define arm_debug_checkwx() ptdump_check_wx() +#ifdef CONFIG_ARM_CHECK_WX +#define arm_pgtable_checkwx() ptdump_check_wx() #else -#define arm_debug_checkwx() do { } while (0) +#define arm_pgtable_checkwx() do { } while (0) #endif #endif /* __ASM_PTDUMP_H */ diff --git a/arch/arm/mm/init.c b/arch/arm/mm/init.c index 0cc1bf04686d83..c515faf22eebe9 100644 --- a/arch/arm/mm/init.c +++ b/arch/arm/mm/init.c @@ -403,7 +403,7 @@ static int __mark_rodata_ro(void *unused) void mark_rodata_ro(void) { stop_machine(__mark_rodata_ro, NULL, NULL); - arm_debug_checkwx(); + arm_pgtable_checkwx(); } #else diff --git a/arch/arm64/Kconfig b/arch/arm64/Kconfig index b6c2dd8b26124d..b51d23a62c9224 100644 --- a/arch/arm64/Kconfig +++ b/arch/arm64/Kconfig @@ -11,7 +11,7 @@ config ARM64 select ACPI_MCFG if (ACPI && PCI) select ACPI_SPCR_TABLE if ACPI select ACPI_PPTT if ACPI - select ARCH_HAS_DEBUG_WX + select ARCH_HAS_CHECK_WX select ARCH_BINFMT_ELF_EXTRA_PHDRS select ARCH_BINFMT_ELF_STATE select ARCH_ENABLE_HUGEPAGE_MIGRATION if HUGETLB_PAGE && MIGRATION diff --git a/arch/powerpc/Kconfig b/arch/powerpc/Kconfig index 0767cfcbaa422b..877393207e3b54 100644 --- a/arch/powerpc/Kconfig +++ b/arch/powerpc/Kconfig @@ -130,7 +130,7 @@ config PPC select ARCH_HAS_CURRENT_STACK_POINTER select ARCH_HAS_DEBUG_VIRTUAL select ARCH_HAS_DEBUG_VM_PGTABLE - select ARCH_HAS_DEBUG_WX if STRICT_KERNEL_RWX + select ARCH_HAS_CHECK_WX if STRICT_KERNEL_RWX select ARCH_HAS_DEVMEM_IS_ALLOWED select ARCH_HAS_DMA_MAP_DIRECT if PPC_PSERIES select ARCH_HAS_DMA_OPS if PPC64 diff --git a/arch/powerpc/configs/ppc64_defconfig b/arch/powerpc/configs/ppc64_defconfig index 1eb8e3457e8bff..5c33f0bba0e361 100644 --- a/arch/powerpc/configs/ppc64_defconfig +++ b/arch/powerpc/configs/ppc64_defconfig @@ -393,7 +393,7 @@ CONFIG_MAGIC_SYSRQ=y CONFIG_PAGE_OWNER=y CONFIG_PAGE_POISONING=y CONFIG_DEBUG_RODATA_TEST=y -CONFIG_DEBUG_WX=y +CONFIG_CHECK_WX=y CONFIG_DEBUG_STACK_USAGE=y CONFIG_DEBUG_VM=y # CONFIG_DEBUG_VM_PGTABLE is not set diff --git a/arch/powerpc/mm/ptdump/ptdump.c b/arch/powerpc/mm/ptdump/ptdump.c index 0d499aebee72ff..3451351b756b4d 100644 --- a/arch/powerpc/mm/ptdump/ptdump.c +++ b/arch/powerpc/mm/ptdump/ptdump.c @@ -191,7 +191,7 @@ static void note_prot_wx(struct pg_state *st, unsigned long addr) if (!pte_write(pte) || !pte_exec(pte)) return; - WARN_ONCE(IS_ENABLED(CONFIG_DEBUG_WX), + WARN_ONCE(IS_ENABLED(CONFIG_CHECK_WX), "powerpc/mm: Found insecure W+X mapping at address %p/%pS\n", (void *)st->start_address, (void *)st->start_address); diff --git a/arch/riscv/Kconfig b/arch/riscv/Kconfig index 0db108ea146626..f409f264d8c413 100644 --- a/arch/riscv/Kconfig +++ b/arch/riscv/Kconfig @@ -29,7 +29,7 @@ config RISCV select ARCH_HAS_CURRENT_STACK_POINTER select ARCH_HAS_DEBUG_VIRTUAL if MMU select ARCH_HAS_DEBUG_VM_PGTABLE - select ARCH_HAS_DEBUG_WX + select ARCH_HAS_CHECK_WX select ARCH_HAS_DELAY_TIMER select ARCH_HAS_ELF_CORE_EFLAGS if BINFMT_ELF && ELF_CORE select ARCH_HAS_FAST_MULTIPLIER diff --git a/arch/s390/Kconfig b/arch/s390/Kconfig index a34376c05f6e3c..4bbb2c89bce057 100644 --- a/arch/s390/Kconfig +++ b/arch/s390/Kconfig @@ -92,7 +92,7 @@ config S390 select ARCH_HAS_CURRENT_STACK_POINTER select ARCH_HAS_DEBUG_VIRTUAL select ARCH_HAS_DEBUG_VM_PGTABLE - select ARCH_HAS_DEBUG_WX + select ARCH_HAS_CHECK_WX select ARCH_HAS_DEVMEM_IS_ALLOWED select ARCH_HAS_DMA_OPS if PCI select ARCH_HAS_ELF_RANDOMIZE diff --git a/arch/s390/configs/debug_defconfig b/arch/s390/configs/debug_defconfig index 3dae7147433305..68d53c0bc8dbe5 100644 --- a/arch/s390/configs/debug_defconfig +++ b/arch/s390/configs/debug_defconfig @@ -841,7 +841,7 @@ CONFIG_DEBUG_PAGEALLOC=y CONFIG_SLUB_DEBUG_ON=y CONFIG_PAGE_OWNER=y CONFIG_DEBUG_RODATA_TEST=y -CONFIG_DEBUG_WX=y +CONFIG_CHECK_WX=y CONFIG_PTDUMP_DEBUGFS=y CONFIG_DEBUG_OBJECTS=y CONFIG_DEBUG_OBJECTS_SELFTEST=y diff --git a/arch/s390/configs/defconfig b/arch/s390/configs/defconfig index 6f5722634b4d39..8e5cfc69512119 100644 --- a/arch/s390/configs/defconfig +++ b/arch/s390/configs/defconfig @@ -820,7 +820,7 @@ CONFIG_DEBUG_INFO_DWARF4=y CONFIG_GDB_SCRIPTS=y CONFIG_DEBUG_SECTION_MISMATCH=y CONFIG_MAGIC_SYSRQ=y -CONFIG_DEBUG_WX=y +CONFIG_CHECK_WX=y CONFIG_PTDUMP_DEBUGFS=y CONFIG_DEBUG_MEMORY_INIT=y CONFIG_PANIC_ON_OOPS=y diff --git a/arch/s390/mm/dump_pagetables.c b/arch/s390/mm/dump_pagetables.c index 89badbe72ae706..a23a0bd4d8a885 100644 --- a/arch/s390/mm/dump_pagetables.c +++ b/arch/s390/mm/dump_pagetables.c @@ -86,7 +86,7 @@ static void note_prot_wx(struct pg_state *st, unsigned long addr) */ if (addr == PAGE_SIZE && (nospec_uses_trampoline() || !cpu_has_bear())) return; - WARN_ONCE(IS_ENABLED(CONFIG_DEBUG_WX), + WARN_ONCE(IS_ENABLED(CONFIG_CHECK_WX), "s390/mm: Found insecure W+X mapping at address %pS\n", (void *)st->start_address); st->wx_pages += (addr - st->start_address) / PAGE_SIZE; diff --git a/arch/x86/Kconfig b/arch/x86/Kconfig index 6e5e462ec059a1..6cb70e47d0bda2 100644 --- a/arch/x86/Kconfig +++ b/arch/x86/Kconfig @@ -109,7 +109,7 @@ config X86 select ARCH_HAS_SYNC_CORE_BEFORE_USERMODE select ARCH_HAS_SYSCALL_WRAPPER select ARCH_HAS_UBSAN - select ARCH_HAS_DEBUG_WX + select ARCH_HAS_CHECK_WX select ARCH_HAS_ZONE_DMA_SET if EXPERT select ARCH_HAVE_NMI_SAFE_CMPXCHG select ARCH_HAVE_EXTRA_ELF_NOTES diff --git a/arch/x86/configs/x86_64_defconfig b/arch/x86/configs/x86_64_defconfig index 269f7d808be4ec..e6896aeb77d81d 100644 --- a/arch/x86/configs/x86_64_defconfig +++ b/arch/x86/configs/x86_64_defconfig @@ -263,7 +263,7 @@ CONFIG_SECURITY_SELINUX_BOOTPARAM=y CONFIG_PRINTK_TIME=y CONFIG_DEBUG_KERNEL=y CONFIG_MAGIC_SYSRQ=y -CONFIG_DEBUG_WX=y +CONFIG_CHECK_WX=y CONFIG_DEBUG_STACK_USAGE=y CONFIG_SCHEDSTATS=y CONFIG_BLK_DEV_IO_TRACE=y diff --git a/arch/x86/include/asm/pgtable.h b/arch/x86/include/asm/pgtable.h index d551120a7c889b..ef0252a09c2804 100644 --- a/arch/x86/include/asm/pgtable.h +++ b/arch/x86/include/asm/pgtable.h @@ -41,10 +41,10 @@ void ptdump_walk_user_pgd_level_checkwx(void); #define pgprot_encrypted(prot) __pgprot(cc_mkenc(pgprot_val(prot))) #define pgprot_decrypted(prot) __pgprot(cc_mkdec(pgprot_val(prot))) -#ifdef CONFIG_DEBUG_WX -#define debug_checkwx_user() ptdump_walk_user_pgd_level_checkwx() +#ifdef CONFIG_CHECK_WX +#define pgtable_checkwx_user() ptdump_walk_user_pgd_level_checkwx() #else -#define debug_checkwx_user() do { } while (0) +#define pgtable_checkwx_user() do { } while (0) #endif extern spinlock_t pgd_lock; diff --git a/arch/x86/mm/pti.c b/arch/x86/mm/pti.c index 598f553cc8713c..31055ee6f1de8e 100644 --- a/arch/x86/mm/pti.c +++ b/arch/x86/mm/pti.c @@ -688,5 +688,5 @@ void pti_finalize(void) pti_clone_entry_text(true); pti_clone_kernel_text(); - debug_checkwx_user(); + pgtable_checkwx_user(); } diff --git a/include/linux/ptdump.h b/include/linux/ptdump.h index 240bd3bff18dd2..af18d1459b2f49 100644 --- a/include/linux/ptdump.h +++ b/include/linux/ptdump.h @@ -31,9 +31,9 @@ bool ptdump_walk_pgd_level_core(struct seq_file *m, void ptdump_walk_pgd(struct ptdump_state *st, struct mm_struct *mm, pgd_t *pgd); bool ptdump_check_wx(void); -static inline void debug_checkwx(void) +static inline void pgtable_checkwx(void) { - if (IS_ENABLED(CONFIG_DEBUG_WX)) + if (IS_ENABLED(CONFIG_CHECK_WX)) ptdump_check_wx(); } diff --git a/init/main.c b/init/main.c index 31f2bf54976ab9..a87a3d52f3e200 100644 --- a/init/main.c +++ b/init/main.c @@ -1535,7 +1535,7 @@ static void mark_readonly(void) flush_module_init_free_work(); jump_label_init_ro(); mark_rodata_ro(); - debug_checkwx(); + pgtable_checkwx(); rodata_test(); } else if (IS_ENABLED(CONFIG_STRICT_KERNEL_RWX)) { pr_info("Kernel memory protection disabled.\n"); diff --git a/kernel/configs/debug.config b/kernel/configs/debug.config index 307c97ac5fa9c3..ac878669c19365 100644 --- a/kernel/configs/debug.config +++ b/kernel/configs/debug.config @@ -50,7 +50,7 @@ CONFIG_DEBUG_NET=y # CONFIG_DEBUG_PAGEALLOC is not set # CONFIG_DEBUG_KMEMLEAK_DEFAULT_OFF is not set # CONFIG_DEBUG_RODATA_TEST is not set -# CONFIG_DEBUG_WX is not set +# CONFIG_CHECK_WX is not set # CONFIG_KFENCE is not set # CONFIG_PAGE_POISONING is not set # CONFIG_SLUB_STATS is not set diff --git a/mm/Kconfig b/mm/Kconfig index edb4a6c0a87021..acefc994d9a8b9 100644 --- a/mm/Kconfig +++ b/mm/Kconfig @@ -1494,6 +1494,47 @@ config LAZY_MMU_MODE_KUNIT_TEST If unsure, say N. +config ARCH_HAS_CHECK_WX + bool + +config CHECK_WX + bool "Warn on W+X mappings at boot" + default STRICT_KERNEL_RWX + depends on ARCH_HAS_CHECK_WX + depends on ARCH_HAS_PTDUMP + depends on MMU + select PTDUMP + help + Generate a warning if any W+X mappings are found at boot. + + This is useful for discovering cases where the kernel is leaving W+X + mappings after applying NX, as such mappings are a security risk. + + Look for a message in dmesg output like this: + + /mm: Checked W+X mappings: passed, no W+X pages found. + + or like this, if the check failed: + + /mm: Checked W+X mappings: failed, W+X pages found. + + Note that even if the check fails, your kernel is possibly + still fine, as W+X mappings are not a security hole in + themselves, what they do is that they make the exploitation + of other unfixed kernel bugs easier. + + There is no runtime or memory usage effect of this option + once the kernel has booted up - it's a one time check. + + If in doubt, say "Y". + +config ARCH_HAS_PTDUMP + bool + +config PTDUMP + bool + + source "mm/damon/Kconfig" endmenu diff --git a/mm/Kconfig.debug b/mm/Kconfig.debug index 9eaa25d1cf2340..9ae75229c76da1 100644 --- a/mm/Kconfig.debug +++ b/mm/Kconfig.debug @@ -180,45 +180,6 @@ config DEBUG_RODATA_TEST help This option enables a testcase for the setting rodata read-only. -config ARCH_HAS_DEBUG_WX - bool - -config DEBUG_WX - bool "Warn on W+X mappings at boot" - depends on ARCH_HAS_DEBUG_WX - depends on ARCH_HAS_PTDUMP - depends on MMU - select PTDUMP - help - Generate a warning if any W+X mappings are found at boot. - - This is useful for discovering cases where the kernel is leaving W+X - mappings after applying NX, as such mappings are a security risk. - - Look for a message in dmesg output like this: - - /mm: Checked W+X mappings: passed, no W+X pages found. - - or like this, if the check failed: - - /mm: Checked W+X mappings: failed, W+X pages found. - - Note that even if the check fails, your kernel is possibly - still fine, as W+X mappings are not a security hole in - themselves, what they do is that they make the exploitation - of other unfixed kernel bugs easier. - - There is no runtime or memory usage effect of this option - once the kernel has booted up - it's a one time check. - - If in doubt, say "Y". - -config ARCH_HAS_PTDUMP - bool - -config PTDUMP - bool - config PTDUMP_DEBUGFS bool "Export kernel pagetable layout to userspace via debugfs" depends on DEBUG_KERNEL From e5655ba605a55b18f960b7e4b86a2980b1bbb96c Mon Sep 17 00:00:00 2001 From: Palla Raghunath Date: Fri, 25 Sep 2026 21:54:49 +0100 Subject: [PATCH 1170/1352] mm/vmalloc: do not warn on -ENOMEM from va_clip() in pcpu_get_vm_areas() When pcpu_get_vm_areas() has to split a free vmap_area in the middle (NE_FIT_TYPE), va_clip() needs an extra vmap_area object. It takes the per-cpu ne_fit_preload_node if one is there, and otherwise falls back to kmem_cache_alloc(GFP_NOWAIT), which may fail and return -ENOMEM. pcpu_get_vm_areas() never preloads, and a single call can do more than one such split: on a NUMA system it places one area per node group, so the first split consumes the preloaded object and the next one depends on the GFP_NOWAIT allocation. That failure is expected and already handled: the recovery path returns the areas clipped so far to the free tree, purges lazily freed areas and retries. But the error is checked with WARN_ON_ONCE(), so a transient allocation failure under memory pressure or fault injection triggers a kernel warning, and a panic with panic_on_warn. syzbot hit this on a two-node VM while creating a per-cpu BPF array map. Keep the WARN_ON_ONCE() for errors other than -ENOMEM, which do indicate a bug, and take the recovery path either way. This matches what commit b9183788a2de ("mm/vmalloc: do not warn on -ENOMEM from va_alloc()") did for the other va_clip() caller. Link: https://lore.kernel.org/20260925205450.21262-1-raghunathpalla.0209@gmail.com Fixes: 1b23ff80b399 ("mm/vmalloc: invoke classify_va_fit_type() in adjust_va_to_fit_type()") Signed-off-by: Palla Raghunath Signed-off-by: Andrew Morton Reported-by: Closes: https://syzkaller.appspot.com/bug?extid=442828bb356b10813a47 Reviewed-by: Andrew Morton Reviewed-by: Uladzislau Rezki (Sony) Reviewed-by: Dev Jain Cc: Shuah Khan Cc: Brigham Campbell Cc: Baoquan He --- mm/vmalloc.c | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index b24896aedac212..4e4cb8d785bdb4 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -5135,9 +5135,14 @@ struct vm_struct **pcpu_get_vm_areas(const unsigned long *offsets, ret = va_clip(&free_vmap_area_root, &free_vmap_area_list, va, start, size); - if (WARN_ON_ONCE(unlikely(ret))) - /* It is a BUG(), but trigger recovery instead. */ + if (unlikely(ret)) { + /* + * -ENOMEM from the GFP_NOWAIT fallback is expected. + * Anything else is a BUG(), but trigger recovery instead. + */ + WARN_ON_ONCE(ret != -ENOMEM); goto recovery; + } /* Allocated area. */ va = vas[area]; From f0fa1e9609ef41bd8c6b8a6480dc3466bd4bb405 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Wed, 30 Sep 2026 18:53:36 +0100 Subject: [PATCH 1171/1352] mm/vma: don't remove VMA from rmap if pgoff unchanged When updating a VMA, vma_prepare() unconditionally removes it from its rmap interval trees under the rmap lock, and vma_complete() reinserts it before releasing the lock. This is wholly unnecessary if its page offset (file rmap) or anonymous page offset (anon rmap) is unchanged. So, track whether they will change in the newly introduced vp->file_pgoff_unchanged and vp->anon_pgoff_unchanged fields, and use them to determine whether to remove the VMA or not. The rmap lock keeps things safe as no rmap walks can concurrently occur during the operation. Additionally, some architectures (arm, parisc, nios2, csky) have dcache flush rmap walkers which take only flush_dcache_mmap_lock(), which is likewise held across the operation. If the VMA remains in the tree, it's necessary to keep the augmented rb_subtree_last field updated to reflect its changed range. Provide mapping_rmap_tree_[pre, post]_update() and anon_rmap_tree_[pre, post]_update_vma() (replacing the existing logic in the anonymous case) to handle both the changed and unchanged cases. For the anon rmap case, with CONFIG_DEBUG_VM_RB set, avc->cached_vma_last is also updated when propagating in place. When performing a VMA shrink or a split where the VMA is the lower one, the page offset cannot change, so set the flags unconditionally in these cases. When merging VMAs the page offset is unchanged only in some cases, so update init_multi_vma_prep() to set the flags only if the page offsets remain the same. Finally, while we're here, also update expand_upwards() similarly. These changes ultimately result in less rmap lock contention. Pan Deng reported results using the UnixBench/excel benchmark on a 2-socket 192 core, 384 thread x86-64 system for v7.3-rc4 with/without the patch applied: Execl Throughput, index score: avg %stdev min max v7.3-rc4 3511.5 0.44% 3494.5 3543.4 + patch 4069.0 0.48% 4047.4 4109.5 (+15.9%) Average wait on file rmap lock in ms, 5 runs per kernel: avg %stdev min max v7.3-rc4 9.470 2.91% 9.070 9.820 + patch 8.420 3.34% 8.100 8.770 (-11.1%) Profiling data obtained during the operation highlighted the file rmap lock as the primary source of contention. Link: https://lore.kernel.org/20260930-speed-up-inplace-rmap-v2-1-ac1aa19708aa@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Suggested-by: Pan Deng Reviewed-by: Rik van Riel Acked-by: Lance Yang Tested-by: Lance Yang Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Harry Yoo Cc: Jann Horn Cc: Pedro Falcato --- include/linux/mm.h | 10 +++ mm/interval_tree.c | 101 ++++++++++++++++++++++++++++++ mm/vma.c | 74 +++++++++------------- mm/vma.h | 2 + tools/testing/vma/include/stubs.h | 22 +++++++ 5 files changed, 164 insertions(+), 45 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index 5e35864eb731c5..e21244ff29117a 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -4358,6 +4358,12 @@ void mapping_rmap_tree_insert_after(struct vm_area_struct *vma, struct address_space *mapping); void mapping_rmap_tree_remove(struct vm_area_struct *vma, struct address_space *mapping); +void mapping_rmap_tree_pre_update(struct vm_area_struct *vma, + struct address_space *mapping, + bool pgoff_unchanged); +void mapping_rmap_tree_post_update(struct vm_area_struct *vma, + struct address_space *mapping, + bool pgoff_unchanged); struct vm_area_struct * mapping_rmap_tree_iter_first(struct address_space *mapping, pgoff_t pgoff_start, pgoff_t pgoff_last); @@ -4375,6 +4381,10 @@ void anon_rmap_tree_insert(struct anon_vma_chain *avc, struct anon_vma *anon_vma); void anon_rmap_tree_remove(struct anon_vma_chain *avc, struct anon_vma *anon_vma); +void anon_rmap_tree_pre_update_vma(struct vm_area_struct *vma, + bool anon_pgoff_unchanged); +void anon_rmap_tree_post_update_vma(struct vm_area_struct *vma, + bool anon_pgoff_unchanged); struct anon_vma_chain * anon_rmap_tree_iter_first(struct anon_vma *anon_vma, pgoff_t pgoff_start, pgoff_t pgoff_last); diff --git a/mm/interval_tree.c b/mm/interval_tree.c index 7bbbf15cfbf0c7..9e24bb99fd80a3 100644 --- a/mm/interval_tree.c +++ b/mm/interval_tree.c @@ -64,6 +64,51 @@ void mapping_rmap_tree_remove(struct vm_area_struct *vma, __mapping_rmap_tree_remove(vma, &mapping->i_mmap); } +static void mapping_rmap_tree_update_inplace(struct vm_area_struct *vma) +{ + /* Propagate all the way up the tree. */ + __mapping_rmap_tree_augment.propagate(&vma->shared.rb, NULL); +} + +/** + * mapping_rmap_tree_pre_update() - Prepare the file rmap tree for a change to + * be made to @vma. + * @vma: The VMA about to be updated. + * @mapping: The file rmap to which @vma belongs. + * @pgoff_unchanged: Whether @vma's page offset will remain unchanged. + * + * The file rmap lock must be held across the entire update. + */ +void mapping_rmap_tree_pre_update(struct vm_area_struct *vma, + struct address_space *mapping, + bool pgoff_unchanged) +{ + /* If the pgoff has changed, then remove and reinsert afterwards. */ + if (!pgoff_unchanged) + mapping_rmap_tree_remove(vma, mapping); +} + +/** + * mapping_rmap_tree_post_update() - Update the file rmap tree to reflect a + * change that has been made to @vma. + * @vma: The VMA that has been updated. + * @mapping: The file rmap to which @vma belongs. + * @pgoff_unchanged: Whether @vma's page offset remained unchanged. + * + * mapping_rmap_tree_pre_update() must have been called prior to this. + * + * The file rmap lock must be held across the entire update. + */ +void mapping_rmap_tree_post_update(struct vm_area_struct *vma, + struct address_space *mapping, + bool pgoff_unchanged) +{ + if (pgoff_unchanged) + mapping_rmap_tree_update_inplace(vma); + else + mapping_rmap_tree_insert(vma, mapping); +} + struct vm_area_struct * mapping_rmap_tree_iter_first(struct address_space *mapping, pgoff_t pgoff_start, pgoff_t pgoff_last) @@ -111,6 +156,62 @@ void anon_rmap_tree_remove(struct anon_vma_chain *avc, __anon_rmap_tree_remove(avc, &anon_vma->rb_root); } +static void anon_rmap_tree_update_inplace(struct anon_vma_chain *avc) +{ +#ifdef CONFIG_DEBUG_VM_RB + avc->cached_vma_last = avc_last_pgoff(avc); +#endif + /* Propagate all the way up the tree. */ + __anon_rmap_tree_augment.propagate(&avc->rb, NULL); +} + +/** + * anon_rmap_tree_pre_update_vma() - Prepare the anon rmap trees for a change + * to be made to @vma. + * @vma: The VMA about to be updated, which has an anon rmap assigned and is + * already inserted on its interval trees. + * @anon_pgoff_unchanged: Whether @vma's anonymous page offset will remain + * unchanged. + * + * The anon rmap lock must be held across the entire update. + */ +void anon_rmap_tree_pre_update_vma(struct vm_area_struct *vma, + bool anon_pgoff_unchanged) +{ + struct anon_vma_chain *avc; + + if (anon_pgoff_unchanged) + return; + + /* If the pgoff has changed, then remove and reinsert afterwards. */ + list_for_each_entry(avc, &vma->anon_vma_chain, same_vma) + anon_rmap_tree_remove(avc, avc->anon_vma); +} + +/** + * anon_rmap_tree_post_update_vma() - Update the anon rmap trees to reflect a + * change that has been made to @vma. + * @vma: The VMA that has been updated. + * @anon_pgoff_unchanged: Whether @vma's anonymous page offset remained + * unchanged. + * + * anon_rmap_tree_pre_update_vma() must have been called prior to this. + * + * The anon rmap lock must be held across the entire update. + */ +void anon_rmap_tree_post_update_vma(struct vm_area_struct *vma, + bool anon_pgoff_unchanged) +{ + struct anon_vma_chain *avc; + + list_for_each_entry(avc, &vma->anon_vma_chain, same_vma) { + if (anon_pgoff_unchanged) + anon_rmap_tree_update_inplace(avc); + else + anon_rmap_tree_insert(avc, avc->anon_vma); + } +} + struct anon_vma_chain * anon_rmap_tree_iter_first(struct anon_vma *anon_vma, pgoff_t pgoff_start, pgoff_t pgoff_last) diff --git a/mm/vma.c b/mm/vma.c index 42e4e674f783ff..bd75b6c7499364 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -201,8 +201,15 @@ static void init_multi_vma_prep(struct vma_prepare *vp, if (vp->file) vp->mapping = vma->vm_file->f_mapping; - if (vmg && vmg->skip_vma_uprobe) + if (!vmg) + return; + + if (vmg->skip_vma_uprobe) vp->skip_vma_uprobe = true; + if (vma_start_pgoff(vma) == vmg_start_pgoff(vmg)) + vp->file_pgoff_unchanged = true; + if (vma_start_anon_pgoff(vma) == vmg_start_anon_pgoff(vmg)) + vp->anon_pgoff_unchanged = true; } /* @@ -299,38 +306,6 @@ static void __remove_shared_vm_struct(struct vm_area_struct *vma, flush_dcache_mmap_unlock(mapping); } -/* - * vma has an anon rmap assigned, and is already inserted on its interval - * trees. - * - * Before updating the vma's vm_start / vm_end / vm_pgoff fields, the - * vma must be removed from the anon rmap's interval trees using - * anon_rmap_tree_pre_update_vma(). - * - * After the update, the vma will be reinserted using - * anon_rmap_tree_post_update_vma(). - * - * The entire update must be protected by exclusive mmap_lock and by - * the anon rmap root lock. - */ -static void -anon_rmap_tree_pre_update_vma(struct vm_area_struct *vma) -{ - struct anon_vma_chain *avc; - - list_for_each_entry(avc, &vma->anon_vma_chain, same_vma) - anon_rmap_tree_remove(avc, avc->anon_vma); -} - -static void -anon_rmap_tree_post_update_vma(struct vm_area_struct *vma) -{ - struct anon_vma_chain *avc; - - list_for_each_entry(avc, &vma->anon_vma_chain, same_vma) - anon_rmap_tree_insert(avc, avc->anon_vma); -} - /* * vma_prepare() - Helper function for handling locking VMAs prior to altering * @vp: The initialized vma_prepare struct @@ -359,16 +334,19 @@ static void vma_prepare(struct vma_prepare *vp) if (vp->anon_vma) { anon_vma_lock_write(vp->anon_vma); - anon_rmap_tree_pre_update_vma(vp->vma); + anon_rmap_tree_pre_update_vma(vp->vma, vp->anon_pgoff_unchanged); + /* The adjacent VMA's start is moved, so its page offset changes. */ if (vp->adj_next) - anon_rmap_tree_pre_update_vma(vp->adj_next); + anon_rmap_tree_pre_update_vma(vp->adj_next, false); } if (vp->file) { flush_dcache_mmap_lock(vp->mapping); - mapping_rmap_tree_remove(vp->vma, vp->mapping); + mapping_rmap_tree_pre_update(vp->vma, vp->mapping, + vp->file_pgoff_unchanged); if (vp->adj_next) - mapping_rmap_tree_remove(vp->adj_next, vp->mapping); + mapping_rmap_tree_pre_update(vp->adj_next, vp->mapping, + false); } } @@ -386,8 +364,10 @@ static void vma_complete(struct vma_prepare *vp, struct vma_iterator *vmi, { if (vp->file) { if (vp->adj_next) - mapping_rmap_tree_insert(vp->adj_next, vp->mapping); - mapping_rmap_tree_insert(vp->vma, vp->mapping); + mapping_rmap_tree_post_update(vp->adj_next, vp->mapping, + false); + mapping_rmap_tree_post_update(vp->vma, vp->mapping, + vp->file_pgoff_unchanged); flush_dcache_mmap_unlock(vp->mapping); } @@ -406,9 +386,9 @@ static void vma_complete(struct vma_prepare *vp, struct vma_iterator *vmi, } if (vp->anon_vma) { - anon_rmap_tree_post_update_vma(vp->vma); + anon_rmap_tree_post_update_vma(vp->vma, vp->anon_pgoff_unchanged); if (vp->adj_next) - anon_rmap_tree_post_update_vma(vp->adj_next); + anon_rmap_tree_post_update_vma(vp->adj_next, false); anon_vma_unlock_write(vp->anon_vma); } @@ -593,6 +573,8 @@ __split_vma(struct vma_iterator *vmi, struct vm_area_struct *vma, init_vma_prep(&vp, vma); vp.insert = new; + vp.file_pgoff_unchanged = !new_below; + vp.anon_pgoff_unchanged = !new_below; vma_prepare(&vp); /* @@ -1346,6 +1328,8 @@ int vma_shrink(struct vma_iterator *vmi, struct vm_area_struct *vma, vma_start_write(vma); init_vma_prep(&vp, vma); + vp.file_pgoff_unchanged = true; + vp.anon_pgoff_unchanged = true; vma_prepare(&vp); vma_adjust_trans_huge(vma, vma->vm_start, end, NULL); @@ -3453,11 +3437,11 @@ int expand_upwards(struct vm_area_struct *vma, unsigned long address) if (vma_test(vma, VMA_LOCKED_BIT)) mm->locked_vm += grow; vm_stat_account(mm, vma->vm_flags, grow); - anon_rmap_tree_pre_update_vma(vma); + anon_rmap_tree_pre_update_vma(vma, true); vma->vm_end = address; /* Overwrite old entry in mtree. */ vma_iter_store_overwrite(&vmi, vma); - anon_rmap_tree_post_update_vma(vma); + anon_rmap_tree_post_update_vma(vma, true); perf_event_mmap(vma); } @@ -3530,12 +3514,12 @@ int expand_downwards(struct vm_area_struct *vma, unsigned long address) if (vma_test(vma, VMA_LOCKED_BIT)) mm->locked_vm += grow; vm_stat_account(mm, vma->vm_flags, grow); - anon_rmap_tree_pre_update_vma(vma); + anon_rmap_tree_pre_update_vma(vma, false); vma->vm_start = address; vma_sub_pgoff(vma, grow); /* Overwrite old entry in mtree. */ vma_iter_store_overwrite(&vmi, vma); - anon_rmap_tree_post_update_vma(vma); + anon_rmap_tree_post_update_vma(vma, false); perf_event_mmap(vma); } diff --git a/mm/vma.h b/mm/vma.h index 7a683272c0a82a..074419f2c102d5 100644 --- a/mm/vma.h +++ b/mm/vma.h @@ -28,6 +28,8 @@ struct vma_prepare { struct vm_area_struct *remove2; bool skip_vma_uprobe :1; + bool file_pgoff_unchanged :1; + bool anon_pgoff_unchanged :1; }; struct unlink_vma_file_batch { diff --git a/tools/testing/vma/include/stubs.h b/tools/testing/vma/include/stubs.h index e4acc6f1fe7bab..5d4944581b8d20 100644 --- a/tools/testing/vma/include/stubs.h +++ b/tools/testing/vma/include/stubs.h @@ -267,6 +267,18 @@ static inline void mapping_rmap_tree_remove(struct vm_area_struct *vma, { } +static inline void mapping_rmap_tree_pre_update(struct vm_area_struct *vma, + struct address_space *mapping, + bool pgoff_unchanged) +{ +} + +static inline void mapping_rmap_tree_post_update(struct vm_area_struct *vma, + struct address_space *mapping, + bool pgoff_unchanged) +{ +} + static inline void flush_dcache_mmap_unlock(struct address_space *mapping) { } @@ -281,6 +293,16 @@ static inline void anon_rmap_tree_remove(struct anon_vma_chain *avc, { } +static inline void anon_rmap_tree_pre_update_vma(struct vm_area_struct *vma, + bool anon_pgoff_unchanged) +{ +} + +static inline void anon_rmap_tree_post_update_vma(struct vm_area_struct *vma, + bool anon_pgoff_unchanged) +{ +} + static inline void uprobe_mmap(struct vm_area_struct *vma) { } From cb21fc154bb6e11e2e8d908d73cad0c82a661d64 Mon Sep 17 00:00:00 2001 From: Zhang Yi Date: Mon, 28 Sep 2026 20:08:30 +0800 Subject: [PATCH 1172/1352] mm/truncate: align truncation boundaries to mapping minimum folio order Patch series "mm/truncate: fix data loss when truncating straddling large folios", v5. This is the fifth version fixing data loss when truncating straddling large folios caught on the upcomming ext4 + iomap buffered I/O conversion. When truncate_inode_pages_range() punches a hole or truncates a file, truncate_inode_partial_folio() splits a large folio so that the caller can drop the in-range sub-folios while keeping the out-of-range tail intact. This series fixes three distinct problems in that path that can each lose the valid out-of-range tail of a straddling folio, plus a follow-up that clarifies the return value semantics. Patch 01 aligns the truncation boundaries inwards to the mapping minimum folio order in truncate_inode_pages_range(). With a non-zero min_order, folio_split() stops at min_order instead of order 0, so a boundary computed at page granularity can land inside a min-order-aligned sub-folio and the truncate loop drops that whole chunk, valid tail included, causing data loss. Patch 02 looks the end-edge straddler up by its page index through __filemap_get_folio() in truncate_inode_partial_folio(). After the first split the straddler is unlocked and only transiently ref'd in the page cache, so the page pointer derived from the original folio can be freed and reallocated as a different folio in the same mapping, and the mapping check cannot catch it, which may cause incorrect splitting and potential data loss. Patch 03 reworks the contract between truncate_inode_partial_folio() and its callers. If the second split of the straddler fails, the function reported success unconditionally, and the leftover incorrect end position could cause the truncate loop to drop that valid tail. After rework, it tells the caller the exact page range safe to discard via new pstart/pend out-parameters, so the truncate loop never touches a straddling folio that still holds valid out-of-range data. Patch 04 clarifies the return value semantics to "at least one split succeeded", which is all the shmem caller needs to decide whether to reset its scan loop. The second patch fixes a pre-existing race issue that is reachable today, so it is Cc'd to stable. Patches 01 and 03 require a dirty large folio that carries no filesystem private data, so they are not reachable on current filesystems. They were found while developing the upcoming ext4 iomap buffered I/O path. [1] In addition, another two pre-existing issues were found during the development of this series: - On 32-bit systems with a non-zero min_order filesystem, truncating or any operation operation involving folio_next_index() on the folio containing MAX_LFS_FILESIZE can cause the index calculation to overflow, posing an unpredictable risk. [2] - In shmem, at the last call site of truncate_inode_partial_folio() in shmem_undo_range(), the restart-loop check is wrong, which leaves split sub-folios behind. [3] These two issues need to be fixed separately. This patch (of 4): When the mapping has a non-zero minimum folio order (min_order), folio_split() in truncate_inode_partial_folio() stops at min_order instead of order 0, so the sub-folio containing a split point stays aligned to 1 << min_order rather than to a single page. The original boundaries in truncate_inode_pages_range() were based on page granularity, so either boundary could land inside the min_order chunk at its edge, and the truncation loop would drop that whole chunk, valid out-of-range tail included. For example, a 64K (order-4) folio with min_order = 2 (16K) punched from offset 0 to 36K: split @p0 -> [p0-p3, p4-p7, p8-p15] # non-uniform, min_order folio2 = p8-p15 # straddles: p8 in range, p9-p15 tail valid 2nd split of folio2 -> [p8-p11, p12-p15] # success end(old) = p9 # BUG: p9 inside [p8-p11] loop truncates ... p8-p11 # p9-p11's valid tail is lost It has gone unnoticed so far for two reasons. A non-zero min_order is only used by filesystems with a block or sector size larger than the page size, and those either always write back the affected range before punching a hole or truncating, or they carry filesystem private data on dirty folios (e.g. buffer_head), which makes filemap_release_folio() fail and folio_split() abort with -EBUSY, so the folio is never split and the old start/end boundaries remain valid. The bug only becomes reachable on paths that truncate dirty large folios without prior writeback and without filesystem private data, such as the upcoming ext4 iomap buffered I/O path. Align both start (rounded up) and end (rounded down) to the mapping minimum folio order so they always fall on a folio boundary. Link: https://lore.kernel.org/20260928120833.3440834-2-yi.zhang@huaweicloud.com Link: https://lore.kernel.org/linux-fsdevel/a638a8fb-c184-4069-ae33-379ec12cd514@huaweicloud.com/ [1] Link: https://lore.kernel.org/linux-mm/5a454f2a-8ae2-491d-b903-750c945cfb9d@huaweicloud.com/ [2] Link: https://lore.kernel.org/linux-mm/5pthbyxtn7q6xi4fmkofvksmcjzfnujcw2g4fxmxjzfin5pbgf@zui3vcimb4cv/ [3] Fixes: e220917fa5077 ("mm: split a folio in minimum folio order chunks") Signed-off-by: Zhang Yi Signed-off-by: Andrew Morton Reported-by: Joanne Koong Closes: https://lore.kernel.org/linux-mm/CAJnrk1bQYUe6+1ryyJur5EEnZYrC+_5AYsy=OWzVRgD4202y1g@mail.gmail.com/ Suggested-by: Zi Yan Reviewed-by: Jan Kara Reviewed-by: Brian Foster Reviewed-by: Zi Yan Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Hugh Dickins Cc: Baolin Wang Cc: Matthew Wilcox (Oracle) Cc: Darrick J. Wong Cc: Cc: Yang Erkun Cc: Zhihao Cheng Cc: Kefeng Wang Cc: --- mm/truncate.c | 18 ++++++++++++------ 1 file changed, 12 insertions(+), 6 deletions(-) diff --git a/mm/truncate.c b/mm/truncate.c index b58ba940be4740..8a28f4a2126776 100644 --- a/mm/truncate.c +++ b/mm/truncate.c @@ -345,9 +345,11 @@ long mapping_evict_folio(struct address_space *mapping, struct folio *folio) * @lstart: offset from which to truncate * @lend: offset to which to truncate (inclusive) * - * Truncate the page cache, removing the pages that are between - * specified offsets (and zeroing out partial pages - * if lstart or lend + 1 is not page aligned). + * Truncate the page cache, removing the folios that are between specified + * offsets (and zeroing out partial folios if lstart or lend + 1 is not + * folio aligned). For mappings with a non-zero minimum folio order, the + * boundaries are aligned inwards to 1 << min_order so the edge sub-folio + * straddling the range is kept. * * Truncate takes two passes - the first pass is nonblocking. It will not * block on page locks and it will not block on writeback. The second pass @@ -366,6 +368,7 @@ long mapping_evict_folio(struct address_space *mapping, struct folio *folio) void truncate_inode_pages_range(struct address_space *mapping, loff_t lstart, uoff_t lend) { + pgoff_t min_nrpages = mapping_min_folio_nrpages(mapping); pgoff_t start; /* inclusive */ pgoff_t end; /* exclusive */ struct folio_batch fbatch; @@ -379,9 +382,8 @@ void truncate_inode_pages_range(struct address_space *mapping, return; /* - * 'start' and 'end' always covers the range of pages to be fully - * truncated. Partial pages are covered with 'partial_start' at the - * start of the range and 'partial_end' at the end of the range. + * 'start' and 'end' always covers the range of folios to be fully + * truncated, with both boundaries aligned inwards to 1 << min_order. * Note that 'end' is exclusive while 'lend' is inclusive. */ start = (lstart + PAGE_SIZE - 1) >> PAGE_SHIFT; @@ -395,6 +397,10 @@ void truncate_inode_pages_range(struct address_space *mapping, else end = (lend + 1) >> PAGE_SHIFT; + start = round_up(start, min_nrpages); + if (end != (pgoff_t)-1) + end = round_down(end, min_nrpages); + folio_batch_init(&fbatch); index = start; while (index < end && find_lock_entries(mapping, &index, end - 1, From afc693b0022666097c59ffd3e92ad630ca531e03 Mon Sep 17 00:00:00 2001 From: Zhang Yi Date: Mon, 28 Sep 2026 20:08:31 +0800 Subject: [PATCH 1173/1352] mm/truncate: look up the end-edge straddler by index In truncate_inode_partial_folio(), after the first split at the start edge, folio_split() unlocks and drops the refcount of the after-split sub-folios. The sub-folio straddling the end of the truncation range is therefore unlocked and only transiently ref'd in the page cache while the code still derives it from a page pointer inside the original folio. Between the first split finishing and page_folio() resolving split_at2, that tail page can be reclaimed, freed and reallocated as a new large folio in the same mapping at a different file offset. folio2 then points at a folio that does not cover the end boundary, yet folio2->mapping == folio->mapping still holds, so the stale pointer passes the mapping check and folio_split_or_unmap() splits a folio at a wrong position (or, with a transient refcount, a use-after-free window opens between try_get and the split). __folio_split()'s own folio != page_folio(split_at) check cannot catch this either since split_at2 has been reallocated as part of the new folio, so page_folio(split_at2) resolves back to folio2. Look the straddler up by its page index instead. __filemap_get_folio() returns the folio currently covering the boundary, ref'd and locked, with the mapping validated under the lock, so the split target is always the real folio at the end edge. Link: https://lore.kernel.org/20260928120833.3440834-3-yi.zhang@huaweicloud.com Link: https://lore.kernel.org/linux-mm/DLGXT0ERY79Z.3C5DYVJVX6S9Z@nvidia.com/ Fixes: 7460b470a131 ("mm/truncate: use folio_split() in truncate operation") Signed-off-by: Zhang Yi Signed-off-by: Andrew Morton Reported-by: Sashiko Closes: https://sashiko.dev/#/message/20260916094500.C30061F00893%40smtp.kernel.org Suggested-by: Jan Kara Suggested-by: Zi Yan Reviewed-by: Jan Kara Reviewed-by: Brian Foster Acked-by: Zi Yan Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Hugh Dickins Cc: Baolin Wang Cc: Matthew Wilcox (Oracle) Cc: Joanne Koong Cc: Darrick J. Wong Cc: Cc: Yang Erkun Cc: Zhihao Cheng Cc: Kefeng Wang Cc: Cc: --- mm/truncate.c | 30 ++++++++++++++++-------------- 1 file changed, 16 insertions(+), 14 deletions(-) diff --git a/mm/truncate.c b/mm/truncate.c index 8a28f4a2126776..f252619ae6c39b 100644 --- a/mm/truncate.c +++ b/mm/truncate.c @@ -259,30 +259,32 @@ bool truncate_inode_partial_folio(struct folio *folio, loff_t start, loff_t end) * for shmem truncate */ struct folio *folio2; + pgoff_t end_idx; if (offset + length == size) goto no_split; - split_at2 = folio_page(folio, - PAGE_ALIGN_DOWN(offset + length) / PAGE_SIZE); - folio2 = page_folio(split_at2); - - if (!folio_try_get(folio2)) + /* + * After the first split at the start edge, the folio at the + * end edge may be freed and reused concurrently. + * __filemap_get_folio() looks up the straddler at end_idx + * and returns it locked and ref'd with the mapping + * validated. + */ + end_idx = (pos + offset + length) >> PAGE_SHIFT; + folio2 = __filemap_get_folio(folio->mapping, end_idx, + FGP_LOCK | FGP_NOWAIT, 0); + if (IS_ERR(folio2)) goto no_split; + /* make sure folio2 is large */ if (!folio_test_large(folio2)) goto out; - if (!folio_trylock(folio2)) - goto out; - - /* make sure folio2 is large and does not change its mapping */ - if (folio_test_large(folio2) && - folio2->mapping == folio->mapping) - folio_split_or_unmap(folio2, split_at2, min_order); - - folio_unlock(folio2); + split_at2 = folio_page(folio2, (end_idx - folio2->index)); + folio_split_or_unmap(folio2, split_at2, min_order); out: + folio_unlock(folio2); folio_put(folio2); no_split: return true; From e6e935887e618da33b3af09112532a3c2be8bd0c Mon Sep 17 00:00:00 2001 From: Zhang Yi Date: Mon, 28 Sep 2026 20:08:32 +0800 Subject: [PATCH 1174/1352] mm/truncate: fix data loss when splitting straddling large folios fails truncate_inode_partial_folio() splits a large folio so that the caller's truncate loop can drop the in-range sub-folios while keeping the out-of-range tail. The first split at the punch start edge is non-uniform, which leaves the sub-folio at the truncation end edge as large as possible, this means it may still straddle the range, holding both zeroed in-range and valid out-of-range data. The function then attempts a second split at offset + length to isolate that tail. If the second split fails the straddling sub-folio stays merged. The function returned true unconditionally on all exit paths of the success block, telling the caller it was fully handled. The caller kept its default end and the truncate loop truncated every sub-folio below it, including the merged straddler, discarding the valid out-of-range tail. For example, a 4-page order-2 folio punched from offset 0 to the middle of the last page: truncate_inode_pages_range() truncate_inode_partial_folio() # same_folio == true 1st split at page0 -> [p0, p1, p2-3] # non-uniform, success folio2 = p2-3 # straddles: p2 zeroed, p3 tail valid 2nd split of folio2 fails / cannot lock return true # BUG: caller keeps default end end = 3 loop truncates p0, p1, p2-3 # p3's valid tail is lost This became reachable after commit 7460b470a131 ("mm/truncate: use folio_split() in truncate operation") replaced the atomic split_folio() with folio_split(), whose non-uniform split can partially split a folio and leave the end edge merged. It has gone unnoticed because a dirty large folio normally carries the filesystem's private data, for example buffer_head, so filemap_release_folio() fails on a dirty folio and folio_split() aborts with -EBUSY before any split, leaving the straddler safely unsplit. The bug is only reachable on paths that produce dirty large folios without filesystem private data, and it was caught on the upcoming ext4 iomap buffered I/O path when no ifs is attached. Rework the contract so the caller is told the folio range to discard: - Add pgoff_t *pstart and *pend out-parameters that receive the folio range fully covered by [lstart, lend] after any split (or none), aligned inwards to min_order, i.e. the folios wholly within the range and safe to discard. - Report a reliable end position to the caller. The straddler is looked up at an index aligned inwards to the mapping minimum folio order, and *pend is set to that boundary on success. If nothing covers the boundary, discarding up to it stays safe. If the straddler is locked by someone else, fall back to folio->index. This best-effort fallback may leave the in-range sub-folios to a later pass but never discards the out-of-range tail. If the straddler cannot be split, fall back to folio2->index so the caller keeps the out-of-range tail. - Rename the byte-range parameters start/end to lstart/lend to better express their semantics. Callers in truncate_inode_pages_range() and shmem_undo_range() pass &pstart for the folio at the start edge and &pend for the folio at the end edge, so the truncate loop drops exactly the fully covered pages and never touches a straddling folio that still holds valid out-of-range data. Link: https://lore.kernel.org/20260928120833.3440834-4-yi.zhang@huaweicloud.com Fixes: 7460b470a131 ("mm/truncate: use folio_split() in truncate operation") Signed-off-by: Zhang Yi Signed-off-by: Andrew Morton Suggested-by: Brian Foster Link: https://lore.kernel.org/linux-fsdevel/anH-WKA1coW6wtfG@bfoster/ Reviewed-by: Jan Kara Reviewed-by: Brian Foster Acked-by: Zi Yan Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Hugh Dickins Cc: Baolin Wang Cc: Matthew Wilcox (Oracle) Cc: Joanne Koong Cc: Darrick J. Wong Cc: Cc: Yang Erkun Cc: Zhihao Cheng Cc: Kefeng Wang Cc: --- mm/internal.h | 4 +-- mm/shmem.c | 13 +++---- mm/truncate.c | 93 +++++++++++++++++++++++++++++++++++---------------- 3 files changed, 71 insertions(+), 39 deletions(-) diff --git a/mm/internal.h b/mm/internal.h index 0434dfcfc36f14..ff28b940e0cd2c 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -625,8 +625,8 @@ unsigned find_lock_entries(struct address_space *mapping, pgoff_t *start, unsigned find_get_entries(struct address_space *mapping, pgoff_t *start, pgoff_t end, struct folio_batch *fbatch, pgoff_t *indices); int truncate_inode_folio(struct address_space *mapping, struct folio *folio); -bool truncate_inode_partial_folio(struct folio *folio, loff_t start, - loff_t end); +bool truncate_inode_partial_folio(struct folio *folio, loff_t lstart, + loff_t lend, pgoff_t *pstart, pgoff_t *pend); long mapping_evict_folio(struct address_space *mapping, struct folio *folio); unsigned long mapping_try_invalidate(struct address_space *mapping, pgoff_t start, pgoff_t end, unsigned long *nr_failed); diff --git a/mm/shmem.c b/mm/shmem.c index 07b2855dfb7bd9..ae08cff4500c3b 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -1386,11 +1386,8 @@ static void shmem_undo_range(struct inode *inode, loff_t lstart, uoff_t lend, if (folio) { same_folio = lend < folio_next_pos(folio); folio_mark_dirty(folio); - if (!truncate_inode_partial_folio(folio, lstart, lend)) { - start = folio_next_index(folio); - if (same_folio) - end = folio->index; - } + truncate_inode_partial_folio(folio, lstart, lend, &start, + same_folio ? &end : NULL); folio_unlock(folio); folio_put(folio); folio = NULL; @@ -1400,8 +1397,7 @@ static void shmem_undo_range(struct inode *inode, loff_t lstart, uoff_t lend, folio = shmem_get_partial_folio(inode, lend >> PAGE_SHIFT); if (folio) { folio_mark_dirty(folio); - if (!truncate_inode_partial_folio(folio, lstart, lend)) - end = folio->index; + truncate_inode_partial_folio(folio, lstart, lend, NULL, &end); folio_unlock(folio); folio_put(folio); } @@ -1469,7 +1465,8 @@ static void shmem_undo_range(struct inode *inode, loff_t lstart, uoff_t lend, if (!folio_test_large(folio)) { truncate_inode_folio(mapping, folio); - } else if (truncate_inode_partial_folio(folio, lstart, lend)) { + } else if (truncate_inode_partial_folio(folio, + lstart, lend, NULL, NULL)) { /* * If we split a page, reset the loop so * that we pick up the new sub pages. diff --git a/mm/truncate.c b/mm/truncate.c index f252619ae6c39b..9eb1087691490e 100644 --- a/mm/truncate.c +++ b/mm/truncate.c @@ -206,30 +206,43 @@ static int folio_split_or_unmap(struct folio *folio, struct page *split_at, /* * Handle partial folios. The folio may be entirely within the * range if a split has raced with us. If not, we zero the part of the - * folio that's within the [start, end] range, and then split the folio if + * folio that's within the [lstart, lend] range, and then split the folio if * it's large. split_page_range() will discard pages which now lie beyond * i_size, and we rely on the caller to discard pages which lie within a * newly created hole. * + * When @pstart and/or @pend are non-NULL they receive the indexes of the + * folio range fully covered by [lstart, lend] after any split (or none), + * aligned inwards to min_order, i.e. the range of folios wholly within + * [lstart, lend] and so safe to discard. + * * Returns false if splitting failed so the caller can avoid * discarding the entire folio which is stubbornly unsplit. */ -bool truncate_inode_partial_folio(struct folio *folio, loff_t start, loff_t end) +bool truncate_inode_partial_folio(struct folio *folio, loff_t lstart, + loff_t lend, pgoff_t *pstart, pgoff_t *pend) { loff_t pos = folio_pos(folio); size_t size = folio_size(folio); unsigned int offset, length; struct page *split_at, *split_at2; + unsigned long min_nrbytes; unsigned int min_order; - if (pos < start) - offset = start - pos; + if (pos < lstart) + offset = lstart - pos; else offset = 0; - if (pos + size <= (u64)end) + if (pos + size <= (u64)lend) length = size - offset; else - length = end + 1 - pos - offset; + length = lend + 1 - pos - offset; + + if (pstart) + *pstart = offset ? folio_next_index(folio) : folio->index; + if (pend) + *pend = (pos + size > (u64)lend) ? folio->index : + folio_next_index(folio); folio_wait_writeback(folio); if (length == size) { @@ -247,10 +260,12 @@ bool truncate_inode_partial_folio(struct folio *folio, loff_t start, loff_t end) if (folio_needs_release(folio)) folio_invalidate(folio, offset, length); - if (!folio_test_large(folio)) - return true; min_order = mapping_min_folio_order(folio->mapping); + min_nrbytes = mapping_min_folio_nrbytes(folio->mapping); + if (folio_order(folio) == min_order) + return true; + split_at = folio_page(folio, PAGE_ALIGN_DOWN(offset) / PAGE_SIZE); if (!folio_split_or_unmap(folio, split_at, min_order)) { /* @@ -259,34 +274,57 @@ bool truncate_inode_partial_folio(struct folio *folio, loff_t start, loff_t end) * for shmem truncate */ struct folio *folio2; - pgoff_t end_idx; + pgoff_t end, aligned_end = round_down(pos + offset + length, + min_nrbytes) >> PAGE_SHIFT; + if (pstart) + *pstart = round_up(pos + offset, + min_nrbytes) >> PAGE_SHIFT; + + end = aligned_end; if (offset + length == size) - goto no_split; + goto out; /* * After the first split at the start edge, the folio at the * end edge may be freed and reused concurrently. - * __filemap_get_folio() looks up the straddler at end_idx + * __filemap_get_folio() looks up the straddler at aligned_end * and returns it locked and ref'd with the mapping * validated. */ - end_idx = (pos + offset + length) >> PAGE_SHIFT; - folio2 = __filemap_get_folio(folio->mapping, end_idx, + folio2 = __filemap_get_folio(folio->mapping, aligned_end, FGP_LOCK | FGP_NOWAIT, 0); - if (IS_ERR(folio2)) - goto no_split; - - /* make sure folio2 is large */ - if (!folio_test_large(folio2)) + if (IS_ERR(folio2)) { + /* + * No sub-folio straddles the boundary when + * aligned_end is empty, so discarding up to it is + * safe. Otherwise the straddler is locked by + * someone else and we cannot obtain a reliable end + * position, so we fall back to folio->index, which + * is safe but leaves the sub-folios split off at + * the offset edge in the page cache. + */ + if (PTR_ERR(folio2) != -ENOENT) + end = folio->index; goto out; + } - split_at2 = folio_page(folio2, (end_idx - folio2->index)); - folio_split_or_unmap(folio2, split_at2, min_order); -out: + /* Already at the minimum order, nothing to split */ + if (folio_order(folio2) == min_order) + goto out_put; + + split_at2 = folio_page(folio2, (aligned_end - folio2->index)); + + /* Split failed, keep the straddler intact */ + if (folio_split_or_unmap(folio2, split_at2, min_order)) + end = folio2->index; + +out_put: folio_unlock(folio2); folio_put(folio2); -no_split: +out: + if (pend) + *pend = end; return true; } if (folio_test_dirty(folio)) @@ -421,11 +459,8 @@ void truncate_inode_pages_range(struct address_space *mapping, folio = __filemap_get_folio(mapping, lstart >> PAGE_SHIFT, FGP_LOCK, 0); if (!IS_ERR(folio)) { same_folio = lend < folio_next_pos(folio); - if (!truncate_inode_partial_folio(folio, lstart, lend)) { - start = folio_next_index(folio); - if (same_folio) - end = folio->index; - } + truncate_inode_partial_folio(folio, lstart, lend, &start, + same_folio ? &end : NULL); folio_unlock(folio); folio_put(folio); folio = NULL; @@ -435,8 +470,8 @@ void truncate_inode_pages_range(struct address_space *mapping, folio = __filemap_get_folio(mapping, lend >> PAGE_SHIFT, FGP_LOCK, 0); if (!IS_ERR(folio)) { - if (!truncate_inode_partial_folio(folio, lstart, lend)) - end = folio->index; + truncate_inode_partial_folio(folio, lstart, lend, + NULL, &end); folio_unlock(folio); folio_put(folio); } From 7300e5e40d2714d7444a07183157408b82e443b2 Mon Sep 17 00:00:00 2001 From: Zhang Yi Date: Mon, 28 Sep 2026 20:08:33 +0800 Subject: [PATCH 1175/1352] mm/truncate: clarify return value of truncate_inode_partial_folio() With the earlier rework the callers no longer rely on the return value of truncate_inode_partial_folio() to decide whether to adjust the truncation range. The pstart/pend out-parameters carry that information instead. The callers now only use the return value as a flag indicating whether the loop should be reset to pick up newly split sub-folios on the shmem path. Return true if at least one split succeeded, and false otherwise. This clarifies the existing confusing return value semantics. Link: https://lore.kernel.org/20260928120833.3440834-5-yi.zhang@huaweicloud.com Signed-off-by: Zhang Yi Signed-off-by: Andrew Morton Reviewed-by: Jan Kara Reviewed-by: Brian Foster Acked-by: Zi Yan Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Hugh Dickins Cc: Baolin Wang Cc: Matthew Wilcox (Oracle) Cc: Joanne Koong Cc: Darrick J. Wong Cc: Cc: Yang Erkun Cc: Zhihao Cheng Cc: Kefeng Wang Cc: --- mm/truncate.c | 14 ++++++-------- 1 file changed, 6 insertions(+), 8 deletions(-) diff --git a/mm/truncate.c b/mm/truncate.c index 9eb1087691490e..5b1b13cf8ff0a3 100644 --- a/mm/truncate.c +++ b/mm/truncate.c @@ -216,8 +216,7 @@ static int folio_split_or_unmap(struct folio *folio, struct page *split_at, * aligned inwards to min_order, i.e. the range of folios wholly within * [lstart, lend] and so safe to discard. * - * Returns false if splitting failed so the caller can avoid - * discarding the entire folio which is stubbornly unsplit. + * Return %true if at least one split succeeded, %false otherwise. */ bool truncate_inode_partial_folio(struct folio *folio, loff_t lstart, loff_t lend, pgoff_t *pstart, pgoff_t *pend) @@ -247,7 +246,7 @@ bool truncate_inode_partial_folio(struct folio *folio, loff_t lstart, folio_wait_writeback(folio); if (length == size) { truncate_inode_folio(folio->mapping, folio); - return true; + return false; } /* @@ -264,7 +263,7 @@ bool truncate_inode_partial_folio(struct folio *folio, loff_t lstart, min_order = mapping_min_folio_order(folio->mapping); min_nrbytes = mapping_min_folio_nrbytes(folio->mapping); if (folio_order(folio) == min_order) - return true; + return false; split_at = folio_page(folio, PAGE_ALIGN_DOWN(offset) / PAGE_SIZE); if (!folio_split_or_unmap(folio, split_at, min_order)) { @@ -327,10 +326,9 @@ bool truncate_inode_partial_folio(struct folio *folio, loff_t lstart, *pend = end; return true; } - if (folio_test_dirty(folio)) - return false; - truncate_inode_folio(folio->mapping, folio); - return true; + if (!folio_test_dirty(folio)) + truncate_inode_folio(folio->mapping, folio); + return false; } /* From 6f3f2342d4feec2385d1f10fc96fe23c06373ad3 Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Mon, 28 Sep 2026 01:58:33 -0700 Subject: [PATCH 1176/1352] mm/damon/core: preserve the quota passed to damon_new_scheme() Patch series "mm/damon/core: preserve quota state when constructing schemes", v3. damon_commit_ctx() first commits the running context's parameters to a temporary context for validating proposed updates. Constructing the temporary schemes clears the running schemes' quota state because damon_new_scheme() initializes the quota passed as a parameter before copying it to the new scheme. Even an update later rejected with -EINVAL loses the running quota state. Initialize the new scheme's copy instead, and add KUnit tests for the constructor and for accepted and rejected context updates. Note: below are test results and changelogs that could be removed from the final commit log. In the v1 live test with damo, a scheme with a plain 64 KiB size quota and a 60-second reset interval uses its quota, and a full "damo tune" with unchanged parameters then lets it try another 64 KiB within the same window. With the fix, sz_tried stays at 64 KiB. KUnit was rerun after the rebase on x86-64 and i386. With only patch 2 applied, the two new tests fail. With the fix, all 46 DAMON KUnit tests pass on both architectures. The v1 DAMON selftests showed no new failures (QEMU TCG guest; the wss_estimation test missed its accuracy bounds with and without the fix). This patch (of 2): damon_commit_ctx() first commits the running context's parameters to a temporary context for validating proposed updates. When damon_commit_schemes() creates the temporary schemes, it passes the running scheme's quota as the quota parameter of damon_new_scheme(). damon_new_scheme() calls damos_quota_init() on that quota before copying it to the new scheme. This clears the running scheme's effective quota, feedback input and charging state. Even an update rejected with -EINVAL loses the running quota state. For a size quota, this discards the bytes already charged and allows the scheme to use a fresh quota before the reset interval has elapsed. For a goal-driven quota, the consist tuner loses its accumulated input and restarts from its minimum input. A time quota loses its throughput estimate and falls back to the initial estimate. The end users will show DAMOS works more or less aggressively than expected for online-commit updates of quotas. DAMON provides best efforts by default. DAMON parameters online commit is supposed to be executed only occasionally. Hence, the issue wouldn't be critical on sane setups. For user_input type quota goals, online commit of the user input score is expected to be frequent. But, for the case commit_schemes_quota_goals command is recommended for optimal execution, and it doesn't have this bug. Commit 60bd24f272d0 ("mm/damon/sysfs: test commit input against realistic destination") introduced this problem in v6.19 when sysfs validation began committing the running context's parameters to a temporary context. Commit b90408ef1163 ("mm/damon/core: safely validate src on damon_commit_ctx()") later moved that validation into the core API, exposing other callers including DAMON_RECLAIM and DAMON_LRU_SORT. Sashiko reported the same side effect [1] on the RFC of the core API change. Copy the quota to the new scheme first, then initialize that copy. Make damos_quota_init() return void, since its return value is no longer needed. Link: https://lore.kernel.org/20260928085835.7675-1-sj@kernel.org Link: https://lore.kernel.org/20260928085835.7675-2-sj@kernel.org Link: https://lore.kernel.org/r/20260702212143.0CB6D1F00A3D@smtp.kernel.org/ [1] Fixes: 60bd24f272d0 ("mm/damon/sysfs: test commit input against realistic destination") Signed-off-by: Karl Mehltretter Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Assisted-by: LLM Cc: Bijan Tabatabai Cc: --- mm/damon/core.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 733025b367457b..3a61e5fdfb625d 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -734,7 +734,7 @@ static bool damos_quota_goals_empty(struct damos_quota *q) } /* initialize fields of @quota that normally API users wouldn't set */ -static struct damos_quota *damos_quota_init(struct damos_quota *quota) +static void damos_quota_init(struct damos_quota *quota) { quota->esz = 0; quota->total_charged_sz = 0; @@ -744,7 +744,6 @@ static struct damos_quota *damos_quota_init(struct damos_quota *quota) quota->charge_target_from = NULL; quota->charge_addr_from = 0; quota->esz_bp = 0; - return quota; } struct damos *damon_new_scheme(struct damos_access_pattern *pattern, @@ -776,7 +775,8 @@ struct damos *damon_new_scheme(struct damos_access_pattern *pattern, scheme->last_applied = NULL; INIT_LIST_HEAD(&scheme->list); - scheme->quota = *(damos_quota_init(quota)); + scheme->quota = *quota; + damos_quota_init(&scheme->quota); /* quota.goals should be separately set by caller */ INIT_LIST_HEAD(&scheme->quota.goals); From c32914ecd3318c645e33de738dc2a65bc1faa805 Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Mon, 28 Sep 2026 01:58:34 -0700 Subject: [PATCH 1177/1352] mm/damon/tests/core-kunit: test preservation of quota state Check that damon_new_scheme() initializes the new scheme's quota without changing the quota passed as a parameter. Cover all eight fields initialized by damos_quota_init(). Also check that damon_commit_ctx() preserves those fields in the destination scheme for both accepted and rejected parameter updates. Use an invalid min_region_sz for the rejected update and confirm that returning -EINVAL leaves the running quota state unchanged. Without the preceding fix, all eight fields are cleared in the constructor test and in both context update cases. Link: https://lore.kernel.org/20260928085835.7675-3-sj@kernel.org Signed-off-by: Karl Mehltretter Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Assisted-by: LLM Cc: Bijan Tabatabai Cc: Brendan Higgins Cc: David Gow --- mm/damon/tests/core-kunit.h | 107 ++++++++++++++++++++++++++++++++++++ 1 file changed, 107 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index df84d9cc7d204a..ce17f44583087e 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -812,6 +812,48 @@ static void damos_test_new_filter(struct kunit *test) damos_destroy_filter(filter); } +static void damos_test_new_scheme_keeps_src_quota(struct kunit *test) +{ + struct damos_access_pattern pattern = {}; + struct damon_target target = {}; + struct damos_quota quota = { + .sz = SZ_64K, + .esz = 123, + .esz_bp = 456, + .total_charged_sz = 789, + .total_charged_ns = 1011, + .charged_sz = 12, + .charged_from = 13, + .charge_target_from = &target, + .charge_addr_from = 14, + }; + struct damos_watermarks wmarks = {}; + struct damos *s; + + s = damon_new_scheme(&pattern, DAMOS_STAT, 0, "a, &wmarks, + NUMA_NO_NODE); + if (!s) + kunit_skip(test, "scheme alloc fail"); + KUNIT_EXPECT_EQ(test, s->quota.sz, (unsigned long)SZ_64K); + KUNIT_EXPECT_EQ(test, s->quota.esz, 0ul); + KUNIT_EXPECT_EQ(test, s->quota.esz_bp, 0ul); + KUNIT_EXPECT_EQ(test, s->quota.total_charged_sz, 0ul); + KUNIT_EXPECT_EQ(test, s->quota.total_charged_ns, 0ul); + KUNIT_EXPECT_EQ(test, s->quota.charged_sz, 0ul); + KUNIT_EXPECT_EQ(test, s->quota.charged_from, 0ul); + KUNIT_EXPECT_PTR_EQ(test, s->quota.charge_target_from, NULL); + KUNIT_EXPECT_EQ(test, s->quota.charge_addr_from, 0ul); + KUNIT_EXPECT_EQ(test, quota.esz, 123ul); + KUNIT_EXPECT_EQ(test, quota.esz_bp, 456ul); + KUNIT_EXPECT_EQ(test, quota.total_charged_sz, 789ul); + KUNIT_EXPECT_EQ(test, quota.total_charged_ns, 1011ul); + KUNIT_EXPECT_EQ(test, quota.charged_sz, 12ul); + KUNIT_EXPECT_EQ(test, quota.charged_from, 13ul); + KUNIT_EXPECT_PTR_EQ(test, quota.charge_target_from, &target); + KUNIT_EXPECT_EQ(test, quota.charge_addr_from, 14ul); + damon_destroy_scheme(s); +} + static void damos_test_commit_quota_goal_for(struct kunit *test, struct damos_quota_goal *dst, struct damos_quota_goal *src) @@ -1572,6 +1614,69 @@ static void damon_test_commit_ctx(struct kunit *test) damon_destroy_ctx(dst); } +static void damon_test_commit_ctx_keeps_quota_for(struct kunit *test, + unsigned long min_region_sz, int expected_err) +{ + struct damos_access_pattern pattern = {}; + struct damos_quota quota = {.sz = SZ_64K}; + struct damos_watermarks wmarks = {}; + struct damon_ctx *src, *dst; + struct damon_target *target; + struct damos *s; + + dst = damon_new_ctx(); + if (!dst) + kunit_skip(test, "dst alloc fail"); + target = damon_new_target(); + if (!target) { + damon_destroy_ctx(dst); + kunit_skip(test, "target alloc fail"); + } + damon_add_target(dst, target); + s = damon_new_scheme(&pattern, DAMOS_STAT, 0, "a, &wmarks, + NUMA_NO_NODE); + if (!s) { + damon_destroy_ctx(dst); + kunit_skip(test, "scheme alloc fail"); + } + damon_add_scheme(dst, s); + + /* Copy the parameters before populating dst's runtime quota state. */ + src = damon_new_test_ctx(dst); + if (!src) { + damon_destroy_ctx(dst); + kunit_skip(test, "src alloc fail"); + } + src->min_region_sz = min_region_sz; + s->quota.esz = 123; + s->quota.esz_bp = 456; + s->quota.total_charged_sz = 789; + s->quota.total_charged_ns = 1011; + s->quota.charged_sz = 12; + s->quota.charged_from = 13; + s->quota.charge_target_from = target; + s->quota.charge_addr_from = 14; + + KUNIT_EXPECT_EQ(test, damon_commit_ctx(dst, src), expected_err); + KUNIT_EXPECT_EQ(test, s->quota.esz, 123ul); + KUNIT_EXPECT_EQ(test, s->quota.esz_bp, 456ul); + KUNIT_EXPECT_EQ(test, s->quota.total_charged_sz, 789ul); + KUNIT_EXPECT_EQ(test, s->quota.total_charged_ns, 1011ul); + KUNIT_EXPECT_EQ(test, s->quota.charged_sz, 12ul); + KUNIT_EXPECT_EQ(test, s->quota.charged_from, 13ul); + KUNIT_EXPECT_PTR_EQ(test, s->quota.charge_target_from, target); + KUNIT_EXPECT_EQ(test, s->quota.charge_addr_from, 14ul); + damon_destroy_ctx(src); + damon_destroy_ctx(dst); +} + +static void damon_test_commit_ctx_keeps_quota(struct kunit *test) +{ + /* Only power of two min_region_sz is allowed. */ + damon_test_commit_ctx_keeps_quota_for(test, 4096, 0); + damon_test_commit_ctx_keeps_quota_for(test, 4095, -EINVAL); +} + static void damon_test_valid_probe_params(struct kunit *test) { struct damon_ctx *ctx; @@ -2259,6 +2364,7 @@ static struct kunit_case damon_test_cases[] = { KUNIT_CASE(damon_test_mvsum), KUNIT_CASE(damon_test_nr_accesses_mvsum), KUNIT_CASE(damos_test_new_filter), + KUNIT_CASE(damos_test_new_scheme_keeps_src_quota), KUNIT_CASE(damos_test_commit_quota_goal), KUNIT_CASE(damos_test_commit_quota_goals), KUNIT_CASE(damos_test_commit_quota), @@ -2270,6 +2376,7 @@ static struct kunit_case damon_test_cases[] = { KUNIT_CASE(damon_test_commit_filter), KUNIT_CASE(damon_test_commit_probes), KUNIT_CASE(damon_test_commit_ctx), + KUNIT_CASE(damon_test_commit_ctx_keeps_quota), KUNIT_CASE(damon_test_valid_probe_params), KUNIT_CASE(damos_test_filter_out), KUNIT_CASE(damos_test_apply_scheme_filtered_sz), From 5d2677078903c005754c98a012841681c6a018bd Mon Sep 17 00:00:00 2001 From: Donggeun Yoo Date: Mon, 28 Sep 2026 01:48:14 -0700 Subject: [PATCH 1178/1352] mm/damon/core: prevent size quota overflow in the temporal goal tuner Patch series "mm/damon: fix the temporal goal tuner's size quota conversion", v5. damos_goal_tune_esz_bp_temporal() hands the size quota to damos_set_effective_quota() through quota->esz_bp in basis points, and the multiply that gets it there is unchecked. On 32-bit it wraps above 429496 bytes, and a wrapped product below 10000 divides to a zero effective quota. damos_quota_is_full() is then true on the first test of every charge window, so the scheme makes no progress for as long as the goal is unachieved. Patch 1 bounds the conversion. Patch 2 pins the boundary in the core kunit suite, where the new test would fail without patch 1 on any word size. This patch (of 2): damos_goal_tune_esz_bp_temporal() converts the scheme's size quota into basis points with "quota->esz_bp = quota->sz * 10000", both unsigned long, and damos_set_effective_quota() divides the result back by 10000. quotas/bytes is unbounded; bytes_store() hands it to kstrtoul() as is. On 32-bit the product wraps for any size quota above ULONG_MAX / 10000, that is 429496 bytes. A wrapped product below 10000 divides to a zero effective quota: 429497 gives 0. damos_quota_is_full() is then true on the first test of every charge window. Other wrapped values are wrong without being zero: 500000 gives 70503. Triggering this needs a scheme with a quota goal, the temporal goal tuner, and a size quota above ULONG_MAX / 10000 -- 429496 bytes on 32-bit, 1844674407370955 on 64-bit. The scheme then makes no progress for as long as the goal is unachieved, which is easy to notice, and writing a smaller size quota restores it. Nothing is corrupted and nothing leaks. This is unlikely to be hit on a tested setup. addr_unit does not cover this. It only scales the numbers a paddr context writes to quotas/bytes, so a large enough scaled value wraps just the same, and vaddr and fvaddr contexts take raw byte values. Bound the multiply. Link: https://lore.kernel.org/20260928084816.5575-1-sj@kernel.org Link: https://lore.kernel.org/20260928084816.5575-2-sj@kernel.org Fixes: af738a6a00c1 ("mm/damon/core: introduce DAMOS_QUOTA_GOAL_TUNER_TEMPORAL") Signed-off-by: Donggeun Yoo Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: # 7.1.x --- mm/damon/core.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 3a61e5fdfb625d..46ef570a4766fd 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -3291,7 +3291,7 @@ static void damos_goal_tune_esz_bp_temporal(struct damon_ctx *c, if (score >= 10000) quota->esz_bp = 0; - else if (quota->sz) + else if (quota->sz && quota->sz <= ULONG_MAX / 10000) quota->esz_bp = quota->sz * 10000; else quota->esz_bp = ULONG_MAX; From 741ac5665c920cb29201cc1a17fefceee3c1a383 Mon Sep 17 00:00:00 2001 From: Donggeun Yoo Date: Mon, 28 Sep 2026 01:48:15 -0700 Subject: [PATCH 1179/1352] mm/damon/tests/core-kunit: test the temporal tuner's size quota conversion damos_goal_tune_esz_bp_temporal() encodes the size quota in basis points, so the conversion is exact only up to ULONG_MAX / 10000. Pin the three sizes around that boundary: the largest one that fits, the first one that does not, and ULONG_MAX. Link: https://lore.kernel.org/20260928084816.5575-3-sj@kernel.org Signed-off-by: Donggeun Yoo Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Brendan Higgins Cc: David Gow --- mm/damon/tests/core-kunit.h | 48 +++++++++++++++++++++++++++++++++++++ 1 file changed, 48 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index ce17f44583087e..a8e22246facf96 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -2347,6 +2347,53 @@ static void damon_test_rand(struct kunit *test) } } +static void damos_test_esz_goal_temporal(struct kunit *test) +{ + struct damos_access_pattern pattern = {}; + struct damos_watermarks wmarks = {}; + struct damos_quota quota = { + .goal_tuner = DAMOS_QUOTA_GOAL_TUNER_TEMPORAL, + }; + struct damos_quota_goal *goal; + struct damon_ctx *ctx; + struct damos *s; + + ctx = damon_new_ctx(); + KUNIT_ASSERT_NOT_NULL(test, ctx); + + s = damon_new_scheme(&pattern, DAMOS_STAT, 0, "a, &wmarks, + NUMA_NO_NODE); + if (!s) { + damon_destroy_ctx(ctx); + kunit_skip(test, "scheme alloc fail"); + } + damon_add_scheme(ctx, s); + + goal = damos_new_quota_goal(DAMOS_QUOTA_USER_INPUT, 10000); + if (!goal) { + damon_destroy_ctx(ctx); + kunit_skip(test, "quota goal alloc fail"); + } + goal->current_value = 0; + damos_add_quota_goal(&s->quota, goal); + + /* The largest size quota the basis-point conversion can hold. */ + s->quota.sz = ULONG_MAX / 10000; + damos_set_effective_quota(ctx, s); + KUNIT_EXPECT_EQ(test, s->quota.esz, ULONG_MAX / 10000); + + /* Any larger one saturates instead of wrapping. */ + s->quota.sz = ULONG_MAX / 10000 + 1; + damos_set_effective_quota(ctx, s); + KUNIT_EXPECT_EQ(test, s->quota.esz, ULONG_MAX / 10000); + + s->quota.sz = ULONG_MAX; + damos_set_effective_quota(ctx, s); + KUNIT_EXPECT_EQ(test, s->quota.esz, ULONG_MAX / 10000); + + damon_destroy_ctx(ctx); +} + static struct kunit_case damon_test_cases[] = { KUNIT_CASE(damon_test_target), KUNIT_CASE(damon_test_regions), @@ -2388,6 +2435,7 @@ static struct kunit_case damon_test_cases[] = { KUNIT_CASE(damon_test_is_last_region), KUNIT_CASE(damon_test_walk_control_obsolete), KUNIT_CASE(damon_test_rand), + KUNIT_CASE(damos_test_esz_goal_temporal), {}, }; From 4e35d3cf0691b13b6f23edd3fc56b41be71a285a Mon Sep 17 00:00:00 2001 From: "Zenghui Yu (Huawei)" Date: Mon, 28 Sep 2026 01:39:55 -0700 Subject: [PATCH 1180/1352] mm/damon/api: remove unused NR_DAMOS_* enumerators Patch series "mm/damon/core: cleanup code, reduce stack usage, and add kunit". Three various improvements. Patch 1 from Zenghui Yu (Huawei) cleans up unused enums. Patch 2 from Arnd Bergmann reduces kdamond's stack usage. Patch 3 from Karl Mehltretter adds a kunit test for uninitialized PSI quota goal stage logic. This patch (of 3): NR_DAMOS_ACTIONS, NR_DAMOS_QUOTA_GOAL_METRICS, and NR_DAMOS_WMARK_METRICS are not referenced by any code. Remove them. They can be reintroduced if a real user comes up. Link: https://lore.kernel.org/20260928083959.4030-1-sj@kernel.org Link: https://lore.kernel.org/20260928083959.4030-2-sj@kernel.org Signed-off-by: Zenghui Yu (Huawei) Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park --- include/linux/damon.h | 8 -------- 1 file changed, 8 deletions(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index 844b175120f09b..6a29dc2ac8db79 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -117,7 +117,6 @@ struct damon_target { * @DAMOS_MIGRATE_HOT: Migrate the regions prioritizing warmer regions. * @DAMOS_MIGRATE_COLD: Migrate the regions prioritizing colder regions. * @DAMOS_STAT: Do nothing but count the stat. - * @NR_DAMOS_ACTIONS: Total number of DAMOS actions * * The support of each action is up to running &struct damon_operations. * Refer to 'Operation Action' section of Documentation/mm/damon/design.rst for @@ -137,7 +136,6 @@ enum damos_action { DAMOS_MIGRATE_HOT, DAMOS_MIGRATE_COLD, DAMOS_STAT, /* Do nothing but only record the stat */ - NR_DAMOS_ACTIONS, }; /** @@ -154,9 +152,6 @@ enum damos_action { * @DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP: Scheme-eligible memory ratio of a * node in basis points (0-10000). * @DAMOS_QUOTA_HUGEPAGE_MEM_BP: Huge page to total used memory ratio. - * @NR_DAMOS_QUOTA_GOAL_METRICS: Number of DAMOS quota goal metrics. - * - * Metrics equal to larger than @NR_DAMOS_QUOTA_GOAL_METRICS are unsupported. */ enum damos_quota_goal_metric { DAMOS_QUOTA_USER_INPUT, @@ -169,7 +164,6 @@ enum damos_quota_goal_metric { DAMOS_QUOTA_INACTIVE_MEM_BP, DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP, DAMOS_QUOTA_HUGEPAGE_MEM_BP, - NR_DAMOS_QUOTA_GOAL_METRICS, }; /** @@ -313,12 +307,10 @@ struct damos_quota { * * @DAMOS_WMARK_NONE: Ignore the watermarks of the given scheme. * @DAMOS_WMARK_FREE_MEM_RATE: Free memory rate of the system in [0,1000]. - * @NR_DAMOS_WMARK_METRICS: Total number of DAMOS watermark metrics */ enum damos_wmark_metric { DAMOS_WMARK_NONE, DAMOS_WMARK_FREE_MEM_RATE, - NR_DAMOS_WMARK_METRICS, }; /** From 48f8c573eea8c9e6a70cab2d742d7b4ff552c172 Mon Sep 17 00:00:00 2001 From: Arnd Bergmann Date: Mon, 28 Sep 2026 01:39:56 -0700 Subject: [PATCH 1181/1352] mm/damon/core: reduce stack usage further In a previous patch, I had annotated kdamond_tune_intervals() as noinline_for_stack in order to not exceed the stack frame warning limit. Unfortunately, my current linux-next randconfig builds show a similar problem again with clang-21: mm/damon/core.c:3953:12: error: stack frame size (1288) exceeds limit (1280) in 'kdamond_fn' [-Werror,-Wframe-larger-than] 3953 | static int kdamond_fn(void *data) Do the same thing with kdamond_apply_schemes(), kdamond_merge_regions(), and kdamond_split_regions(), which also have individually large stacks. This should work to keep the deepest total stack depth down much more as well as avoid the warning. I also tried to reduce the complexity of kdamond_fn() itself by splitting out the while() loop into a separate function. While this arguably led to slightly more readable code, it had no effect on the total stack usage and just made the new function the largest stack user and had a nonzero risk of me getting the conversion wrong. Link: https://lore.kernel.org/20260928083959.4030-3-sj@kernel.org Fixes: 5a00cae64de1 ("mm/damon/core: reduce kernel stack usage") Signed-off-by: Arnd Bergmann Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Nick Desaulniers Cc: Bill Wendling Cc: Justin Stitt Cc: Ravi Jonnalagadda Cc: Nathan Chancellor --- mm/damon/core.c | 9 +++++---- 1 file changed, 5 insertions(+), 4 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 46ef570a4766fd..5ecbea5d71e1dc 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -3433,7 +3433,7 @@ static void damos_trace_stat(struct damon_ctx *c, struct damos *s) trace_call__damos_stat_after_apply_interval(cidx, sidx, &s->stat); } -static void kdamond_apply_schemes(struct damon_ctx *c) +static noinline_for_stack void kdamond_apply_schemes(struct damon_ctx *c) { struct damon_target *t; struct damos *s; @@ -3589,8 +3589,9 @@ static void damon_merge_regions_of(struct damon_target *t, unsigned int thres, * while DAMON is running. For such a case, repeat merging until the limit is * met while increasing @threshold up to possible maximum level. */ -static void kdamond_merge_regions(struct damon_ctx *c, unsigned int threshold, - unsigned long sz_limit) +static noinline_for_stack void kdamond_merge_regions(struct damon_ctx *c, + unsigned int threshold, + unsigned long sz_limit) { struct damon_target *t; unsigned int nr_regions; @@ -3734,7 +3735,7 @@ static void damon_split_some_regions(struct damon_ctx *ctx, * split was unnecessarily made, later 'kdamond_merge_regions()' will revert * it. */ -static void kdamond_split_regions(struct damon_ctx *ctx) +static noinline_for_stack void kdamond_split_regions(struct damon_ctx *ctx) { struct damon_target *t; unsigned long nr_regions = 0; From 6b2f8a9be8eef5715b9583f492c4e4d81f6f663f Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Mon, 28 Sep 2026 01:39:57 -0700 Subject: [PATCH 1182/1352] mm/damon/tests/core-kunit: test PSI goal values with explicit samples Test the PSI current-value helper with explicit totals so the result does not depend on the test system's memory pressure. Cover an unmeasured consist goal, unmeasured temporal goals with zero and nonzero effective quotas, and measured rounds for both tuners. Check last_psi_total after each call. Link: https://lore.kernel.org/20260928083959.4030-4-sj@kernel.org Signed-off-by: Karl Mehltretter Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Assisted-by: LLM Cc: Lian Wang Cc: Kunwu Chan Cc: Brendan Higgins Cc: David Gow --- mm/damon/tests/core-kunit.h | 43 +++++++++++++++++++++++++++++++++++++ 1 file changed, 43 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index a8e22246facf96..2111faa581532a 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -952,6 +952,48 @@ static void damos_test_commit_quota_goal(struct kunit *test) }); } +static void damos_test_set_psi_current_val(struct kunit *test) +{ + struct damos s = { + .quota.goal_tuner = DAMOS_QUOTA_GOAL_TUNER_CONSIST, + }; + struct damos_quota_goal goal = { + .metric = DAMOS_QUOTA_SOME_MEM_PSI_US, + .target_value = 100, + .last_psi_total = U64_MAX, + }; + + /* uninitialized last_psi_total keeps the consist tuner quota */ + damos_set_psi_current_val(1000, &goal, &s); + KUNIT_EXPECT_EQ(test, goal.current_value, 100ul); + KUNIT_EXPECT_EQ(test, goal.last_psi_total, 1000ull); + + /* initialized last_psi_total gives the delta */ + damos_set_psi_current_val(1030, &goal, &s); + KUNIT_EXPECT_EQ(test, goal.current_value, 30ul); + KUNIT_EXPECT_EQ(test, goal.last_psi_total, 1030ull); + + /* temporal tuner keeps a zero quota */ + s.quota.goal_tuner = DAMOS_QUOTA_GOAL_TUNER_TEMPORAL; + s.quota.esz = 0; + goal.last_psi_total = U64_MAX; + damos_set_psi_current_val(2000, &goal, &s); + KUNIT_EXPECT_EQ(test, goal.current_value, 100ul); + KUNIT_EXPECT_EQ(test, goal.last_psi_total, 2000ull); + + /* temporal tuner keeps a non-zero quota */ + s.quota.esz = SZ_64K; + goal.last_psi_total = U64_MAX; + damos_set_psi_current_val(3000, &goal, &s); + KUNIT_EXPECT_EQ(test, goal.current_value, 0ul); + KUNIT_EXPECT_EQ(test, goal.last_psi_total, 3000ull); + + /* temporal tuner uses the measured PSI delta */ + damos_set_psi_current_val(3250, &goal, &s); + KUNIT_EXPECT_EQ(test, goal.current_value, 250ul); + KUNIT_EXPECT_EQ(test, goal.last_psi_total, 3250ull); +} + static void damos_test_commit_quota_goals_for(struct kunit *test, struct damos_quota_goal *dst_goals, int nr_dst_goals, struct damos_quota_goal *src_goals, int nr_src_goals) @@ -2413,6 +2455,7 @@ static struct kunit_case damon_test_cases[] = { KUNIT_CASE(damos_test_new_filter), KUNIT_CASE(damos_test_new_scheme_keeps_src_quota), KUNIT_CASE(damos_test_commit_quota_goal), + KUNIT_CASE(damos_test_set_psi_current_val), KUNIT_CASE(damos_test_commit_quota_goals), KUNIT_CASE(damos_test_commit_quota), KUNIT_CASE(damos_test_commit_dests), From 6ac6e6896041f550e63fb75cab8b9ffc362c19e2 Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Mon, 28 Sep 2026 16:15:54 +0800 Subject: [PATCH 1183/1352] mm/vmalloc: fix vmalloc_dump_obj address alignment for last-page lookups Patch series "mm/vmalloc: fix vmalloc_dump_obj VA lookup", v4. vmalloc_dump_obj() has two bugs that cause it to miss vmalloc allocations. 1. PAGE_ALIGN() rounds up, pushing last-page addresses to va_end and outside the VA lookup range. 2. The function searches only one vmap node, but allocations larger than 64 KiB may span multiple vmap zones whose VA is stored in a different node's rb-tree. Patch 1 fixes the alignment, patch 2 fixes the cross-zone search. This patch (of 2): vmalloc_dump_obj() uses PAGE_ALIGN() to normalize the input address before looking it up in the per-node busy tree. PAGE_ALIGN() rounds up, which can push an address in the last page of a vmalloc allocation to va_end -- outside the [va_start, va_end) range that __find_vmap_area() searches. This causes the lookup to miss the VA and return false, degrading diagnostic output in OOM dumps and KASAN reports to the less informative "vmalloc memory" fallback. The upward alignment can also change the addr_to_node() mapping when the page boundary crosses a vmap zone boundary, causing the search to hit the wrong node entirely. Use PAGE_ALIGN_DOWN() instead, which rounds down to the page containing the address. This keeps the address within the VA range and preserves the correct node mapping. Link: https://lore.kernel.org/20260928-vmalloc_dump_obj-v4-0-6f288a431edc@linux.dev Link: https://lore.kernel.org/20260928-vmalloc_dump_obj-v4-1-6f288a431edc@linux.dev Signed-off-by: Ye Liu Signed-off-by: Andrew Morton Reviewed-by: Uladzislau Rezki (Sony) Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti --- mm/vmalloc.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index 4e4cb8d785bdb4..f5da00821f3fdc 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -5281,7 +5281,7 @@ bool vmalloc_dump_obj(void *object) unsigned long addr; unsigned long nr_pages; - addr = PAGE_ALIGN((unsigned long) object); + addr = PAGE_ALIGN_DOWN((unsigned long) object); vn = addr_to_node(addr); if (!spin_trylock(&vn->busy.lock)) From 85587cdd56392025eea081ba4980cad302cb3015 Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Mon, 28 Sep 2026 16:15:55 +0800 Subject: [PATCH 1184/1352] mm/vmalloc: fix vmalloc_dump_obj cross-zone VA lookup vmalloc_dump_obj() searches only one vmap node (addr_to_node(addr)), but a vmalloc allocation may span multiple vmap zones. The VA is stored in only one node's rb-tree (addr_to_node(va_start)), so an object pointer in a different zone than va_start maps to a different node and the search misses. This affects any allocation larger than vmap_zone_size (64 KiB) on multi-CPU systems. Extract find_vmap_area_lock() from find_vmap_area() to share the cross-node iteration logic. The helper supports both spin_lock and spin_trylock, the latter for atomic dump contexts (OOM, KASAN, RCU). Link: https://lore.kernel.org/20260928-vmalloc_dump_obj-v4-2-6f288a431edc@linux.dev Signed-off-by: Ye Liu Signed-off-by: Andrew Morton Reviewed-by: Uladzislau Rezki (Sony) Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti --- mm/vmalloc.c | 108 +++++++++++++++++++++++++++++++-------------------- 1 file changed, 66 insertions(+), 42 deletions(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index f5da00821f3fdc..a9fce7efe4f496 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -2516,67 +2516,94 @@ static void free_unmap_vmap_area(struct vmap_area *va) free_vmap_area_noflush(va); } -struct vmap_area *find_vmap_area(unsigned long addr) +static inline int next_vmap_node_id(int i) +{ + return (i + nr_vmap_nodes - 1) % nr_vmap_nodes; +} + +enum vmap_lock_mode { + VMAP_LOCK, + VMAP_TRYLOCK, +}; + +/* + * Search for a vmap_area at @addr across all vmap nodes. An + * addr_to_node_id(addr) converts an address to a node index where + * a VA is located. If VA spans several zones and passed addr is not + * the same as va->va_start, what is not common, we may need to scan + * extra nodes. See an example: + * + * <----va----> + * -|-----|-----|-----|-----|- + * 1 2 0 1 + * + * VA resides in node 1 whereas it spans 1, 2 an 0. If passed addr + * is within 2 or 0 nodes we should do extra work. + * + * Returns the VA with @locked_vn->busy.lock held; the caller must + * release it. If @mode is VMAP_TRYLOCK, nodes that cannot be locked + * are skipped. + */ +static struct vmap_area * +find_vmap_area_lock(unsigned long addr, struct vmap_node **locked_vn, + enum vmap_lock_mode mode) { struct vmap_node *vn; struct vmap_area *va; int i, j; + *locked_vn = NULL; + if (unlikely(!vmap_initialized)) return NULL; - /* - * An addr_to_node_id(addr) converts an address to a node index - * where a VA is located. If VA spans several zones and passed - * addr is not the same as va->va_start, what is not common, we - * may need to scan extra nodes. See an example: - * - * <----va----> - * -|-----|-----|-----|-----|- - * 1 2 0 1 - * - * VA resides in node 1 whereas it spans 1, 2 an 0. If passed - * addr is within 2 or 0 nodes we should do extra work. - */ i = j = addr_to_node_id(addr); do { vn = &vmap_nodes[i]; - spin_lock(&vn->busy.lock); - va = __find_vmap_area(addr, &vn->busy.root); - spin_unlock(&vn->busy.lock); + if (mode == VMAP_LOCK) { + spin_lock(&vn->busy.lock); + } else { + if (!spin_trylock(&vn->busy.lock)) + continue; + } - if (va) + va = __find_vmap_area(addr, &vn->busy.root); + if (va) { + *locked_vn = vn; return va; - } while ((i = (i + nr_vmap_nodes - 1) % nr_vmap_nodes) != j); + } + + spin_unlock(&vn->busy.lock); + } while ((i = next_vmap_node_id(i)) != j); return NULL; } -static struct vmap_area *find_unlink_vmap_area(unsigned long addr) +struct vmap_area *find_vmap_area(unsigned long addr) { struct vmap_node *vn; struct vmap_area *va; - int i, j; - - /* - * Check the comment in the find_vmap_area() about the loop. - */ - i = j = addr_to_node_id(addr); - do { - vn = &vmap_nodes[i]; - spin_lock(&vn->busy.lock); - va = __find_vmap_area(addr, &vn->busy.root); - if (va) - unlink_va(va, &vn->busy.root); + va = find_vmap_area_lock(addr, &vn, VMAP_LOCK); + if (va) spin_unlock(&vn->busy.lock); - if (va) - return va; - } while ((i = (i + nr_vmap_nodes - 1) % nr_vmap_nodes) != j); + return va; +} - return NULL; +static struct vmap_area *find_unlink_vmap_area(unsigned long addr) +{ + struct vmap_node *vn; + struct vmap_area *va; + + va = find_vmap_area_lock(addr, &vn, VMAP_LOCK); + if (va) { + unlink_va(va, &vn->busy.root); + spin_unlock(&vn->busy.lock); + } + + return va; } /*** Per cpu kva allocator ***/ @@ -5282,14 +5309,11 @@ bool vmalloc_dump_obj(void *object) unsigned long nr_pages; addr = PAGE_ALIGN_DOWN((unsigned long) object); - vn = addr_to_node(addr); - - if (!spin_trylock(&vn->busy.lock)) - return false; - va = __find_vmap_area(addr, &vn->busy.root); + va = find_vmap_area_lock(addr, &vn, VMAP_TRYLOCK); if (!va || !va->vm) { - spin_unlock(&vn->busy.lock); + if (va) + spin_unlock(&vn->busy.lock); return false; } From 6aaf8b2adfd6b99bcd65f39070e7b5b4bd2b580f Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:24 +0800 Subject: [PATCH 1185/1352] gpu: ipu-v3: don't use GFP_DMA when calling dma_alloc_coherent() Patch series "Don't use GFP_DMA when calling dma_alloc_coherent". This series picks up the first part of an earlier cleanup series [1] that was prepared back in the year of 2022, but for various reasons never made it merged and has been sitting in a local tree since then. This subset only touches the call sites where GFP_DMA is passed to dma_alloc_coherent() (and its dma_alloc_wc()/dmam_alloc_coherent() variants), which is the most self-contained and least risky slice of that work. That GFP_DMA is simply redundant here: the DMA API derives the allocation zone from the device's coherent_dma_mask (together with bus_dma_limit) and ignores the GFP_DMA flag passed by the caller. Removing the redundant GFP_DMA won't harm anything, while keeps it from being blindly copied into new code. This patch (of 13): dma_alloc_coherent() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Remove the redundant GFP_DMA flag. Link: https://lore.kernel.org/20260903111836.1777265-1-hebaoquan@kylinos.cn Link: https://lore.kernel.org/20260903111836.1777265-2-hebaoquan@kylinos.cn Link: https://lore.kernel.org/all/20220219005221.634-1-bhe@redhat.com/T/#u [1] Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: Harry Yoo --- drivers/gpu/ipu-v3/ipu-image-convert.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/gpu/ipu-v3/ipu-image-convert.c b/drivers/gpu/ipu-v3/ipu-image-convert.c index 29b6d36c9bb674..493be4d4b091e3 100644 --- a/drivers/gpu/ipu-v3/ipu-image-convert.c +++ b/drivers/gpu/ipu-v3/ipu-image-convert.c @@ -371,7 +371,7 @@ static int alloc_dma_buf(struct ipu_image_convert_priv *priv, { buf->len = PAGE_ALIGN(size); buf->virt = dma_alloc_coherent(priv->ipu->dev, buf->len, &buf->phys, - GFP_DMA | GFP_KERNEL); + GFP_KERNEL); if (!buf->virt) { dev_err(priv->ipu->dev, "failed to alloc dma buffer\n"); return -ENOMEM; From 65d5a0033825366c977ed1c7a16cf23746562199 Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:25 +0800 Subject: [PATCH 1186/1352] drm/sti: don't use GFP_DMA when calling dma_alloc_wc() dma_alloc_wc() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Remove the redundant GFP_DMA flag. Link: https://lore.kernel.org/20260903111836.1777265-3-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: Harry Yoo --- drivers/gpu/drm/sti/sti_cursor.c | 4 ++-- drivers/gpu/drm/sti/sti_hqvdp.c | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/drivers/gpu/drm/sti/sti_cursor.c b/drivers/gpu/drm/sti/sti_cursor.c index d0b89b28e50cca..a0ec671477fdf2 100644 --- a/drivers/gpu/drm/sti/sti_cursor.c +++ b/drivers/gpu/drm/sti/sti_cursor.c @@ -240,7 +240,7 @@ static int sti_cursor_atomic_check(struct drm_plane *drm_plane, cursor->pixmap.base = dma_alloc_wc(cursor->dev, cursor->pixmap.size, &cursor->pixmap.paddr, - GFP_KERNEL | GFP_DMA); + GFP_KERNEL); if (!cursor->pixmap.base) { DRM_ERROR("Failed to allocate memory for pixmap\n"); return -EINVAL; @@ -380,7 +380,7 @@ struct drm_plane *sti_cursor_create(struct drm_device *drm_dev, /* Allocate clut buffer */ size = 0x100 * sizeof(unsigned short); cursor->clut = dma_alloc_wc(dev, size, &cursor->clut_paddr, - GFP_KERNEL | GFP_DMA); + GFP_KERNEL); if (!cursor->clut) { DRM_ERROR("Failed to allocate memory for cursor clut\n"); diff --git a/drivers/gpu/drm/sti/sti_hqvdp.c b/drivers/gpu/drm/sti/sti_hqvdp.c index cf1ed8a33b8e18..b0d66834bcbe8b 100644 --- a/drivers/gpu/drm/sti/sti_hqvdp.c +++ b/drivers/gpu/drm/sti/sti_hqvdp.c @@ -860,7 +860,7 @@ static void sti_hqvdp_init(struct sti_hqvdp *hqvdp) size = NB_VDP_CMD * sizeof(struct sti_hqvdp_cmd); hqvdp->hqvdp_cmd = dma_alloc_wc(hqvdp->dev, size, &dma_addr, - GFP_KERNEL | GFP_DMA); + GFP_KERNEL); if (!hqvdp->hqvdp_cmd) { DRM_ERROR("Failed to allocate memory for VDP cmd\n"); return; From 3fc14155ff268bc1626662bc2faf27c2f0ceeb17 Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:26 +0800 Subject: [PATCH 1187/1352] ALSA: n64: don't use GFP_DMA when calling dma_alloc_coherent() dma_alloc_coherent() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Remove the redundant GFP_DMA flag. Link: https://lore.kernel.org/20260903111836.1777265-4-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: Harry Yoo --- sound/mips/snd-n64.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/sound/mips/snd-n64.c b/sound/mips/snd-n64.c index f17e63f2ff5a67..e2cf9df15d485c 100644 --- a/sound/mips/snd-n64.c +++ b/sound/mips/snd-n64.c @@ -298,7 +298,7 @@ static int __init n64audio_probe(struct platform_device *pdev) priv->card = card; priv->ring_base = dma_alloc_coherent(card->dev, 32 * 1024, &priv->ring_base_dma, - GFP_DMA|GFP_KERNEL); + GFP_KERNEL); if (!priv->ring_base) { err = -ENOMEM; goto fail_card; From d5bcab18c7f312c56c15f255971f7a67c7d4055d Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:27 +0800 Subject: [PATCH 1188/1352] spi: spi-ti-qspi: don't use GFP_DMA when calling dma_alloc_coherent() dma_alloc_coherent() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Remove the redundant GFP_DMA flag. Link: https://lore.kernel.org/20260903111836.1777265-5-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: Harry Yoo --- drivers/spi/spi-ti-qspi.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/spi/spi-ti-qspi.c b/drivers/spi/spi-ti-qspi.c index 34b154922ff24d..fb0d940883fb93 100644 --- a/drivers/spi/spi-ti-qspi.c +++ b/drivers/spi/spi-ti-qspi.c @@ -855,7 +855,7 @@ static int ti_qspi_probe(struct platform_device *pdev) qspi->rx_bb_addr = dma_alloc_coherent(qspi->dev, QSPI_DMA_BUFFER_SIZE, &qspi->rx_bb_dma_addr, - GFP_KERNEL | GFP_DMA); + GFP_KERNEL); if (!qspi->rx_bb_addr) { dev_err(qspi->dev, "dma_alloc_coherent failed, using PIO mode\n"); From 8e7d315616dc9b012907c872f43f712cfa6a29d4 Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:28 +0800 Subject: [PATCH 1189/1352] fbdev: fsl-diu-fb: don't use GFP_DMA when calling dmam_alloc_coherent() dmam_alloc_coherent() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Remove the redundant GFP_DMA flag. Link: https://lore.kernel.org/20260903111836.1777265-6-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: Harry Yoo --- drivers/video/fbdev/fsl-diu-fb.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/video/fbdev/fsl-diu-fb.c b/drivers/video/fbdev/fsl-diu-fb.c index b71d15794ce8b8..d7ed007915c189 100644 --- a/drivers/video/fbdev/fsl-diu-fb.c +++ b/drivers/video/fbdev/fsl-diu-fb.c @@ -1690,7 +1690,7 @@ static int fsl_diu_probe(struct platform_device *pdev) int ret; data = dmam_alloc_coherent(&pdev->dev, sizeof(struct fsl_diu_data), - &dma_addr, GFP_DMA | __GFP_ZERO); + &dma_addr, __GFP_ZERO); if (!data) return -ENOMEM; data->dma_addr = dma_addr; From 6d070526aa55b4dd963bb82bdf195194ccef45e3 Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:29 +0800 Subject: [PATCH 1190/1352] usb: gadget: lpc32xx_udc: don't use GFP_DMA when calling dma_alloc_coherent() dma_alloc_coherent() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Remove the redundant GFP_DMA flag. Link: https://lore.kernel.org/20260903111836.1777265-7-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: Harry Yoo --- drivers/usb/gadget/udc/lpc32xx_udc.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/usb/gadget/udc/lpc32xx_udc.c b/drivers/usb/gadget/udc/lpc32xx_udc.c index 044c31869cfb26..90d1eef528770e 100644 --- a/drivers/usb/gadget/udc/lpc32xx_udc.c +++ b/drivers/usb/gadget/udc/lpc32xx_udc.c @@ -3080,7 +3080,7 @@ static int lpc32xx_udc_probe(struct platform_device *pdev) /* Allocate memory for the UDCA */ udc->udca_v_base = dma_alloc_coherent(&pdev->dev, UDCA_BUFF_SIZE, &dma_handle, - (GFP_KERNEL | GFP_DMA)); + GFP_KERNEL); if (!udc->udca_v_base) { dev_err(udc->dev, "error getting UDCA region\n"); retval = -ENOMEM; From 28f0067734ab3f794111b19685bd46e49554ca54 Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:30 +0800 Subject: [PATCH 1191/1352] usb: cdns3: don't use GFP_DMA when calling dma_alloc_coherent() dma_alloc_coherent() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Use GFP_KERNEL instead so the allocation may reclaim as usual for probe-time allocations. Link: https://lore.kernel.org/20260903111836.1777265-8-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: Harry Yoo --- drivers/usb/cdns3/cdns3-gadget.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/usb/cdns3/cdns3-gadget.c b/drivers/usb/cdns3/cdns3-gadget.c index 42311c1bfada1d..484d3128cf081c 100644 --- a/drivers/usb/cdns3/cdns3-gadget.c +++ b/drivers/usb/cdns3/cdns3-gadget.c @@ -3376,7 +3376,7 @@ static int cdns3_gadget_start(struct cdns *cdns) /* allocate memory for setup packet buffer */ priv_dev->setup_buf = dma_alloc_coherent(priv_dev->sysdev, 8, - &priv_dev->setup_dma, GFP_DMA); + &priv_dev->setup_dma, GFP_KERNEL); if (!priv_dev->setup_buf) { ret = -ENOMEM; goto err2; From 2d1920d04d694e8844fb98edf6f4bcbda0eaace9 Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:31 +0800 Subject: [PATCH 1192/1352] media: staging: imx: don't use GFP_DMA when calling dma_alloc_coherent() dma_alloc_coherent() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Remove the redundant GFP_DMA flag. Link: https://lore.kernel.org/20260903111836.1777265-9-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: Harry Yoo --- drivers/staging/media/imx/imx-media-utils.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/staging/media/imx/imx-media-utils.c b/drivers/staging/media/imx/imx-media-utils.c index f119477cac6b16..7bdcb88a686b1c 100644 --- a/drivers/staging/media/imx/imx-media-utils.c +++ b/drivers/staging/media/imx/imx-media-utils.c @@ -588,7 +588,7 @@ int imx_media_alloc_dma_buf(struct device *dev, buf->len = PAGE_ALIGN(size); buf->virt = dma_alloc_coherent(dev, buf->len, &buf->phys, - GFP_DMA | GFP_KERNEL); + GFP_KERNEL); if (!buf->virt) return -ENOMEM; From f0d86d0659ddc3f9a523df1a42735d52923a258a Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:32 +0800 Subject: [PATCH 1193/1352] spi: atmel: don't use GFP_DMA when calling dma_alloc_coherent() dma_alloc_coherent() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Remove the redundant GFP_DMA flag. Link: https://lore.kernel.org/20260903111836.1777265-10-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: Harry Yoo --- drivers/spi/spi-atmel.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/drivers/spi/spi-atmel.c b/drivers/spi/spi-atmel.c index c8012c82c3a788..e91e84fdbe8f00 100644 --- a/drivers/spi/spi-atmel.c +++ b/drivers/spi/spi-atmel.c @@ -624,7 +624,7 @@ static int atmel_spi_configure_dma(struct spi_controller *host, if (IS_ENABLED(CONFIG_SOC_SAM_V4_V5)) { as->addr_tx_bbuf = dma_alloc_coherent(dev, SPI_MAX_DMA_XFER, &as->dma_addr_tx_bbuf, - GFP_KERNEL | GFP_DMA); + GFP_KERNEL); if (!as->addr_tx_bbuf) { err = -ENOMEM; goto err_release_dma; @@ -632,7 +632,7 @@ static int atmel_spi_configure_dma(struct spi_controller *host, as->addr_rx_bbuf = dma_alloc_coherent(dev, SPI_MAX_DMA_XFER, &as->dma_addr_rx_bbuf, - GFP_KERNEL | GFP_DMA); + GFP_KERNEL); if (!as->addr_rx_bbuf) { err = -ENOMEM; goto err_release_dma; From 5c0b688701ddea51c58d1cf197ce79e38f53055f Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:33 +0800 Subject: [PATCH 1194/1352] media: imx7-media-csi: don't use GFP_DMA when calling dma_alloc_coherent() dma_alloc_coherent() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Remove the redundant GFP_DMA flag. Link: https://lore.kernel.org/20260903111836.1777265-11-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Reviewed-by: Frank Li Cc: Christoph Hellwig Cc: Harry Yoo --- drivers/media/platform/nxp/imx7-media-csi.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/media/platform/nxp/imx7-media-csi.c b/drivers/media/platform/nxp/imx7-media-csi.c index 7ddc7ba06e3d4e..22c0cbdc92bfb4 100644 --- a/drivers/media/platform/nxp/imx7-media-csi.c +++ b/drivers/media/platform/nxp/imx7-media-csi.c @@ -466,7 +466,7 @@ static int imx7_csi_alloc_dma_buf(struct imx7_csi *csi, buf->len = PAGE_ALIGN(size); buf->virt = dma_alloc_coherent(csi->dev, buf->len, &buf->dma_addr, - GFP_DMA | GFP_KERNEL); + GFP_KERNEL); if (!buf->virt) return -ENOMEM; From 1ef05e12b0b7daddb25c06cde2fdd3b41c501844 Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:34 +0800 Subject: [PATCH 1195/1352] media: nxp: imx8-isi: don't use GFP_DMA when calling dma_alloc_coherent() dma_alloc_coherent() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Remove the redundant GFP_DMA flag. Link: https://lore.kernel.org/20260903111836.1777265-12-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Reviewed-by: Frank Li Cc: Christoph Hellwig Cc: Harry Yoo --- drivers/media/platform/nxp/imx8-isi/imx8-isi-video.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/media/platform/nxp/imx8-isi/imx8-isi-video.c b/drivers/media/platform/nxp/imx8-isi/imx8-isi-video.c index f45c2aae59ce90..fc907d357149ee 100644 --- a/drivers/media/platform/nxp/imx8-isi/imx8-isi-video.c +++ b/drivers/media/platform/nxp/imx8-isi/imx8-isi-video.c @@ -773,7 +773,7 @@ static int mxc_isi_video_alloc_discard_buffers(struct mxc_isi_video *video) buf->size = PAGE_ALIGN(video->pix.plane_fmt[i].sizeimage); buf->addr = dma_alloc_coherent(video->pipe->isi->dev, buf->size, - &buf->dma, GFP_DMA | GFP_KERNEL); + &buf->dma, GFP_KERNEL); if (!buf->addr) { mxc_isi_video_free_discard_buffers(video); return -ENOMEM; From cee8d94fd8fd9711959799dd6fbe5d5d46cace2a Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:35 +0800 Subject: [PATCH 1196/1352] mtd: rawnand: gpmi: don't use GFP_DMA when calling dma_alloc_coherent() dma_alloc_coherent() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Use GFP_KERNEL instead so the allocation may reclaim as usual for probe-time allocations. Link: https://lore.kernel.org/20260903111836.1777265-13-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: Harry Yoo --- drivers/mtd/nand/raw/gpmi-nand/gpmi-nand.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/mtd/nand/raw/gpmi-nand/gpmi-nand.c b/drivers/mtd/nand/raw/gpmi-nand/gpmi-nand.c index 527165ccc839da..0d91950ae9fde6 100644 --- a/drivers/mtd/nand/raw/gpmi-nand/gpmi-nand.c +++ b/drivers/mtd/nand/raw/gpmi-nand/gpmi-nand.c @@ -1387,7 +1387,7 @@ static int gpmi_alloc_dma_buffer(struct gpmi_nand_data *this) goto error_alloc; this->auxiliary_virt = dma_alloc_coherent(dev, geo->auxiliary_size, - &this->auxiliary_phys, GFP_DMA); + &this->auxiliary_phys, GFP_KERNEL); if (!this->auxiliary_virt) goto error_alloc; From ba060c766d42e949be57bca57090a65a7b40441b Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:36 +0800 Subject: [PATCH 1197/1352] usb: cdns2: don't use GFP_DMA when calling dma_alloc_coherent() dma_alloc_coherent() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Use GFP_KERNEL instead so the allocation may reclaim as usual for probe-time allocations. Link: https://lore.kernel.org/20260903111836.1777265-14-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: Harry Yoo --- drivers/usb/gadget/udc/cdns2/cdns2-gadget.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/usb/gadget/udc/cdns2/cdns2-gadget.c b/drivers/usb/gadget/udc/cdns2/cdns2-gadget.c index 308d3c468ab197..8719f1f86e6106 100644 --- a/drivers/usb/gadget/udc/cdns2/cdns2-gadget.c +++ b/drivers/usb/gadget/udc/cdns2/cdns2-gadget.c @@ -2341,7 +2341,7 @@ static int cdns2_gadget_start(struct cdns2_device *pdev) /* Allocate memory for setup packet buffer. */ buf = dma_alloc_coherent(pdev->dev, 8, &pdev->ep0_preq.request.dma, - GFP_DMA); + GFP_KERNEL); pdev->ep0_preq.request.buf = buf; if (!pdev->ep0_preq.request.buf) { From 6b37ebfea4dbe443db0b2973e3baafe737f474fa Mon Sep 17 00:00:00 2001 From: Sang-Heon Jeon Date: Tue, 29 Sep 2026 23:37:06 +0900 Subject: [PATCH 1198/1352] mm/memory: remove unused vmf_insert_mixed_mkwrite() Since commit 38607c62b34b ("fs/dax: properly refcount fs dax pages"), vmf_insert_mixed_mkwrite() has no callers, so the mkwrite argument of __vm_insert_mixed() and insert_pfn() is always false. So remove the function, the mkwrite argument and the unreachable mkwrite branches. Also merge __vm_insert_mixed() into vmf_insert_mixed(). No functional change. Link: https://lore.kernel.org/20260929143707.450805-1-ekffu200098@gmail.com Signed-off-by: Sang-Heon Jeon Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: "Jose A. Perez de Azpillaga" --- include/linux/mm.h | 2 -- mm/memory.c | 76 ++++++++++++---------------------------------- 2 files changed, 19 insertions(+), 59 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index e21244ff29117a..30995f1d3bcba7 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -5015,8 +5015,6 @@ vm_fault_t vmf_insert_pfn_prot(struct vm_area_struct *vma, unsigned long addr, unsigned long pfn, pgprot_t pgprot); vm_fault_t vmf_insert_mixed(struct vm_area_struct *vma, unsigned long addr, unsigned long pfn); -vm_fault_t vmf_insert_mixed_mkwrite(struct vm_area_struct *vma, - unsigned long addr, unsigned long pfn); int vm_iomap_memory(struct vm_area_struct *vma, phys_addr_t start, unsigned long len); static inline vm_fault_t vmf_insert_page(struct vm_area_struct *vma, diff --git a/mm/memory.c b/mm/memory.c index 330cde31bf8b40..1f83a26f873335 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -2453,7 +2453,14 @@ static int insert_page_into_pte_locked(struct vm_area_struct *vma, pte_t *pte, if (!mkwrite) return -EBUSY; - /* see insert_pfn(). */ + /* + * For read faults on private mappings the PFN passed in may + * not match the PFN we have mapped if the mapped PFN is a + * writeable COW page. In the mkwrite case we are creating a + * writable PTE for a shared mapping and we expect the PFNs to + * match. If they don't match, we are likely racing with block + * allocation and mapping invalidation. + */ if (pte_pfn(pteval) != page_to_pfn(page)) { WARN_ON_ONCE(!is_zero_pfn(pte_pfn(pteval))); return -EFAULT; @@ -2858,7 +2865,7 @@ int vm_map_pages_zero(struct vm_area_struct *vma, struct page **pages, EXPORT_SYMBOL(vm_map_pages_zero); static vm_fault_t insert_pfn(struct vm_area_struct *vma, unsigned long addr, - unsigned long pfn, pgprot_t prot, bool mkwrite) + unsigned long pfn, pgprot_t prot) { struct mm_struct *mm = vma->vm_mm; pte_t *pte, entry; @@ -2868,38 +2875,12 @@ static vm_fault_t insert_pfn(struct vm_area_struct *vma, unsigned long addr, if (!pte) return VM_FAULT_OOM; entry = ptep_get(pte); - if (!pte_none(entry)) { - if (mkwrite) { - /* - * For read faults on private mappings the PFN passed - * in may not match the PFN we have mapped if the - * mapped PFN is a writeable COW page. In the mkwrite - * case we are creating a writable PTE for a shared - * mapping and we expect the PFNs to match. If they - * don't match, we are likely racing with block - * allocation and mapping invalidation so just skip the - * update. - */ - if (pte_pfn(entry) != pfn) { - WARN_ON_ONCE(!is_zero_pfn(pte_pfn(entry))); - goto out_unlock; - } - entry = pte_mkyoung(entry); - entry = maybe_mkwrite(pte_mkdirty(entry), vma); - if (ptep_set_access_flags(vma, addr, pte, entry, 1)) - update_mmu_cache(vma, addr, pte); - } + if (!pte_none(entry)) goto out_unlock; - } /* Ok, finally just insert the thing.. */ entry = pte_mkspecial(pfn_pte(pfn, prot)); - if (mkwrite) { - entry = pte_mkyoung(entry); - entry = maybe_mkwrite(pte_mkdirty(entry), vma); - } - set_pte_at(mm, addr, pte, entry); update_mmu_cache(vma, addr, pte); /* XXX: why not for insert_page? */ @@ -2967,7 +2948,7 @@ vm_fault_t vmf_insert_pfn_prot(struct vm_area_struct *vma, unsigned long addr, pfnmap_setup_cachemode_pfn(pfn, &pgprot); - return insert_pfn(vma, addr, pfn, pgprot, false); + return insert_pfn(vma, addr, pfn, pgprot); } EXPORT_SYMBOL(vmf_insert_pfn_prot); @@ -2998,11 +2979,9 @@ vm_fault_t vmf_insert_pfn(struct vm_area_struct *vma, unsigned long addr, } EXPORT_SYMBOL(vmf_insert_pfn); -static bool vm_mixed_ok(struct vm_area_struct *vma, unsigned long pfn, - bool mkwrite) +static bool vm_mixed_ok(struct vm_area_struct *vma, unsigned long pfn) { - if (unlikely(is_zero_pfn(pfn)) && - (mkwrite || !vm_mixed_zeropage_allowed(vma))) + if (unlikely(is_zero_pfn(pfn)) && !vm_mixed_zeropage_allowed(vma)) return false; /* these checks mirror the abort conditions in vm_normal_page */ if (vma->vm_flags & VM_MIXEDMAP) @@ -3012,13 +2991,13 @@ static bool vm_mixed_ok(struct vm_area_struct *vma, unsigned long pfn, return false; } -static vm_fault_t __vm_insert_mixed(struct vm_area_struct *vma, - unsigned long addr, unsigned long pfn, bool mkwrite) +vm_fault_t vmf_insert_mixed(struct vm_area_struct *vma, unsigned long addr, + unsigned long pfn) { pgprot_t pgprot = vma->vm_page_prot; int err; - if (!vm_mixed_ok(vma, pfn, mkwrite)) + if (!vm_mixed_ok(vma, pfn)) return VM_FAULT_SIGBUS; if (addr < vma->vm_start || addr >= vma->vm_end) @@ -3045,9 +3024,9 @@ static vm_fault_t __vm_insert_mixed(struct vm_area_struct *vma, * result in pfn_t_has_page() == false. */ page = pfn_to_page(pfn); - err = insert_page(vma, addr, page, pgprot, mkwrite); + err = insert_page(vma, addr, page, pgprot, false); } else { - return insert_pfn(vma, addr, pfn, pgprot, mkwrite); + return insert_pfn(vma, addr, pfn, pgprot); } if (err == -ENOMEM) @@ -3057,6 +3036,7 @@ static vm_fault_t __vm_insert_mixed(struct vm_area_struct *vma, return VM_FAULT_NOPAGE; } +EXPORT_SYMBOL(vmf_insert_mixed); vm_fault_t vmf_insert_page_mkwrite(struct vm_fault *vmf, struct page *page, bool write) @@ -3078,24 +3058,6 @@ vm_fault_t vmf_insert_page_mkwrite(struct vm_fault *vmf, struct page *page, } EXPORT_SYMBOL_GPL(vmf_insert_page_mkwrite); -vm_fault_t vmf_insert_mixed(struct vm_area_struct *vma, unsigned long addr, - unsigned long pfn) -{ - return __vm_insert_mixed(vma, addr, pfn, false); -} -EXPORT_SYMBOL(vmf_insert_mixed); - -/* - * If the insertion of PTE failed because someone else already added a - * different entry in the mean time, we treat that as success as we assume - * the same entry was actually inserted. - */ -vm_fault_t vmf_insert_mixed_mkwrite(struct vm_area_struct *vma, - unsigned long addr, unsigned long pfn) -{ - return __vm_insert_mixed(vma, addr, pfn, true); -} - /* * maps a range of physical memory into the requested pages. the old * mappings are removed. any references to nonexistent pages results From 3855b40d77de85f8eae41d6ca7b239686d6dc122 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 29 Sep 2026 01:01:04 -0700 Subject: [PATCH 1199/1352] mm/damon/core: introduce damos_quota_goal->complement Patch series "mm/damon: introduce damos quota goal target metric complement flag". Aim-oriented DAMOS quota auto-tuning assumes the aggressiveness of the scheme (quota) and the quota goal metric are directly proportional. Depending on the scheme setup and usage, keeping the relationship can be challenging. For example, let's suppose a memory tiering approach for higher upper tier utilization. One common idea for that (TPP) is utilizing two schemes, one for promotion and the other one for demotion. The promotion scheme migrates hot pages from lower tier to upper tier, aiming for high utilization of the upper tier. The demotion scheme migrates cold pages from upper tier to lower tier, aiming for head room free memory of the upper tier. Both schemes and their goals are in direct proportion. However, for this kind of use case, we need to implement two different goal metrics (per-node memory utilization and free memory ratio) while essentially the free memory ratio is just a complemented value of the utilization. To avoid adding too many new metrics, we are adding metrics that turn out to be really needed for each found use case. For example, some_mem_psi_us, node_eligible_mem_bp and hugepage_mem_bp don't have their complemented value version. But it is not really difficult to expect use cases that their complemented version can be useful. For example, hugepage_mem_bp use case may need a way to reduce the hugepage ratio. That would require a complemented version of hugepage_mem_bp. Adding a new metric for each of such use cases could make the number of metrics unnecessarily high, and discourage flexible usages of DAMOS. Add a new flag, quota goal complement, to allow flexible tuning goal setup without unnecessarily increasing the number of metrics. The flag specifies whether to use the complemented value of the given goal metric for the tuning. For example, if the complement flag is set, the upper tier memory utilization ratio metric works the same as the free memory ratio metric for the tier. This patch (of 8): Introduce damos_quota_goal->complement for specifying whether to use a complemented value of the given goal target metric. Add the field to the data structure and implement essential core support. Handle the flag in the quota goal commit and current quota goal metric value retrieval. Link: https://lore.kernel.org/20260929080113.41708-1-sj@kernel.org Link: https://lore.kernel.org/20260929080113.41708-2-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Gutierrez Asier --- include/linux/damon.h | 2 ++ mm/damon/core.c | 26 +++++++++++++++++++++++++- 2 files changed, 27 insertions(+), 1 deletion(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index 6a29dc2ac8db79..42234839ce29ed 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -169,6 +169,7 @@ enum damos_quota_goal_metric { /** * struct damos_quota_goal - DAMOS scheme quota auto-tuning goal. * @metric: Metric to be used for representing the goal. + * @complement: Use the complement of the metric. * @target_value: Target value of @metric to achieve with the tuning. * @current_value: Current value of @metric. * @nid: Node id. @@ -191,6 +192,7 @@ enum damos_quota_goal_metric { */ struct damos_quota_goal { enum damos_quota_goal_metric metric; + bool complement; unsigned long target_value; unsigned long current_value; /* metric-dependent fields */ diff --git a/mm/damon/core.c b/mm/damon/core.c index 5ecbea5d71e1dc..1ab5154511e614 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -1224,6 +1224,7 @@ static int damos_commit_quota_goal( if (!src->target_value) return -EINVAL; dst->metric = src->metric; + dst->complement = src->complement; dst->target_value = src->target_value; if (dst->metric == DAMOS_QUOTA_USER_INPUT) dst->current_value = src->current_value; @@ -2960,10 +2961,18 @@ static void damos_set_psi_current_val(u64 now_psi_total, struct damos_quota_goal *goal, struct damos *s) { u64 last_psi_total = goal->last_psi_total; + unsigned long val; goal->last_psi_total = now_psi_total; if (last_psi_total != U64_MAX) { - goal->current_value = now_psi_total - last_psi_total; + val = now_psi_total - last_psi_total; + if (goal->complement) { + if (val < s->quota.reset_interval * 1000) + val = s->quota.reset_interval * 1000 - val; + else + val = 0; + } + goal->current_value = val; return; } /* uninitialized last_psi_total; make no effect this round */ @@ -3255,6 +3264,21 @@ static void damos_set_quota_goal_current_value(struct damon_ctx *c, default: break; } + if (!goal->complement) + return; + + /* updte current_value to complemented value */ + + /* for user_input, users set complemented value on their own */ + if (goal->metric == DAMOS_QUOTA_USER_INPUT) + return; + /* damos_set_psi_current_val() handles complement flag itself */ + if (goal->metric == DAMOS_QUOTA_SOME_MEM_PSI_US) + return; + if (goal->current_value < 10000) + goal->current_value = 10000 - goal->current_value; + else + goal->current_value = 0; } /* Return the highest score since it makes schemes least aggressive */ From 006e6a8592e123724500036a809acb9616a5b85a Mon Sep 17 00:00:00 2001 From: Andrew Morton Date: Tue, 29 Sep 2026 13:47:25 -0700 Subject: [PATCH 1200/1352] mm-damon-core-introduce-damos_quota_goal-complement-fix fix comment tpyo, per Gutierrez Cc: Gutierrez Asier Cc: SJ Park Signed-off-by: Andrew Morton --- mm/damon/core.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 1ab5154511e614..0d27f08b350e6e 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -3267,7 +3267,7 @@ static void damos_set_quota_goal_current_value(struct damon_ctx *c, if (!goal->complement) return; - /* updte current_value to complemented value */ + /* update current_value to complemented value */ /* for user_input, users set complemented value on their own */ if (goal->metric == DAMOS_QUOTA_USER_INPUT) From ee0e98d442d083aa19fddb3904e3e0cc8d5bf571 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 29 Sep 2026 01:01:05 -0700 Subject: [PATCH 1201/1352] mm/damon/core: add complement argument to damos_new_quota_goal() damos_quota_goal->complement needs to be manually set by each API callers. It is easy to make mistakes. Extend the quota goal constructor, damos_new_quota_goal() to receive and set the complement flag value. Also update all callers to use the new signature. Link: https://lore.kernel.org/20260929080113.41708-3-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Asier Gutierrez --- include/linux/damon.h | 2 +- mm/damon/core.c | 7 ++++--- mm/damon/lru_sort.c | 5 +++-- mm/damon/reclaim.c | 5 +++-- mm/damon/sysfs-schemes.c | 2 +- mm/damon/tests/core-kunit.h | 3 ++- samples/damon/mtier.c | 2 +- 7 files changed, 15 insertions(+), 11 deletions(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index 42234839ce29ed..63050eb2206a0f 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -1090,7 +1090,7 @@ bool damos_filter_for_ops(enum damos_filter_type type); void damos_destroy_filter(struct damos_filter *f); struct damos_quota_goal *damos_new_quota_goal( - enum damos_quota_goal_metric metric, + enum damos_quota_goal_metric metric, bool complement, unsigned long target_value); void damos_add_quota_goal(struct damos_quota *q, struct damos_quota_goal *g); void damos_destroy_quota_goal(struct damos_quota_goal *goal); diff --git a/mm/damon/core.c b/mm/damon/core.c index 0d27f08b350e6e..b63e60ef899016 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -691,7 +691,7 @@ void damos_destroy_filter(struct damos_filter *f) } struct damos_quota_goal *damos_new_quota_goal( - enum damos_quota_goal_metric metric, + enum damos_quota_goal_metric metric, bool complement, unsigned long target_value) { struct damos_quota_goal *goal; @@ -700,6 +700,7 @@ struct damos_quota_goal *damos_new_quota_goal( if (!goal) return NULL; goal->metric = metric; + goal->complement = complement; goal->target_value = target_value; if (metric == DAMOS_QUOTA_SOME_MEM_PSI_US) goal->last_psi_total = U64_MAX; @@ -1262,8 +1263,8 @@ int damos_commit_quota_goals(struct damos_quota *dst, struct damos_quota *src) damos_for_each_quota_goal_safe(src_goal, next, src) { if (j++ < i) continue; - new_goal = damos_new_quota_goal( - src_goal->metric, src_goal->target_value); + new_goal = damos_new_quota_goal(src_goal->metric, + src_goal->complement, src_goal->target_value); if (!new_goal) return -ENOMEM; err = damos_commit_quota_goal(new_goal, src_goal); diff --git a/mm/damon/lru_sort.c b/mm/damon/lru_sort.c index 273efa3c913ed4..64e086985eb557 100644 --- a/mm/damon/lru_sort.c +++ b/mm/damon/lru_sort.c @@ -233,12 +233,13 @@ static int damon_lru_sort_add_quota_goals(struct damos *hot_scheme, if (!active_mem_bp) return 0; - goal = damos_new_quota_goal(DAMOS_QUOTA_ACTIVE_MEM_BP, active_mem_bp); + goal = damos_new_quota_goal(DAMOS_QUOTA_ACTIVE_MEM_BP, false, + active_mem_bp); if (!goal) return -ENOMEM; damos_add_quota_goal(&hot_scheme->quota, goal); /* aim 0.2 % goal conflict, to keep little ping pong */ - goal = damos_new_quota_goal(DAMOS_QUOTA_INACTIVE_MEM_BP, + goal = damos_new_quota_goal(DAMOS_QUOTA_INACTIVE_MEM_BP, false, 10000 - active_mem_bp + 2); if (!goal) return -ENOMEM; diff --git a/mm/damon/reclaim.c b/mm/damon/reclaim.c index 42a2c9cb134310..014b0779ea6ddd 100644 --- a/mm/damon/reclaim.c +++ b/mm/damon/reclaim.c @@ -233,7 +233,7 @@ static int damon_reclaim_apply_parameters(void) damon_set_schemes(param_ctx, &scheme, 1); if (quota_mem_pressure_us) { - goal = damos_new_quota_goal(DAMOS_QUOTA_SOME_MEM_PSI_US, + goal = damos_new_quota_goal(DAMOS_QUOTA_SOME_MEM_PSI_US, false, quota_mem_pressure_us); if (!goal) goto out; @@ -241,7 +241,8 @@ static int damon_reclaim_apply_parameters(void) } if (quota_autotune_feedback) { - goal = damos_new_quota_goal(DAMOS_QUOTA_USER_INPUT, 10000); + goal = damos_new_quota_goal(DAMOS_QUOTA_USER_INPUT, false, + 10000); if (!goal) goto out; goal->current_value = quota_autotune_feedback; diff --git a/mm/damon/sysfs-schemes.c b/mm/damon/sysfs-schemes.c index bfb6f0bc3f2138..06af417bc9a2f4 100644 --- a/mm/damon/sysfs-schemes.c +++ b/mm/damon/sysfs-schemes.c @@ -2869,7 +2869,7 @@ static int damos_sysfs_add_quota_score( if (!sysfs_goal->target_value) continue; - goal = damos_new_quota_goal(sysfs_goal->metric, + goal = damos_new_quota_goal(sysfs_goal->metric, false, sysfs_goal->target_value); if (!goal) return -ENOMEM; diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index 2111faa581532a..5a924edb171fe5 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -1014,6 +1014,7 @@ static void damos_test_commit_quota_goals_for(struct kunit *test, * Make it kfree()-able. */ goal = damos_new_quota_goal(dst_goals[i].metric, + dst_goals[i].complement, dst_goals[i].target_value); if (!goal) goto out; @@ -2411,7 +2412,7 @@ static void damos_test_esz_goal_temporal(struct kunit *test) } damon_add_scheme(ctx, s); - goal = damos_new_quota_goal(DAMOS_QUOTA_USER_INPUT, 10000); + goal = damos_new_quota_goal(DAMOS_QUOTA_USER_INPUT, false, 10000); if (!goal) { damon_destroy_ctx(ctx); kunit_skip(test, "quota goal alloc fail"); diff --git a/samples/damon/mtier.c b/samples/damon/mtier.c index 27dc88bdf7a0ef..a2e311082cd4b5 100644 --- a/samples/damon/mtier.c +++ b/samples/damon/mtier.c @@ -163,7 +163,7 @@ static struct damon_ctx *damon_sample_mtier_build_ctx(bool promote) damon_set_schemes(ctx, &scheme, 1); quota_goal = damos_new_quota_goal( promote ? DAMOS_QUOTA_NODE_MEM_USED_BP : - DAMOS_QUOTA_NODE_MEM_FREE_BP, + DAMOS_QUOTA_NODE_MEM_FREE_BP, false, promote ? node0_mem_used_bp : node0_mem_free_bp); if (!quota_goal) goto free_out; From 3bc5ea38315c7ade277b26737a8383f4ce9c4b7a Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 29 Sep 2026 01:01:06 -0700 Subject: [PATCH 1202/1352] mm/damon/sysfs-schemes: support quota goal complement flag Add a new sysfs file, complement, under the quota goal directory. It works for setting and getting the quota goal metric complement flag value. Link: https://lore.kernel.org/20260929080113.41708-4-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton --- mm/damon/sysfs-schemes.c | 32 +++++++++++++++++++++++++++++++- 1 file changed, 31 insertions(+), 1 deletion(-) diff --git a/mm/damon/sysfs-schemes.c b/mm/damon/sysfs-schemes.c index 06af417bc9a2f4..8f083611741fdb 100644 --- a/mm/damon/sysfs-schemes.c +++ b/mm/damon/sysfs-schemes.c @@ -1220,6 +1220,7 @@ static const struct kobj_type damon_sysfs_watermarks_ktype = { struct damos_sysfs_quota_goal { struct kobject kobj; enum damos_quota_goal_metric metric; + bool complement; unsigned long target_value; unsigned long current_value; int nid; @@ -1316,6 +1317,30 @@ static ssize_t target_metric_store(struct kobject *kobj, return -EINVAL; } +static ssize_t complement_show(struct kobject *kobj, + struct kobj_attribute *attr, char *buf) +{ + struct damos_sysfs_quota_goal *goal = container_of(kobj, + struct damos_sysfs_quota_goal, kobj); + + return sysfs_emit(buf, "%c\n", goal->complement ? 'Y' : 'N'); +} + +static ssize_t complement_store(struct kobject *kobj, + struct kobj_attribute *attr, const char *buf, size_t count) +{ + struct damos_sysfs_quota_goal *goal = container_of(kobj, + struct damos_sysfs_quota_goal, kobj); + bool complement; + int err = kstrtobool(buf, &complement); + + if (err) + return err; + + goal->complement = complement; + return count; +} + static ssize_t target_value_show(struct kobject *kobj, struct kobj_attribute *attr, char *buf) { @@ -1424,6 +1449,9 @@ static void damos_sysfs_quota_goal_release(struct kobject *kobj) static struct kobj_attribute damos_sysfs_quota_goal_target_metric_attr = __ATTR_RW_MODE(target_metric, 0600); +static struct kobj_attribute damos_sysfs_quota_goal_complement_attr = + __ATTR_RW_MODE(complement, 0600); + static struct kobj_attribute damos_sysfs_quota_goal_target_value_attr = __ATTR_RW_MODE(target_value, 0600); @@ -1438,6 +1466,7 @@ static struct kobj_attribute damos_sysfs_quota_goal_path_attr = static struct attribute *damos_sysfs_quota_goal_attrs[] = { &damos_sysfs_quota_goal_target_metric_attr.attr, + &damos_sysfs_quota_goal_complement_attr.attr, &damos_sysfs_quota_goal_target_value_attr.attr, &damos_sysfs_quota_goal_current_value_attr.attr, &damos_sysfs_quota_goal_nid_attr.attr, @@ -2869,7 +2898,8 @@ static int damos_sysfs_add_quota_score( if (!sysfs_goal->target_value) continue; - goal = damos_new_quota_goal(sysfs_goal->metric, false, + goal = damos_new_quota_goal(sysfs_goal->metric, + sysfs_goal->complement, sysfs_goal->target_value); if (!goal) return -ENOMEM; From e679cd183de8863ab1b9be9251292cbe4cc19e59 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 29 Sep 2026 01:01:07 -0700 Subject: [PATCH 1203/1352] mm/damon/tests/core-kunit: test quota_goal->complement commit Extend existing DAMOS quota goal commit unit test to test the complement flag. Set the source complement flag and confirm the destination is updated to the given input. Link: https://lore.kernel.org/20260929080113.41708-5-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow --- mm/damon/tests/core-kunit.h | 2 ++ 1 file changed, 2 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index 5a924edb171fe5..ef146ca2ae8aa3 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -861,6 +861,7 @@ static void damos_test_commit_quota_goal_for(struct kunit *test, damos_commit_quota_goal(dst, src); KUNIT_EXPECT_EQ(test, dst->metric, src->metric); + KUNIT_EXPECT_EQ(test, dst->complement, src->complement); KUNIT_EXPECT_EQ(test, dst->target_value, src->target_value); if (src->metric == DAMOS_QUOTA_USER_INPUT) KUNIT_EXPECT_EQ(test, dst->current_value, src->current_value); @@ -904,6 +905,7 @@ static void damos_test_commit_quota_goal(struct kunit *test) damos_test_commit_quota_goal_for(test, &dst, &(struct damos_quota_goal){ .metric = DAMOS_QUOTA_USER_INPUT, + .complement = true, .target_value = 789, .current_value = 12}); damos_test_commit_quota_goal_for(test, &dst, From d2c4ff3e35f780f52c96286c6bfecdaa22a7792a Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 29 Sep 2026 01:01:08 -0700 Subject: [PATCH 1204/1352] selftests/damon/sysfs.sh: test quota goal complement flag file Test the existence and the valid input acceptance of the newly added quota goal complement flag sysfs file. Link: https://lore.kernel.org/20260929080113.41708-6-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Shuah Khan --- tools/testing/selftests/damon/sysfs.sh | 3 +++ 1 file changed, 3 insertions(+) diff --git a/tools/testing/selftests/damon/sysfs.sh b/tools/testing/selftests/damon/sysfs.sh index b66593c9ac471f..1cc7ab2d7e22f8 100755 --- a/tools/testing/selftests/damon/sysfs.sh +++ b/tools/testing/selftests/damon/sysfs.sh @@ -212,6 +212,9 @@ test_goal() ensure_write_succ "$fpath" "node_eligible_mem_bp" "valid input" ensure_write_succ "$fpath" "hugepage_mem_bp" "valid input" ensure_write_fail "$fpath" "foo" "invalid input" + ensure_file "$goal_dir/complement" "exist" "600" + ensure_write_succ "$goal_dir/complement" "Y" "valid input" + ensure_write_succ "$goal_dir/complement" "N" "valid input" ensure_file "$goal_dir/nid" "exist" "600" ensure_file "$goal_dir/path" "exist" "600" } From ab6590f22ba10d1f972bd5526618c8d6af2bc10d Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 29 Sep 2026 01:01:09 -0700 Subject: [PATCH 1205/1352] Docs/mm/damon/design: document damos quota goal complement flag Update DAMON design document for the newly added damos quota goal metric value complement flag. Link: https://lore.kernel.org/20260929080113.41708-7-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Liam R. Howlett Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/mm/damon/design.rst | 15 +++++++++++---- 1 file changed, 11 insertions(+), 4 deletions(-) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index e82390e77a70ae..5cd651b1b7aa86 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -702,10 +702,10 @@ There are two such tuning algorithms that users can select as they need. that the zero quota is a valid quota, and therefore ``qt_exceeds`` :ref:`stat ` will keep increasing in this case. -The goal can be specified with five parameters, namely ``target_metric``, -``target_value``, ``current_value``, ``nid`` and ``path``. The auto-tuning -mechanism tries to make ``current_value`` of ``target_metric`` be same to -``target_value``. +The goal can be specified with six parameters, namely ``target_metric``, +``complement``, ``target_value``, ``current_value``, ``nid`` and ``path``. The +auto-tuning mechanism tries to make ``current_value`` of ``complement``-ed +``target_metric`` be same to ``target_value``. - ``user_input``: User-provided value. Users could use any metric that they has interest in for the value. Use space main workload's latency or @@ -733,6 +733,13 @@ mechanism tries to make ``current_value`` of ``target_metric`` be same to - ``hugepage_mem_bp``: Total huge page to total used memory ratio in bp (1/10,000). +``complement`` is a boolean parameter that determines whether to use +complemented value of the target metric. For example, if ``complement`` is set +and target metric is ``active_mem_bp``, it is effectively same to +``inactive_mem_bp``. ``complement`` is no-op when the target metric type is +``user_input``. For the metric type, the user should be able to emit +complemented metric values on their own. + ``nid`` is optionally required for ``node_mem_used_bp``, ``node_mem_free_bp``, ``node_memcg_used_bp``, ``node_memcg_free_bp`` and ``node_eligible_mem_bp`` to point the specific NUMA node. From b57da64859bf3201929d71594803d30de95429b9 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 29 Sep 2026 01:01:10 -0700 Subject: [PATCH 1206/1352] Docs/admin-guide/mm/damon/usage: update for quota goal complement file Update DAMON usage document for the newly added quota goal metric complement sysfs file. Link: https://lore.kernel.org/20260929080113.41708-8-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Liam R. Howlett Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/admin-guide/mm/damon/usage.rst | 15 ++++++++------- 1 file changed, 8 insertions(+), 7 deletions(-) diff --git a/Documentation/admin-guide/mm/damon/usage.rst b/Documentation/admin-guide/mm/damon/usage.rst index ba47255448564b..3d43954c7046c8 100644 --- a/Documentation/admin-guide/mm/damon/usage.rst +++ b/Documentation/admin-guide/mm/damon/usage.rst @@ -98,7 +98,8 @@ comma (","). │ │ │ │ │ │ │ fail_charge_num,fail_charge_denom │ │ │ │ │ │ │ │ weights/sz_permil,nr_accesses_permil,age_permil │ │ │ │ │ │ │ │ :ref:`goals `/nr_goals - │ │ │ │ │ │ │ │ │ 0/target_metric,target_value,current_value,nid,path + │ │ │ │ │ │ │ │ │ 0/target_metric,complement,target_value, + │ │ │ │ │ │ │ │ │ current_value,nid,path │ │ │ │ │ │ │ :ref:`watermarks `/metric,interval_us,high,mid,low │ │ │ │ │ │ │ :ref:`{core_,ops_,}filters `/nr_filters │ │ │ │ │ │ │ │ 0/type,matching,allow,memcg_path,addr_start,addr_end,damon_target_idx,min,max @@ -492,12 +493,12 @@ number (``N``) to the file creates the number of child directories named ``0`` to ``N-1``. Each directory represents each goal and current achievement. Among the multiple feedback, the best one is used. -Each goal directory contains five files, namely ``target_metric``, -``target_value``, ``current_value``, ``nid``, and ``path``. Users can set and -get the five parameters for the quota auto-tuning goals that specified on the -:ref:`design doc ` by writing to and -reading from each of the files. Because the kernel does not update -``current_value``, reading it only makes sense when ``target_metric`` is +Each goal directory contains six files, namely ``target_metric``, +``complement``, ``target_value``, ``current_value``, ``nid``, and ``path``. +Users can set and get the six parameters for the quota auto-tuning goals that +specified on the :ref:`design doc ` by +writing to and reading from each of the files. Because the kernel does not +update ``current_value``, reading it only makes sense when ``target_metric`` is ``user_input``. Note that users should further write ``commit_schemes_quota_goals`` to the ``state`` file of the :ref:`kdamond directory ` to pass the feedback to DAMON. From c34249ca5c09b07f0db0c093a200f575fa9be121 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 29 Sep 2026 01:01:11 -0700 Subject: [PATCH 1207/1352] Docs/ABI/damon: update for quota goal metric complement sysfs file Update DAMON ABI document for the newly added damos quota goal metric complement sysfs file. Link: https://lore.kernel.org/20260929080113.41708-9-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Liam R. Howlett Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/ABI/testing/sysfs-kernel-mm-damon | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/Documentation/ABI/testing/sysfs-kernel-mm-damon b/Documentation/ABI/testing/sysfs-kernel-mm-damon index 55df688ea596ff..4acb2015c2f4f3 100644 --- a/Documentation/ABI/testing/sysfs-kernel-mm-damon +++ b/Documentation/ABI/testing/sysfs-kernel-mm-damon @@ -375,6 +375,12 @@ Contact: SJ Park Description: Writing to and reading from this file sets and gets the quota auto-tuning goal metric. +What: /sys/kernel/mm/damon/admin/kdamonds//contexts//schemes//quotas/goals//complement +Date: Sep 2026 +Contact: SJ Park +Description: Writing to and reading from this file sets and gets whether to + use the complement value of the quota auto-tuning goal metric. + What: /sys/kernel/mm/damon/admin/kdamonds//contexts//schemes//quotas/goals//target_value Date: Nov 2023 Contact: SJ Park From c8df7f10e01272ab93b29814c620d5cb35e89e50 Mon Sep 17 00:00:00 2001 From: Pooyan Azad Date: Tue, 29 Sep 2026 09:18:46 +0200 Subject: [PATCH 1208/1352] zram: fix short reads from block_state read_block_state() formats each entry directly into the buffer supplied by read(). If the remaining buffer is too small for one complete record, snprintf() returns the full record length and the function stops without copying data or advancing the file position. A read from this debugfs file which is smaller than a record therefore returns zero at a non-EOF position and cannot make progress. Convert block_state to seq_file so formatted records are buffered independently of the userspace read size. Keep dev_lock held across each seq_file iteration and continue to protect individual entries with their slot locks. Link: https://lore.kernel.org/20260929071846.24829-1-pooyan.azadparvar@gmail.com Fixes: c0265342bff4 ("zram: introduce zram memory tracking") Signed-off-by: Pooyan Azad Signed-off-by: Andrew Morton Closes: https://lore.kernel.org/r/CANC3H+LdtoydSp+o2ecErAw7k6R2+gRf9LyxcaoHv_mGhJmyQQ@mail.gmail.com/ Reviewed-by: Sergey Senozhatsky Tested-by: Sergey Senozhatsky Cc: Minchan Kim Cc: Jens Axboe --- drivers/block/zram/zram_drv.c | 102 +++++++++++++++++----------------- 1 file changed, 52 insertions(+), 50 deletions(-) diff --git a/drivers/block/zram/zram_drv.c b/drivers/block/zram/zram_drv.c index 024402438bd1d4..ab064c84b0922b 100644 --- a/drivers/block/zram/zram_drv.c +++ b/drivers/block/zram/zram_drv.c @@ -1553,68 +1553,70 @@ static void zram_debugfs_destroy(void) debugfs_remove_recursive(zram_debugfs_root); } -static ssize_t read_block_state(struct file *file, char __user *buf, - size_t count, loff_t *ppos) +static void *zram_block_state_start(struct seq_file *seq, loff_t *pos) { - char *kbuf; - unsigned long index; - ssize_t written = 0; - struct zram *zram = file->private_data; + struct zram *zram = seq->private; unsigned long nr_pages; - kbuf = kvmalloc(count, GFP_KERNEL); - if (!kbuf) - return -ENOMEM; - - guard(rwsem_read)(&zram->dev_lock); - if (!init_done(zram)) { - kvfree(kbuf); - return -EINVAL; - } + down_read(&zram->dev_lock); + if (!init_done(zram)) + return ERR_PTR(-EINVAL); nr_pages = zram->disksize >> PAGE_SHIFT; + if (*pos >= nr_pages) + return NULL; - for (index = *ppos; index < nr_pages; index++) { - int copied; + return pos; +} - slot_lock(zram, index); - if (!slot_allocated(zram, index)) - goto next; +static void *zram_block_state_next(struct seq_file *seq, void *v, loff_t *pos) +{ + struct zram *zram = seq->private; + unsigned long nr_pages = zram->disksize >> PAGE_SHIFT; - copied = snprintf(kbuf + written, count, - "%12lu %12u.%06d %c%c%c%c%c%c\n", - index, zram->table[index].attr.ac_time, 0, - test_slot_flag(zram, index, ZRAM_SAME) ? 's' : '.', - test_slot_flag(zram, index, ZRAM_WB) ? 'w' : '.', - test_slot_flag(zram, index, ZRAM_HUGE) ? 'h' : '.', - test_slot_flag(zram, index, ZRAM_IDLE) ? 'i' : '.', - get_slot_comp_priority(zram, index) ? 'r' : '.', - test_slot_flag(zram, index, - ZRAM_INCOMPRESSIBLE) ? 'n' : '.'); - - if (count <= copied) { - slot_unlock(zram, index); - break; - } - written += copied; - count -= copied; -next: - slot_unlock(zram, index); - *ppos += 1; - } + ++*pos; + if (*pos >= nr_pages) + return NULL; + + return pos; +} + +static void zram_block_state_stop(struct seq_file *seq, void *v) +{ + struct zram *zram = seq->private; - if (copy_to_user(buf, kbuf, written)) - written = -EFAULT; - kvfree(kbuf); + up_read(&zram->dev_lock); +} + +static int zram_block_state_show(struct seq_file *seq, void *v) +{ + struct zram *zram = seq->private; + unsigned long index = *(loff_t *)v; + + slot_lock(zram, index); + if (slot_allocated(zram, index)) { + seq_printf(seq, "%12lu %12u.%06d %c%c%c%c%c%c\n", + index, zram->table[index].attr.ac_time, 0, + test_slot_flag(zram, index, ZRAM_SAME) ? 's' : '.', + test_slot_flag(zram, index, ZRAM_WB) ? 'w' : '.', + test_slot_flag(zram, index, ZRAM_HUGE) ? 'h' : '.', + test_slot_flag(zram, index, ZRAM_IDLE) ? 'i' : '.', + get_slot_comp_priority(zram, index) ? 'r' : '.', + test_slot_flag(zram, index, + ZRAM_INCOMPRESSIBLE) ? 'n' : '.'); + } + slot_unlock(zram, index); - return written; + return 0; } -static const struct file_operations proc_zram_block_state_op = { - .open = simple_open, - .read = read_block_state, - .llseek = default_llseek, +static const struct seq_operations zram_block_state_sops = { + .start = zram_block_state_start, + .next = zram_block_state_next, + .stop = zram_block_state_stop, + .show = zram_block_state_show, }; +DEFINE_SEQ_ATTRIBUTE(zram_block_state); static void zram_debugfs_register(struct zram *zram) { @@ -1624,7 +1626,7 @@ static void zram_debugfs_register(struct zram *zram) zram->debugfs_dir = debugfs_create_dir(zram->disk->disk_name, zram_debugfs_root); debugfs_create_file("block_state", 0400, zram->debugfs_dir, - zram, &proc_zram_block_state_op); + zram, &zram_block_state_fops); } static void zram_debugfs_unregister(struct zram *zram) From 2f8b0f5824dc36ec5b8c73fecc3802320ee12f3e Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Tue, 29 Sep 2026 13:32:26 +0800 Subject: [PATCH 1209/1352] mm/sparse-vmemmap: drop VMEMMAP_POPULATE_DAX Patch series "mm: Unify device DAX and HugeTLB vmemmap population paths", v3. This series is split out from the earlier, larger series "mm: Generalize HVO for HugeTLB and device DAX" [1]. While the parent series generalizes vmemmap optimization across HugeTLB and device DAX, this subset addresses a single, self-contained step: unifying their vmemmap population paths. After the preceding Device DAX conversion, both HugeTLB and Device DAX describe optimized vmemmap mappings through memory-section metadata and use per-zone shared tail vmemmap pages. The generic code, however, still carries a Device DAX-specific population flag and compound-page population path, along with arguments and helpers needed only by that path. This series first removes VMEMMAP_POPULATE_DAX and moves selection and reference handling for the shared tail page into the common vmemmap population path. It then removes the generic Device DAX-specific compound-page population path and routes section vmemmap population through vmemmap_populate(). The powerpc radix path continues to use its architecture-specific compound-page population implementation for optimizable sections. The remaining patches remove the unused ptpfn argument, open-code vmemmap_populate_address() now that no caller needs its returned PTE, and add a warning for inconsistent zone initialization of shared tail vmemmap pages. This is the fourth smaller step toward the broader HVO generalization. After this series, HugeTLB and Device DAX use the same population model instead of parallel generic paths, while powerpc keeps its architecture-specific implementation. This patch (of 6): VMEMMAP_POPULATE_DAX currently distinguishes DAX vmemmap population in two places: it keeps allocations on the normal path and takes a reference when a backing page is supplied for reuse. After Device DAX switched to the common per-zone shared tail page, both conditions can be determined locally. DAX supplies ptpfn for every shared tail mapping and requests an allocation only for compound head mappings, whose PFNs are not optimizable. Therefore, vmemmap_optimizable_pfn() alone selects the correct allocation path. When ptpfn is supplied, the caller is reusing an existing backing page. Once the slab allocator is available, take a reference for each reused mapping to balance the release performed by vmemmap_free(). Although the buddy allocator is available before slab, no vmemmap population occurs in that interval. Earlier mappings are backed by memblock/reserved memory and do not need page reference accounting. Remove VMEMMAP_POPULATE_DAX and the flags argument from the vmemmap population helpers. Link: https://lore.kernel.org/20260929053231.66085-2-songmuchun@bytedance.com Link: https://lore.kernel.org/20260513130542.35604-1-songmuchun@bytedance.com/ [1] Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Reviewed-by: Lance Yang Cc: Madhavan Srinivasan Cc: Mike Rapoport Cc: David Hildenbrand Cc: Michael Ellerman Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Ritesh Harjani (IBM) Cc: Shrikanth Hegde Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko --- mm/sparse-vmemmap.c | 36 +++++++++++++----------------------- 1 file changed, 13 insertions(+), 23 deletions(-) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 09897586152386..b23efff80137f4 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -32,12 +32,6 @@ #include #include -/* - * Flags for vmemmap_populate_range and friends. - */ -/* Vmemmap population for ZONE_DEVICE compound pages */ -#define VMEMMAP_POPULATE_DAX 0x0001 - #include "internal.h" #include "mm_init.h" #include "sparse.h" @@ -241,7 +235,7 @@ struct page __ref *vmemmap_shared_tail_page(unsigned int order, struct zone *zon #endif static __meminit void *vmemmap_alloc_pte(unsigned long pfn, int node, - struct vmem_altmap *altmap, unsigned long flags) + struct vmem_altmap *altmap) { struct zone *zone; struct page *page; @@ -251,7 +245,7 @@ static __meminit void *vmemmap_alloc_pte(unsigned long pfn, int node, * Device DAX still relies on vmemmap_populate_compound_pages() for * head/first-tail allocation and tail-page reuse. */ - if (!vmemmap_optimizable_pfn(pfn) || flags & VMEMMAP_POPULATE_DAX) + if (!vmemmap_optimizable_pfn(pfn)) return vmemmap_alloc_block_buf(PAGE_SIZE, node, altmap); zone = pfn_to_zone(pfn, node); @@ -263,8 +257,7 @@ static __meminit void *vmemmap_alloc_pte(unsigned long pfn, int node, } static pte_t * __meminit vmemmap_pte_populate(pmd_t *pmd, unsigned long addr, int node, - struct vmem_altmap *altmap, - unsigned long ptpfn, unsigned long flags) + struct vmem_altmap *altmap, unsigned long ptpfn) { pte_t *pte = pte_offset_kernel(pmd, addr); unsigned long pfn = page_to_pfn((struct page *)addr); @@ -273,7 +266,7 @@ static pte_t * __meminit vmemmap_pte_populate(pmd_t *pmd, unsigned long addr, in pte_t entry; if (ptpfn == (unsigned long)-1) { - void *p = vmemmap_alloc_pte(pfn, node, altmap, flags); + void *p = vmemmap_alloc_pte(pfn, node, altmap); if (!p) return NULL; @@ -291,7 +284,7 @@ static pte_t * __meminit vmemmap_pte_populate(pmd_t *pmd, unsigned long addr, in * Use try_get_page() to prevent the shared page refcount * from overflowing. */ - if ((flags & VMEMMAP_POPULATE_DAX) && + if (slab_is_available() && !try_get_page(pfn_to_page(ptpfn))) return NULL; } @@ -355,8 +348,7 @@ static pgd_t * __meminit vmemmap_pgd_populate(unsigned long addr, int node) static pte_t * __meminit vmemmap_populate_address(unsigned long addr, int node, struct vmem_altmap *altmap, - unsigned long ptpfn, - unsigned long flags) + unsigned long ptpfn) { pgd_t *pgd; p4d_t *p4d; @@ -376,7 +368,7 @@ static pte_t * __meminit vmemmap_populate_address(unsigned long addr, int node, pmd = vmemmap_pmd_populate(pud, addr, node); if (!pmd) return NULL; - pte = vmemmap_pte_populate(pmd, addr, node, altmap, ptpfn, flags); + pte = vmemmap_pte_populate(pmd, addr, node, altmap, ptpfn); if (!pte) return NULL; vmemmap_verify(pte, node, addr, addr + PAGE_SIZE); @@ -387,15 +379,14 @@ static pte_t * __meminit vmemmap_populate_address(unsigned long addr, int node, static int __meminit vmemmap_populate_range(unsigned long start, unsigned long end, int node, struct vmem_altmap *altmap, - unsigned long ptpfn, - unsigned long flags) + unsigned long ptpfn) { unsigned long addr = start; pte_t *pte; for (; addr < end; addr += PAGE_SIZE) { pte = vmemmap_populate_address(addr, node, altmap, - ptpfn, flags); + ptpfn); if (!pte) return -ENOMEM; } @@ -406,7 +397,7 @@ static int __meminit vmemmap_populate_range(unsigned long start, int __meminit vmemmap_populate_basepages(unsigned long start, unsigned long end, int node, struct vmem_altmap *altmap) { - return vmemmap_populate_range(start, end, node, altmap, -1, 0); + return vmemmap_populate_range(start, end, node, altmap, -1); } /* @@ -532,7 +523,6 @@ static int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, unsigned long end, int node, struct dev_pagemap *pgmap) { - const unsigned long flags = VMEMMAP_POPULATE_DAX; const unsigned int order = pfn_to_section_compound_order(start_pfn); unsigned long size, addr; pte_t *pte; @@ -545,14 +535,14 @@ static int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, if (reuse_compound_section(start_pfn, pgmap)) return vmemmap_populate_range(start, end, node, NULL, - page_to_pfn(page), flags); + page_to_pfn(page)); size = min(end - start, (1UL << order) * sizeof(struct page)); for (addr = start; addr < end; addr += size) { unsigned long next, last = addr + size; /* Populate the head page vmemmap page */ - pte = vmemmap_populate_address(addr, node, NULL, -1, flags); + pte = vmemmap_populate_address(addr, node, NULL, -1); if (!pte) return -ENOMEM; @@ -562,7 +552,7 @@ static int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, */ next = addr + PAGE_SIZE; rc = vmemmap_populate_range(next, last, node, NULL, - page_to_pfn(page), flags); + page_to_pfn(page)); if (rc) return -ENOMEM; } From 72a152523ff347dd253864be045854662dc23b9b Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Tue, 29 Sep 2026 13:32:27 +0800 Subject: [PATCH 1210/1352] mm/sparse-vmemmap: support device DAX in common vmemmap path The common vmemmap population path cannot yet handle optimized Device DAX mappings on its own. It uses pfn_to_zone() to find the shared tail page, but Device DAX populates its vmemmap at runtime before the ZONE_DEVICE span is initialized. Teach the common path to use device_zone() for runtime optimized vmemmap population while retaining pfn_to_zone() for early boot. This allows the same path to support both early boot mappings and Device DAX. The backing PFN supplied by the Device DAX-specific population path is no longer used, allowing the redundant lookup and population code to be removed later. Link: https://lore.kernel.org/20260929053231.66085-3-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Reviewed-by: Lance Yang Cc: Madhavan Srinivasan Cc: Mike Rapoport Cc: David Hildenbrand Cc: Michael Ellerman Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Ritesh Harjani (IBM) Cc: Shrikanth Hegde Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko --- mm/sparse-vmemmap.c | 64 +++++++++++++++++++++++++-------------------- 1 file changed, 35 insertions(+), 29 deletions(-) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index b23efff80137f4..62327ac1520646 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -241,18 +241,43 @@ static __meminit void *vmemmap_alloc_pte(unsigned long pfn, int node, struct page *page; const unsigned int order = pfn_to_section_compound_order(pfn); - /* - * Device DAX still relies on vmemmap_populate_compound_pages() for - * head/first-tail allocation and tail-page reuse. - */ if (!vmemmap_optimizable_pfn(pfn)) return vmemmap_alloc_block_buf(PAGE_SIZE, node, altmap); - zone = pfn_to_zone(pfn, node); + /* + * Before slab is available, vmemmap optimization is used for early + * system RAM, whose zone can be determined from the PFN. + * + * Once slab is available, only ZONE_DEVICE memory reaches this + * optimized population path. Its zone span has not been initialized + * while its vmemmap is being populated, so pfn_to_zone() cannot be + * used. Obtain ZONE_DEVICE directly from the node instead. + */ + zone = slab_is_available() ? device_zone(node) : pfn_to_zone(pfn, node); page = vmemmap_shared_tail_page(order, zone); if (!page) return NULL; + /* + * During early vmemmap population, the shared tail vmemmap backing + * page is allocated from memblock before its struct page can safely + * participate in page refcounting. Therefore, no reference can be + * held for each shared PTE mapping, and the mappings must be unshared + * before the vmemmap is depopulated. + * + * Once slab is available, the shared backing page is allocated from + * the buddy allocator and can be refcounted. Hold one reference for + * each shared PTE mapping. The architecture vmemmap teardown drops + * the reference through __free_pages() when removing the mapping, + * preventing the backing page from being freed while it is shared. + * + * The backing page may be shared by enough PTE mappings to exhaust + * the positive range of its reference count. Stop populating the + * vmemmap if another reference cannot be acquired. + */ + if (slab_is_available() && !try_get_page(page)) + return NULL; + return page_address(page); } @@ -264,31 +289,12 @@ static pte_t * __meminit vmemmap_pte_populate(pmd_t *pmd, unsigned long addr, in if (pte_none(ptep_get(pte))) { pte_t entry; + void *p = vmemmap_alloc_pte(pfn, node, altmap); - if (ptpfn == (unsigned long)-1) { - void *p = vmemmap_alloc_pte(pfn, node, altmap); - - if (!p) - return NULL; - ptpfn = PHYS_PFN(__pa(p)); - } else { - /* - * When a PTE/PMD entry is freed from the init_mm - * there's a free_pages() call to this page allocated - * above. Thus this try_get_page() is paired with the - * put_page_testzero() on the freeing path. - * This can only called by certain ZONE_DEVICE path, - * and through vmemmap_populate_compound_pages() when - * slab is available. - * - * Use try_get_page() to prevent the shared page refcount - * from overflowing. - */ - if (slab_is_available() && - !try_get_page(pfn_to_page(ptpfn))) - return NULL; - } - entry = pfn_pte(ptpfn, PAGE_KERNEL); + if (!p) + return NULL; + + entry = pfn_pte(PHYS_PFN(__pa(p)), PAGE_KERNEL); set_pte_at(&init_mm, addr, pte, entry); } else if (WARN_ON_ONCE(vmemmap_optimizable_pfn(pfn))) return NULL; From 3d1fa884991966394e0ba2c7118cd58c8d1a6b52 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Tue, 29 Sep 2026 13:32:28 +0800 Subject: [PATCH 1211/1352] mm/sparse-vmemmap: drop Device DAX-specific population path The common vmemmap path selects the shared page for optimized mappings itself, so Device DAX no longer needs vmemmap_populate_compound_pages() to find a shared tail page and pass its backing PFN through the generic population helpers. Remove the Device DAX-specific population path and let section memmap population always use vmemmap_populate(). The powerpc retains an architecture-specific compound-page implementation, so select it directly from radix__vmemmap_populate() for optimizable sections. Link: https://lore.kernel.org/20260929053231.66085-4-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Reviewed-by: Lance Yang Cc: Madhavan Srinivasan Cc: Mike Rapoport Cc: David Hildenbrand Cc: Michael Ellerman Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Ritesh Harjani (IBM) Cc: Shrikanth Hegde Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko --- arch/powerpc/mm/book3s64/radix_pgtable.c | 3 + mm/mm_init.c | 2 +- mm/sparse-vmemmap.c | 71 +----------------------- 3 files changed, 5 insertions(+), 71 deletions(-) diff --git a/arch/powerpc/mm/book3s64/radix_pgtable.c b/arch/powerpc/mm/book3s64/radix_pgtable.c index a3332f32ffb25e..44868b62b14ecb 100644 --- a/arch/powerpc/mm/book3s64/radix_pgtable.c +++ b/arch/powerpc/mm/book3s64/radix_pgtable.c @@ -1126,7 +1126,10 @@ int __meminit radix__vmemmap_populate(unsigned long start, unsigned long end, in pud_t *pud; pmd_t *pmd; pte_t *pte; + unsigned long pfn = page_to_pfn((struct page *)start); + if (vmemmap_optimizable_order(pfn_to_section_compound_order(pfn))) + return vmemmap_populate_compound_pages(pfn, start, end, node, NULL); /* * If altmap is present, Make sure we align the start vmemmap addr * to PAGE_SIZE so that we calculate the correct start_pfn in diff --git a/mm/mm_init.c b/mm/mm_init.c index 56bb4567a49405..1650d6bc1211c3 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -1046,7 +1046,7 @@ static void zone_device_page_init_from_template(struct page *page, * initialize is a lot smaller that the total amount of struct pages being * mapped. This is a paired / mild layering violation with explicit knowledge * of how the sparse_vmemmap internals handle compound pages in the lack - * of an altmap. See vmemmap_populate_compound_pages(). + * of an altmap. */ static inline unsigned long compound_nr_pages(unsigned long pfn, struct dev_pagemap *pgmap) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 62327ac1520646..270b5def58e19d 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -503,71 +503,6 @@ int __meminit vmemmap_populate_hugepages(unsigned long start, unsigned long end, return 0; } -#ifndef vmemmap_populate_compound_pages -/* - * For compound pages bigger than section size (e.g. x86 1G compound - * pages with 2M subsection size) fill the rest of sections as tail - * pages. - * - * Note that memremap_pages() resets @nr_range value and will increment - * it after each range successful onlining. Thus the value or @nr_range - * at section memmap populate corresponds to the in-progress range - * being onlined here. - */ -static bool __meminit reuse_compound_section(unsigned long start_pfn, - struct dev_pagemap *pgmap) -{ - unsigned long nr_pages = pgmap_vmemmap_nr(pgmap); - unsigned long offset = start_pfn - - PHYS_PFN(pgmap->ranges[pgmap->nr_range].start); - - return !IS_ALIGNED(offset, nr_pages) && nr_pages > PAGES_PER_SUBSECTION; -} - -static int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, - unsigned long start, - unsigned long end, int node, - struct dev_pagemap *pgmap) -{ - const unsigned int order = pfn_to_section_compound_order(start_pfn); - unsigned long size, addr; - pte_t *pte; - struct page *page; - int rc; - - page = vmemmap_shared_tail_page(order, device_zone(node)); - if (!page) - return -ENOMEM; - - if (reuse_compound_section(start_pfn, pgmap)) - return vmemmap_populate_range(start, end, node, NULL, - page_to_pfn(page)); - - size = min(end - start, (1UL << order) * sizeof(struct page)); - for (addr = start; addr < end; addr += size) { - unsigned long next, last = addr + size; - - /* Populate the head page vmemmap page */ - pte = vmemmap_populate_address(addr, node, NULL, -1); - if (!pte) - return -ENOMEM; - - /* - * Reuse the shared page for the rest of tail pages - * See layout diagram in Documentation/mm/vmemmap_dedup.rst - */ - next = addr + PAGE_SIZE; - rc = vmemmap_populate_range(next, last, node, NULL, - page_to_pfn(page)); - if (rc) - return -ENOMEM; - } - - return 0; -} - -#endif - struct page * __meminit __populate_section_memmap(unsigned long pfn, unsigned long nr_pages, int nid, struct vmem_altmap *altmap, struct dev_pagemap *pgmap) @@ -580,11 +515,7 @@ struct page * __meminit __populate_section_memmap(unsigned long pfn, !IS_ALIGNED(nr_pages, PAGES_PER_SUBSECTION))) return NULL; - if (pgmap && section_vmemmap_optimizable(__pfn_to_section(pfn))) - r = vmemmap_populate_compound_pages(pfn, start, end, nid, pgmap); - else - r = vmemmap_populate(start, end, nid, altmap); - + r = vmemmap_populate(start, end, nid, altmap); if (r < 0) return NULL; From 44e62fee2903925a25c808a34af9946100fc3603 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Tue, 29 Sep 2026 13:32:29 +0800 Subject: [PATCH 1212/1352] mm/sparse-vmemmap: remove the unused ptpfn argument vmemmap_pte_populate() no longer uses ptpfn as an input. Drop the argument to simplify the code. Link: https://lore.kernel.org/20260929053231.66085-5-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Reviewed-by: Lance Yang Cc: Madhavan Srinivasan Cc: Mike Rapoport Cc: David Hildenbrand Cc: Michael Ellerman Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Ritesh Harjani (IBM) Cc: Shrikanth Hegde Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko --- mm/sparse-vmemmap.c | 15 ++++++--------- 1 file changed, 6 insertions(+), 9 deletions(-) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 270b5def58e19d..07188cdf00f659 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -282,7 +282,7 @@ static __meminit void *vmemmap_alloc_pte(unsigned long pfn, int node, } static pte_t * __meminit vmemmap_pte_populate(pmd_t *pmd, unsigned long addr, int node, - struct vmem_altmap *altmap, unsigned long ptpfn) + struct vmem_altmap *altmap) { pte_t *pte = pte_offset_kernel(pmd, addr); unsigned long pfn = page_to_pfn((struct page *)addr); @@ -353,8 +353,7 @@ static pgd_t * __meminit vmemmap_pgd_populate(unsigned long addr, int node) } static pte_t * __meminit vmemmap_populate_address(unsigned long addr, int node, - struct vmem_altmap *altmap, - unsigned long ptpfn) + struct vmem_altmap *altmap) { pgd_t *pgd; p4d_t *p4d; @@ -374,7 +373,7 @@ static pte_t * __meminit vmemmap_populate_address(unsigned long addr, int node, pmd = vmemmap_pmd_populate(pud, addr, node); if (!pmd) return NULL; - pte = vmemmap_pte_populate(pmd, addr, node, altmap, ptpfn); + pte = vmemmap_pte_populate(pmd, addr, node, altmap); if (!pte) return NULL; vmemmap_verify(pte, node, addr, addr + PAGE_SIZE); @@ -384,15 +383,13 @@ static pte_t * __meminit vmemmap_populate_address(unsigned long addr, int node, static int __meminit vmemmap_populate_range(unsigned long start, unsigned long end, int node, - struct vmem_altmap *altmap, - unsigned long ptpfn) + struct vmem_altmap *altmap) { unsigned long addr = start; pte_t *pte; for (; addr < end; addr += PAGE_SIZE) { - pte = vmemmap_populate_address(addr, node, altmap, - ptpfn); + pte = vmemmap_populate_address(addr, node, altmap); if (!pte) return -ENOMEM; } @@ -403,7 +400,7 @@ static int __meminit vmemmap_populate_range(unsigned long start, int __meminit vmemmap_populate_basepages(unsigned long start, unsigned long end, int node, struct vmem_altmap *altmap) { - return vmemmap_populate_range(start, end, node, altmap, -1); + return vmemmap_populate_range(start, end, node, altmap); } /* From 33607100d7e8a4932417e678087a4c206d981766 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Tue, 29 Sep 2026 13:32:30 +0800 Subject: [PATCH 1213/1352] mm/sparse-vmemmap: open-code vmemmap_populate_address() vmemmap_populate_address() no longer has any callers that need the returned PTE. Its only remaining user, vmemmap_populate_range(), only checks whether population succeeded. Open-code vmemmap_populate_address() directly in vmemmap_populate_basepages(), remove the now-redundant range helper, and return -ENOMEM directly on failure. Link: https://lore.kernel.org/20260929053231.66085-6-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Reviewed-by: Lance Yang Cc: Madhavan Srinivasan Cc: Mike Rapoport Cc: David Hildenbrand Cc: Michael Ellerman Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Ritesh Harjani (IBM) Cc: Shrikanth Hegde Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko --- mm/sparse-vmemmap.c | 54 ++++++++++++++------------------------------- 1 file changed, 17 insertions(+), 37 deletions(-) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 07188cdf00f659..40eba70c0048ea 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -352,8 +352,8 @@ static pgd_t * __meminit vmemmap_pgd_populate(unsigned long addr, int node) return pgd; } -static pte_t * __meminit vmemmap_populate_address(unsigned long addr, int node, - struct vmem_altmap *altmap) +int __meminit vmemmap_populate_basepages(unsigned long start, unsigned long end, + int node, struct vmem_altmap *altmap) { pgd_t *pgd; p4d_t *p4d; @@ -361,48 +361,28 @@ static pte_t * __meminit vmemmap_populate_address(unsigned long addr, int node, pmd_t *pmd; pte_t *pte; - pgd = vmemmap_pgd_populate(addr, node); - if (!pgd) - return NULL; - p4d = vmemmap_p4d_populate(pgd, addr, node); - if (!p4d) - return NULL; - pud = vmemmap_pud_populate(p4d, addr, node); - if (!pud) - return NULL; - pmd = vmemmap_pmd_populate(pud, addr, node); - if (!pmd) - return NULL; - pte = vmemmap_pte_populate(pmd, addr, node, altmap); - if (!pte) - return NULL; - vmemmap_verify(pte, node, addr, addr + PAGE_SIZE); - - return pte; -} - -static int __meminit vmemmap_populate_range(unsigned long start, - unsigned long end, int node, - struct vmem_altmap *altmap) -{ - unsigned long addr = start; - pte_t *pte; - - for (; addr < end; addr += PAGE_SIZE) { - pte = vmemmap_populate_address(addr, node, altmap); + for (unsigned long addr = start; addr < end; addr += PAGE_SIZE) { + pgd = vmemmap_pgd_populate(addr, node); + if (!pgd) + return -ENOMEM; + p4d = vmemmap_p4d_populate(pgd, addr, node); + if (!p4d) + return -ENOMEM; + pud = vmemmap_pud_populate(p4d, addr, node); + if (!pud) + return -ENOMEM; + pmd = vmemmap_pmd_populate(pud, addr, node); + if (!pmd) + return -ENOMEM; + pte = vmemmap_pte_populate(pmd, addr, node, altmap); if (!pte) return -ENOMEM; + vmemmap_verify(pte, node, addr, addr + PAGE_SIZE); } return 0; } -int __meminit vmemmap_populate_basepages(unsigned long start, unsigned long end, - int node, struct vmem_altmap *altmap) -{ - return vmemmap_populate_range(start, end, node, altmap); -} - /* * Write protect the mirrored tail page structs for HVO. This will be * called from the hugetlb code when gathering and initializing the From 89ffc0ff2b1d09b51f7b5f1ec78c0d9e5c980fa7 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Tue, 29 Sep 2026 13:32:31 +0800 Subject: [PATCH 1214/1352] mm/mm_init: add zone mismatch warning during page init For vmemmap-optimized sections, tail struct pages may be backed by shared vmemmap pages. Those shared pages must carry the same page zone ID as the struct pages initialized for the section. Warn in __init_single_page() if the shared tail page has a different page_zone_id(), which would indicate inconsistent initialization. Link: https://lore.kernel.org/20260929053231.66085-7-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Reviewed-by: Lance Yang Cc: Madhavan Srinivasan Cc: Mike Rapoport Cc: David Hildenbrand Cc: Michael Ellerman Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Ritesh Harjani (IBM) Cc: Shrikanth Hegde Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko --- mm/mm_init.c | 3 +++ 1 file changed, 3 insertions(+) diff --git a/mm/mm_init.c b/mm/mm_init.c index 1650d6bc1211c3..bd02e8d0696581 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -609,6 +609,9 @@ void __meminit __init_single_page(struct page *page, unsigned long pfn, if (!is_highmem_idx(zone)) set_page_address(page, __va(pfn << PAGE_SHIFT)); #endif + VM_WARN_ON_ONCE(vmemmap_optimizable_order(pfn_to_section_compound_order(pfn)) && + page_zone_id(page + VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES) != + page_zone_id(page)); } #ifdef CONFIG_NUMA From 40cdf2b57d6c6914170fbbd006aff29febfb3382 Mon Sep 17 00:00:00 2001 From: Ren Tamura Date: Wed, 30 Sep 2026 15:03:17 +0900 Subject: [PATCH 1215/1352] tools/cgroup: sum shrinker object counts across NUMA nodes Each row of a memcg-aware shrinker's debugfs count file contains the cgroup ID followed by one object count per NUMA node. memcg_shrinker.py uses only the first count when sorting and displaying entries. This undercounts objects on NUMA systems. An entry with counts of 0 and 23 is treated as zero and omitted from the output, even though its total is 23. Sum all count fields after the cgroup ID so that the displayed totals and sort order account for every node. Single-node input retains the same result. Link: https://lore.kernel.org/179074810112.139422.9274126921568167859@gmail.com Fixes: d261ea23533b ("tools: add memcg_shrinker.py") Signed-off-by: Ren Tamura Signed-off-by: Andrew Morton Assisted-by: LLM Cc: Roman Gushchin --- tools/cgroup/memcg_shrinker.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tools/cgroup/memcg_shrinker.py b/tools/cgroup/memcg_shrinker.py index e81c3017ada964..714772449c6ba9 100644 --- a/tools/cgroup/memcg_shrinker.py +++ b/tools/cgroup/memcg_shrinker.py @@ -31,7 +31,7 @@ def scan_shrinkers(shrinker_debugfs): items = line.split(' ') ino = int(items[0]) # (count, shrinker, memcg ino) - shrinkers.append((int(items[1]), shrinker, ino)) + shrinkers.append((sum(map(int, items[1:])), shrinker, ino)) return shrinkers From a5c18dba7bc6a20b697e4e0924ad57eaee43cd0f Mon Sep 17 00:00:00 2001 From: DAI RENJIE Date: Fri, 21 Aug 2026 14:08:17 +0000 Subject: [PATCH 1216/1352] resource: fix lost wakeup when waiting for a muxed region A task waiting for a muxed region can sleep forever in TASK_UNINTERRUPTIBLE even though the region it waits for is already free. __request_region_locked() queues itself on muxed_resource_wait and drops resource_lock before setting TASK_UNINTERRUPTIBLE, while __release_region() wakes the queue after dropping the same lock. A wakeup landing in between finds TASK_RUNNING, does not match TASK_NORMAL and is discarded; callers hold a muxed region only across a bounded transaction, so no further release is coming. The task is unkillable and its caller never returns. The window is one store wide, but an interrupt is enough to hold the waiter in it, and the machine this was seen on runs PREEMPT_DYNAMIC in its voluntary default. Since v6.11 spd5118 exports the DDR5 sensors of AMD boards through i2c-piix4, which takes a muxed region per SMBus transaction; a third of the in-tree users of request_muxed_region() are hwmon drivers, so reading a world-readable attribute is all an unprivileged user needs to drive the contention. The blocked task sleeps holding the i2c adapter bus lock, and 27 more piled up behind it. Reproduced by building a kernel with the two orderings selectable at runtime and a 2ms delay inside the window. Switching only that knob, a two-thread barriered reproducer loses the wakeup 200 times out of 200 before the fix and 0 out of 200 after it; without the delay it goes 20000 times through the wait path and loses none. Fix it by setting the task state before dropping resource_lock, as prepare_to_wait() does: the releasing side needs resource_lock to unlink the resource, so it cannot reach the wakeup before the state is published. Link: https://lore.kernel.org/20260821-b4-resource-muxed-lost-wakeup-v1-1-37eb6473a76c@gmail.com Fixes: 8b6d043b7ee2 ("resource: shared I/O region support") Signed-off-by: DAI RENJIE Signed-off-by: Andrew Morton Reviewed-by: Bradley Morgan Reviewed-by: Andrew Morton Assisted-by: Claude:claude-opus-5 Cc: Mark Brown Cc: Kees Cook Cc: Bjorn Helgaas Cc: --- kernel/resource.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/kernel/resource.c b/kernel/resource.c index e60539a55541df..54d7199695fbe9 100644 --- a/kernel/resource.c +++ b/kernel/resource.c @@ -1350,8 +1350,8 @@ static int __request_region_locked(struct resource *res, struct resource *parent } if (conflict->flags & flags & IORESOURCE_MUXED) { add_wait_queue(&muxed_resource_wait, &wait); - write_unlock(&resource_lock); set_current_state(TASK_UNINTERRUPTIBLE); + write_unlock(&resource_lock); schedule(); remove_wait_queue(&muxed_resource_wait, &wait); write_lock(&resource_lock); From 3ae04413386920029b5e97f57b126f45c66b9ed4 Mon Sep 17 00:00:00 2001 From: OGAWA Hirofumi Date: Tue, 25 Aug 2026 21:11:32 +0900 Subject: [PATCH 1217/1352] fat: fix fat_ent_write() for reverting the value commit 64d9183203ee ("fat: restore original value when fat_ent_write failed") try to revert the fatent value to old value when got the error on mirror FAT. However it didn't work if the error is when writing the fatent bh. In that case, the bh is cleared the uptodate flag, so reuse bh is invalid. Fix this by reverting the fatent only if got the error on mirror FAT. Link: https://lore.kernel.org/87ik4yz9fv.fsf_-_@mail.parknet.co.jp Fixes: 64d9183203ee ("fat: restore original value when fat_ent_write failed") Signed-off-by: OGAWA Hirofumi Signed-off-by: Andrew Morton Reported-by: syzbot+e64c6472a3d96a75172a@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=e64c6472a3d96a75172a Reported-by: syzbot+26461e903494e689c24f@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=26461e903494e689c24f Cc: Yemu Lu Cc: Ren Wei Cc: Yuan Tan Cc: Yifan Wu Cc: Juefei Pu Cc: Xin Liu Cc: --- fs/fat/fat.h | 2 +- fs/fat/fatent.c | 21 ++++++++++++++++++--- fs/fat/file.c | 3 ++- fs/fat/misc.c | 6 ++---- 4 files changed, 23 insertions(+), 9 deletions(-) diff --git a/fs/fat/fat.h b/fs/fat/fat.h index 61338413d9f3e4..fbd207c55859e7 100644 --- a/fs/fat/fat.h +++ b/fs/fat/fat.h @@ -392,7 +392,7 @@ extern void fat_ent_access_init(struct super_block *sb); extern int fat_ent_read(struct inode *inode, struct fat_entry *fatent, int entry); extern int fat_ent_write(struct inode *inode, struct fat_entry *fatent, - int new, int wait); + int new, int old, int wait); extern int fat_alloc_clusters(struct inode *inode, int *cluster, int nr_cluster); extern int fat_free_clusters(struct inode *inode, int cluster); diff --git a/fs/fat/fatent.c b/fs/fat/fatent.c index f0801d99dd62ae..df23fc85f31307 100644 --- a/fs/fat/fatent.c +++ b/fs/fat/fatent.c @@ -413,7 +413,7 @@ static int fat_mirror_bhs(struct super_block *sb, struct buffer_head **bhs, } int fat_ent_write(struct inode *inode, struct fat_entry *fatent, - int new, int wait) + int new, int old, int wait) { struct super_block *sb = inode->i_sb; const struct fatent_operations *ops = MSDOS_SB(sb)->fatent_ops; @@ -422,10 +422,25 @@ int fat_ent_write(struct inode *inode, struct fat_entry *fatent, ops->ent_put(fatent, new); if (wait) { err = fat_sync_bhs(fatent->bhs, fatent->nr_bhs); - if (err) + if (err) { + /* + * bhs are not uptodate after I/O error. So we + * can't simply re-dirty to revert. And it + * would not have value to write again on I/O + * error. + */ return err; + } } - return fat_mirror_bhs(sb, fatent->bhs, fatent->nr_bhs); + + err = fat_mirror_bhs(sb, fatent->bhs, fatent->nr_bhs); + if (err) { + /* Try to revert if got the error on mirror FAT */ + ops->ent_put(fatent, old); + if (wait) + fat_sync_bhs(fatent->bhs, fatent->nr_bhs); + } + return err; } static inline int fat_ent_next(struct msdos_sb_info *sbi, diff --git a/fs/fat/file.c b/fs/fat/file.c index 1c835ca5f21a51..6c475c53334c6c 100644 --- a/fs/fat/file.c +++ b/fs/fat/file.c @@ -363,7 +363,8 @@ static int fat_free(struct inode *inode, int skip) __func__, MSDOS_I(inode)->i_pos); ret = -EIO; } else if (ret > 0) { - err = fat_ent_write(inode, &fatent, FAT_ENT_EOF, wait); + err = fat_ent_write(inode, &fatent, FAT_ENT_EOF, ret, + wait); if (err) ret = err; } diff --git a/fs/fat/misc.c b/fs/fat/misc.c index e79762cf19754d..c44296756eae61 100644 --- a/fs/fat/misc.c +++ b/fs/fat/misc.c @@ -133,11 +133,9 @@ int fat_chain_add(struct inode *inode, int new_dclus, int nr_cluster) ret = fat_ent_read(inode, &fatent, last); if (ret >= 0) { int wait = inode_needs_sync(inode); - int old = ret; - ret = fat_ent_write(inode, &fatent, new_dclus, wait); - if (ret < 0) - fat_ent_write(inode, &fatent, old, wait); + ret = fat_ent_write(inode, &fatent, new_dclus, ret, + wait); fatent_brelse(&fatent); } if (ret < 0) From a3e3527ac09d0f3ba2f96f5f9c2d0c7b0c428b67 Mon Sep 17 00:00:00 2001 From: Michael Liang Date: Fri, 21 Aug 2026 12:15:27 -0600 Subject: [PATCH 1218/1352] fault-inject: fix dentry leak fault_create_debugfs_attr() has always taken an extra dentry reference on the created directory (attr->dname = dget(dir)) so that fail_dump() could print the name via %pd from any context. Nothing anywhere in the tree ever calls dput() on attr->dname. For callers with a matching teardown, that unmatched reference causes one dentry plus its attached inode to leak per fault_create_debugfs_attr / debugfs_remove_recursive cycle. simple_recursive_removal() drops debugfs's own +1 ref on the child dentry, but the dget()'d ref keeps its refcount at 1: the dentry ends up unhashed but pinned, and its inode is never freed. Boot-once callers (mm/failslab, block/blk-core, etc.) leak exactly once at init and never destroy the tree, so the impact there is bounded. But per-lifecycle callers (drivers/nvme, drivers/infiniband/hw/hfi1, drivers/mmc, drivers/iommu/iommufd, drivers/media, drivers/misc, drivers/gpu/drm/msm, drivers/crypto, net/sunrpc) leak on every create/destroy cycle. We observed this in production: an NVMe/RDMA host repeatedly reconnecting to a target that rejected the CRTO Property Get went through ~50 nvme controller create/destroy cycles per second, and dentry and inode_cache grew by ~13k pinned objects per 240 s -- unrecoverable through drop_caches. Byte math matched a per-cycle 1-dentry / 1-inode leak from the "fault_inject" directory dentry. Fix this by not holding any external reference in fault_attr. Embed the directory name as a fixed-size char array (FAULT_ATTR_DNAME_LEN, 64 bytes) inside struct fault_attr, copied by strscpy() at fault_create_debugfs_attr() time. fail_dump() prints it via %s. Advantages of an embedded array over kstrdup() + kfree() paired with a new destroy API: - Zero API footprint. No new export and no caller changes required: callers already own their fault_attr's memory and free it when they are done, and now that suffices. - No allocation on the create path. - fault_create_debugfs_attr() cannot fail from the name-copy step. - No lifetime coupling between attr->dname and debugfs; the string is valid for exactly as long as the containing struct. The 64-byte length accommodates every in-tree caller with generous headroom (the longest current name is "fail_dma_array_full", 19 chars). The user-visible fail_dump() format changes from "name %pd" to "name %s", but the printed content is identical -- %pd on the created directory renders the same string that was passed in as @name. drivers/infiniband/hw/hfi1/fault.c drops a now-invalid "attr.dname = NULL" statement; the surrounding kzalloc() already zero-initialises the array. Link: https://lore.kernel.org/20260821181527.3271414-1-mliang@purestorage.com Fixes: 6adc4a22f20b ("fault-inject: add ratelimit option") Signed-off-by: Michael Liang Signed-off-by: Andrew Morton Reviewed-by: Andrew Morton Cc: Akinbou Mita Cc: Dennis Dalessandro Cc: Jason Gunthorpe Cc: Leon Romanovsky Cc: Vlastimil Babka Cc: --- drivers/infiniband/hw/hfi1/fault.c | 1 - include/linux/fault-inject.h | 10 ++++++++-- lib/fault-inject.c | 7 +++++-- 3 files changed, 13 insertions(+), 5 deletions(-) diff --git a/drivers/infiniband/hw/hfi1/fault.c b/drivers/infiniband/hw/hfi1/fault.c index 4ab72ef03ba11b..941a0b96590b60 100644 --- a/drivers/infiniband/hw/hfi1/fault.c +++ b/drivers/infiniband/hw/hfi1/fault.c @@ -216,7 +216,6 @@ int hfi1_fault_init_debugfs(struct hfi1_ibdev *ibd) ibd->fault->attr.interval = 1; ibd->fault->attr.require_end = ULONG_MAX; ibd->fault->attr.stacktrace_depth = 32; - ibd->fault->attr.dname = NULL; ibd->fault->attr.verbose = 0; ibd->fault->enable = false; ibd->fault->opcode = false; diff --git a/include/linux/fault-inject.h b/include/linux/fault-inject.h index 58fd14c8227080..5c74748a53f38f 100644 --- a/include/linux/fault-inject.h +++ b/include/linux/fault-inject.h @@ -18,6 +18,13 @@ enum fault_flags { #include #include +/* + * Length of the debugfs directory name embedded in struct fault_attr. + * Chosen to accommodate every in-tree caller of fault_create_debugfs_attr() + * (the longest is "fail_dma_array_full", 19 chars) with generous headroom. + */ +#define FAULT_ATTR_DNAME_LEN 64 + /* * For explanation of the elements of this struct, see * Documentation/fault-injection/fault-injection.rst @@ -37,7 +44,7 @@ struct fault_attr { unsigned long count; struct ratelimit_state ratelimit_state; - struct dentry *dname; + char dname[FAULT_ATTR_DNAME_LEN]; }; #define FAULT_ATTR_INITIALIZER { \ @@ -47,7 +54,6 @@ struct fault_attr { .stacktrace_depth = 32, \ .ratelimit_state = RATELIMIT_STATE_INIT_DISABLED, \ .verbose = 2, \ - .dname = NULL, \ } #define DECLARE_FAULT_ATTR(name) struct fault_attr name = FAULT_ATTR_INITIALIZER diff --git a/lib/fault-inject.c b/lib/fault-inject.c index 999053fa133e3f..02916ef2761c60 100644 --- a/lib/fault-inject.c +++ b/lib/fault-inject.c @@ -5,6 +5,7 @@ #include #include #include +#include #include #include #include @@ -64,7 +65,7 @@ static void fail_dump(struct fault_attr *attr) { if (attr->verbose > 0 && __ratelimit(&attr->ratelimit_state)) { printk(KERN_NOTICE "FAULT_INJECTION: forcing a failure.\n" - "name %pd, interval %lu, probability %lu, " + "name %s, interval %lu, probability %lu, " "space %d, times %d\n", attr->dname, attr->interval, attr->probability, atomic_read(&attr->space), @@ -261,7 +262,9 @@ struct dentry *fault_create_debugfs_attr(const char *name, debugfs_create_xul("reject-end", mode, dir, &attr->reject_end); #endif /* CONFIG_FAULT_INJECTION_STACKTRACE_FILTER */ - attr->dname = dget(dir); + if (strscpy(attr->dname, name, sizeof(attr->dname)) == -E2BIG) + pr_warn("FAULT_INJECTION: name '%s' truncated to '%s'\n", + name, attr->dname); return dir; } EXPORT_SYMBOL_GPL(fault_create_debugfs_attr); From 5bdddab7aec9f8508d06eacb891548781f1d0ced Mon Sep 17 00:00:00 2001 From: Andrew Morton Date: Tue, 26 May 2026 15:14:09 -0700 Subject: [PATCH 1219/1352] drivers/media/v4l2-core/v4l2-vp9.c: reduce inlining csky allmodconfig, gcc-15.2.0: drivers/media/v4l2-core/v4l2-vp9.c: In function 'v4l2_vp9_adapt_noncoef_probs': drivers/media/v4l2-core/v4l2-vp9.c:1834:1: error: the frame size of 1436 bytes is larger than 1280 bytes [-Werror=frame-larger-than=] The amount of inlining in there is simply nuts. This patch semi-randomly uninlines various things and fixes the above. Ad the .text size reduction is tremendous: ts:/usr/src/25> size drivers/media/v4l2-core/v4l2-vp9.o text data bss dec hex filename 22450 36 0 22486 57d6 drivers/media/v4l2-core/v4l2-vp9.o-before 16144 36 0 16180 3f34 drivers/media/v4l2-core/v4l2-vp9.o-after Reviewed-by: Daniel Almeida Cc: Mauro Carvalho Chehab Signed-off-by: Andrew Morton --- drivers/media/v4l2-core/v4l2-vp9.c | 30 +++++++++++++++--------------- 1 file changed, 15 insertions(+), 15 deletions(-) diff --git a/drivers/media/v4l2-core/v4l2-vp9.c b/drivers/media/v4l2-core/v4l2-vp9.c index 859589f1fd35f5..e965ffbd9b8aa2 100644 --- a/drivers/media/v4l2-core/v4l2-vp9.c +++ b/drivers/media/v4l2-core/v4l2-vp9.c @@ -1582,25 +1582,25 @@ static inline u8 noncoef_merge_prob(u8 pre_prob, u32 ct0, u32 ct1) * merge_prob(p[9], c[9], [10]) */ -static inline void merge_probs_variant_a(u8 *p, const u32 *c, u16 count_sat, u32 update_factor) +static noinline_for_stack void merge_probs_variant_a(u8 *p, const u32 *c, u16 count_sat, u32 update_factor) { p[1] = merge_prob(p[1], c[0], c[1] + c[2], count_sat, update_factor); p[2] = merge_prob(p[2], c[1], c[2], count_sat, update_factor); } -static inline void merge_probs_variant_b(u8 *p, const u32 *c, u16 count_sat, u32 update_factor) +static noinline_for_stack void merge_probs_variant_b(u8 *p, const u32 *c, u16 count_sat, u32 update_factor) { p[0] = merge_prob(p[0], c[0], c[1], count_sat, update_factor); } -static inline void merge_probs_variant_c(u8 *p, const u32 *c) +static noinline_for_stack void merge_probs_variant_c(u8 *p, const u32 *c) { p[0] = noncoef_merge_prob(p[0], c[2], c[1] + c[0] + c[3]); p[1] = noncoef_merge_prob(p[1], c[0], c[1] + c[3]); p[2] = noncoef_merge_prob(p[2], c[1], c[3]); } -static void merge_probs_variant_d(u8 *p, const u32 *c) +static noinline_for_stack void merge_probs_variant_d(u8 *p, const u32 *c) { u32 sum = 0, s2; @@ -1624,20 +1624,20 @@ static void merge_probs_variant_d(u8 *p, const u32 *c) p[8] = noncoef_merge_prob(p[8], c[6], c[7]); } -static inline void merge_probs_variant_e(u8 *p, const u32 *c) +static noinline_for_stack void merge_probs_variant_e(u8 *p, const u32 *c) { p[0] = noncoef_merge_prob(p[0], c[0], c[1] + c[2] + c[3]); p[1] = noncoef_merge_prob(p[1], c[1], c[2] + c[3]); p[2] = noncoef_merge_prob(p[2], c[2], c[3]); } -static inline void merge_probs_variant_f(u8 *p, const u32 *c) +static noinline_for_stack void merge_probs_variant_f(u8 *p, const u32 *c) { p[0] = noncoef_merge_prob(p[0], c[0], c[1] + c[2]); p[1] = noncoef_merge_prob(p[1], c[1], c[2]); } -static void merge_probs_variant_g(u8 *p, const u32 *c) +static noinline_for_stack void merge_probs_variant_g(u8 *p, const u32 *c) { u32 sum; @@ -1659,12 +1659,12 @@ static void merge_probs_variant_g(u8 *p, const u32 *c) } /* 8.4.3 Coefficient probability adaptation process */ -static inline void adapt_probs_variant_a_coef(u8 *p, const u32 *c, u32 update_factor) +static noinline_for_stack void adapt_probs_variant_a_coef(u8 *p, const u32 *c, u32 update_factor) { merge_probs_variant_a(p, c, 24, update_factor); } -static inline void adapt_probs_variant_b_coef(u8 *p, const u32 *c, u32 update_factor) +static noinline_for_stack void adapt_probs_variant_b_coef(u8 *p, const u32 *c, u32 update_factor) { merge_probs_variant_b(p, c, 24, update_factor); } @@ -1724,33 +1724,33 @@ static inline void adapt_probs_variant_b(u8 *p, const u32 *c) merge_probs_variant_b(p, c, 20, 128); } -static inline void adapt_probs_variant_c(u8 *p, const u32 *c) +static noinline_for_stack void adapt_probs_variant_c(u8 *p, const u32 *c) { merge_probs_variant_c(p, c); } -static inline void adapt_probs_variant_d(u8 *p, const u32 *c) +static noinline_for_stack void adapt_probs_variant_d(u8 *p, const u32 *c) { merge_probs_variant_d(p, c); } -static inline void adapt_probs_variant_e(u8 *p, const u32 *c) +static noinline_for_stack void adapt_probs_variant_e(u8 *p, const u32 *c) { merge_probs_variant_e(p, c); } -static inline void adapt_probs_variant_f(u8 *p, const u32 *c) +static noinline_for_stack void adapt_probs_variant_f(u8 *p, const u32 *c) { merge_probs_variant_f(p, c); } -static inline void adapt_probs_variant_g(u8 *p, const u32 *c) +static noinline_for_stack void adapt_probs_variant_g(u8 *p, const u32 *c) { merge_probs_variant_g(p, c); } /* 8.4.4 Non coefficient probability adaptation process, adapt_prob() */ -static inline u8 adapt_prob(u8 prob, const u32 counts[2]) +static noinline_for_stack u8 adapt_prob(u8 prob, const u32 counts[2]) { return noncoef_merge_prob(prob, counts[0], counts[1]); } From 4f86a014e7b7a1610acf092232411330192ebd32 Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Tue, 4 Aug 2026 09:34:00 +0000 Subject: [PATCH 1220/1352] taskstats: copy signal->stats under siglock in taskstats_exit taskstats_exit() copies tsk->signal->stats into the exit reply without taking any lock. Every other writer of this struct holds sighand->siglock before touching it, and this copy does not. The copy happens on the last thread of a thread group that exits. group_dead being 1 only says that every thread has dropped signal->live, it does not say how far the other threads got in do_exit(). One of them can still be inside fill_tgid_exit() adding its counters to the struct while the last thread copies it out, so the copy can read the struct in the middle of an update. The commit that added the copy assumed no locking was needed because the group was dead: /* No locking needed for tsk->signal->stats since group is dead */ but at that point the other threads have not necessarily finished their exit path. cpu0 (thread A, not last) cpu1 (thread B, last) =========================== ============================== atomic_dec(&signal->live) atomic_dec(&signal->live) -> 0 group_dead = 0 group_dead = 1 ... taskstats_exit(tsk, 1) taskstats_exit(tsk, 0) fill_tgid_exit(tsk) [siglock] fill_tgid_exit(tsk) memcpy(stats, signal->stats) spin_lock(siglock) reads ac_utime (new) stats->ac_utime += x reads ac_stime (old) stats->ac_stime += y torn snapshot -> netlink spin_unlock(siglock) The listeners receive a partially updated tgid snapshot, with some fields from before the concurrent update and some from after. There is no crash or splat, which is likely why this went unnoticed since 2006. A userspace model of the same shape, writer under a lock and a lockless memcpy reader, produces millions of torn reads in a few seconds. Take siglock around the copy like every other access does. sighand is still alive here because taskstats_exit() runs before exit_notify(), and fill_tgid_exit() already takes this same lock earlier in this function. Link: https://lore.kernel.org/20260804093400.3922-1-include@grrlz.net Fixes: ad4ecbcba728 ("[PATCH] delay accounting taskstats interface send tgid once") Signed-off-by: Bradley Morgan Signed-off-by: Andrew Morton Acked-by: Oleg Nesterov Cc: Balbir Singh --- kernel/taskstats.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/kernel/taskstats.c b/kernel/taskstats.c index f31df72f0e9df1..9a48827e22bce3 100644 --- a/kernel/taskstats.c +++ b/kernel/taskstats.c @@ -590,6 +590,7 @@ void taskstats_exit(struct task_struct *tsk, int group_dead) struct sk_buff *rep_skb; size_t size; int is_thread_group; + unsigned long flags; if (!family_registered) return; @@ -635,7 +636,10 @@ void taskstats_exit(struct task_struct *tsk, int group_dead) if (!stats) goto err; + /* This was racy before, copy the stats under siglock. */ + spin_lock_irqsave(&tsk->sighand->siglock, flags); memcpy(stats, tsk->signal->stats, sizeof(*stats)); + spin_unlock_irqrestore(&tsk->sighand->siglock, flags); stats->version = TASKSTATS_VERSION; send: From d2f1a67a911cd32c2859c0007d30d638e8b27d00 Mon Sep 17 00:00:00 2001 From: Petr Vorel Date: Mon, 10 Aug 2026 18:11:59 +0200 Subject: [PATCH 1221/1352] checkpatch: skip CamelCase cache for --no-tree without root Running outside tree (--no-tree) without git root (--root DIR) is not doable because we have no include/ directory which could be cached. But 3445686af721 expected that we are always in Linux tree (w/a git). But --no-tree does not require --root. Therefore skip whole caching in that case. This fixes perl and find errors when running checkpatch.pl *with* --no-tree --strict and *without* --root: No structs that should be const will be found - file 'scripts/const_structs.checkpatch': No such file or directory Use of uninitialized value $root in concatenation (.) or string at scripts/checkpatch.pl line 1213. find: `/include': No such file or directory Link: https://lore.kernel.org/20260810161159.1044160-1-pvorel@suse.cz Fixes: 3445686af721 ("checkpatch: ignore existing CamelCase uses from include/...") Signed-off-by: Petr Vorel Signed-off-by: Andrew Morton Cc: Andy Whitcroft Cc: Dwaipayan Ray Cc: Joe Perches Cc: Lukas Bulwahn --- scripts/checkpatch.pl | 2 ++ 1 file changed, 2 insertions(+) diff --git a/scripts/checkpatch.pl b/scripts/checkpatch.pl index 8a7787d228a63d..f424dafce5bce7 100755 --- a/scripts/checkpatch.pl +++ b/scripts/checkpatch.pl @@ -1206,6 +1206,8 @@ sub seed_camelcase_includes { my $git_last_include_commit = `${git_command} log --no-merges --pretty=format:"%h%n" -1 -- include`; chomp $git_last_include_commit; $camelcase_cache = ".checkpatch-camelcase.git.$git_last_include_commit"; + } elsif (not defined $root) { + return; } else { my $last_mod_date = 0; $files = `find $root/include -name "*.h"`; From 5c6dbea74880997e7a121653cadd0690bd677f2e Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Thu, 20 Aug 2026 09:19:12 -0400 Subject: [PATCH 1222/1352] USB: gadgetfs: do not WARN about excessively large memory allocations GadgetFS passes an excessively large user input len to kmalloc and kmalloc gives a WARN (see below for details). Suppress it by passing __GFP_NOWARN to kmalloc used by both ep_write_iter() and ep_read_iter(). Follow the same method as commit 4f2629ea67e72 ("USB: usbfs: Don't WARN about excessively large memory allocations"). kmalloc is used to allocate physically contiguous memory for kernel allocations. For requests larger than KMALLOC_MAX_CACHE_SIZE, kmalloc uses the page allocator and can only support up to KMALLOC_MAX_SIZE. For request sizes bigger than KMALLOC_MAX_SIZE, the page allocator can emit a WARN because kmalloc allocates an order greater than MAX_PAGE_ORDER. Link: https://lore.kernel.org/DKTTMAS94IMH.2C6ERY0ZIVWVZ@nvidia.com Fixes: b3c466ce5129 ("page allocator: do not sanity check order in the fast path") Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Reported-by: syzbot+805630f1453e490427fa@syzkaller.appspotmail.com Closes: https://lore.kernel.org/all/6a820ebc.9ebadd4d.20b15e.001b.GAE@google.com/ Tested-by: syzbot+805630f1453e490427fa@syzkaller.appspotmail.com Acked-by: Alan Stern Acked-by: Vlastimil Babka (SUSE) Cc: Greg Kroah-Hartman Cc: --- drivers/usb/gadget/legacy/inode.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/drivers/usb/gadget/legacy/inode.c b/drivers/usb/gadget/legacy/inode.c index 67c6ffaf4f72d6..7e383e3e0f664d 100644 --- a/drivers/usb/gadget/legacy/inode.c +++ b/drivers/usb/gadget/legacy/inode.c @@ -613,7 +613,7 @@ ep_read_iter(struct kiocb *iocb, struct iov_iter *to) return -EBADMSG; } - buf = kmalloc(len, GFP_KERNEL); + buf = kmalloc(len, GFP_KERNEL | __GFP_NOWARN); if (unlikely(!buf)) { mutex_unlock(&epdata->lock); return -ENOMEM; @@ -675,7 +675,7 @@ ep_write_iter(struct kiocb *iocb, struct iov_iter *from) return -EBADMSG; } - buf = kmalloc(len, GFP_KERNEL); + buf = kmalloc(len, GFP_KERNEL | __GFP_NOWARN); if (unlikely(!buf)) { mutex_unlock(&epdata->lock); return -ENOMEM; From 0828105b4327f292d25a413bfff668e0907debc5 Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Tue, 25 Aug 2026 15:11:57 +0000 Subject: [PATCH 1223/1352] mailmap: update email address for Bradley Morgan I switched to brads@mainlining.org for kernel work, so map the old include@grrlz.net address over to keep shortlog and blame from splitting commits between the two. Link: https://lore.kernel.org/20260825151157.4533-1-brads@mainlining.org Signed-off-by: Bradley Morgan Signed-off-by: Andrew Morton --- .mailmap | 1 + 1 file changed, 1 insertion(+) diff --git a/.mailmap b/.mailmap index 389b94a0124e31..3940b0a12c2820 100644 --- a/.mailmap +++ b/.mailmap @@ -171,6 +171,7 @@ Boris Brezillon Boris Brezillon Boris Brezillon Boris Brezillon +Bradley Morgan Brendan Higgins Brendan Jackman Brian Avery From a56f27bef6c24d5841c2a34ecc9f1a59cd31a4a9 Mon Sep 17 00:00:00 2001 From: Thorsten Blum Date: Wed, 26 Aug 2026 12:09:03 +0200 Subject: [PATCH 1224/1352] init: fix early boot crash with bare hostname parameter When a bare hostname parameter is specified on the kernel command line without the '=' separator, early parameter parsing passes NULL to early_hostname(), which dereferences it in strscpy() and can crash the system during early boot. Reject NULL values in early_hostname() and return -EINVAL instead. Link: https://lore.kernel.org/20260826100904.296151-2-blum@kernel.org Fixes: 5a704629f2c1 ("init: add "hostname" kernel parameter") Signed-off-by: Thorsten Blum Signed-off-by: Andrew Morton Cc: Dan Moulding Cc: --- init/version.c | 3 +++ 1 file changed, 3 insertions(+) diff --git a/init/version.c b/init/version.c index 94c96f6fbfe6a2..0bd5c45aabc463 100644 --- a/init/version.c +++ b/init/version.c @@ -23,6 +23,9 @@ static int __init early_hostname(char *arg) size_t maxlen = bufsize - 1; ssize_t arglen; + if (!arg) + return -EINVAL; + arglen = strscpy(init_uts_ns.name.nodename, arg, bufsize); if (arglen < 0) { pr_warn("hostname parameter exceeds %zd characters and will be truncated", From 363fdeb600f19dfa244184875f197eae0deb4ef1 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Wed, 26 Aug 2026 19:26:59 +0800 Subject: [PATCH 1225/1352] ocfs2: fix deadlock in inline-data truncate transactions Updating an inode xattr can cause an ABBA deadlock with inline file truncation: ocfs2_truncate_file() down_write(&oi->ip_alloc_sem) ocfs2_truncate_inline() ocfs2_start_trans() ocfs2_xattr_set() ocfs2_start_trans() ocfs2_xattr_ibody_set() down_write(&oi->ip_alloc_sem) The xattr set path starts the merged transaction before the inode-body xattr helper acquires ip_alloc_sem, reversing the ip_alloc_sem -> transaction order used by the allocation and truncate paths. The transaction merge in commit 85db90e77806 ("ocfs2/xattr: Merge xattr set transaction.") introduced this ordering. Fix it by acquiring ip_alloc_sem once in ocfs2_xattr_set(), before xattr preparation, allocation reservations and ocfs2_start_trans(), and removing the per-helper acquisition from ocfs2_xattr_ibody_find(), ocfs2_xattr_ibody_set() and ocfs2_xattr_create_index_block(). These helpers now assert via lockdep that the caller holds ip_alloc_sem. ocfs2_xattr_set_handle(), which only sets initial ACL or security xattrs on unpublished inodes inside the create transaction, takes ip_alloc_sem under a dedicated lockdep subclass so that the assertions hold without creating a transaction -> ip_alloc_sem cycle against the ip_alloc_sem -> transaction order. The inode is unpublished, so the acquisition can never contend. This keeps the established ip_alloc_sem -> transaction order and makes the locking unconditional, so lockdep can verify a single plain ordering instead of conditional acquisitions. Link: https://lore.kernel.org/20260826112659.246574-1-joseph.qi@linux.alibaba.com Fixes: 85db90e77806 ("ocfs2/xattr: Merge xattr set transaction.") Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Cc: ZhengYuan Huang Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi Cc: Changwei Ge Cc: Jun Piao Cc: Heming Zhao --- fs/ocfs2/xattr.c | 79 ++++++++++++++++++++++++++++++++---------------- 1 file changed, 53 insertions(+), 26 deletions(-) diff --git a/fs/ocfs2/xattr.c b/fs/ocfs2/xattr.c index bfafe059bedff0..3f83a75d29a6da 100644 --- a/fs/ocfs2/xattr.c +++ b/fs/ocfs2/xattr.c @@ -2909,6 +2909,9 @@ static int ocfs2_xattr_has_space_inline(struct inode *inode, * * Find extended attribute in inode block and * fill search info into struct ocfs2_xattr_search. + * + * The inline free-space check races with truncate and allocation, so + * callers must hold ip_alloc_sem for writing. */ static int ocfs2_xattr_ibody_find(struct inode *inode, int name_index, @@ -2920,13 +2923,13 @@ static int ocfs2_xattr_ibody_find(struct inode *inode, int ret; int has_space = 0; + lockdep_assert_held_write(&oi->ip_alloc_sem); + if (inode->i_sb->s_blocksize == OCFS2_MIN_BLOCKSIZE) return 0; if (!(oi->ip_dyn_features & OCFS2_INLINE_XATTR_FL)) { - down_read(&oi->ip_alloc_sem); has_space = ocfs2_xattr_has_space_inline(inode, di); - up_read(&oi->ip_alloc_sem); if (!has_space) return 0; } @@ -3007,6 +3010,7 @@ static int ocfs2_xattr_ibody_init(struct inode *inode, * * Set, replace or remove an extended attribute into inode block. * + * Callers must hold ip_alloc_sem for writing. */ static int ocfs2_xattr_ibody_set(struct inode *inode, struct ocfs2_xattr_info *xi, @@ -3017,16 +3021,17 @@ static int ocfs2_xattr_ibody_set(struct inode *inode, struct ocfs2_inode_info *oi = OCFS2_I(inode); struct ocfs2_xa_loc loc; + lockdep_assert_held_write(&oi->ip_alloc_sem); + if (inode->i_sb->s_blocksize == OCFS2_MIN_BLOCKSIZE) return -ENOSPC; - down_write(&oi->ip_alloc_sem); if (!(oi->ip_dyn_features & OCFS2_INLINE_XATTR_FL)) { ret = ocfs2_xattr_ibody_init(inode, xs->inode_bh, ctxt); if (ret) { if (ret != -ENOSPC) mlog_errno(ret); - goto out; + return ret; } } @@ -3036,13 +3041,10 @@ static int ocfs2_xattr_ibody_set(struct inode *inode, if (ret) { if (ret != -ENOSPC) mlog_errno(ret); - goto out; + return ret; } xs->here = loc.xl_entry; -out: - up_write(&oi->ip_alloc_sem); - return ret; } @@ -3686,6 +3688,18 @@ static int __ocfs2_xattr_set_handle(struct inode *inode, return ret; } +/* + * ip_alloc_sem subclass for inodes being initialized before publication. + * ocfs2_xattr_set_handle() runs inside the create transaction, so taking + * ip_alloc_sem there adds a transaction -> ip_alloc_sem order that would + * form a lockdep cycle with the ip_alloc_sem -> transaction order used + * elsewhere, if not for this separate subclass. The inode is unpublished + * so the acquisition can never contend. + */ +enum { + OCFS2_IP_ALLOC_SEM_UNPUBLISHED = 1, +}; + /* * This helper is only for setting initial ACL or security xattrs on an inode * that is still unpublished, unhashed, and unattached to a dentry. @@ -3747,6 +3761,13 @@ int ocfs2_xattr_set_handle(handle_t *handle, xis.inode_bh = xbs.inode_bh = di_bh; di = (struct ocfs2_dinode *)di_bh->b_data; + /* + * The inode is unpublished and cannot contend, but take the + * semaphore anyway so the helpers' lockdep assertions hold. + */ + down_write_nested(&OCFS2_I(inode)->ip_alloc_sem, + OCFS2_IP_ALLOC_SEM_UNPUBLISHED); + ret = ocfs2_xattr_ibody_find(inode, name_index, name, &xis); if (ret) goto cleanup; @@ -3759,6 +3780,7 @@ int ocfs2_xattr_set_handle(handle_t *handle, ret = __ocfs2_xattr_set_handle(inode, di, &xi, &xis, &xbs, &ctxt); cleanup: + up_write(&OCFS2_I(inode)->ip_alloc_sem); brelse(xbs.xattr_bh); ocfs2_xattr_bucket_free(xbs.bucket); @@ -3827,30 +3849,38 @@ int ocfs2_xattr_set(struct inode *inode, di = (struct ocfs2_dinode *)di_bh->b_data; down_write(&OCFS2_I(inode)->ip_xattr_sem); + /* + * The allocation and truncate paths take ip_alloc_sem before + * starting a transaction, so take it here before xattr + * preparation, allocation reservations and ocfs2_start_trans() + * to keep that order. The xattr helpers below no longer take + * it themselves. + */ + down_write(&OCFS2_I(inode)->ip_alloc_sem); /* * Scan inode and external block to find the same name * extended attribute and collect search information. */ ret = ocfs2_xattr_ibody_find(inode, name_index, name, &xis); if (ret) - goto cleanup; + goto out_free_ac; if (xis.not_found) { ret = ocfs2_xattr_block_find(inode, name_index, name, &xbs); if (ret) - goto cleanup; + goto out_free_ac; } if (xis.not_found && xbs.not_found) { ret = -ENODATA; if (flags & XATTR_REPLACE) - goto cleanup; + goto out_free_ac; ret = 0; if (!value) - goto cleanup; + goto out_free_ac; } else { ret = -EEXIST; if (flags & XATTR_CREATE) - goto cleanup; + goto out_free_ac; } /* Check whether the value is refcounted and do some preparation. */ @@ -3861,7 +3891,7 @@ int ocfs2_xattr_set(struct inode *inode, &ref_meta, &ref_credits); if (ret) { mlog_errno(ret); - goto cleanup; + goto out_free_ac; } } @@ -3872,7 +3902,7 @@ int ocfs2_xattr_set(struct inode *inode, if (ret < 0) { inode_unlock(tl_inode); mlog_errno(ret); - goto cleanup; + goto out_free_ac; } } inode_unlock(tl_inode); @@ -3881,7 +3911,7 @@ int ocfs2_xattr_set(struct inode *inode, &xbs, &ctxt, ref_meta, &credits); if (ret) { mlog_errno(ret); - goto cleanup; + goto out_free_ac; } /* we need to update inode's ctime field, so add credit for it. */ @@ -3899,6 +3929,7 @@ int ocfs2_xattr_set(struct inode *inode, ocfs2_commit_trans(osb, ctxt.handle); out_free_ac: + up_write(&OCFS2_I(inode)->ip_alloc_sem); if (ctxt.data_ac) ocfs2_free_alloc_context(ctxt.data_ac); if (ctxt.meta_ac) @@ -3907,7 +3938,6 @@ int ocfs2_xattr_set(struct inode *inode, ocfs2_schedule_truncate_log_flush(osb, 1); ocfs2_run_deallocs(osb, &ctxt.dealloc); -cleanup: if (ref_tree) ocfs2_unlock_refcount_tree(osb, ref_tree, 1); up_write(&OCFS2_I(inode)->ip_xattr_sem); @@ -4508,6 +4538,10 @@ static void ocfs2_xattr_update_xattr_search(struct inode *inode, xs->here = &xs->header->xh_entries[i]; } +/* + * Caller must hold ip_alloc_sem for writing, since a new xattr block + * is allocated and the xattr block header is rewritten. + */ static int ocfs2_xattr_create_index_block(struct inode *inode, struct ocfs2_xattr_search *xs, struct ocfs2_xattr_set_ctxt *ctxt) @@ -4523,19 +4557,14 @@ static int ocfs2_xattr_create_index_block(struct inode *inode, struct ocfs2_xattr_tree_root *xr; u16 xb_flags = le16_to_cpu(xb->xb_flags); + lockdep_assert_held_write(&oi->ip_alloc_sem); + trace_ocfs2_xattr_create_index_block_begin( (unsigned long long)xb_bh->b_blocknr); BUG_ON(xb_flags & OCFS2_XATTR_INDEXED); BUG_ON(!xs->bucket); - /* - * XXX: - * We can use this lock for now, and maybe move to a dedicated mutex - * if performance becomes a problem later. - */ - down_write(&oi->ip_alloc_sem); - ret = ocfs2_journal_access_xb(handle, INODE_CACHE(inode), xb_bh, OCFS2_JOURNAL_ACCESS_WRITE); if (ret) { @@ -4597,8 +4626,6 @@ static int ocfs2_xattr_create_index_block(struct inode *inode, ocfs2_journal_dirty(handle, xb_bh); out: - up_write(&oi->ip_alloc_sem); - return ret; } From 7bd0a78969c84ea888151127ce2f2d3185e23f25 Mon Sep 17 00:00:00 2001 From: ZhengYuan Huang Date: Thu, 6 Aug 2026 16:50:12 +0800 Subject: [PATCH 1226/1352] ocfs2: reject inconsistent local xattr entries [BUG] A corrupt OCFS2 xattr entry can set OCFS2_XATTR_ENTRY_LOCAL while keeping xe_value_size larger than OCFS2_XATTR_INLINE_SIZE. When that entry reaches namevalue_size_xe(), the filesystem hits its BUG_ON: kernel BUG at fs/ocfs2/xattr.c:231! Oops: invalid opcode: 0000 [#1] SMP KASAN NOPTI RIP: 0010:namevalue_size_xe fs/ocfs2/xattr.c:231 [inline] RIP: 0010:ocfs2_xa_block_wipe_namevalue+0x2e4/0x330 fs/ocfs2/xattr.c:1638 Call Trace: ocfs2_xa_wipe_namevalue fs/ocfs2/xattr.c:1470 [inline] ocfs2_xa_remove_entry+0xae/0x1d0 fs/ocfs2/xattr.c:1941 ocfs2_xa_remove fs/ocfs2/xattr.c:2043 [inline] ocfs2_xa_set+0x11a8/0x30a0 fs/ocfs2/xattr.c:2247 ocfs2_xattr_ibody_set+0x302/0xc50 fs/ocfs2/xattr.c:2795 __ocfs2_xattr_set_handle+0x7e6/0xdb0 fs/ocfs2/xattr.c:3416 ocfs2_xattr_set+0x1447/0x2610 fs/ocfs2/xattr.c:3650 ocfs2_xattr_security_set+0x37/0x50 fs/ocfs2/xattr.c:7241 __vfs_removexattr+0x14d/0x1d0 fs/xattr.c:518 cap_inode_killpriv+0x29/0x50 security/commoncap.c:355 security_inode_killpriv+0x105/0x220 security/security.c:2724 setattr_prepare+0x147/0x8a0 fs/attr.c:219 ocfs2_setattr+0x504/0x1fd0 fs/ocfs2/file.c:1148 notify_change+0x4b5/0x1030 fs/attr.c:546 do_truncate+0x1d2/0x230 fs/open.c:68 handle_truncate fs/namei.c:3596 [inline] do_open fs/namei.c:3979 [inline] path_openat+0x260f/0x2ce0 fs/namei.c:4134 do_filp_open+0x1f6/0x430 fs/namei.c:4161 do_sys_openat2+0x117/0x1c0 fs/open.c:1437 ... [CAUSE] namevalue_size_xe() assumes that local entries contain an inline value no larger than OCFS2_XATTR_INLINE_SIZE. Existing xattr metadata validation only checks whether the value fits the storage region, and cached entries can reach lookup and bucket maintenance paths without a semantic check. A corrupt entry can therefore be passed to namevalue_size_xe(). [FIX] Validate the local/value-size invariant in the existing flat and bucket metadata validators and before accepting matched entries or traversing bucket entries in paths that call namevalue_size_xe(). Return an OCFS2 corruption error instead of firing the assertion. Link: https://lore.kernel.org/20260806085012.2650042-1-gality369@gmail.com Signed-off-by: ZhengYuan Huang Signed-off-by: Andrew Morton Reviewed-by: Joseph Qi Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi Cc: Changwei Ge Cc: Jun Piao Cc: Heming Zhao --- fs/ocfs2/xattr.c | 53 +++++++++++++++++++++++++++++++++++++++++------- 1 file changed, 46 insertions(+), 7 deletions(-) diff --git a/fs/ocfs2/xattr.c b/fs/ocfs2/xattr.c index 3f83a75d29a6da..740d4bb3890f9c 100644 --- a/fs/ocfs2/xattr.c +++ b/fs/ocfs2/xattr.c @@ -237,6 +237,21 @@ static int namevalue_size_xe(struct ocfs2_xattr_entry *xe) return namevalue_size(xe->xe_name_len, value_len); } +static int ocfs2_validate_xattr_entry(struct super_block *sb, u64 blkno, + struct ocfs2_xattr_entry *xe) +{ + u64 value_len = le64_to_cpu(xe->xe_value_size); + + if (value_len > OCFS2_XATTR_INLINE_SIZE && + ocfs2_xattr_is_local(xe)) + return ocfs2_error(sb, + "Invalid local xattr in block %llu: value size %llu\n", + (unsigned long long)blkno, + (unsigned long long)value_len); + + return 0; +} + static int ocfs2_xattr_bucket_get_name_value(struct super_block *sb, struct ocfs2_xattr_header *xh, @@ -989,7 +1004,7 @@ static int ocfs2_validate_xattr_entries_flat(struct super_block *sb, u64 blkno, size_t entries_limit = region_size; size_t nv_limit = region_size; size_t max_entries; - int i; + int i, ret; if (region_size < sizeof(*xh)) return ocfs2_error(sb, @@ -1009,6 +1024,11 @@ static int ocfs2_validate_xattr_entries_flat(struct super_block *sb, u64 blkno, struct ocfs2_xattr_entry *xe = &xh->xh_entries[i]; size_t name_offset = le16_to_cpu(xe->xe_name_offset); size_t value_offset; + u64 value_len = le64_to_cpu(xe->xe_value_size); + + ret = ocfs2_validate_xattr_entry(sb, blkno, xe); + if (ret) + return ret; if (name_offset > nv_limit || xe->xe_name_len > nv_limit - name_offset) @@ -1023,8 +1043,7 @@ static int ocfs2_validate_xattr_entries_flat(struct super_block *sb, u64 blkno, (unsigned long long)blkno, i); if (ocfs2_xattr_is_local(xe)) { - if (le64_to_cpu(xe->xe_value_size) > - nv_limit - value_offset) + if (value_len > nv_limit - value_offset) return ocfs2_error(sb, "Invalid xattr in block %llu: entry %d value is out of bounds\n", (unsigned long long)blkno, @@ -1109,7 +1128,7 @@ static int ocfs2_validate_xattr_bucket(struct ocfs2_xattr_bucket *bucket, size_t entries_limit = sb->s_blocksize; size_t nv_limit = sb->s_blocksize; size_t max_entries; - int i; + int i, ret; if (region_size < sizeof(*xh)) return ocfs2_error(sb, @@ -1137,6 +1156,11 @@ static int ocfs2_validate_xattr_bucket(struct ocfs2_xattr_bucket *bucket, size_t block_off = name_offset >> sb->s_blocksize_bits; size_t block_offset = name_offset % nv_limit; size_t value_offset; + u64 value_len = le64_to_cpu(xe->xe_value_size); + + ret = ocfs2_validate_xattr_entry(sb, blkno, xe); + if (ret) + return ret; if (name_offset >= region_size || block_off >= bucket->bu_blocks) return ocfs2_error(sb, @@ -1155,8 +1179,7 @@ static int ocfs2_validate_xattr_bucket(struct ocfs2_xattr_bucket *bucket, (unsigned long long)blkno, i); if (ocfs2_xattr_is_local(xe)) { - if (le64_to_cpu(xe->xe_value_size) > - nv_limit - value_offset) + if (value_len > nv_limit - value_offset) return ocfs2_error(sb, "Invalid xattr bucket %llu: entry %d value is out of bounds\n", (unsigned long long)blkno, @@ -1304,7 +1327,7 @@ static int ocfs2_xattr_find_entry(struct inode *inode, int name_index, { struct ocfs2_xattr_entry *entry; size_t name_len; - int i, name_offset, cmp = 1; + int i, name_offset, cmp = 1, ret; if (name == NULL) return -EINVAL; @@ -1327,6 +1350,12 @@ static int ocfs2_xattr_find_entry(struct inode *inode, int name_index, return -EFSCORRUPTED; } cmp = memcmp(name, (xs->base + name_offset), name_len); + if (!cmp) { + ret = ocfs2_validate_xattr_entry(inode->i_sb, + OCFS2_I(inode)->ip_blkno, entry); + if (ret) + return ret; + } } if (cmp == 0) break; @@ -4068,6 +4097,10 @@ static int ocfs2_find_xe_in_bucket(struct inode *inode, xe_name = bucket_block(bucket, block_off) + new_offset; if (!memcmp(name, xe_name, name_len)) { + ret = ocfs2_validate_xattr_entry(inode->i_sb, + OCFS2_I(inode)->ip_blkno, xe); + if (ret) + break; *xe_index = i; *found = 1; ret = 0; @@ -4705,6 +4738,9 @@ static int ocfs2_defrag_xattr_bucket(struct inode *inode, xe = xh->xh_entries; end = OCFS2_XATTR_BUCKET_SIZE; for (i = 0; i < le16_to_cpu(xh->xh_count); i++, xe++) { + ret = ocfs2_validate_xattr_entry(inode->i_sb, blkno, xe); + if (ret) + goto out; offset = le16_to_cpu(xe->xe_name_offset); len = namevalue_size_xe(xe); @@ -4987,6 +5023,9 @@ static int ocfs2_divide_xattr_bucket(struct inode *inode, name_value_len = 0; for (i = 0; i < start; i++) { xe = &xh->xh_entries[i]; + ret = ocfs2_validate_xattr_entry(inode->i_sb, blk, xe); + if (ret) + goto out; name_value_len += namevalue_size_xe(xe); if (le16_to_cpu(xe->xe_name_offset) < name_offset) name_offset = le16_to_cpu(xe->xe_name_offset); From 4ec990fe041280902fe9e95549477f38e260c59c Mon Sep 17 00:00:00 2001 From: Geert Uytterhoeven Date: Mon, 24 Aug 2026 17:14:01 +0200 Subject: [PATCH 1227/1352] raid/kunit: enable RAID6 PQ and XOR benchmarks if KUNIT_ALL_TESTS=m Enabling the (possibly long-running benchmarks) by default may cause a big delay in boot time in case of built-in tests. However, they can still safely be enabled by default if all tests are modular, as they would only run when requested explicitly by the system administrator. Link: https://lore.kernel.org/64c6e0191bd8ccef0074ffbb09bd0584680d710b.1787584360.git.geert@linux-m68k.org Signed-off-by: Geert Uytterhoeven Signed-off-by: Andrew Morton Reviewed-by: Christoph Hellwig Acked-by: Ard Biesheuvel Cc: Eric Biggers --- lib/raid/Kconfig | 2 ++ 1 file changed, 2 insertions(+) diff --git a/lib/raid/Kconfig b/lib/raid/Kconfig index 01f007b2522cfc..563a178aa930c4 100644 --- a/lib/raid/Kconfig +++ b/lib/raid/Kconfig @@ -32,6 +32,7 @@ config XOR_KUNIT_TEST config XOR_BENCHMARK bool "Benchmark for xor_gen" depends on XOR_KUNIT_TEST + default y if KUNIT_ALL_TESTS=m help Include benchmarks in the KUnit test suite for xor_gen. @@ -63,6 +64,7 @@ config RAID6_PQ_KUNIT_TEST config RAID6_PQ_KUNIT_BENCHMARK bool "Benchmark for RAID6 PQ" depends on RAID6_PQ_KUNIT_TEST + default y if KUNIT_ALL_TESTS=m help Include benchmarks in the KUnit test suite for raid P/Q generation. From a6b6caf59aa8207a61a7bdd20fabf46f44d47277 Mon Sep 17 00:00:00 2001 From: Daehyeon Ko <4ncienth@gmail.com> Date: Mon, 24 Aug 2026 13:23:59 +0900 Subject: [PATCH 1228/1352] ipc/mqueue: release notification resources during inode eviction mqueue_flush_file() removes an mq_notify() registration only when the closing task belongs to the thread group stored in notify_owner. A task in a separate thread group created with CLONE_FILES can register SIGEV_THREAD notification and exit without closing the shared file table. If another thread group then unlinks and last-closes the queue, ->flush() skips the registration and inode eviction loses the only pointers to its resources. The orphaned registration permanently retains the notification skb, its netlink socket, a pid reference and a user namespace reference. An unprivileged process can repeat the sequence with new queues and sockets. No inode users remain during eviction. Remove any stale registration there after dropping info->lock, since netlink_sendskb() may release the final socket reference. This bug creates an unkillable kernel resource leak by failing to free netlink socket, PID, and user namespace references when a POSIX message queue is evicted. An unprivileged process can exploit this leak repeatedly to cause kernel memory exhaustion and lead to a DoS. Link: https://lore.kernel.org/20260824042359.925145-1-4ncienth@gmail.com Fixes: 1da177e4c3f4 ("Linux-2.6.12-rc2") Signed-off-by: Daehyeon Ko <4ncienth@gmail.com> Signed-off-by: Andrew Morton Assisted-by: LLM Cc: Davidlohr Bueso Cc: Al Viro Cc: Christian Brauner Cc: Jan Kara Cc: Manfred Spraul Cc: NeilBrown Cc: --- ipc/mqueue.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/ipc/mqueue.c b/ipc/mqueue.c index d1dd36a651b0d0..d1a1965c98118c 100644 --- a/ipc/mqueue.c +++ b/ipc/mqueue.c @@ -528,6 +528,13 @@ static void mqueue_evict_inode(struct inode *inode) list_add_tail(&msg->m_list, &tmp_msg); kfree(info->node_cache); spin_unlock(&info->lock); + /* + * A shared file table can let the notification owner exit without + * running ->flush(). No users of the inode remain during eviction, so + * tear down any stale notification after dropping info->lock because + * netlink_sendskb() may release the final socket reference. + */ + remove_notification(info); list_for_each_entry_safe(msg, nmsg, &tmp_msg, m_list) { list_del(&msg->m_list); From d5d9f3f78084f41dc45961e0112d6b8b2e58d746 Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Thu, 27 Aug 2026 06:17:08 +0200 Subject: [PATCH 1229/1352] klist: avoid accesses after waking klist_remove() klist_remove() waits until a node is unreferenced so that its caller can free the containing object. klist_release() currently publishes waiter->woken and wakes the waiter before its final accesses to the waiter and node. klist_remove() can then return, allowing its stack waiter and the containing object to be freed or reused while klist_release() is still running. In particular, bus_remove_driver() can free drv->p while __device_attach() walks the same bus klist with bus_for_each_drv(). On an arm64 Cortex-A72 system, an unpatched 7.2.0-rc3 kernel with CONFIG_PREEMPT_RT=y and CONFIG_KASAN=y reproduced the bug through the in-tree I2C/at24 path. KASAN reported a use-after-free in klist_dec_and_del() reached from klist_next()/bus_for_each_drv() while at24 was being unregistered. Clear n_klist and take a task reference before publishing woken. Use release/acquire accesses for that publication and wake the referenced task. The task reference keeps the waiter task alive if it returns and exits before wake_up_process(). Link: https://lore.kernel.org/20260827041708.31682-1-kmehltretter@gmail.com Fixes: 8b0c250be489 ("[PATCH] add klist_node_attached() to determine if a node is on a list or not.") Fixes: 210272a28465 ("driver core: Remove completion from struct klist_node") Signed-off-by: Karl Mehltretter Signed-off-by: Andrew Morton Assisted-by: LLM Cc: Danilo Krummrich Cc: Greg Kroah-Hartman Cc: Matthew Wilcox (Oracle) Cc: "Rafael J. Wysocki" Cc: --- lib/klist.c | 16 ++++++++++++---- 1 file changed, 12 insertions(+), 4 deletions(-) diff --git a/lib/klist.c b/lib/klist.c index 332a4fbf18ff08..f133740b1c2cb7 100644 --- a/lib/klist.c +++ b/lib/klist.c @@ -36,6 +36,7 @@ #include #include #include +#include /* * Use the lowest bit of n_klist to mark deleted nodes and exclude @@ -187,18 +188,24 @@ static void klist_release(struct kref *kref) WARN_ON(!knode_dead(n)); list_del(&n->n_node); + knode_set_klist(n, NULL); spin_lock(&klist_remove_lock); list_for_each_entry_safe(waiter, tmp, &klist_remove_waiters, list) { + struct task_struct *p; + if (waiter->node != n) continue; + p = waiter->process; + get_task_struct(p); list_del(&waiter->list); - waiter->woken = 1; + /* Publish only after the final waiter and n accesses */ + smp_store_release(&waiter->woken, 1); mb(); - wake_up_process(waiter->process); + wake_up_process(p); + put_task_struct(p); } spin_unlock(&klist_remove_lock); - knode_set_klist(n, NULL); } static int klist_dec_and_del(struct klist_node *n) @@ -250,7 +257,8 @@ void klist_remove(struct klist_node *n) for (;;) { set_current_state(TASK_UNINTERRUPTIBLE); - if (waiter.woken) + /* Pairs with the release store in klist_release() */ + if (smp_load_acquire(&waiter.woken)) break; schedule(); } From a9d27c1bbaceec00747a8de618418301366ecc51 Mon Sep 17 00:00:00 2001 From: Vishal Badole Date: Wed, 26 Aug 2026 22:45:37 +0530 Subject: [PATCH 1230/1352] lib/group_cpus: snapshot cluster masks to keep grouping hotplug invariant group_cpus_evenly() builds the managed-IRQ affinity spread used by multi-queue devices such as NVMe. That spread is meant to be a property of the static CPU topology: it walks cpu_present_mask and then cpu_possible_mask so every hardware queue owns a fixed set of CPUs, including CPUs that are offline at the time. A driver depends on that partition staying stable across re-computation - the CPUs a queue is given at probe must still describe the same queue after the device is later reset and its affinity recomputed. On an AMD system that stability breaks across an s2idle cycle. With CPUs 3-11 offlined and only CPUs 0-2 left online, the machine is suspended to s2idle and resumed. The NVMe controller uses the simple-suspend quirk, so resume fully re-initialises it and recomputes the affinity spread. The system then hangs for roughly two minutes and stays sluggish afterwards, the controller only making progress through its command-timeout poll: nvme nvme0: I/O tag 898 (3382) QID 9 timeout, completion polled nvme nvme0: I/O tag 398 (618e) QID 11 timeout, completion polled QID 9 and QID 11 are the queues whose CPUs were offline when the spread was recomputed. "completion polled" means the commands did finish in hardware, but their interrupts were never delivered to a CPU that was watching the queue, so nothing reaped them until the timeout fired. It happens because commit 89802ca36c96 ("lib/group_cpus: make group CPU cluster aware") derives the cluster groups from topology_cluster_cpumask(), which lists only the cluster siblings that are online when it is called. The resulting partition therefore depends on the transient online mask rather than on the topology alone. Recomputed on resume while the non-boot CPUs are still offline, it no longer matches the boot-time partition, and a queue is left with an affinity that does not cover the CPU it is meant to serve once that CPU comes back online. The dependence is on the online mask, not on any AMD-specific behaviour, so the same stall is reproducible on Intel platforms as well. Make the cluster grouping depend on the complete cluster topology rather than on whichever CPUs happen to be online. Snapshot the cluster masks once while every present CPU is online and reuse that view for every later spread. Every spread then groups from the same masks, so the partition computed when the controller is reset matches the one computed at probe and each queue's IRQ still covers the CPUs it serves. If the snapshot was never taken, the cluster path is skipped and the plain present/possible spread is used. Link: https://lore.kernel.org/20260826171537.4167367-1-Vishal.Badole@amd.com Fixes: 89802ca36c96 ("lib/group_cpus: make group CPU cluster aware") Signed-off-by: Vishal Badole Signed-off-by: Andrew Morton Cc: "Borislav Petkov (AMD)" Cc: Radu Rendec Cc: Thomas Gleixner Cc: Tim Chen Cc: Wangyang Guo Cc: Tianyou Li Cc: Dan Liang Cc: --- lib/group_cpus.c | 87 ++++++++++++++++++++++++++++++++++++++++++++++-- 1 file changed, 85 insertions(+), 2 deletions(-) diff --git a/lib/group_cpus.c b/lib/group_cpus.c index e6e18d7a49bba6..3c2229feb9c3d2 100644 --- a/lib/group_cpus.c +++ b/lib/group_cpus.c @@ -6,6 +6,7 @@ #include #include #include +#include #include #include @@ -286,6 +287,77 @@ static void assign_cpus_to_groups(unsigned int ncpus, } } +/* + * topology_cluster_cpumask() only lists the cluster siblings that are online, + * so group_cpus_evenly() would compute a different managed-IRQ partition when + * recomputed with CPUs offline (e.g. an NVMe reset across s2idle), steering a + * queue's IRQ away from the CPU it serves. + * + * Snapshot the cluster masks once, on the first spread seen with every + * present CPU online, and reuse it so the grouping stays stable. If no + * snapshot exists (partial boot via maxcpus=/nosmp, or allocation failure) + * the cluster path is skipped and the plain present/possible spread is + * used. Only the cluster path is stabilised; grp_spread_init_one()'s + * sibling mask is unchanged. The snapshot lives for the system lifetime + * and is not refreshed for CPUs hot-added after boot. + */ +static cpumask_var_t *cluster_snapshot; +static bool cluster_snapshot_ready; +static DEFINE_MUTEX(cluster_snapshot_lock); + +static void capture_cluster_snapshot(void) +{ + cpumask_var_t *snapshot; + unsigned int cpu; + + /* Pairs with the smp_store_release() below. */ + if (smp_load_acquire(&cluster_snapshot_ready)) + return; + + /* Only capture when all present CPUs are online. */ + if (!data_race(cpumask_equal(cpu_present_mask, cpu_online_mask))) + return; + + mutex_lock(&cluster_snapshot_lock); + if (cluster_snapshot_ready) + goto out; + + snapshot = kcalloc(nr_cpu_ids, sizeof(*snapshot), GFP_KERNEL); + if (!snapshot) + goto out; + + for_each_possible_cpu(cpu) + if (!zalloc_cpumask_var(&snapshot[cpu], GFP_KERNEL)) + goto free_snapshot; + + /* Trylock: a caller may hold a lock the hotplug writer needs. */ + if (!cpus_read_trylock()) + goto free_snapshot; + + /* Recheck under the lock, which also pins the cluster masks. */ + if (!data_race(cpumask_equal(cpu_present_mask, cpu_online_mask))) { + cpus_read_unlock(); + goto free_snapshot; + } + + for_each_possible_cpu(cpu) + cpumask_copy(snapshot[cpu], topology_cluster_cpumask(cpu)); + cpus_read_unlock(); + + cluster_snapshot = snapshot; + /* Publish the filled snapshot before the ready flag. */ + smp_store_release(&cluster_snapshot_ready, true); + goto out; + +free_snapshot: + /* Unallocated entries are NULL, which free_cpumask_var() ignores. */ + for_each_possible_cpu(cpu) + free_cpumask_var(snapshot[cpu]); + kfree(snapshot); +out: + mutex_unlock(&cluster_snapshot_lock); +} + static int alloc_cluster_groups(unsigned int ncpus, unsigned int ngroups, struct cpumask *node_cpumask, @@ -299,6 +371,17 @@ static int alloc_cluster_groups(unsigned int ncpus, const struct cpumask **clusters; struct node_groups *cluster_groups; + /* + * Capture on the first spread with every present CPU online (normally + * the first device probe); later spreads reuse it. Sample the ready + * flag once so both loops below use one consistent source. + */ + capture_cluster_snapshot(); + + /* Pairs with the smp_store_release() in capture_cluster_snapshot(). */ + if (!smp_load_acquire(&cluster_snapshot_ready)) + goto no_cluster; + cpumask_copy(msk, node_cpumask); /* Probe how many clusters in this node. */ @@ -307,7 +390,7 @@ static int alloc_cluster_groups(unsigned int ncpus, if (cpu >= nr_cpu_ids) break; - cluster_mask = topology_cluster_cpumask(cpu); + cluster_mask = cluster_snapshot[cpu]; if (!cpumask_weight(cluster_mask)) goto no_cluster; /* Clean out CPUs on the same cluster. */ @@ -331,7 +414,7 @@ static int alloc_cluster_groups(unsigned int ncpus, cpumask_copy(msk, node_cpumask); for (n = 0; n < ncluster; n++) { cpu = cpumask_first(msk); - cluster_mask = topology_cluster_cpumask(cpu); + cluster_mask = cluster_snapshot[cpu]; nc = cpumask_weight_and(cluster_mask, node_cpumask); clusters[n] = cluster_mask; cluster_groups[n].id = n; From 4d37865993268b19160ae48f6c640525514b4049 Mon Sep 17 00:00:00 2001 From: Chris Gellermann Date: Mon, 3 Aug 2026 14:48:59 +0200 Subject: [PATCH 1231/1352] selftests/membarrier: introduce helper to get membarrier command registrations Patch series "selftests/membarrier: Skip an unregistered memory barrier test on Musl". The membarrier test "membarrier MEMBARRIER_CMD_PRIVATE_EXPEDITED not registered failure" fails in the multithreaded test scenario when using Musl libc as the command gets preregistered implicitly during thread creation. Skip the test if command registration is detected. This patch (of 2): Add a new membarrier_get_registrations() for reusage. Link: https://lore.kernel.org/20260803124900.3328789-1-christian.gellermann@codasip.com Link: https://lore.kernel.org/20260803124900.3328789-2-christian.gellermann@codasip.com Signed-off-by: Chris Gellermann Signed-off-by: Andrew Morton Tested-by: Michael Jeanson Cc: Ben Segall Cc: Dietmar Eggemann Cc: Ingo Molnar Cc: Juri Lelli Cc: K Prateek Nayak Cc: Mathieu Desnoyers Cc: Mel Gorman Cc: "Paul E . McKenney" Cc: Peter Zijlstra Cc: Shuah Khan Cc: Steven Rostedt Cc: Valentin Schneider Cc: Vincent Guittot Cc: Wei Yang --- tools/testing/selftests/membarrier/membarrier_test_impl.h | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/tools/testing/selftests/membarrier/membarrier_test_impl.h b/tools/testing/selftests/membarrier/membarrier_test_impl.h index f6d7c44b2288da..29aac3bc498871 100644 --- a/tools/testing/selftests/membarrier/membarrier_test_impl.h +++ b/tools/testing/selftests/membarrier/membarrier_test_impl.h @@ -16,6 +16,11 @@ static int sys_membarrier(int cmd, int flags) return syscall(__NR_membarrier, cmd, flags); } +static int membarrier_get_registrations(void) +{ + return sys_membarrier(MEMBARRIER_CMD_GET_REGISTRATIONS, 0); +} + static int test_membarrier_get_registrations(int cmd) { int ret, flags = 0; @@ -24,7 +29,7 @@ static int test_membarrier_get_registrations(int cmd) registrations |= cmd; - ret = sys_membarrier(MEMBARRIER_CMD_GET_REGISTRATIONS, 0); + ret = membarrier_get_registrations(); if (ret < 0) { ksft_exit_fail_msg( "%s test: flags = %d, errno = %d\n", From b318d61efcc9b83a3b97d75ba4468d397154e482 Mon Sep 17 00:00:00 2001 From: Chris Gellermann Date: Mon, 3 Aug 2026 14:49:00 +0200 Subject: [PATCH 1232/1352] selftests/membarrier: skip unpermitted membarrier command test if preregistered by libc On thread creation, Musl registers the private expedited memory barrier, see pthread_create [1]. Thus, invoking the barrier command will no longer be rejected by the kernel with EPERM. The test checking this will fail. Check if the memory barrier command has been registered and skip the test in this case. Link: https://git.musl-libc.org/cgit/musl/tree/src/thread/pthread_create.c#n260 [1] Link: https://lore.kernel.org/20260803124900.3328789-3-christian.gellermann@codasip.com Signed-off-by: Chris Gellermann Signed-off-by: Andrew Morton Tested-by: Michael Jeanson Cc: Ben Segall Cc: Dietmar Eggemann Cc: Ingo Molnar Cc: Juri Lelli Cc: K Prateek Nayak Cc: Mathieu Desnoyers Cc: Mel Gorman Cc: "Paul E . McKenney" Cc: Peter Zijlstra Cc: Shuah Khan Cc: Steven Rostedt Cc: Valentin Schneider Cc: Vincent Guittot Cc: Wei Yang --- .../selftests/membarrier/membarrier_test_impl.h | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/tools/testing/selftests/membarrier/membarrier_test_impl.h b/tools/testing/selftests/membarrier/membarrier_test_impl.h index 29aac3bc498871..b4dcbb32538d47 100644 --- a/tools/testing/selftests/membarrier/membarrier_test_impl.h +++ b/tools/testing/selftests/membarrier/membarrier_test_impl.h @@ -113,6 +113,16 @@ static int test_membarrier_private_expedited_fail(void) int cmd = MEMBARRIER_CMD_PRIVATE_EXPEDITED, flags = 0; const char *test_name = "sys membarrier MEMBARRIER_CMD_PRIVATE_EXPEDITED not registered failure"; + /* + * Some C libraries, like Musl, register the private expedited barrier + * command when creating a thread. Expecting an EPERM on an unregistered + * command will therefore no longer work. Skip the test in this case. + */ + if (MEMBARRIER_CMD_REGISTER_PRIVATE_EXPEDITED & membarrier_get_registrations()) { + ksft_test_result_skip("%s test: Command already registered\n", test_name); + return 0; + } + if (sys_membarrier(cmd, flags) != -1) { ksft_exit_fail_msg( "%s test: flags = %d. Should fail, but passed\n", From b2cad0557528a0cd8fdbe32fb685f983928b0424 Mon Sep 17 00:00:00 2001 From: Florian Schmaus Date: Fri, 28 Aug 2026 17:54:07 +0200 Subject: [PATCH 1233/1352] selftests/epoll: fix race condition in multi-waiter wakeup tests In tests with multiple concurrent waiters on edge-triggered epoll instances where an emitter writes to multiple sockets (epoll16, epoll56, epoll58): When the emitter performs its first write(), ep_poll_callback() fires and wakes up both waiters because one waiter uses epoll_wait() and the other one uses poll(). This translates to different wait queues, ep->wq for epoll and ep->poll_wait for poll/select, which are both awoken by the kernel because of that single write. Next, both waiter threads invoke epoll_wait(), but since there is only one event, only one epoll_wait() will return non-zero because of the edge-triggered mode being used (in level-triggered mode, the kernel would re-queue the event because of remaining unread data). Since the second waiter sees an empty ready list, it does not increment ctx.count and the test fails spuriously with ctx.count == 1 instead of 2. Emitter (CPU 0) Thread 0 (CPU 1) Thread 1 (CPU 2) =============== ================ ================ epoll_wait(e0, -1) poll(e0, -1) [on e0->wq] [on e0->poll_wait] write(sfd[1]) | +--(Kernel wakes BOTH e0->wq and e0->poll_wait via callback)--+ | | | wakes up wakes up | | epoll_wait() reaps e1 poll() returns 1 | | (e1 removed via ET) (wants event) | | e0->rdllist is EMPTY | | | count++ (count = 1) v | | epoll_wait(e0, 0) | | sees EMPTY list! | | returns 0! | | thread exits | v | write(sfd[3]) | (event arrives too late!) v EXPECT_EQ(count, 2) <-- SPURIOUS FAILURE! Introduce waiter_entry1ap_loop() to retry poll() if the initial epoll_wait(..., 0) yielded no events. This ensures the thread waits for the subsequent write rather than failing immediately. Apply this helper in epoll16, epoll56, and for both waiter threads in epoll58. Link: https://lore.kernel.org/20260828-selftest-epoll-fix-race-v2-1-953ab57fd60a@codasip.com Fixes: f2728fe80cef ("selftests: add epoll selftests") Signed-off-by: Florian Schmaus Signed-off-by: Andrew Morton Cc: Heiher Cc: Roman Penyaev Cc: Shuah Khan Cc: Christian Brauner --- .../filesystems/epoll/epoll_wakeup_test.c | 32 +++++++++++++------ 1 file changed, 22 insertions(+), 10 deletions(-) diff --git a/tools/testing/selftests/filesystems/epoll/epoll_wakeup_test.c b/tools/testing/selftests/filesystems/epoll/epoll_wakeup_test.c index 81a994943e121b..b4dcbcd79773a7 100644 --- a/tools/testing/selftests/filesystems/epoll/epoll_wakeup_test.c +++ b/tools/testing/selftests/filesystems/epoll/epoll_wakeup_test.c @@ -74,6 +74,24 @@ static void *waiter_entry1ap(void *data) return NULL; } +static void *waiter_entry1ap_loop(void *data) +{ + struct pollfd pfd; + struct epoll_event e; + struct epoll_mtcontext *ctx = data; + + pfd.fd = ctx->efd[0]; + pfd.events = POLLIN; + while (poll(&pfd, 1, 2000) > 0) { + if (epoll_wait(ctx->efd[0], &e, 1, 0) > 0) { + __sync_fetch_and_add(&ctx->count, 1); + break; + } + } + + return NULL; +} + static void *waiter_entry1o(void *data) { struct epoll_event e; @@ -809,7 +827,7 @@ TEST(epoll16) ASSERT_EQ(epoll_ctl(ctx.efd[0], EPOLL_CTL_ADD, ctx.sfd[2], events), 0); ctx.main = pthread_self(); - ASSERT_EQ(pthread_create(&ctx.waiter, NULL, waiter_entry1ap, &ctx), 0); + ASSERT_EQ(pthread_create(&ctx.waiter, NULL, waiter_entry1ap_loop, &ctx), 0); ASSERT_EQ(pthread_create(&emitter, NULL, emitter_entry2, &ctx), 0); if (epoll_wait(ctx.efd[0], events, 1, -1) > 0) @@ -2925,7 +2943,7 @@ TEST(epoll56) ASSERT_EQ(epoll_ctl(ctx.efd[0], EPOLL_CTL_ADD, ctx.efd[2], &e), 0); ctx.main = pthread_self(); - ASSERT_EQ(pthread_create(&ctx.waiter, NULL, waiter_entry1ap, &ctx), 0); + ASSERT_EQ(pthread_create(&ctx.waiter, NULL, waiter_entry1ap_loop, &ctx), 0); ASSERT_EQ(pthread_create(&emitter, NULL, emitter_entry2, &ctx), 0); if (epoll_wait(ctx.efd[0], &e, 1, -1) > 0) @@ -3030,7 +3048,6 @@ TEST(epoll57) TEST(epoll58) { pthread_t emitter; - struct pollfd pfd; struct epoll_event e; struct epoll_mtcontext ctx = { 0 }; @@ -3061,15 +3078,10 @@ TEST(epoll58) ASSERT_EQ(epoll_ctl(ctx.efd[0], EPOLL_CTL_ADD, ctx.efd[2], &e), 0); ctx.main = pthread_self(); - ASSERT_EQ(pthread_create(&ctx.waiter, NULL, waiter_entry1ap, &ctx), 0); + ASSERT_EQ(pthread_create(&ctx.waiter, NULL, waiter_entry1ap_loop, &ctx), 0); ASSERT_EQ(pthread_create(&emitter, NULL, emitter_entry2, &ctx), 0); - pfd.fd = ctx.efd[0]; - pfd.events = POLLIN; - if (poll(&pfd, 1, -1) > 0) { - if (epoll_wait(ctx.efd[0], &e, 1, 0) > 0) - __sync_fetch_and_add(&ctx.count, 1); - } + waiter_entry1ap_loop(&ctx); ASSERT_EQ(pthread_join(ctx.waiter, NULL), 0); EXPECT_EQ(ctx.count, 2); From c3f697c48f59ebdeeb0c805a1143de04e13da0ab Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Fri, 28 Aug 2026 19:28:23 +0800 Subject: [PATCH 1234/1352] ocfs2: exit recovery thread on mount error path When a mount fails after the cluster connection has been established, e.g. in ocfs2_mount_volume(), ocfs2_fill_super() unwinds via out_debugfs/out_super and frees the osb without disabling recovery. A node failure event can concurrently launch the recovery thread, which blocks in __ocfs2_wait_on_mount() waiting for the volume state to become VOLUME_MOUNTED or VOLUME_DISABLED. As the mount error path neither sets VOLUME_DISABLED nor wakes osb_mount_event, the thread can never make progress: the kthread leaks and stays blocked on the wait queue embedded in the freed osb, which may then be accessed as freed memory. Fix it by setting VOLUME_DISABLED and waking osb_mount_event on this path so the thread bails out, and replace the plain kfree(osb->recovery_map) with ocfs2_recovery_exit(), which waits for a running recovery thread to exit before the recovery map is freed. Link: https://lore.kernel.org/20260828112825.666097-1-joseph.qi@linux.alibaba.com Fixes: f1e75d128b46 ("ocfs2: rewrite error handling of ocfs2_fill_super") Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Reviewed-by: Heming Zhao Cc: Changwei Ge Cc: Joel Becker Cc: Jun Piao Cc: Junxiao Bi Cc: Mark Fasheh Cc: --- fs/ocfs2/super.c | 11 ++++++++++- 1 file changed, 10 insertions(+), 1 deletion(-) diff --git a/fs/ocfs2/super.c b/fs/ocfs2/super.c index c62e389d4dd659..f785c39d1fb8a5 100644 --- a/fs/ocfs2/super.c +++ b/fs/ocfs2/super.c @@ -1169,8 +1169,17 @@ static int ocfs2_fill_super(struct super_block *sb, struct fs_context *fc) out_debugfs: debugfs_remove_recursive(osb->osb_debug_root); out_super: + /* + * A recovery thread launched by a node failure event may still be + * waiting for the volume to be mounted. Set VOLUME_DISABLED and + * wake it up, then wait for it to exit before osb is freed, + * otherwise the kthread would leak and stay blocked on the wait + * queue embedded in the freed osb. + */ + atomic_set(&osb->vol_state, VOLUME_DISABLED); + wake_up(&osb->osb_mount_event); ocfs2_release_system_inodes(osb); - kfree(osb->recovery_map); + ocfs2_recovery_exit(osb); ocfs2_delete_osb(osb); kfree(osb); out: From 6a5493ce687e912cbcecdfd2943109b7e57b1af5 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Fri, 28 Aug 2026 19:28:24 +0800 Subject: [PATCH 1235/1352] ocfs2: free replay slots in ocfs2_recovery_exit() Commit ce2fcf1516d6 ("ocfs2: fix memory leak in ocfs2_mount_volume()") added ocfs2_free_replay_slots() calls to the mount error paths out_dismount and out_check_volume to fix a leak of osb->replay_map. However these calls are unlocked while the bail path of the recovery thread, which is woken up by out_dismount right before the call, also frees the replay slots under osb->recovery_lock. Both sides can thus observe a non-NULL osb->replay_map and trigger a double free. Fix this by moving ocfs2_free_replay_slots() into ocfs2_recovery_exit() after ocfs2_recovery_disable(), which waits for a running recovery thread to exit under osb->recovery_lock, and drop the unlocked call sites. Both ocfs2_dismount_volume() and the out_super path of ocfs2_fill_super() call ocfs2_recovery_exit(), so the replay slots are freed on every path. Since super.c no longer references it, make ocfs2_free_replay_slots() static again. Link: https://lore.kernel.org/20260828112825.666097-2-joseph.qi@linux.alibaba.com Fixes: ce2fcf1516d6 ("ocfs2: fix memory leak in ocfs2_mount_volume()") Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Reviewed-by: Heming Zhao Cc: Changwei Ge Cc: Joel Becker Cc: Jun Piao Cc: Junxiao Bi Cc: Mark Fasheh Cc: --- fs/ocfs2/journal.c | 3 ++- fs/ocfs2/journal.h | 1 - fs/ocfs2/super.c | 5 +---- 3 files changed, 3 insertions(+), 6 deletions(-) diff --git a/fs/ocfs2/journal.c b/fs/ocfs2/journal.c index d8afbc1a76bb8a..a3938a03e93bf7 100644 --- a/fs/ocfs2/journal.c +++ b/fs/ocfs2/journal.c @@ -156,7 +156,7 @@ static void ocfs2_queue_replay_slots(struct ocfs2_super *osb, replay_map->rm_state = REPLAY_DONE; } -void ocfs2_free_replay_slots(struct ocfs2_super *osb) +static void ocfs2_free_replay_slots(struct ocfs2_super *osb) { struct ocfs2_replay_map *replay_map = osb->replay_map; @@ -243,6 +243,7 @@ void ocfs2_recovery_exit(struct ocfs2_super *osb) /* XXX: Should we bug if there are dirty entries? */ kfree(rm); + ocfs2_free_replay_slots(osb); } static int __ocfs2_recovery_map_test(struct ocfs2_super *osb, diff --git a/fs/ocfs2/journal.h b/fs/ocfs2/journal.h index f8b3b2a3d6309e..19fc920d26b1cd 100644 --- a/fs/ocfs2/journal.h +++ b/fs/ocfs2/journal.h @@ -151,7 +151,6 @@ void ocfs2_recovery_exit(struct ocfs2_super *osb); void ocfs2_recovery_disable_quota(struct ocfs2_super *osb); int ocfs2_compute_replay_slots(struct ocfs2_super *osb); -void ocfs2_free_replay_slots(struct ocfs2_super *osb); /* * Journal Control: * Initialize, Load, Shutdown, Wipe a journal. diff --git a/fs/ocfs2/super.c b/fs/ocfs2/super.c index f785c39d1fb8a5..6a8092b65bb558 100644 --- a/fs/ocfs2/super.c +++ b/fs/ocfs2/super.c @@ -1162,7 +1162,6 @@ static int ocfs2_fill_super(struct super_block *sb, struct fs_context *fc) out_dismount: atomic_set(&osb->vol_state, VOLUME_DISABLED); wake_up(&osb->osb_mount_event); - ocfs2_free_replay_slots(osb); ocfs2_dismount_volume(sb, 1); goto out; @@ -1776,14 +1775,12 @@ static int ocfs2_mount_volume(struct super_block *sb) status = ocfs2_truncate_log_init(osb); if (status < 0) { mlog_errno(status); - goto out_check_volume; + goto out_system_inodes; } ocfs2_super_unlock(osb, 1); return 0; -out_check_volume: - ocfs2_free_replay_slots(osb); out_system_inodes: if (osb->local_alloc_state == OCFS2_LA_ENABLED) ocfs2_shutdown_local_alloc(osb); From 58842d0cd6800ee0e45673ccdde6be3113a8157e Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Fri, 28 Aug 2026 19:28:25 +0800 Subject: [PATCH 1236/1352] ocfs2: defer suballocator block group reclaim to workqueue When the last bit in a suballocator block group is freed, _ocfs2_free_suballoc_bits() reclaims the group back to the global bitmap. The reclaim takes inode_lock() on the global bitmap inode while running inside the freeing transaction, adding a lock dependency of j_trans_barrier -> global bitmap inode i_rwsem This forms a circular dependency with paths such as ocfs2_shutdown_local_alloc(), which take the global bitmap inode lock before starting a transaction: Task1 (dealloc): ocfs2_run_deallocs ocfs2_free_cached_blocks ocfs2_start_trans down_read(j_trans_barrier) _ocfs2_free_suballoc_bits _ocfs2_reclaim_suballoc_to_main inode_lock(main_bm_inode) <- wait on Task2 Task2 (dismount): ocfs2_shutdown_local_alloc inode_lock(main_bm_inode) ocfs2_start_trans down_read(j_trans_barrier) <- wait on Task3 Task3 (ocfs2cmt): ocfs2_commit_cache down_write(j_trans_barrier) <- wait on Task1's handle jbd2_journal_flush Task1 waits for Task2's inode_lock(), Task2 waits for the j_trans_barrier down_write() held by ocfs2cmt, and ocfs2cmt waits for Task1's running transaction to commit - a real deadlock, observed with aio-stress direct IO writes racing dismount. Fix it by deferring the reclaim to the per-superblock ocfs2_wq workqueue, so the freeing transaction no longer takes the global bitmap inode lock. The worker re-checks under the suballocator locks that the block group is still fully freed (it may have been allocated from again in the meantime), takes the global bitmap inode locks before starting its own transaction, and performs the same suballocator cleanup and space return. The inode lock order (suballocator inode -> global bitmap inode) is consistent with the existing "inode lock before transaction" order, breaking the cycle. Reclaim work can still be queued late in dismount, e.g. when the truncate log is flushed or orphan dir recovery frees inode bits, so both ocfs2_dismount_volume() and the mount error path flush ocfs2_wq right before the system inodes are released, while the journal is still alive, to make sure no reclaim work is left running. The worker also bails out if the journal is already gone. Tested with the ocfs2 testsuite (including aio-stress direct IO) and umount/mount cycles on a CONFIG_PROVE_LOCKING kernel: the circular locking dependency is gone and freed block groups are still returned to the global bitmap. Link: https://lore.kernel.org/20260828112825.666097-3-joseph.qi@linux.alibaba.com Fixes: 4a54331616b3 ("ocfs2: give ocfs2 the ability to reclaim suballocator free bg") Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Reviewed-by: Heming Zhao Assisted-by: Qoder:Qwen3.8-Max Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi Cc: Changwei Ge Cc: Jun Piao --- fs/ocfs2/ocfs2.h | 5 ++ fs/ocfs2/suballoc.c | 202 +++++++++++++++++++++++++++++++++++++------- fs/ocfs2/suballoc.h | 2 +- fs/ocfs2/super.c | 16 ++++ 4 files changed, 195 insertions(+), 30 deletions(-) diff --git a/fs/ocfs2/ocfs2.h b/fs/ocfs2/ocfs2.h index 62cad6522c7a31..b747cdec178758 100644 --- a/fs/ocfs2/ocfs2.h +++ b/fs/ocfs2/ocfs2.h @@ -502,6 +502,11 @@ struct ocfs2_super */ struct workqueue_struct *ocfs2_wq; + /* deferred reclaim of fully freed suballocator block groups */ + spinlock_t os_suballoc_reclaim_lock; + struct list_head os_suballoc_reclaim_list; + struct work_struct os_suballoc_reclaim_work; + /* sysfs directory per partition */ struct kset *osb_dev_kset; diff --git a/fs/ocfs2/suballoc.c b/fs/ocfs2/suballoc.c index 20c3aec6b9873c..453b56be9624c6 100644 --- a/fs/ocfs2/suballoc.c +++ b/fs/ocfs2/suballoc.c @@ -2687,16 +2687,24 @@ static int ocfs2_block_group_clear_bits(handle_t *handle, * cleanup rec/alloc_inode job, then switches to the main bitmap * to reclaim released space. * + * Callers must hold inode_lock() and ocfs2_inode_lock() on + * main_bm_inode, i.e. the global bitmap inode locks must be taken + * before starting the transaction. + * * handle: The transaction handle * alloc_inode: The suballoc inode * alloc_bh: The buffer_head of suballoc inode * group_bh: The group descriptor buffer_head of suballocator managed. - * Caller should release the input group_bh. + * This function takes ownership of it and will release it. + * main_bm_inode: The global bitmap inode + * main_bm_bh: The buffer_head of the global bitmap inode */ static int _ocfs2_reclaim_suballoc_to_main(handle_t *handle, struct inode *alloc_inode, struct buffer_head *alloc_bh, - struct buffer_head *group_bh) + struct buffer_head *group_bh, + struct inode *main_bm_inode, + struct buffer_head *main_bm_bh) { int idx, status = 0; int i, next_free_rec, len = 0; @@ -2706,8 +2714,6 @@ static int _ocfs2_reclaim_suballoc_to_main(handle_t *handle, u64 bg_blkno, start_blk; unsigned int count; struct ocfs2_chain_rec *rec; - struct buffer_head *main_bm_bh = NULL; - struct inode *main_bm_inode = NULL; struct ocfs2_super *osb = OCFS2_SB(alloc_inode->i_sb); struct ocfs2_dinode *fe = (struct ocfs2_dinode *) alloc_bh->b_data; struct ocfs2_chain_list *cl = &fe->id2.i_chain; @@ -2794,24 +2800,12 @@ static int _ocfs2_reclaim_suballoc_to_main(handle_t *handle, ocfs2_remove_from_cache(INODE_CACHE(alloc_inode), group_bh); memset(group, 0, sizeof(struct ocfs2_group_desc)); - /* prepare job for reclaim clusters */ - main_bm_inode = ocfs2_get_system_file_inode(osb, - GLOBAL_BITMAP_SYSTEM_INODE, - OCFS2_INVALID_SLOT); - if (!main_bm_inode) - goto bail; /* ignore the error in reclaim path */ - - inode_lock(main_bm_inode); - - status = ocfs2_inode_lock(main_bm_inode, &main_bm_bh, 1); - if (status < 0) - goto free_bm_inode; /* ignore the error in reclaim path */ - ocfs2_block_to_cluster_group(main_bm_inode, start_blk, &bg_blkno, &start_bit); fe = (struct ocfs2_dinode *) main_bm_bh->b_data; cl = &fe->id2.i_chain; - /* reuse group_bh, caller will release the input group_bh */ + /* release the suballocator group descriptor before reuse */ + brelse(group_bh); group_bh = NULL; /* reclaim clusters to global_bitmap */ @@ -2819,7 +2813,7 @@ static int _ocfs2_reclaim_suballoc_to_main(handle_t *handle, &group_bh); if (status < 0) { mlog_errno(status); - goto free_bm_bh; + goto bail; } group = (struct ocfs2_group_desc *) group_bh->b_data; @@ -2827,7 +2821,7 @@ static int _ocfs2_reclaim_suballoc_to_main(handle_t *handle, ocfs2_error(alloc_inode->i_sb, "reclaim length (%d) beyands block group length (%d)", count + start_bit, le16_to_cpu(group->bg_bits)); - goto free_group_bh; + goto bail; } old_bg_contig_free_bits = group->bg_contig_free_bits; @@ -2837,7 +2831,7 @@ static int _ocfs2_reclaim_suballoc_to_main(handle_t *handle, _ocfs2_clear_bit); if (status < 0) { mlog_errno(status); - goto free_group_bh; + goto bail; } status = ocfs2_journal_access_di(handle, INODE_CACHE(main_bm_inode), @@ -2847,7 +2841,7 @@ static int _ocfs2_reclaim_suballoc_to_main(handle_t *handle, ocfs2_block_group_set_bits(handle, main_bm_inode, group, group_bh, start_bit, count, le16_to_cpu(old_bg_contig_free_bits), 1); - goto free_group_bh; + goto bail; } idx = le16_to_cpu(group->bg_chain); @@ -2858,19 +2852,168 @@ static int _ocfs2_reclaim_suballoc_to_main(handle_t *handle, fe->id1.bitmap1.i_used = cpu_to_le32(tmp_used - count); ocfs2_journal_dirty(handle, main_bm_bh); -free_group_bh: +bail: brelse(group_bh); + return status; +} + +/* + * When a suballocator block group becomes fully freed, its space is + * reclaimed back to the global bitmap. Taking the global bitmap inode + * lock inside the freeing transaction would create a lock dependency + * of "j_trans_barrier -> global bitmap inode i_rwsem", which forms a + * circular dependency with paths like ocfs2_shutdown_local_alloc() that + * take the inode lock before starting a transaction, and can lead to a + * real deadlock with the ocfs2cmt journal commit thread. So queue the + * reclaim to the workqueue and let it run outside the freeing + * transaction. + */ +struct ocfs2_suballoc_reclaim_work { + struct list_head list; + struct inode *alloc_inode; + u64 bg_blkno; +}; + +static void ocfs2_queue_suballoc_reclaim(struct ocfs2_super *osb, + struct inode *alloc_inode, + u64 bg_blkno) +{ + struct ocfs2_suballoc_reclaim_work *reclaim_work; + + reclaim_work = kmalloc_obj(*reclaim_work, GFP_NOFS); + if (!reclaim_work) { + /* + * Reclaim is only a space return optimization. If we can't + * queue it, the freed block group just stays owned by the + * suballocator. + */ + return; + } + + igrab(alloc_inode); + reclaim_work->alloc_inode = alloc_inode; + reclaim_work->bg_blkno = bg_blkno; + + spin_lock(&osb->os_suballoc_reclaim_lock); + list_add_tail(&reclaim_work->list, &osb->os_suballoc_reclaim_list); + spin_unlock(&osb->os_suballoc_reclaim_lock); -free_bm_bh: + queue_work(osb->ocfs2_wq, &osb->os_suballoc_reclaim_work); +} + +static void ocfs2_do_suballoc_reclaim(struct ocfs2_super *osb, + struct ocfs2_suballoc_reclaim_work *reclaim_work) +{ + int status, i; + handle_t *handle; + struct inode *alloc_inode = reclaim_work->alloc_inode; + struct inode *main_bm_inode; + struct buffer_head *alloc_bh = NULL, *group_bh = NULL; + struct buffer_head *main_bm_bh = NULL; + struct ocfs2_dinode *fe; + struct ocfs2_chain_list *cl; + struct ocfs2_chain_rec *rec; + + /* journal already gone, e.g. during dismount cleanup */ + if (!osb->journal) + return; + + inode_lock(alloc_inode); + status = ocfs2_inode_lock(alloc_inode, &alloc_bh, 1); + if (status < 0) + goto out_alloc; + + fe = (struct ocfs2_dinode *) alloc_bh->b_data; + cl = &fe->id2.i_chain; + + /* + * The block group may have been allocated from again since the + * reclaim work was queued, re-check that it is still fully freed. + * A stale work item can also reference a group that is no longer + * chained, whose descriptor would fail validation and trigger a + * spurious ocfs2_error(), so verify chain membership first. + */ + for (i = 0; i < le16_to_cpu(cl->cl_next_free_rec); i++) { + rec = &cl->cl_recs[i]; + if (le64_to_cpu(rec->c_blkno) == reclaim_work->bg_blkno) + break; + } + if (i == le16_to_cpu(cl->cl_next_free_rec) || + ocfs2_is_cluster_bitmap(alloc_inode) || + (le32_to_cpu(rec->c_free) != (le32_to_cpu(rec->c_total) - 1)) || + (le16_to_cpu(cl->cl_next_free_rec) == 1)) + goto out_alloc_unlock; + + status = ocfs2_read_group_descriptor(alloc_inode, fe, + reclaim_work->bg_blkno, &group_bh); + if (status < 0) + goto out_alloc_unlock; + + main_bm_inode = ocfs2_get_system_file_inode(osb, + GLOBAL_BITMAP_SYSTEM_INODE, + OCFS2_INVALID_SLOT); + if (!main_bm_inode) + goto out_group; + + inode_lock(main_bm_inode); + status = ocfs2_inode_lock(main_bm_inode, &main_bm_bh, 1); + if (status < 0) + goto out_main; + + handle = ocfs2_start_trans(osb, OCFS2_SUBALLOC_FREE); + if (IS_ERR(handle)) { + status = PTR_ERR(handle); + mlog_errno(status); + goto out_main_unlock; + } + + status = _ocfs2_reclaim_suballoc_to_main(handle, alloc_inode, + alloc_bh, group_bh, + main_bm_inode, main_bm_bh); + /* group_bh ownership passed to _ocfs2_reclaim_suballoc_to_main() */ + group_bh = NULL; + if (status < 0) + mlog_errno(status); + + ocfs2_commit_trans(osb, handle); + +out_main_unlock: ocfs2_inode_unlock(main_bm_inode, 1); brelse(main_bm_bh); - -free_bm_inode: +out_main: inode_unlock(main_bm_inode); iput(main_bm_inode); +out_group: + brelse(group_bh); +out_alloc_unlock: + ocfs2_inode_unlock(alloc_inode, 1); + brelse(alloc_bh); +out_alloc: + inode_unlock(alloc_inode); +} -bail: - return status; +void ocfs2_suballoc_reclaim_worker(struct work_struct *work) +{ + struct ocfs2_super *osb = container_of(work, struct ocfs2_super, + os_suballoc_reclaim_work); + struct ocfs2_suballoc_reclaim_work *reclaim_work; + + while (1) { + spin_lock(&osb->os_suballoc_reclaim_lock); + if (list_empty(&osb->os_suballoc_reclaim_list)) { + spin_unlock(&osb->os_suballoc_reclaim_lock); + break; + } + reclaim_work = list_first_entry(&osb->os_suballoc_reclaim_list, + struct ocfs2_suballoc_reclaim_work, + list); + list_del(&reclaim_work->list); + spin_unlock(&osb->os_suballoc_reclaim_lock); + + ocfs2_do_suballoc_reclaim(osb, reclaim_work); + iput(reclaim_work->alloc_inode); + kfree(reclaim_work); + } } /* @@ -2955,7 +3098,8 @@ static int _ocfs2_free_suballoc_bits(handle_t *handle, goto bail; } - _ocfs2_reclaim_suballoc_to_main(handle, alloc_inode, alloc_bh, group_bh); + ocfs2_queue_suballoc_reclaim(OCFS2_SB(alloc_inode->i_sb), alloc_inode, + bg_blkno); bail: brelse(group_bh); diff --git a/fs/ocfs2/suballoc.h b/fs/ocfs2/suballoc.h index bcf2ed4a86310b..6042abc032f96e 100644 --- a/fs/ocfs2/suballoc.h +++ b/fs/ocfs2/suballoc.h @@ -206,7 +206,7 @@ int ocfs2_lock_allocators(struct inode *inode, struct ocfs2_extent_tree *et, int ocfs2_test_inode_bit(struct ocfs2_super *osb, u64 blkno, int *res); - +void ocfs2_suballoc_reclaim_worker(struct work_struct *work); /* * The following two interfaces are for ocfs2_create_inode_in_orphan(). diff --git a/fs/ocfs2/super.c b/fs/ocfs2/super.c index 6a8092b65bb558..1e76b1d9fe0af4 100644 --- a/fs/ocfs2/super.c +++ b/fs/ocfs2/super.c @@ -1784,6 +1784,9 @@ static int ocfs2_mount_volume(struct super_block *sb) out_system_inodes: if (osb->local_alloc_state == OCFS2_LA_ENABLED) ocfs2_shutdown_local_alloc(osb); + /* Drain pending suballoc reclaim work before the journal goes away */ + if (osb->ocfs2_wq) + flush_workqueue(osb->ocfs2_wq); ocfs2_release_system_inodes(osb); /* before journal shutdown, we should release slot_info */ ocfs2_free_slot_info(osb); @@ -1854,6 +1857,14 @@ static void ocfs2_dismount_volume(struct super_block *sb, int mnt_err) if (osb->cconn) ocfs2_super_unlock(osb, 1); + /* + * Drain pending suballoc reclaim work while the system inodes and + * the journal are still alive, since the worker needs to look up + * the global bitmap inode and start a transaction. + */ + if (osb->ocfs2_wq) + flush_workqueue(osb->ocfs2_wq); + ocfs2_release_system_inodes(osb); ocfs2_journal_shutdown(osb); @@ -2140,6 +2151,11 @@ static int ocfs2_initialize_super(struct super_block *sb, INIT_WORK(&osb->dquot_drop_work, ocfs2_drop_dquot_refs); init_llist_head(&osb->dquot_drop_list); + spin_lock_init(&osb->os_suballoc_reclaim_lock); + INIT_LIST_HEAD(&osb->os_suballoc_reclaim_list); + INIT_WORK(&osb->os_suballoc_reclaim_work, + ocfs2_suballoc_reclaim_worker); + /* get some pseudo constants for clustersize bits */ osb->s_clustersize_bits = le32_to_cpu(di->id2.i_super.s_clustersize_bits); From cc9f46d41502fde24ea46b46058f2bbddfa2e88e Mon Sep 17 00:00:00 2001 From: Thorsten Blum Date: Fri, 28 Aug 2026 19:43:37 +0200 Subject: [PATCH 1237/1352] init: simplify early_hostname() Inline the strscpy() check and remove the redundant arglen variable. Use %zu to format the unsigned maxlen argument and add a newline after the truncation warning. Link: https://lore.kernel.org/20260828174337.609333-2-blum@kernel.org Signed-off-by: Thorsten Blum Signed-off-by: Andrew Morton --- init/version.c | 8 +++----- 1 file changed, 3 insertions(+), 5 deletions(-) diff --git a/init/version.c b/init/version.c index 0bd5c45aabc463..cbae11b6f7688b 100644 --- a/init/version.c +++ b/init/version.c @@ -21,16 +21,14 @@ static int __init early_hostname(char *arg) { size_t bufsize = sizeof(init_uts_ns.name.nodename); size_t maxlen = bufsize - 1; - ssize_t arglen; if (!arg) return -EINVAL; - arglen = strscpy(init_uts_ns.name.nodename, arg, bufsize); - if (arglen < 0) { - pr_warn("hostname parameter exceeds %zd characters and will be truncated", + if (strscpy(init_uts_ns.name.nodename, arg, bufsize) < 0) + pr_warn("hostname parameter exceeds %zu characters and will be truncated\n", maxlen); - } + return 0; } early_param("hostname", early_hostname); From 1c162b1717fb099c13f8ce92c69ee724e62e3ec7 Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Sat, 22 Aug 2026 16:33:27 +0200 Subject: [PATCH 1238/1352] squashfs: fix fragment index table sizing overflow on 32-bit Patch series "squashfs: harden fragment index table sizing". Two integer overflows undermine fragment index table handling. One is in the original fragment sizing macros. The other is in a bounds check added by commit 1cac63cc9b2f ("Squashfs: add sanity checks to fragment reading at mount time"). Patch 1: the fragment byte count wraps on 32-bit, so the index table is allocated too small and squashfs_frag_lookup() reads out of bounds. A crafted image triggers a KASAN out-of-bounds read on a 32-bit build. With the fix the same image fails cleanly at mount. Patch 2: the check that the table fits before the next one adds two u64 values controlled by the filesystem image and can wrap. This patch (of 2): SQUASHFS_FRAGMENT_BYTES() multiplies the on-disk fragment count (an unsigned int) by sizeof(struct squashfs_fragment_entry), a size_t. On a 32-bit kernel that product is 32-bit and can wrap. squashfs_read_fragment_index_table() sizes the fragment index table from it, but squashfs_frag_lookup() bounds the fragment number against msblk->fragments, the unwrapped superblock value. The two disagree: an image declaring 0x10000001 fragments wraps the product to 16, so a single index entry is allocated, yet the lookup still accepts fragment 0x0fffffff: if (fragment >= msblk->fragments) return -EIO; block = SQUASHFS_FRAGMENT_INDEX(fragment); ... start_block = le64_to_cpu(msblk->fragment_index[block]); block is then 524287 and the read lands ~4MB past an 8-byte allocation. On a 32-bit build KASAN catches it when the crafted image is mounted and the file is stat'd. Cast to u64 in the macro so the multiplication is 64-bit on all targets. After conversion to index-table entries, SQUASHFS_FRAGMENT_INDEX_BYTES() is at most 64 MiB for any u32 count, so it fits both the unsigned int local and the int argument it feeds. 64-bit builds are unchanged. Link: https://lore.kernel.org/20260822143328.68867-1-kmehltretter@gmail.com Link: https://lore.kernel.org/20260822143328.68867-2-kmehltretter@gmail.com Fixes: ffae2cd73a9e ("Squashfs: header files") Signed-off-by: Karl Mehltretter Signed-off-by: Andrew Morton Assisted-by: Claude:claude-opus-5 Cc: Phillip Lougher Cc: --- fs/squashfs/squashfs_fs.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/squashfs/squashfs_fs.h b/fs/squashfs/squashfs_fs.h index a955d9369749f2..93436c7d80c973 100644 --- a/fs/squashfs/squashfs_fs.h +++ b/fs/squashfs/squashfs_fs.h @@ -136,7 +136,7 @@ static inline int squashfs_block_size(__le32 raw) /* fragment and fragment table defines */ #define SQUASHFS_FRAGMENT_BYTES(A) \ - ((A) * sizeof(struct squashfs_fragment_entry)) + ((u64)(A) * sizeof(struct squashfs_fragment_entry)) #define SQUASHFS_FRAGMENT_INDEX(A) (SQUASHFS_FRAGMENT_BYTES(A) / \ SQUASHFS_METADATA_SIZE) From 22c1126020b267dfde632f6b5f57b1fa1f20764b Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Sat, 22 Aug 2026 16:33:28 +0200 Subject: [PATCH 1239/1352] squashfs: make the fragment index table bounds check overflow-safe squashfs_read_fragment_index_table() checks that the table fits before the next one with: if (fragment_table_start + length > next_table) return ERR_PTR(-EINVAL); fragment_table_start comes from the superblock and is not validated before this point. A start of 2^64 - length wraps the sum to zero, so the check passes regardless of next_table and fails to reject the invalid table ordering. length then reaches kmalloc() through squashfs_read_table(). A fragment count of 0xffffffff asks for 64MB, order 14. GFP_KERNEL does not include __GFP_NOWARN, so the page allocator warns before the mount fails with -ENOMEM. With panic_on_warn, the warning panics the kernel. Compare the operands instead of adding them. id.c and export.c avoid the same wrap with an exact-size check. Keep the inequality here because a gap before the next table is still allowed. Link: https://lore.kernel.org/20260822143328.68867-3-kmehltretter@gmail.com Fixes: 1cac63cc9b2f ("Squashfs: add sanity checks to fragment reading at mount time") Signed-off-by: Karl Mehltretter Signed-off-by: Andrew Morton Assisted-by: Claude:claude-opus-5 Cc: Phillip Lougher Cc: --- fs/squashfs/fragment.c | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/fs/squashfs/fragment.c b/fs/squashfs/fragment.c index 49602b9a42e19e..c46e946fa47447 100644 --- a/fs/squashfs/fragment.c +++ b/fs/squashfs/fragment.c @@ -69,9 +69,11 @@ __le64 *squashfs_read_fragment_index_table(struct super_block *sb, /* * Sanity check, length bytes should not extend into the next table - * this check also traps instances where fragment_table_start is - * incorrectly larger than the next table start + * incorrectly larger than the next table start. Both values are read + * from the filesystem image, so compare without adding them. */ - if (fragment_table_start + length > next_table) + if (fragment_table_start > next_table || + length > next_table - fragment_table_start) return ERR_PTR(-EINVAL); table = squashfs_read_table(sb, fragment_table_start, length); From b54e27891ab321e993045d8b589dbea5138b7ec3 Mon Sep 17 00:00:00 2001 From: Aaron Tomlin Date: Sat, 29 Aug 2026 10:53:20 -0400 Subject: [PATCH 1240/1352] hung_task: reset warning budget when problem gets resolved Patch series "hung_task: Improve warning budget handling and task reporting", v10. The hung_task watchdog detects tasks stuck in TASK_UNINTERRUPTIBLE (D) state for longer than CONFIG_DEFAULT_HUNG_TASK_TIMEOUT seconds. To prevent log spam during system spikes, sysctl_hung_task_warnings enforces a budget on the number of logged warnings. However, the current implementation has two major limitations: 1. Permanent exhaustion of warning budget sysctl_hung_task_warnings is decremented directly when printing warnings. Once this budget hits zero, no further warnings are reported until an administrator manually updates the sysctl value or reboots the system. Consequently, a single temporary hang episode permanently blinds the kernel watchdog to any subsequent hung tasks after system recovery. 2. Total log suppression when budget is exhausted Once the warning budget reaches zero, hung_task_info() completely suppresses all output, including the basic single-line alert. While suppressing verbose stack dumps and lock debugging is desirable to prevent dmesg flooding, hiding basic task alerts leaves administrators entirely unaware that tasks are hanging. This patch series resolves both limitations by decoupling the configured warning limit from the active runtime budget, automatically resetting the budget upon system recovery or sysctl updates, and emitting a single aggregate summary line when hung tasks are detected under an exhausted warning budget. Patch 1 separates the configured sysctl hung_task_warnings from the runtime budget, making khungtaskd the sole owner of runtime budget updates. The budget is reloaded directly when a scan finds zero hung tasks, or via an atomic reset request published on sysctl write. Patch 2 prevents dmesg flooding during system-wide hangs by keeping non-panic per-task stack dumps budgeted, while providing ongoing visibility by logging a single aggregate summary line at the end of each scan iteration when the warning budget is exhausted. This patch (of 2): The sysctl hung_task_warnings currently holds both the configured warning limit and the remaining budget. Each detailed report decrements the sysctl, so once it reaches zero, the configured limit is lost and cannot be restored automatically. Keep sysctl_hung_task_warnings as the configured warning limit and make khungtaskd the sole owner of the remaining budget. A check that finds no hung tasks reloads the budget directly from the configured limit. A successful sysctl write publishes an atomic reset request, which khungtaskd consumes at the start of the next check. Link: https://lore.kernel.org/20260829145321.18423-1-atomlin@atomlin.com Link: https://lore.kernel.org/20260829145321.18423-2-atomlin@atomlin.com Signed-off-by: Aaron Tomlin Signed-off-by: Andrew Morton Suggested-by: Petr Mladek Suggested-by: Lance Yang Tested-by: Lance Yang Reviewed-by: Lance Yang Reviewed-by: Bradley Morgan Cc: David Laight Cc: "Masami Hiramatsu (Google)" --- Documentation/admin-guide/sysctl/kernel.rst | 5 ++- kernel/hung_task.c | 49 +++++++++++++++++---- 2 files changed, 44 insertions(+), 10 deletions(-) diff --git a/Documentation/admin-guide/sysctl/kernel.rst b/Documentation/admin-guide/sysctl/kernel.rst index ffea61d448ebb1..c03369e234a928 100644 --- a/Documentation/admin-guide/sysctl/kernel.rst +++ b/Documentation/admin-guide/sysctl/kernel.rst @@ -459,8 +459,9 @@ hung_task_warnings ================== The maximum number of warnings to report. During a check interval -if a hung task is detected, this value is decreased by 1. -When this value reaches 0, no more warnings will be reported. +if a hung task is detected, the internal warning budget is decreased by 1. +When this budget reaches 0, no more detailed warnings will be reported. The +warning budget is reset to the configured limit when no hung task is found. This file shows up if ``CONFIG_DETECT_HUNG_TASK`` is enabled. -1: report an infinite number of warnings. diff --git a/kernel/hung_task.c b/kernel/hung_task.c index 6fcc94ce4ca9d2..a5043188456d42 100644 --- a/kernel/hung_task.c +++ b/kernel/hung_task.c @@ -57,8 +57,20 @@ unsigned long __read_mostly sysctl_hung_task_timeout_secs = CONFIG_DEFAULT_HUNG_ */ static unsigned long __read_mostly sysctl_hung_task_check_interval_secs; +/* + * Limit the number of printed hung tasks to prevent printing + * the same or similar backtraces repeatedly. + */ static int __read_mostly sysctl_hung_task_warnings = 10; +/* + * The number of hung tasks which still can be reported. + * The budget gets restored to the original limit when + * the previous stall is resolved. + */ +static int hung_task_warnings_budget = 10; +static atomic_t reset_hung_task_warnings = ATOMIC_INIT(0); + static int __read_mostly did_panic; static bool hung_task_call_panic; @@ -245,11 +257,11 @@ static void hung_task_info(struct task_struct *t, unsigned long timeout, /* * The given task did not get scheduled for more than * CONFIG_DEFAULT_HUNG_TASK_TIMEOUT. Therefore, complain - * accordingly + * accordingly with full details if the budget is not exhausted. */ - if (sysctl_hung_task_warnings || hung_task_call_panic) { - if (sysctl_hung_task_warnings > 0) - sysctl_hung_task_warnings--; + if (hung_task_warnings_budget || hung_task_call_panic) { + if (hung_task_warnings_budget > 0) + hung_task_warnings_budget--; pr_err("INFO: task %s:%d blocked%s for more than %ld seconds.\n", t->comm, t->pid, t->in_iowait ? " in I/O wait" : "", (jiffies - t->last_switch_time) / HZ); @@ -264,7 +276,7 @@ static void hung_task_info(struct task_struct *t, unsigned long timeout, sched_show_task(t); debug_show_blocker(t, timeout); - if (!sysctl_hung_task_warnings) + if (!hung_task_warnings_budget) pr_info("Future hung task reports are suppressed, see sysctl kernel.hung_task_warnings\n"); } @@ -304,7 +316,7 @@ static void check_hung_uninterruptible_tasks(unsigned long timeout) unsigned long last_break = jiffies; struct task_struct *g, *t; unsigned long this_round_count; - int need_warning = sysctl_hung_task_warnings; + int need_warning; unsigned long si_mask = hung_task_si_mask; /* @@ -314,6 +326,11 @@ static void check_hung_uninterruptible_tasks(unsigned long timeout) if (test_taint(TAINT_DIE) || did_panic) return; + if (atomic_xchg_acquire(&reset_hung_task_warnings, 0)) + hung_task_warnings_budget = + READ_ONCE(sysctl_hung_task_warnings); + need_warning = hung_task_warnings_budget; + this_round_count = 0; rcu_read_lock(); for_each_process_thread(g, t) { @@ -340,8 +357,11 @@ static void check_hung_uninterruptible_tasks(unsigned long timeout) unlock: rcu_read_unlock(); - if (!this_round_count) + if (!this_round_count) { + hung_task_warnings_budget = + READ_ONCE(sysctl_hung_task_warnings); return; + } if (need_warning || hung_task_call_panic) { si_mask |= SYS_INFO_LOCKS; @@ -425,6 +445,19 @@ static int proc_dohung_task_timeout_secs(const struct ctl_table *table, int writ return ret; } +static int proc_dohung_task_warnings(const struct ctl_table *table, int write, + void *buffer, + size_t *lenp, loff_t *ppos) +{ + int ret; + + ret = proc_dointvec_minmax(table, write, buffer, lenp, ppos); + if (!ret && write) + atomic_set_release(&reset_hung_task_warnings, 1); + + return ret; +} + /* * This is needed for proc_doulongvec_minmax of sysctl_hung_task_timeout_secs * and hung_task_check_interval_secs @@ -480,7 +513,7 @@ static const struct ctl_table hung_task_sysctls[] = { .data = &sysctl_hung_task_warnings, .maxlen = sizeof(int), .mode = 0644, - .proc_handler = proc_dointvec_minmax, + .proc_handler = proc_dohung_task_warnings, .extra1 = SYSCTL_NEG_ONE, }, { From 792477c71f1db409c00ce85f70df73eb7942ddc9 Mon Sep 17 00:00:00 2001 From: Aaron Tomlin Date: Sat, 29 Aug 2026 10:53:21 -0400 Subject: [PATCH 1241/1352] hung_task: log summary line when warning budget is exhausted Once the warning budget is exhausted, hung_task_info() normally stops printing per-task details. When panic is triggered, full details are still printed so diagnostics remain available before panic. To retain visibility without restoring per-task output after budget exhaustion, emit a single aggregate summary line at the end of each watchdog scan that detects hung tasks with an exhausted budget. This keeps non-panic per-task reports budgeted during system-wide hangs. Link: https://lore.kernel.org/20260829145321.18423-3-atomlin@atomlin.com Signed-off-by: Aaron Tomlin Signed-off-by: Andrew Morton Suggested-by: Petr Mladek Suggested-by: Lance Yang Reviewed-by: Petr Mladek Reviewed-by: Lance Yang Reviewed-by: Bradley Morgan Cc: David Laight Cc: "Masami Hiramatsu (Google)" --- kernel/hung_task.c | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/kernel/hung_task.c b/kernel/hung_task.c index a5043188456d42..49d47ae475ab1c 100644 --- a/kernel/hung_task.c +++ b/kernel/hung_task.c @@ -277,7 +277,7 @@ static void hung_task_info(struct task_struct *t, unsigned long timeout, debug_show_blocker(t, timeout); if (!hung_task_warnings_budget) - pr_info("Future hung task reports are suppressed, see sysctl kernel.hung_task_warnings\n"); + pr_info("hung_task: further per-task details suppressed until warning budget is reset or panic is triggered (see sysctl kernel.hung_task_warnings)\n"); } touch_nmi_watchdog(); @@ -363,6 +363,10 @@ static void check_hung_uninterruptible_tasks(unsigned long timeout) return; } + if (!hung_task_warnings_budget && !hung_task_call_panic) + pr_info("hung_task: %lu hung tasks detected, warning budget exhausted\n", + this_round_count); + if (need_warning || hung_task_call_panic) { si_mask |= SYS_INFO_LOCKS; From b8c2cf27978cffc1d23806df4aa7f2db3b846c14 Mon Sep 17 00:00:00 2001 From: Wilson Felipe Pereira Date: Tue, 18 Aug 2026 23:16:32 +0000 Subject: [PATCH 1242/1352] init, arch: make CONFIG_COMMAND_LINE_SIZE globally configurable MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Currently, s390 has the ability to configure the maximum kernel command line size via Kconfig (CONFIG_COMMAND_LINE_SIZE). Other architectures define a hardcoded COMMAND_LINE_SIZE macro in their setup.h headers. In some use cases, such as netboot kernels, rootfs configurations, or larger initramfs setups, a larger command line size is required. While for embedded workloads, it can be reduced to save memory. Move CONFIG_COMMAND_LINE_SIZE out of arch/s390/Kconfig and into init/Kconfig under General setup, and update every architecture's setup.h header to define COMMAND_LINE_SIZE as CONFIG_COMMAND_LINE_SIZE. For user-space API (uapi) headers, wrap the definition in an `#ifdef __KERNEL__` guard and retain the historical hardcoded default in the `#else` block. When user-space headers are installed via `make headers_install`, unifdef strips out the kernel section, ensuring the same value as before for user-space applications including ``. For S390, the range is kept the same, but other architectures have varying constraints. S390 requires a minimum of 896 bytes to protect legacy bootloaders from overwriting the .text section. ARM, M68K, and NIOS2 allocate the command line directly on severely constrained decompressor stacks, so their ranges are strictly capped at 2048 bytes to prevent deterministic stack exhaustion and boot panics. PowerPC (PPC) boot wrappers silently truncate arguments past 2048 bytes, so it is also capped at 2048 to prevent silent parameter loss. The SuperH (SUPERH) boot parameter page allocates exactly PAGE_SIZE (typically 4096 bytes), and placing a 4096-byte command line starting at offset 256 would cause strscpy() to read out of bounds; it is capped at 3840 bytes. Alpha physically limits its boot parameter block to 256 bytes, so its limit is strictly locked to 256. All other architectures are capped at 4096 bytes to prevent unreasonable allocations. Link: https://lore.kernel.org/20260818231646.804507-2-wfelipe@google.com Signed-off-by: Maciej Żenczykowski Signed-off-by: Wilson Felipe Pereira Signed-off-by: Andrew Morton Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Arnd Bergmann Cc: Christian Borntraeger Cc: Heiko Carstens Cc: Palmer Dabbelt Cc: Sven Schnelle Cc: Vasily Gorbik --- arch/alpha/include/uapi/asm/setup.h | 4 ++++ arch/arc/include/asm/setup.h | 2 +- arch/arm/include/uapi/asm/setup.h | 6 +++++- arch/arm64/include/uapi/asm/setup.h | 4 ++++ arch/loongarch/include/uapi/asm/setup.h | 4 ++++ arch/m68k/include/uapi/asm/setup.h | 6 +++++- arch/microblaze/include/uapi/asm/setup.h | 4 ++++ arch/mips/include/uapi/asm/setup.h | 4 ++++ arch/parisc/include/uapi/asm/setup.h | 4 ++++ arch/powerpc/include/uapi/asm/setup.h | 4 ++++ arch/riscv/include/uapi/asm/setup.h | 4 ++++ arch/s390/Kconfig | 8 -------- arch/sparc/include/uapi/asm/setup.h | 10 +++++++--- arch/um/include/asm/setup.h | 2 +- arch/x86/include/asm/setup.h | 2 +- arch/xtensa/include/uapi/asm/setup.h | 4 ++++ include/uapi/asm-generic/setup.h | 4 ++++ init/Kconfig | 16 ++++++++++++++++ 18 files changed, 76 insertions(+), 16 deletions(-) diff --git a/arch/alpha/include/uapi/asm/setup.h b/arch/alpha/include/uapi/asm/setup.h index f881ea5947cbc1..169f743ef7658d 100644 --- a/arch/alpha/include/uapi/asm/setup.h +++ b/arch/alpha/include/uapi/asm/setup.h @@ -2,6 +2,10 @@ #ifndef _UAPI__ALPHA_SETUP_H #define _UAPI__ALPHA_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 256 +#endif #endif /* _UAPI__ALPHA_SETUP_H */ diff --git a/arch/arc/include/asm/setup.h b/arch/arc/include/asm/setup.h index 1c6db599e1fcc9..60e158d58cec77 100644 --- a/arch/arc/include/asm/setup.h +++ b/arch/arc/include/asm/setup.h @@ -9,7 +9,7 @@ #include #include -#define COMMAND_LINE_SIZE 256 +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE /* * Data structure to map a ID to string diff --git a/arch/arm/include/uapi/asm/setup.h b/arch/arm/include/uapi/asm/setup.h index 8e50e034fec73a..4aa93558af1e7f 100644 --- a/arch/arm/include/uapi/asm/setup.h +++ b/arch/arm/include/uapi/asm/setup.h @@ -17,7 +17,11 @@ #include -#define COMMAND_LINE_SIZE 1024 +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else +#define COMMAND_LINE_SIZE 1024 +#endif /* The list ends with an ATAG_NONE node. */ #define ATAG_NONE 0x00000000 diff --git a/arch/arm64/include/uapi/asm/setup.h b/arch/arm64/include/uapi/asm/setup.h index 5d703888f35110..2236890175a5ab 100644 --- a/arch/arm64/include/uapi/asm/setup.h +++ b/arch/arm64/include/uapi/asm/setup.h @@ -22,6 +22,10 @@ #include +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 2048 +#endif #endif diff --git a/arch/loongarch/include/uapi/asm/setup.h b/arch/loongarch/include/uapi/asm/setup.h index d46363ce3e024c..03c7bfa1e5d9f2 100644 --- a/arch/loongarch/include/uapi/asm/setup.h +++ b/arch/loongarch/include/uapi/asm/setup.h @@ -3,6 +3,10 @@ #ifndef _UAPI_ASM_LOONGARCH_SETUP_H #define _UAPI_ASM_LOONGARCH_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 4096 +#endif #endif /* _UAPI_ASM_LOONGARCH_SETUP_H */ diff --git a/arch/m68k/include/uapi/asm/setup.h b/arch/m68k/include/uapi/asm/setup.h index 25fe26d5597cc6..2d5b24a5345f92 100644 --- a/arch/m68k/include/uapi/asm/setup.h +++ b/arch/m68k/include/uapi/asm/setup.h @@ -12,6 +12,10 @@ #ifndef _UAPI_M68K_SETUP_H #define _UAPI_M68K_SETUP_H -#define COMMAND_LINE_SIZE 256 +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else +#define COMMAND_LINE_SIZE 256 +#endif #endif /* _UAPI_M68K_SETUP_H */ diff --git a/arch/microblaze/include/uapi/asm/setup.h b/arch/microblaze/include/uapi/asm/setup.h index 16c56807f86a2d..e4b253064c7d99 100644 --- a/arch/microblaze/include/uapi/asm/setup.h +++ b/arch/microblaze/include/uapi/asm/setup.h @@ -12,6 +12,10 @@ #ifndef _UAPI_ASM_MICROBLAZE_SETUP_H #define _UAPI_ASM_MICROBLAZE_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 256 +#endif #endif /* _UAPI_ASM_MICROBLAZE_SETUP_H */ diff --git a/arch/mips/include/uapi/asm/setup.h b/arch/mips/include/uapi/asm/setup.h index 7d48c433b0c27d..8d6c474835aece 100644 --- a/arch/mips/include/uapi/asm/setup.h +++ b/arch/mips/include/uapi/asm/setup.h @@ -2,7 +2,11 @@ #ifndef _UAPI_MIPS_SETUP_H #define _UAPI_MIPS_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 4096 +#endif #endif /* _UAPI_MIPS_SETUP_H */ diff --git a/arch/parisc/include/uapi/asm/setup.h b/arch/parisc/include/uapi/asm/setup.h index 78b2f4ec7d6522..cfa77e84205dc6 100644 --- a/arch/parisc/include/uapi/asm/setup.h +++ b/arch/parisc/include/uapi/asm/setup.h @@ -2,6 +2,10 @@ #ifndef _PARISC_SETUP_H #define _PARISC_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 1024 +#endif #endif /* _PARISC_SETUP_H */ diff --git a/arch/powerpc/include/uapi/asm/setup.h b/arch/powerpc/include/uapi/asm/setup.h index c54940b09d065c..daa15ac4e94c6e 100644 --- a/arch/powerpc/include/uapi/asm/setup.h +++ b/arch/powerpc/include/uapi/asm/setup.h @@ -2,6 +2,10 @@ #ifndef _UAPI_ASM_POWERPC_SETUP_H #define _UAPI_ASM_POWERPC_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 2048 +#endif #endif /* _UAPI_ASM_POWERPC_SETUP_H */ diff --git a/arch/riscv/include/uapi/asm/setup.h b/arch/riscv/include/uapi/asm/setup.h index eb4f0209c6960e..a6c1f4b0987e35 100644 --- a/arch/riscv/include/uapi/asm/setup.h +++ b/arch/riscv/include/uapi/asm/setup.h @@ -3,6 +3,10 @@ #ifndef _UAPI_ASM_RISCV_SETUP_H #define _UAPI_ASM_RISCV_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 2048 +#endif #endif /* _UAPI_ASM_RISCV_SETUP_H */ diff --git a/arch/s390/Kconfig b/arch/s390/Kconfig index 4b51bc6e8948d7..026ba041ca9509 100644 --- a/arch/s390/Kconfig +++ b/arch/s390/Kconfig @@ -521,14 +521,6 @@ endchoice config 64BIT def_bool y -config COMMAND_LINE_SIZE - int "Maximum size of kernel command line" - default 4096 - range 896 1048576 - help - This allows you to specify the maximum length of the kernel command - line. - config SMP def_bool y diff --git a/arch/sparc/include/uapi/asm/setup.h b/arch/sparc/include/uapi/asm/setup.h index 3c208a4dd46405..7054f5249a3a68 100644 --- a/arch/sparc/include/uapi/asm/setup.h +++ b/arch/sparc/include/uapi/asm/setup.h @@ -6,10 +6,14 @@ #ifndef _UAPI_SPARC_SETUP_H #define _UAPI_SPARC_SETUP_H -#if defined(__sparc__) && defined(__arch64__) -# define COMMAND_LINE_SIZE 2048 +#ifdef __KERNEL__ +# define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE #else -# define COMMAND_LINE_SIZE 256 +# if defined(__sparc__) && defined(__arch64__) +# define COMMAND_LINE_SIZE 2048 +# else +# define COMMAND_LINE_SIZE 256 +# endif #endif diff --git a/arch/um/include/asm/setup.h b/arch/um/include/asm/setup.h index 80ada899f25426..bc83dc4d467d30 100644 --- a/arch/um/include/asm/setup.h +++ b/arch/um/include/asm/setup.h @@ -6,6 +6,6 @@ * command line, so this choice is ok. */ -#define COMMAND_LINE_SIZE 4096 +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE #endif /* SETUP_H_INCLUDED */ diff --git a/arch/x86/include/asm/setup.h b/arch/x86/include/asm/setup.h index 895d09faaf832e..1b333bb091d7f2 100644 --- a/arch/x86/include/asm/setup.h +++ b/arch/x86/include/asm/setup.h @@ -4,7 +4,7 @@ #include -#define COMMAND_LINE_SIZE 2048 +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE #include #include diff --git a/arch/xtensa/include/uapi/asm/setup.h b/arch/xtensa/include/uapi/asm/setup.h index 5356a5fd4d1737..dcf33a403527a3 100644 --- a/arch/xtensa/include/uapi/asm/setup.h +++ b/arch/xtensa/include/uapi/asm/setup.h @@ -12,6 +12,10 @@ #ifndef _XTENSA_SETUP_H #define _XTENSA_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 256 +#endif #endif diff --git a/include/uapi/asm-generic/setup.h b/include/uapi/asm-generic/setup.h index 88ac5100df3598..b8d06e6d56bd7e 100644 --- a/include/uapi/asm-generic/setup.h +++ b/include/uapi/asm-generic/setup.h @@ -2,6 +2,10 @@ #ifndef __ASM_GENERIC_SETUP_H #define __ASM_GENERIC_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 512 +#endif #endif /* __ASM_GENERIC_SETUP_H */ diff --git a/init/Kconfig b/init/Kconfig index 8583d9f06c522e..edc79893e1ac4f 100644 --- a/init/Kconfig +++ b/init/Kconfig @@ -1614,6 +1614,22 @@ config CMDLINE_FROM_BOOTCONFIG If unsure, say N. +config COMMAND_LINE_SIZE + int "Maximum size of kernel command line" + default 4096 if S390 || LOONGARCH || MIPS || UML + default 2048 if X86 || ARM64 || PPC || SPARC64 || RISCV + default 1024 if ARM || PARISC + default 256 if ALPHA || ARC || M68K || MICROBLAZE || SPARC32 || XTENSA + default 512 + range 896 1048576 if S390 + range 256 2048 if ARM || M68K || NIOS2 || PPC + range 256 3840 if SUPERH + range 256 256 if ALPHA + range 256 4096 + help + This allows you to specify the maximum length of the kernel command + line. + config CMDLINE_LOG_WRAP_IDEAL_LEN int "Length to try to wrap the cmdline when logged at boot" default 1021 From c13fd394a9e1b011d952cb316e0f00a297db40eb Mon Sep 17 00:00:00 2001 From: Wilson Felipe Pereira Date: Tue, 18 Aug 2026 23:16:33 +0000 Subject: [PATCH 1243/1352] init/Kconfig: make config INIT_ENV_ARG_LIMIT user-configurable MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Patch series "init, arch: make command line size and init arg limit configurable", v2. This patch series promotes COMMAND_LINE_SIZE from arch/s390/Kconfig to init/Kconfig to be generally available to other architectures. In some use cases, such as netboot kernels, rootfs configurations, larger initramfs setups, require larger sizes. While for embedded workloads, it can be reduced to save memory. Since COMMAND_LINE_SIZE can be larger, it also makes sense to allow INIT_ENV_ARG_LIMIT to be configured. This patch (of 2): INIT_ENV_ARG_LIMIT is defined without a prompt string (`int`), making it a hidden Kconfig symbol that defaults to 32 (or 128 for UML) and cannot be configured in `make menuconfig`. Now that CONFIG_COMMAND_LINE_SIZE is configurable across all architectures, users who select larger kernel command lines (e.g., 4096 bytes) may pass more than 32 command-line arguments or environment variables (`foo=bar`) to `/sbin/init`. If INIT_ENV_ARG_LIMIT remains hardcoded at 32, any argument after the 32nd sets the panic_later flag and causes a hard kernel panic on boot. Add a prompt string ("Maximum number of kernel command line arguments") and a `range 32 4096` to `config INIT_ENV_ARG_LIMIT` so that users can configure their init argument and environment variable limit when needed, while preserving the existing default of 32 for standard builds. Link: https://lore.kernel.org/20260818231646.804507-1-wfelipe@google.com Link: https://lore.kernel.org/20260818231646.804507-3-wfelipe@google.com Signed-off-by: Maciej Żenczykowski Signed-off-by: Wilson Felipe Pereira Signed-off-by: Andrew Morton Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Arnd Bergmann Cc: Christian Borntraeger Cc: Heiko Carstens Cc: Palmer Dabbelt Cc: Sven Schnelle Cc: Vasily Gorbik --- init/Kconfig | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/init/Kconfig b/init/Kconfig index edc79893e1ac4f..3b141ba427d5e0 100644 --- a/init/Kconfig +++ b/init/Kconfig @@ -231,9 +231,10 @@ config BROKEN_ON_SMP default y config INIT_ENV_ARG_LIMIT - int + int "Maximum number of kernel command line arguments" default 32 if !UML default 128 if UML + range 32 4096 help Maximum of each of the number of arguments and environment variables passed to init from the kernel command line. From 5626e0aee919241f62c9d9a0503feebf2d8598ce Mon Sep 17 00:00:00 2001 From: Zhan Xusheng Date: Mon, 17 Aug 2026 20:16:13 +0800 Subject: [PATCH 1244/1352] minmax.h: update the stale 'x' versus 'ux' comment Commit b280bb27a9f7 ("minmax.h: reduce the #define expansion of min(), max() and clamp()") made __sign_use(), __is_nonneg() and __types_ok() take only 'ux', and commit a5743f32baec ("minmax.h: use BUILD_BUG_ON_MSG() for the lo < hi test in clamp()") did the same for the clamp() limit test. The comment describing the old split was added one patch earlier and was never updated. 'ux' now carries the value check too, since __is_nonneg() tests it rather than the original expression, and nothing here looks at the value of 'x' any more: it is expanded only to initialise 'ux' and in the error message, as the first of those changes intended. Link: https://lore.kernel.org/20260817121613.3846511-1-zhanxusheng@xiaomi.com Signed-off-by: Zhan Xusheng Signed-off-by: Andrew Morton Cc: David Laight Cc: "H. Peter Anvin" --- include/linux/minmax.h | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/include/linux/minmax.h b/include/linux/minmax.h index a0158db54a0411..5ef4d58c0c4291 100644 --- a/include/linux/minmax.h +++ b/include/linux/minmax.h @@ -38,9 +38,9 @@ * Note that 'x' is the original expression, and 'ux' is the unique variable * that contains the value. * - * We use 'ux' for pure type checking, and 'x' for when we need to look at the - * value (but without evaluating it for side effects! - * Careful to only ever evaluate it with sizeof() or __builtin_constant_p() etc). + * We use 'ux' for both the type and the value checks, so 'x' itself is only + * expanded twice: once to initialise 'ux', and once quoted in the error + * message. * * Pointers end up being checked by the normal C type rules at the actual * comparison, and these expressions only need to be careful to not cause From 9d31a3f098b00fe41047aa68d3d3a34e61dd27c0 Mon Sep 17 00:00:00 2001 From: Julian Braha Date: Mon, 17 Aug 2026 22:03:11 +0100 Subject: [PATCH 1245/1352] lib: cleanup "fake" tristates in Kconfig These 7 DECOMPRESS_ options (e.g. DECOMPRESS_GZIP) currently have the tristate type, but can never be set to M. Their only valid values are Y and N, making them effectively booleans. Let's make their types more accurate by changing them to 'bool'. Note that this is only a code cleanup, there is no functional change. These bistates were found by kconfirm, a static analysis tool for Kconfig. Link: https://lore.kernel.org/20260817210311.2142999-1-julianbraha@gmail.com Signed-off-by: Julian Braha Signed-off-by: Andrew Morton Cc: Arnd Bergmann Cc: Jani Nikula Cc: Julia Lawall --- lib/Kconfig | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/lib/Kconfig b/lib/Kconfig index 4e6b34c3346d56..1e42f5167c68bc 100644 --- a/lib/Kconfig +++ b/lib/Kconfig @@ -219,29 +219,29 @@ source "lib/xz/Kconfig" # config DECOMPRESS_GZIP select ZLIB_INFLATE - tristate + bool config DECOMPRESS_BZIP2 - tristate + bool config DECOMPRESS_LZMA - tristate + bool config DECOMPRESS_XZ select XZ_DEC - tristate + bool config DECOMPRESS_LZO select LZO_DECOMPRESS - tristate + bool config DECOMPRESS_LZ4 select LZ4_DECOMPRESS - tristate + bool config DECOMPRESS_ZSTD select ZSTD_DECOMPRESS - tristate + bool # # Generic allocator support is selected if needed From abf855160a55cda018d11ca031581e59e0bce0ea Mon Sep 17 00:00:00 2001 From: Wilson Felipe Pereira Date: Tue, 18 Aug 2026 04:53:46 +0000 Subject: [PATCH 1246/1352] init/main: fix off-by-one in argv_init cleanup Patch series "init: fix array boundary bugs in boot parameter parsing". This series fixes two distinct boundary logic edge-case bugs in `init/main.c` related to parsing boot command-line arguments and environment variables. Both bugs have been present since the early git history (Linux-2.6.12-rc2). 1. The first patch fixes an off-by-one error in `init_setup()` where the final slot of the `argv_init` array was left uncleared. This allowed a stale kernel parameter to leak into the `init` process's user-space command line if exactly `MAX_INIT_ARGS` unknown parameters were passed. 2. The second patch fixes a false-positive kernel panic in `unknown_bootoption()`. If a user filled the environment variable array up to its exact limit (32) and then attempted to overwrite the final variable, the kernel would panic before evaluating whether it was a harmless duplicate. Exact QEMU reproduction steps for both edge cases are documented inside their respective commit descriptions. This patch (of 2): When cleaning up argv_init in init_setup() and rdinit_setup(), the loop terminates one element early due to using '<' instead of '<='. Since argv_init is sized MAX_INIT_ARGS+2, index MAX_INIT_ARGS is a valid element that should be cleared to NULL. If exactly MAX_INIT_ARGS unknown arguments are passed before 'init=', the uncleared argv_init[MAX_INIT_ARGS] can act as a ghost argument to /sbin/init or cause a spurious kernel panic when later appended to. To verify the argument leak, boot a VM into a shell with 32 unknown kernel arguments, the init parameter, and 31 user arguments: STALE_ARGS=$(for i in {1..32}; do echo -n "stale$i "; done) USER_ARGS=$(for i in {1..31}; do echo -n "user$i "; done) qemu-system-x86_64 -kernel bzImage \ -append "$STALE_ARGS init=/bin/sh $USER_ARGS" Running `cat /proc/1/cmdline` inside the shell reveals that the 32nd kernel argument ('stale32') incorrectly leaked into the init process's command line. This patch zeroes the final slot, cleanly terminating the array. Link: https://lore.kernel.org/20260818045357.4123784-1-wfelipe@google.com Link: https://lore.kernel.org/20260818045357.4123784-2-wfelipe@google.com Fixes: 1da177e4c3f4 ("Linux-2.6.12-rc2") Fixes: ffdfc40976dd ("[PATCH] Add rdinit parameter to pick early userspace init") Signed-off-by: Wilson Felipe Pereira Signed-off-by: Andrew Morton --- init/main.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/init/main.c b/init/main.c index 31f2bf54976ab9..517b76447950f9 100644 --- a/init/main.c +++ b/init/main.c @@ -581,7 +581,7 @@ static int __init init_setup(char *str) * the shell think it should execute a script with such name. * So we ignore all arguments entered _before_ init=... [MJ] */ - for (i = 1; i < MAX_INIT_ARGS; i++) + for (i = 1; i <= MAX_INIT_ARGS; i++) argv_init[i] = NULL; return 1; } @@ -594,7 +594,7 @@ static int __init rdinit_setup(char *str) ramdisk_execute_command = str; ramdisk_execute_command_set = true; /* See "auto" comment in init_setup */ - for (i = 1; i < MAX_INIT_ARGS; i++) + for (i = 1; i <= MAX_INIT_ARGS; i++) argv_init[i] = NULL; return 1; } From 3b95152ac1df6d3cf836e3700cfaba635ac2c0fd Mon Sep 17 00:00:00 2001 From: Wilson Felipe Pereira Date: Tue, 18 Aug 2026 04:53:47 +0000 Subject: [PATCH 1247/1352] init/main: fix false-positive kernel panic on environment variable overwrite In unknown_bootoption(), the limit checking for environment variables sets panic_later *before* checking if the variable already exists in envp_init. If a user passes exactly MAX_INIT_ENVS custom variables and then overwrites the final variable by matching its key, it causes a false-positive hard panic on boot despite not actually exceeding the array bounds or increasing the total variable count. Swapping the order of these checks allows the duplicate check to break out of the loop before the panic flag is erroneously latched. To verify, boot a VM with 31 custom variables (filling the array up to its limit of 32) and then overwrite the very last variable: ENV_VARS=$(for i in {1..31}; do echo -n "var$i=$i "; done) qemu-system-x86_64 -kernel bzImage -append "$ENV_VARS var31=overwrite" Without this patch, the kernel crashes instantly with: Kernel panic - not syncing: Too many boot env vars at 'var31=overwrite' With this patch, the kernel safely overwrites the variable and boots. Link: https://lore.kernel.org/20260818045357.4123784-3-wfelipe@google.com Fixes: 1da177e4c3f4 ("Linux-2.6.12-rc2") Signed-off-by: Wilson Felipe Pereira Signed-off-by: Andrew Morton --- init/main.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/init/main.c b/init/main.c index 517b76447950f9..d1cd8efb823860 100644 --- a/init/main.c +++ b/init/main.c @@ -548,12 +548,12 @@ static int __init unknown_bootoption(char *param, char *val, /* Environment option */ unsigned int i; for (i = 0; envp_init[i]; i++) { + if (!strncmp(param, envp_init[i], len+1)) + break; if (i == MAX_INIT_ENVS) { panic_later = "env"; panic_param = param; } - if (!strncmp(param, envp_init[i], len+1)) - break; } envp_init[i] = param; } else { From 90e3253ee364698504a501da091e97749ba03c53 Mon Sep 17 00:00:00 2001 From: Andrei Vagin Date: Sun, 16 Aug 2026 16:12:15 +0000 Subject: [PATCH 1248/1352] proc: report SIGEV_NONE in /proc/pid/timers if target task has died When a posix timer is created targeting a specific thread (using SIGEV_SIGNAL | SIGEV_THREAD_ID), it takes a reference to the target struct pid in timer->it_pid. If the target thread subsequently terminates, its numeric tid is freed and can be recycled for an unrelated task. However, the timer holds its reference to the original struct pid. show_timer() in /proc/[pid]/timers previously called pid_nr_ns() directly on timer->it_pid without checking whether any task remained attached to that struct pid. As a result: 1. It reported the stale tid, which could mistakenly refer to a recycled pid. 2. In the kernel, expired signals for dead target threads are dropped by posixtimer_send_sigqueue() because posixtimer_get_target() returns NULL, so the timer functionally acts as SIGEV_NONE. 3. Checkpoint/restore tools (CRIU) parsing /proc/[pid]/timers would try to restore a timer with SIGEV_SIGNAL | SIGEV_THREAD_ID targeting a non-existent or unrelated thread. Check pid_has_task(timer->it_pid, timer->it_pid_type) in show_timer(). If the target task has died, override notify to SIGEV_NONE and report PID 0 (e.g., 'notify: none/pid.0'). Link: https://lore.kernel.org/20260816161216.984580-1-avagin@google.com Fixes: 57b8015e07a7 ("posix-timers: Show sigevent info in proc file") Signed-off-by: Andrei Vagin Signed-off-by: Andrew Morton Reviewed-by: Pavel Tikhomirov Cc: Thomas Gleixner --- fs/proc/base.c | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/fs/proc/base.c b/fs/proc/base.c index 6a39de424f62a1..1455de58e53cf0 100644 --- a/fs/proc/base.c +++ b/fs/proc/base.c @@ -2519,17 +2519,23 @@ static int show_timer(struct seq_file *m, void *v) struct k_itimer *timer = hlist_entry((struct hlist_node *)v, struct k_itimer, list); struct timers_private *tp = m->private; int notify = timer->it_sigev_notify; + pid_t nr = 0; guard(spinlock_irq)(&timer->it_lock); if (!posixtimer_valid(timer)) return 0; + if (timer->it_pid && pid_has_task(timer->it_pid, timer->it_pid_type)) + nr = pid_nr_ns(timer->it_pid, tp->ns); + else + notify = SIGEV_NONE; + seq_printf(m, "ID: %d\n", timer->it_id); seq_printf(m, "signal: %d/%px\n", timer->sigq.info.si_signo, timer->sigq.info.si_value.sival_ptr); seq_printf(m, "notify: %s/%s.%d\n", nstr[notify & ~SIGEV_THREAD_ID], (notify & SIGEV_THREAD_ID) ? "tid" : "pid", - pid_nr_ns(timer->it_pid, tp->ns)); + nr); seq_printf(m, "ClockID: %d\n", timer->it_clock); return 0; From daa8f6d5667051cc68a665ccbd4e635494c90ef1 Mon Sep 17 00:00:00 2001 From: Konstantin Khorenko Date: Fri, 14 Aug 2026 18:57:09 +0200 Subject: [PATCH 1249/1352] selftests/core: fix unshare_test with large fs.nr_open The test assumes fs.nr_open is close to the default 1048576, but some systems set it much higher (e.g. 1073741816). This is systemd's doing: since systemd v240 (2018), PID 1 bumps fs.nr_open and fs.file-max to their largest possible values on boot, as file descriptors are already accounted for by memcg [1]. In that case, dup2() to nr_open + 64 requires the kernel to allocate a file descriptor table with ~1 billion entries, which fails with ENOMEM. On a kernel that already carries 04a2c4b4511d1, dup2() no longer fails with ENOMEM. The allocation is now rejected up front and the caller gets EMFILE instead, without the WARNING, but the test still fails. Cap the nr_open value used for the test's own arithmetic to a known reasonable base value (1048576) and restore the true original value once the test has completed. Link: https://lore.kernel.org/20260814165709.513263-1-khorenko@virtuozzo.com Link: https://github.com/systemd/systemd/commit/a8b627aaed409a15260c25988970c795bf963812 [1] Signed-off-by: Konstantin Khorenko Signed-off-by: Eva Kurchatova Signed-off-by: Andrew Morton Cc: Shuah Khan Cc: Wei Yang Cc: --- tools/testing/selftests/core/unshare_test.c | 19 +++++++++++++++++-- 1 file changed, 17 insertions(+), 2 deletions(-) diff --git a/tools/testing/selftests/core/unshare_test.c b/tools/testing/selftests/core/unshare_test.c index ffce75a6c228f1..d40e963dd5205e 100644 --- a/tools/testing/selftests/core/unshare_test.c +++ b/tools/testing/selftests/core/unshare_test.c @@ -40,6 +40,14 @@ TEST(unshare_EMFILE) ASSERT_EQ(sscanf(buf, "%d", &nr_open), 1); + /* + * Cap nr_open for the duration of the test to avoid ENOMEM from a + * huge fd table allocation; buf/n keep the real original value so + * fs.nr_open can be restored to it once the test is done. + */ + if (nr_open > 1024 * 1024) + nr_open = 1024 * 1024; + ASSERT_EQ(0, getrlimit(RLIMIT_NOFILE, &rlimit)); /* bump fs.nr_open */ @@ -73,10 +81,13 @@ TEST(unshare_EMFILE) if (pid == 0) { int err; + char buf3[32]; + ssize_t n3; - /* restore fs.nr_open */ + /* restore fs.nr_open to the (possibly capped) test baseline */ + n3 = sprintf(buf3, "%d\n", nr_open); lseek(fd, 0, SEEK_SET); - write(fd, buf, n); + write(fd, buf3, n3); /* ... and now unshare(CLONE_FILES) must fail with EMFILE */ err = unshare(CLONE_FILES); EXPECT_EQ(err, -1) @@ -89,6 +100,10 @@ TEST(unshare_EMFILE) EXPECT_EQ(waitpid(pid, &status, 0), pid); EXPECT_EQ(true, WIFEXITED(status)); EXPECT_EQ(0, WEXITSTATUS(status)); + + /* restore the real fs.nr_open value */ + lseek(fd, 0, SEEK_SET); + write(fd, buf, n); } TEST_HARNESS_MAIN From d8088f3be8341fced13c947544d8b6a97fce7951 Mon Sep 17 00:00:00 2001 From: Thomas Maarseveen Date: Wed, 12 Aug 2026 20:55:33 +0200 Subject: [PATCH 1250/1352] lib/tests: add KUnit tests for errseq The errseq_t infrastructure (lib/errseq.c) underpins writeback error reporting but has no regression tests. Its semantics are subtle enough to have needed fixing before: commit b4678df184b3 ("errseq: Always report a writeback error once") changed how unseen errors reach new samplers. Add a KUnit suite covering the documented single-threaded semantics: - a zeroed errseq_t is the "no error yet" epoch - errors are recorded, overwrite one another, and both ends of the valid errno range round-trip exactly - an error nobody has seen samples as zero, so a check against a fresh sample still reports it - errseq_check_and_advance() reports a given error exactly once per cursor and leaves the cursor in place when nothing has changed - once an error has been seen, a fresh sample is current and a check against it reports nothing - the same error recorded again after being seen is reported again, even to a cursor that consumed the first occurrence while another cursor marked the repeat as seen - independent cursors each observe each error The lockless behaviour of errseq_t under concurrent updates and the WARN path for invalid error values are deliberately out of scope. Tested with ./tools/testing/kunit/kunit.py run, with a kunitconfig enabling CONFIG_KUNIT=y and CONFIG_ERRSEQ_KUNIT_TEST=y; all 13 tests pass under ARCH=um. Link: https://lore.kernel.org/20260812-errseq-kunit-v1-1-312be4c3aa0d@gmail.com Signed-off-by: Thomas Maarseveen Signed-off-by: Andrew Morton Acked-by: Jeff Layton Cc: David Gow --- MAINTAINERS | 1 + lib/Kconfig.debug | 15 +++ lib/tests/Makefile | 1 + lib/tests/errseq_kunit.c | 237 +++++++++++++++++++++++++++++++++++++++ 4 files changed, 254 insertions(+) create mode 100644 lib/tests/errseq_kunit.c diff --git a/MAINTAINERS b/MAINTAINERS index 360977678f707e..67c42e55d60afd 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -9700,6 +9700,7 @@ M: Jeff Layton S: Maintained F: include/linux/errseq.h F: lib/errseq.c +F: lib/tests/errseq_kunit.c ESD CAN NETWORK DRIVERS M: Stefan Mätje diff --git a/lib/Kconfig.debug b/lib/Kconfig.debug index 134b15a44625e8..c3f448f3b8f13c 100644 --- a/lib/Kconfig.debug +++ b/lib/Kconfig.debug @@ -2797,6 +2797,21 @@ config SYSCTL_KUNIT_TEST If unsure, say N. +config ERRSEQ_KUNIT_TEST + tristate "KUnit test for errseq" if !KUNIT_ALL_TESTS + depends on KUNIT + default KUNIT_ALL_TESTS + help + This builds the errseq KUnit test suite. + It tests the documented semantics of the errseq_t error-tracking + infrastructure (lib/errseq.c), which underpins writeback error + reporting. + + For more information on KUnit and unit tests in general please refer + to the KUnit documentation in Documentation/dev-tools/kunit/. + + If unsure, say N. + config KFIFO_KUNIT_TEST tristate "KUnit Test for the generic kernel FIFO implementation" if !KUNIT_ALL_TESTS depends on KUNIT diff --git a/lib/tests/Makefile b/lib/tests/Makefile index 3cac3b63a7522c..8e11b125433bf1 100644 --- a/lib/tests/Makefile +++ b/lib/tests/Makefile @@ -13,6 +13,7 @@ obj-$(CONFIG_BLACKHOLE_DEV_KUNIT_TEST) += blackhole_dev_kunit.o obj-$(CONFIG_CHECKSUM_KUNIT) += checksum_kunit.o obj-$(CONFIG_CMDLINE_KUNIT_TEST) += cmdline_kunit.o obj-$(CONFIG_CPUMASK_KUNIT_TEST) += cpumask_kunit.o +obj-$(CONFIG_ERRSEQ_KUNIT_TEST) += errseq_kunit.o obj-$(CONFIG_FFS_KUNIT_TEST) += ffs_kunit.o CFLAGS_fortify_kunit.o += $(call cc-disable-warning, unsequenced) CFLAGS_fortify_kunit.o += $(call cc-disable-warning, stringop-overread) diff --git a/lib/tests/errseq_kunit.c b/lib/tests/errseq_kunit.c new file mode 100644 index 00000000000000..8f39ebc4a2488e --- /dev/null +++ b/lib/tests/errseq_kunit.c @@ -0,0 +1,237 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * KUnit tests for the errseq_t error-tracking infrastructure. + * + * These exercise the documented single-threaded semantics of the errseq + * API (see Documentation/core-api/errseq.rst and lib/errseq.c): error + * recording and overwriting, the "seen" handoff between errseq_sample() + * and errseq_check_and_advance(), and the re-reporting of an error that + * is recorded again after it has been seen. + * + * The lockless properties of errseq_t under concurrent updates are + * outside the scope of these deterministic tests, as is the WARN path + * for invalid error values. + */ +#include + +#include +#include +#include + +/* + * A zeroed errseq_t is the "no error has ever occurred" epoch: it + * samples as zero and no check against it reports anything. + */ +static void errseq_test_zero_epoch_reports_no_error(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t since = 0; + + KUNIT_EXPECT_EQ(test, errseq_sample(&eseq), 0); + KUNIT_EXPECT_EQ(test, errseq_check(&eseq, 0), 0); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), 0); + KUNIT_EXPECT_EQ(test, since, 0); +} + +static void errseq_test_set_records_error(struct kunit *test) +{ + errseq_t eseq = 0; + + /* errseq_set() returns the previous value; the epoch is zero. */ + KUNIT_EXPECT_EQ(test, errseq_set(&eseq, -EIO), 0); + KUNIT_EXPECT_EQ(test, errseq_check(&eseq, 0), -EIO); +} + +/* Any error set always overwrites an existing error. */ +static void errseq_test_set_overwrites_error(struct kunit *test) +{ + errseq_t eseq = 0; + + errseq_set(&eseq, -EIO); + errseq_set(&eseq, -ENOSPC); + KUNIT_EXPECT_EQ(test, errseq_check(&eseq, 0), -ENOSPC); +} + +/* Both ends of the valid error range are recorded exactly. */ +static void errseq_test_errno_range_extremes(struct kunit *test) +{ + errseq_t lo = 0; + errseq_t hi = 0; + + errseq_set(&lo, -1); + KUNIT_EXPECT_EQ(test, errseq_check(&lo, 0), -1); + + errseq_set(&hi, -MAX_ERRNO); + KUNIT_EXPECT_EQ(test, errseq_check(&hi, 0), -MAX_ERRNO); +} + +/* + * An error nobody has seen yet samples as zero, so that a check against + * the sample still reports it (see commit b4678df184b3 ("errseq: Always + * report a writeback error once")). + */ +static void errseq_test_sample_of_unseen_error_is_zero(struct kunit *test) +{ + errseq_t eseq = 0; + + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_sample(&eseq), 0); +} + +static void errseq_test_new_sampler_sees_unseen_error(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t since; + + errseq_set(&eseq, -EIO); + since = errseq_sample(&eseq); + KUNIT_EXPECT_EQ(test, errseq_check(&eseq, since), -EIO); +} + +/* A given error is reported exactly once per advancing cursor. */ +static void errseq_test_check_and_advance_reports_once(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t since = errseq_sample(&eseq); + + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), 0); +} + +/* + * Once an error has been seen, a fresh sample is non-zero and checking + * against it reports nothing: handled errors do not reach new samplers. + */ +static void errseq_test_sample_after_seen_is_current(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t since = 0; + errseq_t sample; + + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), -EIO); + + sample = errseq_sample(&eseq); + KUNIT_EXPECT_NE(test, sample, 0); + KUNIT_EXPECT_EQ(test, errseq_check(&eseq, sample), 0); +} + +static void errseq_test_new_error_after_advance(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t since = 0; + + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), -EIO); + + errseq_set(&eseq, -ENOSPC); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), -ENOSPC); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), 0); +} + +/* + * Recording the same error again after it has been seen must bump the + * sequence, so cursors that consumed the first occurrence see the + * second one too. + */ +static void errseq_test_same_error_reported_again_after_seen(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t since = 0; + errseq_t seen_cursor; + + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), -EIO); + + seen_cursor = since; + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_check(&eseq, since), -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), -EIO); + /* The repeat must advance the sequence, not just re-toggle "seen". */ + KUNIT_EXPECT_NE(test, since, seen_cursor); +} + +/* + * A cursor that consumed an error must still observe a repeat of that + * error even when another cursor has already marked the repeat seen: + * recording over a seen value must advance the sequence. + */ +static void errseq_test_repeat_error_visible_to_all_cursors(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t cursor_a = 0; + errseq_t cursor_b = 0; + + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &cursor_a), -EIO); + + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &cursor_b), -EIO); + + KUNIT_EXPECT_EQ(test, errseq_check(&eseq, cursor_a), -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &cursor_a), -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &cursor_a), 0); +} + +/* An advance with no new error reports nothing and leaves the cursor put. */ +static void errseq_test_advance_stable_when_unchanged(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t since = 0; + errseq_t cursor; + + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), -EIO); + + cursor = since; + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), 0); + KUNIT_EXPECT_EQ(test, since, cursor); +} + +/* + * Cursors are independent: one subscriber consuming an error does not + * consume it for another, and each subscriber sees each error once. + */ +static void errseq_test_two_subscribers_independent(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t cursor_a = errseq_sample(&eseq); + errseq_t cursor_b = errseq_sample(&eseq); + + errseq_set(&eseq, -EIO); + + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &cursor_a), -EIO); + KUNIT_EXPECT_EQ(test, errseq_check(&eseq, cursor_b), -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &cursor_b), -EIO); + + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &cursor_a), 0); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &cursor_b), 0); +} + +static struct kunit_case errseq_test_cases[] = { + KUNIT_CASE(errseq_test_zero_epoch_reports_no_error), + KUNIT_CASE(errseq_test_set_records_error), + KUNIT_CASE(errseq_test_set_overwrites_error), + KUNIT_CASE(errseq_test_errno_range_extremes), + KUNIT_CASE(errseq_test_sample_of_unseen_error_is_zero), + KUNIT_CASE(errseq_test_new_sampler_sees_unseen_error), + KUNIT_CASE(errseq_test_check_and_advance_reports_once), + KUNIT_CASE(errseq_test_sample_after_seen_is_current), + KUNIT_CASE(errseq_test_new_error_after_advance), + KUNIT_CASE(errseq_test_same_error_reported_again_after_seen), + KUNIT_CASE(errseq_test_repeat_error_visible_to_all_cursors), + KUNIT_CASE(errseq_test_advance_stable_when_unchanged), + KUNIT_CASE(errseq_test_two_subscribers_independent), + {} +}; + +static struct kunit_suite errseq_test_suite = { + .name = "errseq", + .test_cases = errseq_test_cases, +}; + +kunit_test_suite(errseq_test_suite); + +MODULE_DESCRIPTION("KUnit tests for the errseq infrastructure"); +MODULE_LICENSE("GPL"); From 66d727e839eae503112349cba16a71f781cd8e18 Mon Sep 17 00:00:00 2001 From: Eric Biggers Date: Mon, 31 Aug 2026 14:22:48 -0700 Subject: [PATCH 1251/1352] xor: add missing vzeroupper to AVX code Since the AVX optimized XOR code uses YMM registers, execute vzeroupper before returning from it. This is needed to avoid degrading the performance of any later SSE code that may happen to be executed. Link: https://lore.kernel.org/20260831212248.213805-1-ebiggers@kernel.org Fixes: ea4d26ae24e5 ("raid5: add AVX optimized RAID5 checksumming") Signed-off-by: Eric Biggers Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: --- lib/raid/xor/x86/xor-avx.c | 1 + 1 file changed, 1 insertion(+) diff --git a/lib/raid/xor/x86/xor-avx.c b/lib/raid/xor/x86/xor-avx.c index f7777d7aa269bd..95b21e7225e8d7 100644 --- a/lib/raid/xor/x86/xor-avx.c +++ b/lib/raid/xor/x86/xor-avx.c @@ -147,6 +147,7 @@ static void xor_gen_avx(void *dest, void **srcs, unsigned int src_cnt, { kernel_fpu_begin(); xor_gen_avx_inner(dest, srcs, src_cnt, bytes); + asm volatile("vzeroupper"); kernel_fpu_end(); } From 4877e82423497e768331071bf5f3696ff3c92e5c Mon Sep 17 00:00:00 2001 From: Eric Biggers Date: Mon, 31 Aug 2026 14:23:08 -0700 Subject: [PATCH 1252/1352] raid6: add missing vzeroupper to AVX2 code Since the AVX2 optimized RAID6 code uses YMM registers, execute vzeroupper before returning from it. This is needed to avoid degrading the performance of any later SSE code that may happen to be executed. Link: https://lore.kernel.org/20260831212308.213855-1-ebiggers@kernel.org Fixes: 2c935842bdb4 ("lib/raid6: Add AVX2 optimized gen_syndrome functions") Fixes: 7056741fd9fc ("lib/raid6: Add AVX2 optimized recovery functions") Signed-off-by: Eric Biggers Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: --- lib/raid/raid6/x86/avx2.c | 6 ++++++ lib/raid/raid6/x86/recov_avx2.c | 2 ++ 2 files changed, 8 insertions(+) diff --git a/lib/raid/raid6/x86/avx2.c b/lib/raid/raid6/x86/avx2.c index 7d829c669ea795..3cc2fe7ac42c57 100644 --- a/lib/raid/raid6/x86/avx2.c +++ b/lib/raid/raid6/x86/avx2.c @@ -67,6 +67,7 @@ static void raid6_avx21_gen_syndrome(int disks, size_t bytes, void **ptrs) } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -115,6 +116,7 @@ static void raid6_avx21_xor_syndrome(int disks, int start, int stop, } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -175,6 +177,7 @@ static void raid6_avx22_gen_syndrome(int disks, size_t bytes, void **ptrs) } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -243,6 +246,7 @@ static void raid6_avx22_xor_syndrome(int disks, int start, int stop, } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -334,6 +338,7 @@ static void raid6_avx24_gen_syndrome(int disks, size_t bytes, void **ptrs) } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -444,6 +449,7 @@ static void raid6_avx24_xor_syndrome(int disks, int start, int stop, asm volatile("vmovntdq %%ymm14,%0" : "=m" (q[d+96])); } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } diff --git a/lib/raid/raid6/x86/recov_avx2.c b/lib/raid/raid6/x86/recov_avx2.c index a714a780a2d8f6..820871046e3058 100644 --- a/lib/raid/raid6/x86/recov_avx2.c +++ b/lib/raid/raid6/x86/recov_avx2.c @@ -176,6 +176,7 @@ static void raid6_2data_recov_avx2(int disks, size_t bytes, int faila, #endif } + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -293,6 +294,7 @@ static void raid6_datap_recov_avx2(int disks, size_t bytes, int faila, #endif } + asm volatile("vzeroupper"); kernel_fpu_end(); } From 50b44694f3c7b4464004e533d14659c221cbc188 Mon Sep 17 00:00:00 2001 From: Eric Biggers Date: Mon, 31 Aug 2026 14:23:16 -0700 Subject: [PATCH 1253/1352] raid6: add missing vzeroupper to AVX-512 code Since the AVX-512 optimized RAID6 code uses ZMM registers, execute vzeroupper before returning from it. This is needed to avoid degrading the performance of any later SSE code that may happen to be executed. Link: https://lore.kernel.org/20260831212316.213896-1-ebiggers@kernel.org Fixes: e0a491c12968 ("lib/raid6: Add AVX512 optimized gen_syndrome functions") Fixes: 13c520b2993c ("lib/raid6: Add AVX512 optimized recovery functions") Signed-off-by: Eric Biggers Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: --- lib/raid/raid6/x86/avx512.c | 6 ++++++ lib/raid/raid6/x86/recov_avx512.c | 2 ++ 2 files changed, 8 insertions(+) diff --git a/lib/raid/raid6/x86/avx512.c b/lib/raid/raid6/x86/avx512.c index e671eb5bde63e4..772bfc4af6dfd7 100644 --- a/lib/raid/raid6/x86/avx512.c +++ b/lib/raid/raid6/x86/avx512.c @@ -78,6 +78,7 @@ static void raid6_avx5121_gen_syndrome(int disks, size_t bytes, void **ptrs) } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -137,6 +138,7 @@ static void raid6_avx5121_xor_syndrome(int disks, int start, int stop, } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -208,6 +210,7 @@ static void raid6_avx5122_gen_syndrome(int disks, size_t bytes, void **ptrs) } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -292,6 +295,7 @@ static void raid6_avx5122_xor_syndrome(int disks, int start, int stop, } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -396,6 +400,7 @@ static void raid6_avx5124_gen_syndrome(int disks, size_t bytes, void **ptrs) } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -529,6 +534,7 @@ static void raid6_avx5124_xor_syndrome(int disks, int start, int stop, "m" (q[d+128]), "m" (q[d+192])); } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } const struct raid6_calls raid6_avx512x4 = { diff --git a/lib/raid/raid6/x86/recov_avx512.c b/lib/raid/raid6/x86/recov_avx512.c index ec72d5a30c01ef..299a3f044d6162 100644 --- a/lib/raid/raid6/x86/recov_avx512.c +++ b/lib/raid/raid6/x86/recov_avx512.c @@ -211,6 +211,7 @@ static void raid6_2data_recov_avx512(int disks, size_t bytes, int faila, #endif } + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -353,6 +354,7 @@ static void raid6_datap_recov_avx512(int disks, size_t bytes, int faila, #endif } + asm volatile("vzeroupper"); kernel_fpu_end(); } From dbd59579f35286ff988e5bfd21c743c6c8e58fc1 Mon Sep 17 00:00:00 2001 From: Hrushiraj Gandhi Date: Mon, 31 Aug 2026 20:19:53 +0530 Subject: [PATCH 1254/1352] gcov: use strscpy() instead of strcpy() in init_node() node->name is a flexible array member sized to exactly strlen(name) + 1 bytes at allocation time in new_node(), so this copy can never actually overflow. Still, prefer the bounded strscpy() over strcpy() on general principle; pass the same strlen(name) + 1 bound the allocation used, since sizeof() cannot be applied to a flexible array member. No functional change. Link: https://lore.kernel.org/20260831144953.324441-1-hrushirajg23@gmail.com Signed-off-by: Hrushiraj Gandhi Signed-off-by: Andrew Morton Reviewed-by: Bradley Morgan Cc: Peter Oberparleiter --- kernel/gcov/fs.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/kernel/gcov/fs.c b/kernel/gcov/fs.c index 1d19b1be207a7d..764918570de13d 100644 --- a/kernel/gcov/fs.c +++ b/kernel/gcov/fs.c @@ -529,7 +529,7 @@ static void init_node(struct gcov_node *node, struct gcov_info *info, } node->parent = parent; if (name) - strcpy(node->name, name); + strscpy(node->name, name, strlen(name) + 1); } /* From 1189eaed308243c3a831b5c5416a4f9e28444c77 Mon Sep 17 00:00:00 2001 From: Mete Durlu Date: Mon, 31 Aug 2026 11:57:15 +0200 Subject: [PATCH 1255/1352] panic: introduce arch_do_panic Patch series "Introduce arch_do_panic", v6. Replace architecture-specific ifdef sections in vpanic() with a clean arch_do_panic() hook. Currently s390 and sparc embed their panic handlers directly in vpanic() using preprocessor conditionals, making the common code path harder to maintain. Introduce arch_do_panic() as an architecture extension point called at the end of vpanic(). Architectures can use this hook to implement their specific panic handling without polluting the generic panic code. Remove s390s ifdef block in vpanic() and move the corresponding code block to s390s own arch_do_panic() implementation in architecture specific code. Move sparc panic handling from ifdef blocks to arch_do_panic(). Remove the preprocessor conditionals from vpanic() and place the Stop-A enablement code in architecture-specific files where it belongs. Stop-A enablement markers are now printed after "end Kernel panic" line. To me, there are no better alternatives other than setup.c to put sparc's arch_do_panic() implementation. The other files under arch/sparc/kernel are either divided to *_32.c and *_64.c variants, which mean code duplication, or unrelated. The cleanup reduces vpanic() complexity and establishes a pattern for other architectures needing custom panic behavior. No functional changes, only minor print order changes. This patch (of 3): Introduce a hook for architectures to put their specific panic handlers. s390 and sparc already have ifdef preprocessor checks to execute architecture specific code. Pave the way for vpanic() cleanup. Link: https://lore.kernel.org/20260831-arch_do_panic-v6-1-a1e170a9e7fd@linux.ibm.com Link: https://lore.kernel.org/all/20260730-arch_do_panic-v3-0-d5401e683cdb@linux.ibm.com/ [1] Signed-off-by: Mete Durlu Signed-off-by: Andrew Morton Reviewed-by: Bradley Morgan Suggested-by: Sven Schnelle Reviewed-by: Andrew Morton Cc: Alexander Gordeev Cc: Andreas Larsson Cc: Christian Borntraeger Cc: David S. Miller Cc: Heiko Carstens Cc: Petr Mladek Cc: Vasily Gorbik --- include/linux/panic.h | 2 ++ kernel/panic.c | 3 +++ 2 files changed, 5 insertions(+) diff --git a/include/linux/panic.h b/include/linux/panic.h index f1dd417e54b294..98dd7dfd27de7a 100644 --- a/include/linux/panic.h +++ b/include/linux/panic.h @@ -110,4 +110,6 @@ extern void add_taint(unsigned flag, enum lockdep_ok); extern int test_taint(unsigned flag); extern unsigned long get_taint(void); +void arch_do_panic(void); + #endif /* _LINUX_PANIC_H */ diff --git a/kernel/panic.c b/kernel/panic.c index 213725b612aa11..726a978422326f 100644 --- a/kernel/panic.c +++ b/kernel/panic.c @@ -567,6 +567,8 @@ static void panic_other_cpus_shutdown(bool crash_kexec) crash_smp_send_stop(); } +void __weak arch_do_panic(void) {} + /** * vpanic - halt the system * @fmt: The text string to print @@ -756,6 +758,7 @@ void vpanic(const char *fmt, va_list args) #endif pr_emerg("---[ end Kernel panic - not syncing: %s ]---\n", buf); + arch_do_panic(); /* Do not scroll important messages printed above */ suppress_printk = 1; From 064b1d161ce17f1bb8ed35505b6ff94f108432b0 Mon Sep 17 00:00:00 2001 From: Mete Durlu Date: Mon, 31 Aug 2026 11:57:16 +0200 Subject: [PATCH 1256/1352] s390: implement arch_do_panic Implement s390 specific arch_do_panic() instead of using s390 specific ifdef sections in vpanic() code. disabled_wait() is now called after "end Kernel panic" marker. No functional changes. Link: https://lore.kernel.org/20260831-arch_do_panic-v6-2-a1e170a9e7fd@linux.ibm.com Signed-off-by: Mete Durlu Signed-off-by: Andrew Morton Acked-by: Heiko Carstens Reviewed-by: Bradley Morgan Reviewed-by: Andrew Morton Cc: Alexander Gordeev Cc: Andreas Larsson Cc: Christian Borntraeger Cc: David S. Miller Cc: Petr Mladek Cc: Sven Schnelle Cc: Vasily Gorbik --- arch/s390/kernel/traps.c | 7 +++++++ kernel/panic.c | 3 --- 2 files changed, 7 insertions(+), 3 deletions(-) diff --git a/arch/s390/kernel/traps.c b/arch/s390/kernel/traps.c index b6ba4465f59dea..115cb337324769 100644 --- a/arch/s390/kernel/traps.c +++ b/arch/s390/kernel/traps.c @@ -26,6 +26,7 @@ #include #include #include +#include #include #include #include @@ -33,6 +34,7 @@ #include #include #include +#include #include "entry.h" struct pgm_stat { @@ -283,6 +285,11 @@ static void monitor_event_exception(struct pt_regs *regs) } } +void arch_do_panic(void) +{ + disabled_wait(); +} + void kernel_stack_invalid(struct pt_regs *regs) { /* diff --git a/kernel/panic.c b/kernel/panic.c index 726a978422326f..ee6e3f9e39002e 100644 --- a/kernel/panic.c +++ b/kernel/panic.c @@ -752,9 +752,6 @@ void vpanic(const char *fmt, va_list args) pr_emerg("Press Stop-A (L1-A) from sun keyboard or send break\n" "twice on console to return to the boot prom\n"); } -#endif -#if defined(CONFIG_S390) - disabled_wait(); #endif pr_emerg("---[ end Kernel panic - not syncing: %s ]---\n", buf); From a153871ef120d2394a201064528055518bd58b32 Mon Sep 17 00:00:00 2001 From: Mete Durlu Date: Mon, 31 Aug 2026 11:57:17 +0200 Subject: [PATCH 1257/1352] sparc: implement arch_do_panic Implement sparc specific arch_do_panic() instead of using sparc specific ifdef sections in vpanic() code. Reorder arch specific panic handling, sparc's Stop-A messages are now printed after "end Kernel panic" marker. Link: https://lore.kernel.org/20260831-arch_do_panic-v6-3-a1e170a9e7fd@linux.ibm.com Signed-off-by: Mete Durlu Signed-off-by: Andrew Morton Reviewed-by: Bradley Morgan Reviewed-by: Andrew Morton Cc: Alexander Gordeev Cc: Andreas Larsson Cc: Christian Borntraeger Cc: David S. Miller Cc: Heiko Carstens Cc: Petr Mladek Cc: Sven Schnelle Cc: Vasily Gorbik --- arch/sparc/kernel/setup.c | 9 +++++++++ kernel/panic.c | 9 --------- 2 files changed, 9 insertions(+), 9 deletions(-) diff --git a/arch/sparc/kernel/setup.c b/arch/sparc/kernel/setup.c index 4975867d9001b6..5f43cef8063825 100644 --- a/arch/sparc/kernel/setup.c +++ b/arch/sparc/kernel/setup.c @@ -2,6 +2,8 @@ #include #include +#include +#include static const struct ctl_table sparc_sysctl_table[] = { { @@ -36,6 +38,13 @@ static const struct ctl_table sparc_sysctl_table[] = { #endif }; +void arch_do_panic(void) +{ + /* Make sure the user can actually press Stop-A (L1-A) */ + stop_a_enabled = 1; + pr_emerg("Press Stop-A (L1-A) from sun keyboard or send break\n" + "twice on console to return to the boot prom\n"); +} static int __init init_sparc_sysctls(void) { diff --git a/kernel/panic.c b/kernel/panic.c index ee6e3f9e39002e..7dda841c16f9cc 100644 --- a/kernel/panic.c +++ b/kernel/panic.c @@ -744,15 +744,6 @@ void vpanic(const char *fmt, va_list args) reboot_mode = panic_reboot_mode; emergency_restart(); } -#ifdef __sparc__ - { - extern int stop_a_enabled; - /* Make sure the user can actually press Stop-A (L1-A) */ - stop_a_enabled = 1; - pr_emerg("Press Stop-A (L1-A) from sun keyboard or send break\n" - "twice on console to return to the boot prom\n"); - } -#endif pr_emerg("---[ end Kernel panic - not syncing: %s ]---\n", buf); arch_do_panic(); From 2a1751aa3707793e07199e796ba00648b6f2b7e7 Mon Sep 17 00:00:00 2001 From: Ivy Lopez Date: Mon, 31 Aug 2026 19:41:38 -0600 Subject: [PATCH 1258/1352] lib: decompress_unxz: make it obvious that there is no memory leak Calling __decompress() or unxz() with fill == NULL && flush == NULL && in == NULL is invalid, thus there were no memory leaks even though it might have looked like that. Move the conditional free() calls so that it's obvious that there are no leaks. Link: https://lore.kernel.org/20260901014138.22699-1-skunkolee@gmail.com Link: https://lore.kernel.org/lkml/20241006072542.66442-2-t.v.s10123@gmail.com/T/ Link: https://lore.kernel.org/lkml/20260825191333.34276-1-skunkolee@gmail.com/T/ Signed-off-by: Ivy Lopez Signed-off-by: Andrew Morton Closes: https://bugzilla.kernel.org/show_bug.cgi?id=207113 Reviewed-by: Lasse Collin --- lib/decompress_unxz.c | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/lib/decompress_unxz.c b/lib/decompress_unxz.c index 05d5cb490a44e7..9ccded9934c667 100644 --- a/lib/decompress_unxz.c +++ b/lib/decompress_unxz.c @@ -342,13 +342,13 @@ STATIC int INIT unxz(unsigned char *in, long in_size, b.out_pos = 0; } } while (ret == XZ_OK); + } - if (must_free_in) - free(in); + if (must_free_in) + free(in); - if (flush != NULL) - free(b.out); - } + if (flush != NULL) + free(b.out); if (in_used != NULL) *in_used += b.in_pos; From b85f1900d477c7619bd10bc5dc3effedd66e2e36 Mon Sep 17 00:00:00 2001 From: Julian Braha Date: Mon, 31 Aug 2026 00:00:16 +0100 Subject: [PATCH 1259/1352] arch/Kconfig: fix dead conditions by removing dead options These two 'int' options: ARCH_MMAP_RND_BITS_DEFAULT ARCH_MMAP_RND_COMPAT_BITS_DEFAULT are used directly as conditions for defaults. 'int' options should not be used as conditions, because they will always evaluate to false. Let's remove these options, because they are not used anywhere else. This dead code was found by kconfirm, a static analysis tool for Kconfig. Link: https://lore.kernel.org/20260830230016.2730093-1-julianbraha@gmail.com Signed-off-by: Julian Braha Signed-off-by: Andrew Morton Reviewed-by: Arnd Bergmann Reviewed-by: Jinjie Ruan --- arch/Kconfig | 8 -------- 1 file changed, 8 deletions(-) diff --git a/arch/Kconfig b/arch/Kconfig index 45c65777236231..72890200d049cc 100644 --- a/arch/Kconfig +++ b/arch/Kconfig @@ -1238,13 +1238,9 @@ config ARCH_MMAP_RND_BITS_MIN config ARCH_MMAP_RND_BITS_MAX int -config ARCH_MMAP_RND_BITS_DEFAULT - int - config ARCH_MMAP_RND_BITS int "Number of bits to use for ASLR of mmap base address" if EXPERT range ARCH_MMAP_RND_BITS_MIN ARCH_MMAP_RND_BITS_MAX - default ARCH_MMAP_RND_BITS_DEFAULT if ARCH_MMAP_RND_BITS_DEFAULT default ARCH_MMAP_RND_BITS_MIN depends on HAVE_ARCH_MMAP_RND_BITS help @@ -1272,13 +1268,9 @@ config ARCH_MMAP_RND_COMPAT_BITS_MIN config ARCH_MMAP_RND_COMPAT_BITS_MAX int -config ARCH_MMAP_RND_COMPAT_BITS_DEFAULT - int - config ARCH_MMAP_RND_COMPAT_BITS int "Number of bits to use for ASLR of mmap base address for compatible applications" if EXPERT range ARCH_MMAP_RND_COMPAT_BITS_MIN ARCH_MMAP_RND_COMPAT_BITS_MAX - default ARCH_MMAP_RND_COMPAT_BITS_DEFAULT if ARCH_MMAP_RND_COMPAT_BITS_DEFAULT default ARCH_MMAP_RND_COMPAT_BITS_MIN depends on HAVE_ARCH_MMAP_RND_COMPAT_BITS help From 862622f49cc99630a7b8f546db1b4117cec11751 Mon Sep 17 00:00:00 2001 From: Yuntao Wang Date: Tue, 11 Aug 2026 20:18:30 +0800 Subject: [PATCH 1260/1352] dyndbg: fix incorrect mod_ct value in dynamic_debug_init() Patch series "dyndbg: fix incorrect mod_ct value in dynamic_debug_init()". Fix and clean up dynamic_debug_init(). This patch (of 2): Suppose all `struct _ddebug` instances belong to the same module, mod_ct should be 1, but it is currently 0. mod_ct is incremented only when iter->modname changes, i.e. when the loop encounters the first _ddebug entry of a new module: if (strcmp(modname, iter->modname)) { mod_ct++; ... } If all _ddebug entries belong to the same module, strcmp() never returns nonzero, so mod_ct remains 0. However, the last (and in this case only) module is added after the loop: di.num_descs = mod_sites; di.descs = iter_mod_start; ret = ddebug_add_module(&di, modname); Thus, mod_ct should be incremented before adding this final module. The bug only affects the diagnostic message printed by vpr_info(): "%d prdebugs in %d modules, ..." It reports one fewer module than the actual number of modules. There is no userspace-visible runtime effect; the dynamic debug tables themselves are initialized correctly. Fix it. Link: https://lore.kernel.org/20260811121831.577848-1-yuntao.wang@linux.dev Link: https://lore.kernel.org/20260811121831.577848-2-yuntao.wang@linux.dev Signed-off-by: Yuntao Wang Signed-off-by: Andrew Morton Cc: Jason Baron Cc: Jim Cromie --- lib/dynamic_debug.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/lib/dynamic_debug.c b/lib/dynamic_debug.c index 18a71a9108d3e5..16fad5454d6a50 100644 --- a/lib/dynamic_debug.c +++ b/lib/dynamic_debug.c @@ -1456,6 +1456,8 @@ static int __init dynamic_debug_init(void) iter_mod_start = iter; } } + + mod_ct++; di.num_descs = mod_sites; di.descs = iter_mod_start; ret = ddebug_add_module(&di, modname); From 9bc00d7da2f64571d40dcfab818247e0506fa2b0 Mon Sep 17 00:00:00 2001 From: Yuntao Wang Date: Tue, 11 Aug 2026 20:18:31 +0800 Subject: [PATCH 1261/1352] dyndbg: clean up dynamic_debug_init() to improve readability Keep variable assignments in the same order throughout the function to make the code easier to follow. No functional changes. Link: https://lore.kernel.org/20260811121831.577848-3-yuntao.wang@linux.dev Signed-off-by: Yuntao Wang Signed-off-by: Andrew Morton Cc: Jason Baron Cc: Jim Cromie --- lib/dynamic_debug.c | 15 ++++++++------- 1 file changed, 8 insertions(+), 7 deletions(-) diff --git a/lib/dynamic_debug.c b/lib/dynamic_debug.c index 16fad5454d6a50..49334d1aa4b3af 100644 --- a/lib/dynamic_debug.c +++ b/lib/dynamic_debug.c @@ -1442,28 +1442,29 @@ static int __init dynamic_debug_init(void) i = mod_sites = mod_ct = 0; for (; iter < __stop___dyndbg; iter++, i++, mod_sites++) { - if (strcmp(modname, iter->modname)) { - mod_ct++; - di.num_descs = mod_sites; di.descs = iter_mod_start; + di.num_descs = mod_sites; ret = ddebug_add_module(&di, modname); if (ret) goto out_err; - mod_sites = 0; - modname = iter->modname; + mod_ct++; + iter_mod_start = iter; + modname = iter->modname; + mod_sites = 0; } } - mod_ct++; - di.num_descs = mod_sites; di.descs = iter_mod_start; + di.num_descs = mod_sites; ret = ddebug_add_module(&di, modname); if (ret) goto out_err; + mod_ct++; + ddebug_init_success = 1; vpr_info("%d prdebugs in %d modules, %d KiB in ddebug tables, %d kiB in __dyndbg section\n", i, mod_ct, (int)((mod_ct * sizeof(struct ddebug_table)) >> 10), From db31e34f2f75c6f8ef36ab376c8702c407844605 Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Fri, 10 Jul 2026 14:39:57 +0200 Subject: [PATCH 1262/1352] fork: honor task_struct's declared alignment Since commit cb7ca40a3882 ("x86/fpu: Make task_struct::thread constant size"), struct task_struct is declared __attribute__((aligned(64))) on all architectures. But fork_init() still sets the task_struct slab cache's alignment to align = max(L1_CACHE_BYTES, ARCH_MIN_TASKALIGN) which is smaller than 64 on architectures whose cache lines are below 64 bytes: e.g. 32 on ARMv5. In practice plain SLUB happens to hand out 64-byte-aligned objects anyway. With CONFIG_SLUB_DEBUG_ON the red-zone padding shifts objects to the requested alignment. With CONFIG_UBSAN_ALIGNMENT=y a boot on QEMU versatilepb (ARM926EJ-S, v7.2-rc2, gcc 13.3) floods the console with reports like: UBSAN: misaligned-access in include/linux/sched.h:2087:9 member access within misaligned address c295d7e0 for type 'struct task_struct' which requires 64 byte alignment CPU: 0 UID: 0 PID: 15 Comm: pr/ttyAMA-1 Not tainted 7.2.0-rc2 #1 VOLUNTARY Set the slab alignment to at least the type's declared alignment. Link: https://lore.kernel.org/20260710123957.31774-1-kmehltretter@gmail.com Fixes: cb7ca40a3882 ("x86/fpu: Make task_struct::thread constant size") Signed-off-by: Karl Mehltretter Signed-off-by: Andrew Morton Reviewed-by: Bradley Morgan Assisted-by: Claude:claude-fable-5 Cc: Ingo Molnar Cc: Kees Cook Cc: Peter Zijlstra Cc: Vlastimil Babka Cc: --- kernel/fork.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/kernel/fork.c b/kernel/fork.c index 10f2d05d816a5f..4558150fd133a9 100644 --- a/kernel/fork.c +++ b/kernel/fork.c @@ -858,7 +858,8 @@ void __init fork_init(void) #ifndef ARCH_MIN_TASKALIGN #define ARCH_MIN_TASKALIGN 0 #endif - int align = max_t(int, L1_CACHE_BYTES, ARCH_MIN_TASKALIGN); + int align = max3(L1_CACHE_BYTES, ARCH_MIN_TASKALIGN, + __alignof__(struct task_struct)); unsigned long useroffset, usersize; /* create a slab on which task_structs can be allocated */ From 52a7d910c62aaa17c701ef3f665e617e946822f0 Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Sat, 1 Aug 2026 22:58:44 +0000 Subject: [PATCH 1263/1352] taskstats: fold the two pid/tgid handlers into one cmd_attr_pid() and cmd_attr_tgid() are copy paste. fold them into one handler that takes the attr type and fill function as parameters, same pattern as the cpumask fold in 59c0bc949c5e. No functional change. Link: https://lore.kernel.org/20260801225845.23855-1-include@grrlz.net Signed-off-by: Bradley Morgan Signed-off-by: Andrew Morton Reviewed-by: Andrew Morton Cc: Balbir Singh Cc: Balbir Singh --- kernel/taskstats.c | 47 ++++++++++++---------------------------------- 1 file changed, 12 insertions(+), 35 deletions(-) diff --git a/kernel/taskstats.c b/kernel/taskstats.c index 9a48827e22bce3..598d9cd8325018 100644 --- a/kernel/taskstats.c +++ b/kernel/taskstats.c @@ -473,7 +473,8 @@ static size_t taskstats_packet_size(void) return size; } -static int cmd_attr_pid(struct genl_info *info) +static int cmd_attr_pid_tgid(struct genl_info *info, int attr, + int (*fill)(pid_t, struct taskstats *)) { struct taskstats *stats; struct sk_buff *rep_skb; @@ -488,41 +489,15 @@ static int cmd_attr_pid(struct genl_info *info) return rc; rc = -EINVAL; - pid = nla_get_u32(info->attrs[TASKSTATS_CMD_ATTR_PID]); - stats = mk_reply(rep_skb, TASKSTATS_TYPE_PID, pid); + pid = nla_get_u32(info->attrs[attr]); + stats = mk_reply(rep_skb, + attr == TASKSTATS_CMD_ATTR_PID + ? TASKSTATS_TYPE_PID : TASKSTATS_TYPE_TGID, + pid); if (!stats) goto err; - rc = fill_stats_for_pid(pid, stats); - if (rc < 0) - goto err; - return send_reply(rep_skb, info); -err: - nlmsg_free(rep_skb); - return rc; -} - -static int cmd_attr_tgid(struct genl_info *info) -{ - struct taskstats *stats; - struct sk_buff *rep_skb; - size_t size; - u32 tgid; - int rc; - - size = taskstats_packet_size(); - - rc = prepare_reply(info, TASKSTATS_CMD_NEW, &rep_skb, size); - if (rc < 0) - return rc; - - rc = -EINVAL; - tgid = nla_get_u32(info->attrs[TASKSTATS_CMD_ATTR_TGID]); - stats = mk_reply(rep_skb, TASKSTATS_TYPE_TGID, tgid); - if (!stats) - goto err; - - rc = fill_stats_for_tgid(tgid, stats); + rc = fill(pid, stats); if (rc < 0) goto err; return send_reply(rep_skb, info); @@ -542,9 +517,11 @@ static int taskstats_user_cmd(struct sk_buff *skb, struct genl_info *info) TASKSTATS_CMD_ATTR_DEREGISTER_CPUMASK, DEREGISTER); else if (info->attrs[TASKSTATS_CMD_ATTR_PID]) - return cmd_attr_pid(info); + return cmd_attr_pid_tgid(info, TASKSTATS_CMD_ATTR_PID, + fill_stats_for_pid); else if (info->attrs[TASKSTATS_CMD_ATTR_TGID]) - return cmd_attr_tgid(info); + return cmd_attr_pid_tgid(info, TASKSTATS_CMD_ATTR_TGID, + fill_stats_for_tgid); else return -EINVAL; } From 99466beef31df2b0fa8891b556f0ae18705385c4 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Tue, 1 Sep 2026 20:52:18 +0800 Subject: [PATCH 1264/1352] ocfs2: restrict OCFS2_INVALID_SLOT suballoc slot to system inodes Patch series "ocfs2: validate suballoc slot and bit of metadata blocks", v3. The ocfs2 metadata validators trust the on-disk suballoc slot and bit without checking them against the slot range of the mounted filesystem and the capacity of the block group bitmap. A corrupted image can carry OCFS2_INVALID_SLOT or another out-of-range slot, or a suballoc bit beyond the bitmap, and once the corresponding inode, extent block, xattr block, dir index root or refcount block gets freed, the bad value goes straight into ocfs2_get_system_file_inode() or _ocfs2_free_suballoc_bits() and hits a BUG_ON() or runs off the end of local_system_inodes[]. This series rejects such values at read time, in the existing validators, so corrupted objects fail with -EROFS (and a read-only remount) instead of crashing: patch 1 restricts OCFS2_INVALID_SLOT dinodes to system inodes, completing fe7a283b3916 ("ocfs2: add suballoc slot check in ocfs2_validate_inode_block()"), and turns the "system file state is ambiguous" BUG_ON() in ocfs2_read_locked_inode() into an ocfs2_error(); patch 2 rejects oversized dinode suballoc bits; patch 3 validates the suballoc slot and bit of xattr and dir index blocks; patch 4 validates the suballoc slot and bit of extent and refcount blocks. The checks only enforce what the kernel and mkfs.ocfs2 already write: a valid slot from meta_ac->ac_alloc_slot and a bit within the block group bitmap, with system inodes carrying OCFS2_INVALID_SLOT plus OCFS2_SYSTEM_FL and extent blocks using slot 0. Nothing changes for healthy filesystems. Each new check was exercised under QEMU by corrupting the field in question with an out-of-range value; with the series applied the access fails with -EROFS and the filesystem remounts read-only instead of hitting the BUG_ON(). This patch (of 4): ocfs2_validate_inode_block() currently permits i_suballoc_slot to be OCFS2_INVALID_SLOT for any dinode. Only system inodes created by mkfs.ocfs2 are allocated from the global allocator and thus legitimately carry this value; regular inodes are always allocated from a per-slot suballocator and hence must have a valid slot. If a corrupted regular inode with OCFS2_INVALID_SLOT is accepted, ocfs2_remove_inode() will pass the slot to ocfs2_get_system_file_inode() and get_local_system_inode() will hit BUG_ON(slot == OCFS2_INVALID_SLOT) when the inode is deleted. This can be triggered by an unprivileged user unlinking such a corrupted file. Reject OCFS2_INVALID_SLOT for non-system dinodes during validation, while still accepting it for system inodes. Note that a crafted dinode carrying OCFS2_SYSTEM_FL passes the check above, yet a plain lookup of it still used to BUG() in ocfs2_read_locked_inode() ("system file state is ambiguous"). Since i_flags comes from disk, handle that mismatch with ocfs2_error() instead of BUG_ON() as well. Link: https://lore.kernel.org/20260901125221.1634686-1-joseph.qi@linux.alibaba.com Link: https://lore.kernel.org/20260901125221.1634686-2-joseph.qi@linux.alibaba.com Fixes: fe7a283b3916 ("ocfs2: add suballoc slot check in ocfs2_validate_inode_block()") Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Cc: Changwei Ge Cc: Heming Zhao Cc: Joel Becker Cc: Jun Piao Cc: Junxiao Bi Cc: Mark Fasheh Cc: --- fs/ocfs2/inode.c | 37 ++++++++++++++++++++++++++++--------- 1 file changed, 28 insertions(+), 9 deletions(-) diff --git a/fs/ocfs2/inode.c b/fs/ocfs2/inode.c index 180107a11046c5..9228d6ef23c24d 100644 --- a/fs/ocfs2/inode.c +++ b/fs/ocfs2/inode.c @@ -638,14 +638,18 @@ static int ocfs2_read_locked_inode(struct inode *inode, fe = (struct ocfs2_dinode *) bh->b_data; /* - * This is a code bug. Right now the caller needs to - * understand whether it is asking for a system file inode or - * not so the proper lock names can be built. + * The caller must know whether it is asking for a system file inode + * or not so the proper lock names can be built. Since i_flags comes + * from disk, a mismatch is filesystem corruption instead of a code + * bug, so handle it with ocfs2_error() rather than BUG_ON(). */ - mlog_bug_on_msg(!!(fe->i_flags & cpu_to_le32(OCFS2_SYSTEM_FL)) != - !!(args->fi_flags & OCFS2_FI_FLAG_SYSFILE), - "Inode %llu: system file state is ambiguous\n", - (unsigned long long)args->fi_blkno); + if (!!(fe->i_flags & cpu_to_le32(OCFS2_SYSTEM_FL)) != + !!(args->fi_flags & OCFS2_FI_FLAG_SYSFILE)) { + status = ocfs2_error(osb->sb, + "Inode %llu: system file state is ambiguous\n", + (unsigned long long)args->fi_blkno); + goto bail; + } if (S_ISCHR(le16_to_cpu(fe->i_mode)) || S_ISBLK(le16_to_cpu(fe->i_mode))) @@ -1520,8 +1524,23 @@ int ocfs2_validate_inode_block(struct super_block *sb, goto bail; } - if (le16_to_cpu(di->i_suballoc_slot) != (u16)OCFS2_INVALID_SLOT && - (u32)le16_to_cpu(di->i_suballoc_slot) > OCFS2_SB(sb)->max_slots - 1) { + /* + * Only system inodes created by mkfs.ocfs2 are allocated from the + * global allocator and thus legitimately carry OCFS2_INVALID_SLOT. + * Regular inodes are always allocated from a per-slot suballocator. + * If a regular inode with OCFS2_INVALID_SLOT was accepted here, + * deleting it would pass the slot to get_local_system_inode() via + * ocfs2_remove_inode() and trigger BUG_ON(slot == OCFS2_INVALID_SLOT). + */ + if (le16_to_cpu(di->i_suballoc_slot) == (u16)OCFS2_INVALID_SLOT) { + if (!(le32_to_cpu(di->i_flags) & OCFS2_SYSTEM_FL)) { + rc = ocfs2_error(sb, + "Invalid dinode %llu: suballoc slot %u for non-system inode\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(di->i_suballoc_slot)); + goto bail; + } + } else if ((u32)le16_to_cpu(di->i_suballoc_slot) > OCFS2_SB(sb)->max_slots - 1) { rc = ocfs2_error(sb, "Invalid dinode %llu: suballoc slot %u\n", (unsigned long long)bh->b_blocknr, le16_to_cpu(di->i_suballoc_slot)); From ad55f18b896388d649fae780d33f0561bda8faa4 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Tue, 1 Sep 2026 20:52:19 +0800 Subject: [PATCH 1265/1352] ocfs2: validate suballoc bit during inode read i_suballoc_bit of a dinode is currently not validated at all. A corrupted dinode can carry an abnormally large i_suballoc_bit, which bypasses ocfs2_validate_inode_block(). When the inode is deleted, ocfs2_remove_inode() calls ocfs2_free_dinode(), which passes the unvalidated bit to _ocfs2_free_suballoc_bits() and triggers BUG_ON((count + start_bit) > ocfs2_bits_per_group(cl)). A suballocator block group bitmap is contained in a single block and starts after the group descriptor header, so a valid suballoc bit must be smaller than the number of bits fitting in the remaining space. Reject oversized i_suballoc_bit values during dinode validation. The bound is derived from ocfs2_group_bitmap_size() so it is also tight when discontig_bg caps the suballocator bitmap at OCFS2_MAX_BG_BITMAP_SIZE. Note the above check alone is not sufficient since the freeing path compares the bit against ocfs2_bits_per_group(), which is derived from cl_cpg/cl_bpc of the allocator dinode that is not validated against the actual group capacity and can be artificially smaller on a corrupted image. Convert this BUG_ON in _ocfs2_free_suballoc_bits() to ocfs2_error() as well. Link: https://lore.kernel.org/20260901125221.1634686-3-joseph.qi@linux.alibaba.com Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Cc: Changwei Ge Cc: Heming Zhao Cc: Joel Becker Cc: Jun Piao Cc: Junxiao Bi Cc: Mark Fasheh --- fs/ocfs2/inode.c | 16 ++++++++++++++++ fs/ocfs2/ocfs2.h | 12 ++++++++++++ fs/ocfs2/suballoc.c | 18 +++++++++++++++--- 3 files changed, 43 insertions(+), 3 deletions(-) diff --git a/fs/ocfs2/inode.c b/fs/ocfs2/inode.c index 9228d6ef23c24d..92f3450010fbb0 100644 --- a/fs/ocfs2/inode.c +++ b/fs/ocfs2/inode.c @@ -1547,6 +1547,22 @@ int ocfs2_validate_inode_block(struct super_block *sb, goto bail; } + /* + * A suballocator block group bitmap is contained in a single block + * and starts after the group descriptor header, so a valid suballoc + * bit can never exceed ocfs2_suballoc_bits_per_block(). Otherwise + * deleting the inode will pass the oversized bit to + * _ocfs2_free_suballoc_bits() via ocfs2_free_dinode() and trigger + * BUG_ON((count + start_bit) > ocfs2_bits_per_group(cl)), since any + * group holds at most ocfs2_suballoc_bits_per_block() bits. + */ + if (le16_to_cpu(di->i_suballoc_bit) >= ocfs2_suballoc_bits_per_block(sb)) { + rc = ocfs2_error(sb, "Invalid dinode %llu: suballoc bit %u\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(di->i_suballoc_bit)); + goto bail; + } + if ((le32_to_cpu(di->i_flags) & OCFS2_ORPHANED_FL) && le16_to_cpu(di->i_orphaned_slot) >= OCFS2_SB(sb)->max_slots) { rc = ocfs2_error(sb, "Invalid dinode %llu: orphaned slot %u\n", diff --git a/fs/ocfs2/ocfs2.h b/fs/ocfs2/ocfs2.h index b747cdec178758..e3bb3cc0b25a29 100644 --- a/fs/ocfs2/ocfs2.h +++ b/fs/ocfs2/ocfs2.h @@ -593,6 +593,18 @@ static inline int ocfs2_supports_discontig_bg(struct ocfs2_super *osb) return 0; } +/* + * A suballocator block group bitmap starts right after the group + * descriptor header, so a suballoc bit can never exceed this number + * of bits. Derive it from ocfs2_group_bitmap_size() which also caps + * it at OCFS2_MAX_BG_BITMAP_SIZE when discontig_bg is enabled. + */ +static inline u32 ocfs2_suballoc_bits_per_block(struct super_block *sb) +{ + return ocfs2_group_bitmap_size(sb, 1, + OCFS2_SB(sb)->s_feature_incompat) * 8; +} + static inline unsigned int ocfs2_link_max(struct ocfs2_super *osb) { if (ocfs2_supports_indexed_dirs(osb)) diff --git a/fs/ocfs2/suballoc.c b/fs/ocfs2/suballoc.c index 453b56be9624c6..ce22d0c3d28748 100644 --- a/fs/ocfs2/suballoc.c +++ b/fs/ocfs2/suballoc.c @@ -3040,10 +3040,22 @@ static int _ocfs2_free_suballoc_bits(handle_t *handle, /* The alloc_bh comes from ocfs2_free_dinode() or * ocfs2_free_clusters(). The callers have all locked the * allocator and gotten alloc_bh from the lock call. This - * validates the dinode buffer. Any corruption that has happened - * is a code bug. */ + * validates the dinode buffer. */ BUG_ON(!OCFS2_IS_VALID_DINODE(fe)); - BUG_ON((count + start_bit) > ocfs2_bits_per_group(cl)); + + /* + * ocfs2_bits_per_group() is derived from cl_cpg and cl_bpc of the + * allocator dinode, which are not validated against the volume + * geometry. A corrupted image can carry a suballoc bit beyond it, + * so error out instead of crashing. + */ + if ((count + start_bit) > ocfs2_bits_per_group(cl)) { + return ocfs2_error(alloc_inode->i_sb, + "Allocator #%llu: freeing bits %u+%u exceeds bits per group %u\n", + (unsigned long long)le64_to_cpu(fe->i_blkno), + count, start_bit, + ocfs2_bits_per_group(cl)); + } trace_ocfs2_free_suballoc_bits( (unsigned long long)OCFS2_I(alloc_inode)->ip_blkno, From f084bd23248501e34c22213fa93951b4bc989ace Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Tue, 1 Sep 2026 20:52:20 +0800 Subject: [PATCH 1266/1352] ocfs2: validate suballoc slot and bit of xattr and dir index blocks ocfs2_validate_xattr_block() and ocfs2_validate_dx_root() do not validate xb_suballoc_slot, xb_suballoc_bit, dr_suballoc_slot and dr_suballoc_bit at all. Since xattr blocks and dir index root blocks are allocated from a per-slot suballocator at runtime, their suballoc slots must be within range and their suballoc bits must fit in a block group bitmap. Otherwise a corrupted image can carry an out-of-range slot. When the xattr block or dir index is removed, ocfs2_xattr_block_remove() or ocfs2_dx_dir_remove_index() passes the unvalidated slot to ocfs2_get_system_file_inode() and get_local_system_inode() will either hit BUG_ON(slot == OCFS2_INVALID_SLOT) or compute an out-of-bounds index into the local_system_inodes array. Similarly an oversized suballoc bit will error out the filesystem in _ocfs2_free_suballoc_bits(). Furthermore ocfs2_validate_dx_root() does not verify dr_blkno against the physical block number like the extent and xattr block validators do, so a misplaced dir index root block can pass validation. Reject misplaced dir index root blocks, out-of-range suballoc slots and oversized suballoc bits during validation. Link: https://lore.kernel.org/20260901125221.1634686-4-joseph.qi@linux.alibaba.com Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Cc: Changwei Ge Cc: Heming Zhao Cc: Joel Becker Cc: Jun Piao Cc: Junxiao Bi Cc: Mark Fasheh --- fs/ocfs2/dir.c | 35 +++++++++++++++++++++++++++++++++++ fs/ocfs2/xattr.c | 25 +++++++++++++++++++++++++ 2 files changed, 60 insertions(+) diff --git a/fs/ocfs2/dir.c b/fs/ocfs2/dir.c index 0075e1624310e8..6bb6aa133f0150 100644 --- a/fs/ocfs2/dir.c +++ b/fs/ocfs2/dir.c @@ -605,6 +605,41 @@ static int ocfs2_validate_dx_root(struct super_block *sb, goto bail; } + if (le64_to_cpu(dx_root->dr_blkno) != bh->b_blocknr) { + ret = ocfs2_error(sb, + "Dir Index Root # %llu has an invalid dr_blkno of %llu\n", + (unsigned long long)bh->b_blocknr, + (unsigned long long)le64_to_cpu(dx_root->dr_blkno)); + goto bail; + } + + /* + * Dir index root blocks are allocated from a per-slot suballocator, + * so the slot must be in range. Otherwise removing the index passes + * it to get_local_system_inode(), which hits BUG_ON() for + * OCFS2_INVALID_SLOT or computes an out-of-bounds index otherwise. + */ + if ((u32)le16_to_cpu(dx_root->dr_suballoc_slot) >= OCFS2_SB(sb)->max_slots) { + ret = ocfs2_error(sb, + "Dir Index Root # %llu has invalid dr_suballoc_slot %u\n", + (unsigned long long)le64_to_cpu(dx_root->dr_blkno), + le16_to_cpu(dx_root->dr_suballoc_slot)); + goto bail; + } + + /* + * Similarly the suballoc bit must fit in a block group bitmap. + * Otherwise removing the index will pass the oversized bit to + * _ocfs2_free_suballoc_bits() and trigger ocfs2_error() there. + */ + if (le16_to_cpu(dx_root->dr_suballoc_bit) >= ocfs2_suballoc_bits_per_block(sb)) { + ret = ocfs2_error(sb, + "Dir Index Root # %llu has invalid dr_suballoc_bit %u\n", + (unsigned long long)le64_to_cpu(dx_root->dr_blkno), + le16_to_cpu(dx_root->dr_suballoc_bit)); + goto bail; + } + if (!(dx_root->dr_flags & OCFS2_DX_FLAG_INLINE)) { struct ocfs2_extent_list *el = &dx_root->dr_list; diff --git a/fs/ocfs2/xattr.c b/fs/ocfs2/xattr.c index 740d4bb3890f9c..e27f5f925df52e 100644 --- a/fs/ocfs2/xattr.c +++ b/fs/ocfs2/xattr.c @@ -532,6 +532,31 @@ static int ocfs2_validate_xattr_block(struct super_block *sb, le32_to_cpu(xb->xb_fs_generation)); } + /* + * Xattr blocks are allocated from a per-slot suballocator, so the + * slot must be in range. Otherwise freeing the block passes it to + * get_local_system_inode(), which hits BUG_ON() for + * OCFS2_INVALID_SLOT or computes an out-of-bounds index otherwise. + */ + if ((u32)le16_to_cpu(xb->xb_suballoc_slot) >= OCFS2_SB(sb)->max_slots) { + return ocfs2_error(sb, + "Extended attribute block #%llu has an invalid xb_suballoc_slot of %u\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(xb->xb_suballoc_slot)); + } + + /* + * Similarly the suballoc bit must fit in a block group bitmap. + * Otherwise freeing the block will pass the oversized bit to + * _ocfs2_free_suballoc_bits() and trigger ocfs2_error() there. + */ + if (le16_to_cpu(xb->xb_suballoc_bit) >= ocfs2_suballoc_bits_per_block(sb)) { + return ocfs2_error(sb, + "Extended attribute block #%llu has an invalid xb_suballoc_bit of %u\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(xb->xb_suballoc_bit)); + } + if (!(le16_to_cpu(xb->xb_flags) & OCFS2_XATTR_INDEXED)) { size_t region_offset = offsetof(struct ocfs2_xattr_block, xb_attrs.xb_header); From 56369bd70030e0566fe7877c3a32e7589236c81e Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Tue, 1 Sep 2026 20:52:21 +0800 Subject: [PATCH 1267/1352] ocfs2: validate suballoc slot and bit of extent and refcount blocks ocfs2_validate_extent_block() and ocfs2_validate_refcount_block() do not validate h_suballoc_slot, h_suballoc_bit, rf_suballoc_slot and rf_suballoc_bit at all. Since extent blocks and refcount blocks are allocated from a per-slot suballocator at runtime, their suballoc slots must be within range and their suballoc bits must fit in a block group bitmap. Otherwise a corrupted image can carry an out-of-range slot. When the extent block is freed, ocfs2_cache_extent_block_free() caches it and ocfs2_free_cached_blocks() later passes the unvalidated slot to ocfs2_get_system_file_inode(); when the refcount block is freed, ocfs2_remove_refcount_extent() passes it via ocfs2_cache_block_dealloc(). get_local_system_inode() will then either hit BUG_ON(slot == OCFS2_INVALID_SLOT) or compute an out-of-bounds index into the local_system_inodes array. Similarly an oversized suballoc bit will error out the filesystem in _ocfs2_free_suballoc_bits(). Furthermore group descriptor validation only guarantees bg_bits within the physical bitmap size, so a corrupted image can still carry a suballoc bit beyond bg_bits, which would let ocfs2_block_group_clear_bits() clear bits beyond bg_bitmap. Convert the remaining BUG_ON against group->bg_bits in _ocfs2_free_suballoc_bits() to ocfs2_error() as well. Reject out-of-range suballoc slots and oversized suballoc bits during validation. Link: https://lore.kernel.org/20260901125221.1634686-5-joseph.qi@linux.alibaba.com Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi Cc: Changwei Ge Cc: Jun Piao Cc: Heming Zhao --- fs/ocfs2/alloc.c | 27 +++++++++++++++++++++++++++ fs/ocfs2/refcounttree.c | 27 +++++++++++++++++++++++++++ fs/ocfs2/suballoc.c | 15 ++++++++++++++- 3 files changed, 68 insertions(+), 1 deletion(-) diff --git a/fs/ocfs2/alloc.c b/fs/ocfs2/alloc.c index be09e766ac1fc9..2fdc5403b10b04 100644 --- a/fs/ocfs2/alloc.c +++ b/fs/ocfs2/alloc.c @@ -925,6 +925,33 @@ static int ocfs2_validate_extent_block(struct super_block *sb, goto bail; } + /* + * Extent blocks are allocated from a per-slot suballocator, so the + * slot must be in range. Otherwise freeing the block passes it to + * get_local_system_inode(), which hits BUG_ON() for + * OCFS2_INVALID_SLOT or computes an out-of-bounds index otherwise. + */ + if ((u32)le16_to_cpu(eb->h_suballoc_slot) >= OCFS2_SB(sb)->max_slots) { + rc = ocfs2_error(sb, + "Extent block #%llu has an invalid h_suballoc_slot of %u\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(eb->h_suballoc_slot)); + goto bail; + } + + /* + * Similarly the suballoc bit must fit in a block group bitmap. + * Otherwise freeing the block will pass the oversized bit to + * _ocfs2_free_suballoc_bits() and trigger ocfs2_error() there. + */ + if (le16_to_cpu(eb->h_suballoc_bit) >= ocfs2_suballoc_bits_per_block(sb)) { + rc = ocfs2_error(sb, + "Extent block #%llu has an invalid h_suballoc_bit of %u\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(eb->h_suballoc_bit)); + goto bail; + } + if (le16_to_cpu(eb->h_list.l_count) != ocfs2_extent_recs_per_eb(sb)) { rc = ocfs2_error(sb, "Extent block #%llu has invalid l_count %u (expected %u)\n", diff --git a/fs/ocfs2/refcounttree.c b/fs/ocfs2/refcounttree.c index d9f22b4a265461..3e9cccf06e48cb 100644 --- a/fs/ocfs2/refcounttree.c +++ b/fs/ocfs2/refcounttree.c @@ -117,6 +117,33 @@ static int ocfs2_validate_refcount_block(struct super_block *sb, goto out; } + /* + * Refcount blocks are allocated from a per-slot suballocator, so the + * slot must be in range. Otherwise freeing the block passes it to + * get_local_system_inode(), which hits BUG_ON() for + * OCFS2_INVALID_SLOT or computes an out-of-bounds index otherwise. + */ + if ((u32)le16_to_cpu(rb->rf_suballoc_slot) >= OCFS2_SB(sb)->max_slots) { + rc = ocfs2_error(sb, + "Refcount block #%llu has an invalid rf_suballoc_slot of %u\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(rb->rf_suballoc_slot)); + goto out; + } + + /* + * Similarly the suballoc bit must fit in a block group bitmap. + * Otherwise freeing the block will pass the oversized bit to + * _ocfs2_free_suballoc_bits() and trigger ocfs2_error() there. + */ + if (le16_to_cpu(rb->rf_suballoc_bit) >= ocfs2_suballoc_bits_per_block(sb)) { + rc = ocfs2_error(sb, + "Refcount block #%llu has an invalid rf_suballoc_bit of %u\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(rb->rf_suballoc_bit)); + goto out; + } + /* * rf_records (rl_count/rl_used/rl_recs[]) is only meaningful when * this block is not an interior tree block (OCFS2_REFCOUNT_TREE_FL); diff --git a/fs/ocfs2/suballoc.c b/fs/ocfs2/suballoc.c index ce22d0c3d28748..624152e4f7fedf 100644 --- a/fs/ocfs2/suballoc.c +++ b/fs/ocfs2/suballoc.c @@ -3070,7 +3070,20 @@ static int _ocfs2_free_suballoc_bits(handle_t *handle, } group = (struct ocfs2_group_desc *) group_bh->b_data; - BUG_ON((count + start_bit) > le16_to_cpu(group->bg_bits)); + /* + * Group descriptor validation only guarantees bg_bits within the + * physical bitmap size, so double check the freeing range here. + * Otherwise ocfs2_block_group_clear_bits() would clear bits beyond + * bg_bitmap. + */ + if ((count + start_bit) > le16_to_cpu(group->bg_bits)) { + status = ocfs2_error(alloc_inode->i_sb, + "Group descriptor #%llu has %u bits, cannot free bits %u+%u\n", + (unsigned long long)le64_to_cpu(group->bg_blkno), + le16_to_cpu(group->bg_bits), + count, start_bit); + goto bail; + } if (ocfs2_is_cluster_bitmap(alloc_inode)) old_bg_contig_free_bits = group->bg_contig_free_bits; From 2e9b8abda33cf21964ec1a86d3dfe0b706ecde1a Mon Sep 17 00:00:00 2001 From: Feng Tang Date: Wed, 2 Sep 2026 19:48:51 +0800 Subject: [PATCH 1268/1352] panic: remove the unneeded panic_print_get() panic_print_get() was introduced in commit 2683df6539cb ("panic: add note that 'panic_print' parameter is deprecated") to print out warning message of the deprecation of 'panic_print' on read access. Since commit 90f3c123247e ("panic: only warn about deprecated panic_print on write access"), panic_print_get() wrapper is not needed anymore for read access, so remove it and use param_get_ulong() instead. Link: https://lore.kernel.org/20260902114851.77062-1-feng.tang@linux.alibaba.com Signed-off-by: Feng Tang Signed-off-by: Andrew Morton Reviewed-by: Bradley Morgan Reviewed-by: Andrew Morton Reviewed-by: Petr Mladek --- kernel/panic.c | 7 +------ 1 file changed, 1 insertion(+), 6 deletions(-) diff --git a/kernel/panic.c b/kernel/panic.c index 7dda841c16f9cc..50715f14cf04ef 100644 --- a/kernel/panic.c +++ b/kernel/panic.c @@ -1216,14 +1216,9 @@ static int panic_print_set(const char *val, const struct kernel_param *kp) return param_set_ulong(val, kp); } -static int panic_print_get(char *val, const struct kernel_param *kp) -{ - return param_get_ulong(val, kp); -} - static const struct kernel_param_ops panic_print_ops = { .set = panic_print_set, - .get = panic_print_get, + .get = param_get_ulong, }; __core_param_cb(panic_print, &panic_print_ops, &panic_print, 0644); From 263e980bfac517eed714645187f760170fc7e0fe Mon Sep 17 00:00:00 2001 From: Nick Desaulniers Date: Wed, 2 Sep 2026 14:01:53 -0700 Subject: [PATCH 1269/1352] scripts/checkstack.pl: support llvm-objdump disassembly for x86 When running `make LLVM=1 checkstack`, OBJDUMP is set to llvm-objdump. llvm-objdump outputs disassembly with instruction size suffixes (such as subq/addq/subl/addl), tabs/whitespace differences, spaces after commas, and trailing comments (e.g. `# imm = 0x...`). Because scripts/checkstack.pl used rigid regular expressions specifically tuned to GNU objdump format (e.g. requiring exactly four spaces, no suffix, and no space after comma), checkstack.pl failed to match any stack adjustment instructions and yielded no output when using llvm-objdump on x86. Update the regular expressions for x86 to match optional suffixes, variable whitespace, and trailing comments. Link: https://lore.kernel.org/20260902-checkstack_llvm_objdump-v1-1-edb4eca5f163@google.com Signed-off-by: Nick Desaulniers Signed-off-by: Andrew Morton Assisted-by: LLM Gemini Cc: Bill Wendling Cc: Justin Stitt Cc: Nathan Chancellor --- scripts/checkstack.pl | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/scripts/checkstack.pl b/scripts/checkstack.pl index 14ce31f732ee8a..f8011e789a9b45 100755 --- a/scripts/checkstack.pl +++ b/scripts/checkstack.pl @@ -66,8 +66,10 @@ #c0105234: 81 ec ac 05 00 00 sub $0x5ac,%esp # or # 2f60: 48 81 ec e8 05 00 00 sub $0x5e8,%rsp - $re = qr/^.*[as][du][db] \$(0x$x{1,8}),\%(e|r)sp$/o; - $dre = qr/^.*[as][du][db] (%.*),\%(e|r)sp$/o; + # or + # 0: 48 81 ec e8 05 00 00 subq $0x5e8, %rsp + $re = qr/^.*[as][du][db][ql]?\s+\$(0x$x{1,8}),\s*\%(e|r)sp/o; + $dre = qr/^.*[as][du][db][ql]?\s+(%.*),\s*\%(e|r)sp/o; } elsif ($arch eq 'm68k') { # 2b6c: 4e56 fb70 linkw %fp,#-1168 # 1df770: defc ffe4 addaw #-28,%sp From a25ed0b09bcd6ebdf41f5199bfa8f3293b254a65 Mon Sep 17 00:00:00 2001 From: Maximilian Heyne Date: Fri, 19 Jun 2026 11:24:29 +0000 Subject: [PATCH 1270/1352] selftests: uevent filtering: don't shrink the socket buffer The uevent_filtering test shrinks the uevent socket buffer to 4 KB although the default socket buffer size is much higher. This leads to this test being flaky when too many unrelated uevents are fired on the machine. They might fill up the netlink receive buffer leading to ENOBUFS errors when trying to receive the uevents. For example, I could trigger test failures when running triggering a lot of udev events in the background: $ # run multiple of that in the background: $ while :; do sudo udevadm trigger --action=change; done & $ sudo ./uevent_filtering # Starting 1 tests from 1 test cases. # RUN global.uevent_filtering ... add@/devices/virtual/mem/fullACTION=addDEVPATH=/devices/virtual/mem/fullSUBSYSTEM=memSYNTH_UUID=0MAJOR=1MINOR=7DEVNAME=fullDEVMODE=0666SEQNUM=304458 add@/devices/virtual/mem/fullACTION=addDEVPATH=/devices/virtual/mem/fullSUBSYSTEM=memSYNTH_UUID=0MAJOR=1MINOR=7DEVNAME=fullDEVMODE=0666SEQNUM=304471 add@/devices/virtual/mem/fullACTION=addDEVPATH=/devices/virtual/mem/fullSUBSYSTEM=memSYNTH_UUID=0MAJOR=1MINOR=7DEVNAME=fullDEVMODE=0666SEQNUM=304481 add@/devices/virtual/mem/fullACTION=addDEVPATH=/devices/virtual/mem/fullSUBSYSTEM=memSYNTH_UUID=0MAJOR=1MINOR=7DEVNAME=fullDEVMODE=0666SEQNUM=349156 No buffer space available - Failed to receive uevent # uevent_filtering.c:463:uevent_filtering:Expected 0 (0) == ret (-1) # uevent_filtering: Test failed # FAIL global.uevent_filtering not ok 1 global.uevent_filtering The default receive buffer size (SK_RMEM_MAX) is far larger than the requested 4 KB, so keep this to make the test less flaky. Link: https://lore.kernel.org/20260619-get-swam-a1cd4cca@mheyne-amazon Fixes: 9d3df886d17b ("selftests: uevent filtering") Signed-off-by: Maximilian Heyne Signed-off-by: Andrew Morton Cc: Christian Brauner Cc: David S. Miller Cc: Shuah Khan Cc: Wei Yang Cc: --- tools/testing/selftests/uevent/uevent_filtering.c | 8 -------- 1 file changed, 8 deletions(-) diff --git a/tools/testing/selftests/uevent/uevent_filtering.c b/tools/testing/selftests/uevent/uevent_filtering.c index 33a09f66d7e22f..8c623845e3d042 100644 --- a/tools/testing/selftests/uevent/uevent_filtering.c +++ b/tools/testing/selftests/uevent/uevent_filtering.c @@ -78,7 +78,6 @@ static int uevent_listener(unsigned long post_flags, bool expect_uevent, { int sk_fd, ret; socklen_t sk_addr_len; - int rcv_buf_sz = __UEVENT_BUFFER_SIZE; uint64_t sync_add = 1; struct sockaddr_nl sk_addr = { 0 }, rcv_addr = { 0 }; char buf[__UEVENT_BUFFER_SIZE] = { 0 }; @@ -96,13 +95,6 @@ static int uevent_listener(unsigned long post_flags, bool expect_uevent, return -1; } - ret = setsockopt(sk_fd, SOL_SOCKET, SO_RCVBUF, &rcv_buf_sz, - sizeof(rcv_buf_sz)); - if (ret < 0) { - fprintf(stderr, "%s - Failed to set socket options\n", strerror(errno)); - goto on_error; - } - sk_addr.nl_family = AF_NETLINK; sk_addr.nl_groups = __UEVENT_LISTEN_ALL; From 4669a5c237b18884af37e09772d31e3c07494bf9 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Thu, 3 Sep 2026 21:13:12 +0800 Subject: [PATCH 1271/1352] ocfs2: allow xattr bucket entries to span multiple blocks Patch series "ocfs2: xattr bucket validation fixes", v2. This series fixes two problems around xattr bucket validation. Patch 1 fixes a false-corruption failure on blocksize-512 volumes: the bucket validator limited the entry array to the first bucket block while the write path stores entries across the whole 4096-byte bucket region, so a legitimately written, fsck-clean bucket could be rejected and force the filesystem read-only. It also adds an alignment check on the bucket block number, since the entry array is accessed as one contiguous region and a corrupted xattr tree could otherwise point a bucket at blocks straddling a page boundary. Patch 2 converts two mlog_bug_on_msg() checks in the bucket defrag path to ocfs2_error() returns, so that a corrupt bucket holding overlapping entries or an inflated xh_free_start marks the filesystem read-only and fails the setxattr instead of panicking the kernel. Both patches have been tested in QEMU: the blocksize-512 reproducer (40 xattrs with 100-byte values, previously failing with "entry count 32 exceeds maximum 31") now passes with a clean fsck.ocfs2 result, and the ocfs2 testsuite xattr tests pass 48/48 across blocksize combinations. This patch (of 2): ocfs2_validate_xattr_bucket() limits the entry array to the first bucket block, but the write path stores entries across the whole OCFS2_XATTR_BUCKET_SIZE region. With 512-byte blocks a bucket spans eight blocks, and a bucket filled with small xattrs places its last entries past offset 512. Reading such a bucket back errors out: OCFS2: ERROR (device loop0): ocfs2_validate_xattr_bucket: Invalid xattr bucket 86072: entry count 32 exceeds maximum 31 On-disk corruption discovered. Please run fsck.ocfs2 once the filesystem is unmounted. OCFS2: File system is now read-only. This is reproducible by setting ~33 xattrs with 100-byte values on a file on a blocksize-512 volume; fsck.ocfs2 reports the resulting image clean. Check the entry count against the full bucket region instead. The per-block bounds checks for names and values stay as they are, since ocfs2_bucket_align_free_start() keeps each name+value pair within a single block. The entry array is one contiguous region, so a bucket from a corrupted xattr tree whose first block is not aligned to OCFS2_XATTR_BUCKET_SIZE could straddle a page and make the validation loop read out of bounds. Buckets allocated within clusters are always aligned, so reject any other block number while validating. Link: https://lore.kernel.org/20260903131313.2396208-1-joseph.qi@linux.alibaba.com Link: https://lore.kernel.org/20260903131313.2396208-2-joseph.qi@linux.alibaba.com Fixes: 2cf82b46d5e4 ("ocfs2: validate external xattr entries when reading metadata") Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi Cc: Changwei Ge Cc: Jun Piao Cc: Heming Zhao --- fs/ocfs2/xattr.c | 21 ++++++++++++++++++++- 1 file changed, 20 insertions(+), 1 deletion(-) diff --git a/fs/ocfs2/xattr.c b/fs/ocfs2/xattr.c index e27f5f925df52e..5f65ae6ebb92bd 100644 --- a/fs/ocfs2/xattr.c +++ b/fs/ocfs2/xattr.c @@ -1150,11 +1150,30 @@ static int ocfs2_validate_xattr_bucket(struct ocfs2_xattr_bucket *bucket, struct ocfs2_xattr_header *xh = bucket_xh(bucket); u16 xattr_count = le16_to_cpu(xh->xh_count); size_t region_size = (size_t)sb->s_blocksize * bucket->bu_blocks; - size_t entries_limit = sb->s_blocksize; + /* + * The entry array grows up from the header across the whole + * bucket region, so it may extend beyond the first bucket block + * when the blocksize is smaller than OCFS2_XATTR_BUCKET_SIZE. + * Name/value pairs, however, always live within a single block. + */ + size_t entries_limit = region_size; size_t nv_limit = sb->s_blocksize; size_t max_entries; int i, ret; + /* + * The entry array is one contiguous region that may span the + * bucket's buffer_heads. Buckets are allocated within clusters, + * so their first block is always aligned to + * OCFS2_XATTR_BUCKET_SIZE and the whole bucket fits in one page. + * A corrupted xattr tree can point a bucket at blocks straddling + * a page, so reject it before touching the entry array. + */ + if (blkno & (bucket->bu_blocks - 1)) + return ocfs2_error(sb, + "Invalid xattr bucket %llu: unaligned block number\n", + (unsigned long long)blkno); + if (region_size < sizeof(*xh)) return ocfs2_error(sb, "Invalid xattr bucket %llu: region size %zu is too small\n", From a2364736914d665ab742da670156b6fb91cc4699 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Thu, 3 Sep 2026 21:13:13 +0800 Subject: [PATCH 1272/1352] ocfs2: reject inconsistent xattr bucket during defrag ocfs2_defrag_xattr_bucket() has two mlog_bug_on_msg() checks that assume the name/value pairs in a bucket are disjoint and that xh_free_start is not below the compacted region. ocfs2_validate_xattr_bucket() only checks each entry in isolation, so a corrupt bucket holding overlapping entries, or one with an inflated xh_free_start, passes validation and then hits BUG() in defrag when a setxattr triggers it. Defrag works on a linear copy of the bucket and does not touch the real blocks before the copy back, so the checks can return an error instead of calling BUG(). Link: https://lore.kernel.org/20260903131313.2396208-3-joseph.qi@linux.alibaba.com Fixes: 012255961c9e ("ocfs2: Enable xattr set in index btree") Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Cc: Changwei Ge Cc: Heming Zhao Cc: Joel Becker Cc: Jun Piao Cc: Junxiao Bi Cc: Mark Fasheh --- fs/ocfs2/xattr.c | 16 +++++++++++----- 1 file changed, 11 insertions(+), 5 deletions(-) diff --git a/fs/ocfs2/xattr.c b/fs/ocfs2/xattr.c index 5f65ae6ebb92bd..a2d6c3d0f2e8d8 100644 --- a/fs/ocfs2/xattr.c +++ b/fs/ocfs2/xattr.c @@ -4801,16 +4801,22 @@ static int ocfs2_defrag_xattr_bucket(struct inode *inode, memmove(bucket_buf + end - len, bucket_buf + offset, len); xe->xe_name_offset = cpu_to_le16(end - len); + } else if (end < offset + len) { + ret = ocfs2_error(inode->i_sb, + "Defrag check failed for bucket %llu\n", + (unsigned long long)blkno); + goto out; } - mlog_bug_on_msg(end < offset + len, "Defrag check failed for " - "bucket %llu\n", (unsigned long long)blkno); - end -= len; } - mlog_bug_on_msg(xh_free_start > end, "Defrag check failed for " - "bucket %llu\n", (unsigned long long)blkno); + if (xh_free_start > end) { + ret = ocfs2_error(inode->i_sb, + "Defrag check failed for bucket %llu\n", + (unsigned long long)blkno); + goto out; + } if (xh_free_start == end) goto out; From 147ba9899a61b70fe048c9ce4d4f63c30b3e5fb1 Mon Sep 17 00:00:00 2001 From: Hengyu Liang Date: Wed, 2 Sep 2026 13:01:15 -0400 Subject: [PATCH 1273/1352] fat: calculate data area start without overflow On 32-bit architectures, sbi->fat_length, sbi->dir_start and sbi->data_start are unsigned long. The number of FATs is an 8-bit BPB field, while the FAT32 length is a 32-bit BPB field. Therefore, the calculation sbi->fat_start + sbi->fats * sbi->fat_length can wrap before data_start is checked against total_sectors. For example, with fat_start=32, fats=2 and fat_length=0x80000001, the unwrapped data area start is 0x100000022 (4294967330), but the calculation wraps to 34 on i386. With total_sectors=36, the validation then incorrectly passes. The following script creates an image that demonstrates the problem: python3 - <<'PY' import struct S = 512 b = bytearray(36 * S) def p(off, fmt, value): struct.pack_into(fmt, b, off, value) # FAT32 BPB b[0:3] = b'\xeb\x58\x90' b[3:11] = b'MSWIN4.1' p(11, ' Signed-off-by: Andrew Morton Acked-by: OGAWA Hirofumi --- fs/fat/inode.c | 23 +++++++++++++++++------ 1 file changed, 17 insertions(+), 6 deletions(-) diff --git a/fs/fat/inode.c b/fs/fat/inode.c index f775a004cae1e2..0b0bbe777842da 100644 --- a/fs/fat/inode.c +++ b/fs/fat/inode.c @@ -20,6 +20,7 @@ #include #include #include +#include #include #include #include @@ -1577,6 +1578,7 @@ int fat_fill_super(struct super_block *sb, struct fs_context *fc, struct msdos_sb_info *sbi; u16 logical_sector_size; u32 total_sectors, total_clusters, fat_clusters, rootdir_sectors; + u32 dir_start, data_start; long error; char buf[50]; struct timespec64 ts; @@ -1752,7 +1754,6 @@ int fat_fill_super(struct super_block *sb, struct fs_context *fc, sbi->dir_per_block = sb->s_blocksize / sizeof(struct msdos_dir_entry); sbi->dir_per_block_bits = ffs(sbi->dir_per_block) - 1; - sbi->dir_start = sbi->fat_start + sbi->fats * sbi->fat_length; sbi->dir_entries = bpb.fat_dir_entries; if (sbi->dir_entries & (sbi->dir_per_block - 1)) { if (!silent) @@ -1763,20 +1764,30 @@ int fat_fill_super(struct super_block *sb, struct fs_context *fc, rootdir_sectors = sbi->dir_entries * sizeof(struct msdos_dir_entry) / sb->s_blocksize; - sbi->data_start = sbi->dir_start + rootdir_sectors; + if (check_mul_overflow(sbi->fats, sbi->fat_length, &dir_start) || + check_add_overflow(sbi->fat_start, dir_start, &dir_start) || + check_add_overflow(dir_start, rootdir_sectors, &data_start)) { + if (!silent) + fat_msg(sb, KERN_ERR, + "overflow of root dir or data layout"); + goto out_invalid; + } + total_sectors = bpb.fat_sectors; if (total_sectors == 0) total_sectors = bpb.fat_total_sect; - if (total_sectors < sbi->data_start) { + if (total_sectors < data_start) { if (!silent) fat_msg(sb, KERN_ERR, - "data area starts beyond volume (%lu > %u)", - sbi->data_start, total_sectors); + "data area starts beyond volume (%u > %u)", + data_start, total_sectors); goto out_invalid; } - total_clusters = (total_sectors - sbi->data_start) / sbi->sec_per_clus; + sbi->dir_start = dir_start; + sbi->data_start = data_start; + total_clusters = (total_sectors - data_start) / sbi->sec_per_clus; if (!is_fat32(sbi)) sbi->fat_bits = (total_clusters > MAX_FAT12) ? 16 : 12; From c74fb27d0c5f4cc625f2c14ba980909e21fca955 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Thu, 3 Sep 2026 18:47:44 +0800 Subject: [PATCH 1274/1352] ocfs2: skip uninitialized lockres in ocfs2_mark_lockres_freeing() A hard readonly mount skips ocfs2_dlm_init(), so the per-osb lock resources are never initialized and osb->cconn stays NULL. Before commit 550842cc60987 ("ocfs2: fix freeing uninitialized resource on ocfs2_dlm_shutdown") ocfs2_dismount_volume() only called ocfs2_dlm_shutdown() when osb->cconn was set. It now calls it unconditionally, so unmounting a hard readonly mount drops the osb locks and takes the never initialized l_lock in ocfs2_mark_lockres_freeing(). With lockdep enabled this triggers: INFO: trying to register non-static key. The code is fine but needs lockdep annotation, or maybe you didn't initialize this object before use? turning off the locking correctness validator. ocfs2_drop_lock() and ocfs2_lock_res_free() already skip lock resources without OCFS2_LOCK_INITIALIZED. Add the same check to ocfs2_mark_lockres_freeing(), which is reachable before them through ocfs2_simple_drop_lockres(), so an uninitialized lockres is never touched. Link: https://lore.kernel.org/20260903104744.2164235-1-joseph.qi@linux.alibaba.com Fixes: 550842cc6098 ("ocfs2: fix freeing uninitialized resource on ocfs2_dlm_shutdown") Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Reported-by: ZW Tang Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi Cc: Changwei Ge Cc: Jun Piao Cc: Heming Zhao Cc: --- fs/ocfs2/dlmglue.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/fs/ocfs2/dlmglue.c b/fs/ocfs2/dlmglue.c index a23dd8f86c8950..cf3318b0d3a8c9 100644 --- a/fs/ocfs2/dlmglue.c +++ b/fs/ocfs2/dlmglue.c @@ -3525,6 +3525,10 @@ void ocfs2_mark_lockres_freeing(struct ocfs2_super *osb, struct ocfs2_mask_waiter mw; unsigned long flags, flags2; + /* We didn't get anywhere near actually using this lockres. */ + if (!(lockres->l_flags & OCFS2_LOCK_INITIALIZED)) + return; + ocfs2_init_mask_waiter(&mw); spin_lock_irqsave(&lockres->l_lock, flags); From 6012ff64a2cfdce56c47c8e7447f91721b14ea31 Mon Sep 17 00:00:00 2001 From: Adam Harshbarger Date: Thu, 3 Sep 2026 17:24:56 -0500 Subject: [PATCH 1275/1352] lib/plist: fix plist_requeue() corrupting order in the last bucket plist_requeue() is meant to move a node to the end of its own priority run. When the node heads the *last* priority bucket it is instead placed at the head of the whole list, leaving the plist unsorted: built: A(prio 0) B(prio 1) C(prio 1) requeue(B): B(prio 1) A(prio 0) C(prio 1) expected: A(prio 0) C(prio 1) B(prio 1) prio_list is a *headless* circular ring of the nodes that lead each priority bucket. The shortcut added by commit 95d4b3450ebe ("lib/plist.c: add shortcut for plist_requeue()") takes iter = list_entry(iter->prio_list.next, struct plist_node, prio_list); node_next = &iter->node_list; which from the last bucket wraps round to the *first* bucket, so node_next ends up pointing at the head of the list rather than at its end. The plist_for_each_continue() loop immediately below it computes the correct answer (&head->node_list) for that case. With any bucket after it the shortcut is correct, which is why this went unnoticed: the benchmark in that commit measured elapsed time and never checked the resulting order. Keep the shortcut -- it is a real win -- but exclude the case where iter's bucket is the last one, which is exactly when its ring successor is the first bucket again. Reachable from mm/swapfile.c, which rotates swap_avail_heads[] with plist_requeue(). It takes three or more swap devices: at least two distinct priorities, so that a later bucket exists for the ring to wrap round from, and two or more devices sharing the lowest priority, so that plist_requeue() does not return early. One device per priority returns early at the node->prio != iter->prio test. A single priority is also safe, but for a different reason worth stating: with one bucket no node is ever linked onto prio_list at all -- plist_add() skips it for the first node and for every node whose predecessor shares its priority -- so list_empty(&iter->prio_list) holds and the shortcut is never entered. Tested by driving three implementations -- the pre-95d4b3450ebe code, current mainline, and this patch -- through 1,084,492 identical random add/del/requeue operations over 24 nodes and 1..5 distinct priorities, comparing the resulting node_list node for node after every operation: variant differs from pre-95d4b3450ebe left list unsorted pre-95d4b3450ebe -- (reference) 0 mainline 289,297 276,660 this patch 0 0 Link: https://lore.kernel.org/20260903222456.1881786-1-handyhandyman.adam@gmail.com Fixes: 95d4b3450ebe ("lib/plist.c: add shortcut for plist_requeue()") Signed-off-by: Adam Harshbarger Signed-off-by: Andrew Morton Assisted-by: Claude:claude-opus-5 Cc: I Hsin Cheng Cc: # v6.15+ --- lib/plist.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/lib/plist.c b/lib/plist.c index a5bef38add431d..b0273533c04d2a 100644 --- a/lib/plist.c +++ b/lib/plist.c @@ -174,8 +174,15 @@ void plist_requeue(struct plist_node *node, struct plist_head *head) /* * After plist_del(), iter is the replacement of the node. If the node * was on prio_list, take shortcut to find node_next instead of looping. + * + * prio_list is a headless ring, so from the LAST bucket ->next wraps + * round to the first one; in that case node_next is the list head. */ if (!list_empty(&iter->prio_list)) { + struct plist_node *first = plist_first(head); + + if (iter->prio_list.next == &first->prio_list) + goto queue; iter = list_entry(iter->prio_list.next, struct plist_node, prio_list); node_next = &iter->node_list; From 352939c193a5b997a156a566af8bbd855d868ea3 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ali=20Ahmet=20Memi=C5=9F?= Date: Sat, 5 Sep 2026 22:49:39 +0300 Subject: [PATCH 1276/1352] =?UTF-8?q?mailmap:=20update=20email=20address?= =?UTF-8?q?=20for=20Ali=20Ahmet=20Memi=C5=9F?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit I'm switching to my disroot address for kernel contributions. Map the address my earlier patches were sent from to it, so that get_maintainer.pl stops offering the old one as a recipient. Also switch to the proper spelling of my name with diacritics, which is what I use with the new address. Link: https://lore.kernel.org/20260905195000.556185-1-aliamemis@disroot.org Signed-off-by: Ali Ahmet Memiş Signed-off-by: Andrew Morton --- .mailmap | 1 + 1 file changed, 1 insertion(+) diff --git a/.mailmap b/.mailmap index 3940b0a12c2820..b399a10744fc08 100644 --- a/.mailmap +++ b/.mailmap @@ -67,6 +67,7 @@ Alex Hung Alex Shi Alex Shi Alex Shi +Ali Ahmet Memiş Alice Mikityanska Alice Mikityanska Alice Mikityanska From 20cac81fb8752bd3525defb5d4c7ba52d0e58e76 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Sat, 5 Sep 2026 22:21:43 +0800 Subject: [PATCH 1277/1352] ocfs2: validate dr_fs_generation of dir index root blocks ocfs2_validate_dx_root() does not verify dr_fs_generation against the superblock generation, unlike the extent and xattr block validators which check h_fs_generation and xb_fs_generation respectively. The field is documented as "Must match super block". Without the check, a stale dir index root block left on the device from a previously formatted filesystem at the same physical block number can pass validation as long as its signature, dr_blkno and checksum match. Its index entries and suballocator information would then be used in the new filesystem context. Reject dir index root blocks whose dr_fs_generation does not match the mounted filesystem, like the extent and xattr block validators do. Link: https://lore.kernel.org/20260905142144.2869105-1-joseph.qi@linux.alibaba.com Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Reviewed-by: Heming Zhao Cc: Changwei Ge Cc: Joel Becker Cc: Jun Piao Cc: Junxiao Bi Cc: Mark Fasheh --- fs/ocfs2/dir.c | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/fs/ocfs2/dir.c b/fs/ocfs2/dir.c index 6bb6aa133f0150..329680b4622739 100644 --- a/fs/ocfs2/dir.c +++ b/fs/ocfs2/dir.c @@ -613,6 +613,14 @@ static int ocfs2_validate_dx_root(struct super_block *sb, goto bail; } + if (le32_to_cpu(dx_root->dr_fs_generation) != OCFS2_SB(sb)->fs_generation) { + ret = ocfs2_error(sb, + "Dir Index Root # %llu has an invalid dr_fs_generation of #%u\n", + (unsigned long long)bh->b_blocknr, + le32_to_cpu(dx_root->dr_fs_generation)); + goto bail; + } + /* * Dir index root blocks are allocated from a per-slot suballocator, * so the slot must be in range. Otherwise removing the index passes From 8dbb70adc5e674a431cda8dc710e03328db30d30 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Sat, 5 Sep 2026 22:21:44 +0800 Subject: [PATCH 1278/1352] ocfs2: validate dl_blkno and dl_fs_generation of dir index leaf blocks ocfs2_validate_dx_leaf() checks the checksum, the signature and the entry list counts, but it never checks dl_blkno or dl_fs_generation. The inode, extent block, xattr block, refcount block and dir index root validators all check the on-disk block number against bh->b_blocknr and the generation against the superblock, and both dir index leaf fields are documented as "Must match super block". Without the checks, a stale dir index leaf block left on the device from a previously formatted filesystem at the same physical block number can pass validation as long as its signature, entry counts and checksum match. Its index entries would then be used in the new filesystem context. Both fields are written unconditionally when a leaf block is formatted in ocfs2_dx_dir_format_cluster(), from the live superblock generation and the real block number, so a correctly formatted filesystem cannot trip the new checks. The leaf block number read back here comes from on-disk dir index root extent records. Reject dir index leaf blocks whose dl_blkno or dl_fs_generation does not match, like the dir index root validator does. Link: https://lore.kernel.org/20260905142144.2869105-2-joseph.qi@linux.alibaba.com Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Reviewed-by: Heming Zhao Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi Cc: Changwei Ge Cc: Jun Piao --- fs/ocfs2/dir.c | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/fs/ocfs2/dir.c b/fs/ocfs2/dir.c index 329680b4622739..55c4a305a2823e 100644 --- a/fs/ocfs2/dir.c +++ b/fs/ocfs2/dir.c @@ -733,6 +733,18 @@ static int ocfs2_validate_dx_leaf(struct super_block *sb, return ocfs2_error(sb, "Dir Index Leaf has bad signature %.*s\n", 7, dx_leaf->dl_signature); + if (le64_to_cpu(dx_leaf->dl_blkno) != bh->b_blocknr) + return ocfs2_error(sb, + "Dir Index Leaf # %llu has an invalid dl_blkno of %llu\n", + (unsigned long long)bh->b_blocknr, + (unsigned long long)le64_to_cpu(dx_leaf->dl_blkno)); + + if (le32_to_cpu(dx_leaf->dl_fs_generation) != OCFS2_SB(sb)->fs_generation) + return ocfs2_error(sb, + "Dir Index Leaf # %llu has an invalid dl_fs_generation of #%u\n", + (unsigned long long)bh->b_blocknr, + le32_to_cpu(dx_leaf->dl_fs_generation)); + if (le16_to_cpu(dx_leaf->dl_list.de_count) != ocfs2_dx_entries_per_leaf(sb)) return ocfs2_error(sb, From 7d70ff225262b78a68bdb04f9e8f67d9aac31835 Mon Sep 17 00:00:00 2001 From: Petr Vorel Date: Fri, 4 Sep 2026 13:02:27 +0200 Subject: [PATCH 1279/1352] checkpatch: add more userspace directories to is_userspace() Patch series "checkpatch: userspace improvements", v6. Few improvements for user space code + --userspace option for projects which vendored checkpatch.pl. There could be probably more checks which are kernel space only. This patch (of 4): arch/ directory contains subdirectories with userspace tools (at least arch/*/tools/ and arch/*/boot/tools/). Add check to consider any arch/.*/tools/ subdirectory as userspace tools directory. This helps not only to strscpy() checks but also to CamelCase checks in the next commit to be more precise. This is a follow-up to 99b70ece33d8 ("checkpatch: suppress strscpy warnings for userspace tools"). Link: https://lore.kernel.org/20260904110230.1219037-1-pvorel@suse.cz Link: https://lore.kernel.org/20260904110230.1219037-2-pvorel@suse.cz Signed-off-by: Petr Vorel Signed-off-by: Andrew Morton Cc: Andy Whitcroft Cc: Dwaipayan Ray Cc: Joe Perches Cc: Lukas Bulwahn --- scripts/checkpatch.pl | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/scripts/checkpatch.pl b/scripts/checkpatch.pl index f424dafce5bce7..b8702f6bc9b572 100755 --- a/scripts/checkpatch.pl +++ b/scripts/checkpatch.pl @@ -2667,7 +2667,9 @@ sub exclude_global_initialisers { sub is_userspace { my ($realfile) = @_; - return ($realfile =~ m@^tools/@ || $realfile =~ m@^scripts/@); + return ($realfile =~ m@^tools/@ || + $realfile =~ m@^scripts/@ || + $realfile =~ m@^arch/.*/tools/@); } sub process { From a7e10bbff672896f8552859e26958b235cf65bf5 Mon Sep 17 00:00:00 2001 From: Petr Vorel Date: Fri, 4 Sep 2026 13:02:28 +0200 Subject: [PATCH 1280/1352] checkpatch: ignore format macros for userspace tools Constants from are used only in userspace tools, they are from ISO C99, let's don't report it: arch/mips/boot/tools/relocs.c:572: CHECK: Avoid CamelCase: arch/s390/tools/relocs.c:52: CHECK: Avoid CamelCase: tools/testing/selftests/mm/vm_util.c:244: CHECK: Avoid CamelCase: Link: https://lore.kernel.org/20260904110230.1219037-3-pvorel@suse.cz Signed-off-by: Petr Vorel Signed-off-by: Andrew Morton Cc: Andy Whitcroft Cc: Dwaipayan Ray Cc: Joe Perches Cc: Lukas Bulwahn --- scripts/checkpatch.pl | 2 ++ 1 file changed, 2 insertions(+) diff --git a/scripts/checkpatch.pl b/scripts/checkpatch.pl index b8702f6bc9b572..b458c7f2268484 100755 --- a/scripts/checkpatch.pl +++ b/scripts/checkpatch.pl @@ -5950,6 +5950,8 @@ sub process { #Ignore SI style variants like nS, mV and dB #(ie: max_uV, regulator_min_uA_show, RANGE_mA_VALUE) $var !~ /^(?:[a-z0-9_]*|[A-Z0-9_]*)?_?[a-z][A-Z](?:_[a-z0-9_]+|_[A-Z0-9_]+)?$/ && +#Ignore format macros (e.g. PRIu64, SCNu64) + (is_userspace($realfile) ? $var !~ /^(?:PRI|SCN)[dioux][A-Z0-9]+$/ : 1) && #Ignore some three character SI units explicitly, like MiB and KHz $var !~ /^(?:[a-z_]*?)_?(?:[KMGT]iB|[KMGT]?Hz)(?:_[a-z_]+)?$/) { while ($var =~ m{\b($Ident)}g) { From 53b34d70fc501227d81ed42b9fd36352a230cb80 Mon Sep 17 00:00:00 2001 From: Petr Vorel Date: Fri, 4 Sep 2026 13:02:29 +0200 Subject: [PATCH 1281/1352] checkpatch: add --userspace to force userspace rules Also allow to use --no-userspace for userspace projects which vendored checkpatch.pl and use --userspace globally to be able switch it off for files with kernel code. Link: https://lore.kernel.org/20260904110230.1219037-4-pvorel@suse.cz Signed-off-by: Petr Vorel Signed-off-by: Andrew Morton Cc: Andy Whitcroft Cc: Dwaipayan Ray Cc: Joe Perches Cc: Lukas Bulwahn --- scripts/checkpatch.pl | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/scripts/checkpatch.pl b/scripts/checkpatch.pl index b458c7f2268484..ead35e6abba7ba 100755 --- a/scripts/checkpatch.pl +++ b/scripts/checkpatch.pl @@ -63,6 +63,7 @@ my $max_line_length = 100; my $ignore_perl_version = 0; my $spdx_cxx_comments = 0; +my $userspace; my $minimum_perl_version = 5.10.0; my $min_conf_desc_length = 4; my $spelling_file = "$D/spelling.txt"; @@ -143,6 +144,7 @@ sub help { (required by old toolchains), allow also C++ comments (//). NOTE: it should *not* be used for Linux mainline. + --userspace Force rules specific for userspace. --codespell Use the codespell dictionary for spelling/typos (default:$codespellfile) --codespellfile Use this codespell dictionary @@ -358,6 +360,7 @@ sub load_docs { 'codespell!' => \$codespell, 'codespellfile=s' => \$user_codespellfile, 'typedefsfile=s' => \$typedefsfile, + 'userspace!' => \$userspace, 'color=s' => \$color, 'no-color' => \$color, #keep old behaviors of -nocolor 'nocolor' => \$color, #keep old behaviors of -nocolor @@ -2667,6 +2670,9 @@ sub exclude_global_initialisers { sub is_userspace { my ($realfile) = @_; + + return $userspace if (defined $userspace); + return ($realfile =~ m@^tools/@ || $realfile =~ m@^scripts/@ || $realfile =~ m@^arch/.*/tools/@); From c07d9714eeca738e0f28ca191522dccf6c2ea140 Mon Sep 17 00:00:00 2001 From: Petr Vorel Date: Fri, 4 Sep 2026 13:02:30 +0200 Subject: [PATCH 1282/1352] checkpatch: skip kernel specific checks for userspace These check are kernel specific, do not warn about it when testing userspace code: * BIT_MACRO * LONG_UDELAY * MSLEEP * PREFER_KERNEL_TYPES * USLEEP_RANGE This is a follow-up to 99b70ece33d8 ("checkpatch: suppress strscpy warnings for userspace tools"). Link: https://lore.kernel.org/20260904110230.1219037-5-pvorel@suse.cz Signed-off-by: Petr Vorel Signed-off-by: Andrew Morton Cc: Andy Whitcroft Cc: Dwaipayan Ray Cc: Joe Perches Cc: Lukas Bulwahn --- scripts/checkpatch.pl | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/scripts/checkpatch.pl b/scripts/checkpatch.pl index ead35e6abba7ba..3614cfe4dcbb45 100755 --- a/scripts/checkpatch.pl +++ b/scripts/checkpatch.pl @@ -6696,7 +6696,8 @@ sub process { } # prefer usleep_range over udelay - if ($line =~ /\budelay\s*\(\s*(\d+)\s*\)/) { + if (!is_userspace($realfile) && + $line =~ /\budelay\s*\(\s*(\d+)\s*\)/) { my $delay = $1; # ignore udelay's < 10, however if (! ($delay < 10) ) { @@ -6710,7 +6711,8 @@ sub process { } # warn about unexpectedly long msleep's - if ($line =~ /\bmsleep\s*\((\d+)\);/) { + if (!is_userspace($realfile) && + $line =~ /\bmsleep\s*\((\d+)\);/) { if ($1 < 20) { WARN("MSLEEP", "msleep < 20ms can sleep for up to 20ms; see function description of msleep().\n" . $herecurr); @@ -6932,7 +6934,7 @@ sub process { # check for c99 types like uint8_t used outside of uapi/ and tools/ if ($realfile !~ m@\binclude/uapi/@ && - $realfile !~ m@\btools/@ && + !is_userspace($realfile) && $line =~ /\b($Declare)\s*$Ident\s*[=;,\[]/) { my $type = $1; if ($type =~ /\b($typeC99Typedefs)\b/) { @@ -7182,6 +7184,7 @@ sub process { # check usleep_range arguments if ($perl_version_ok && defined $stat && + !is_userspace($realfile) && $stat =~ /^\+(?:.*?)\busleep_range\s*\(\s*($FuncArg)\s*,\s*($FuncArg)\s*\)/) { my $min = $1; my $max = $7; @@ -7425,6 +7428,7 @@ sub process { # check for #defines like: 1 << that could be BIT(digit), it is not exported to uapi if ($realfile !~ m@^include/uapi/@ && + !is_userspace($realfile) && $line =~ /#\s*define\s+\w+\s+\(?\s*1\s*([ulUL]*)\s*\<\<\s*(?:\d+|$Ident)\s*\)?/) { my $ull = ""; $ull = "_ULL" if (defined($1) && $1 =~ /ll/i); From 3cbd8a545e89f575d2f9e244ff0e3efd53299480 Mon Sep 17 00:00:00 2001 From: Kazuki Hanai Date: Fri, 28 Aug 2026 00:25:16 +0900 Subject: [PATCH 1283/1352] tmpfs: fix unicode_map leaks in casefold option handling MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit shmem_parse_opt_casefold() stores the unicode_map returned by utf8_load() in ctx->encoding. The casefold parameter can be supplied more than once for the same filesystem context, but replacing the stored map does not release the previous reference. The final reference is also leaked when an unmounted filesystem context is freed. Release the previous map before replacing it, clear ctx->encoding after transferring ownership to the superblock, and release any remaining reference from shmem_free_fc(). An unprivileged user can repeatedly set the casefold parameter on a tmpfs filesystem context from a user namespace. This causes unbounded kernel memory consumption and can result in a local denial of service. Link: https://lore.kernel.org/20260827152516.805622-1-hnkz.64@gmail.com Fixes: 58e55efd6c72 ("tmpfs: Add casefold lookup support") Signed-off-by: Kazuki Hanai Signed-off-by: Andrew Morton Reviewed-by: Andrew Morton Cc: Baolin Wang Cc: Hugh Dickins Cc: André Almeida Cc: Christian Brauner Cc: --- mm/shmem.c | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/mm/shmem.c b/mm/shmem.c index 848316eaa7f4fb..c6be5961956256 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -4530,6 +4530,7 @@ static int shmem_parse_opt_casefold(struct fs_context *fc, struct fs_parameter * pr_info("tmpfs: Using encoding : utf8-%u.%u.%u\n", unicode_major(version), unicode_minor(version), unicode_rev(version)); + utf8_unload(ctx->encoding); ctx->encoding = encoding; return 0; @@ -4998,6 +4999,7 @@ static int shmem_fill_super(struct super_block *sb, struct fs_context *fc) if (ctx->encoding) { sb->s_encoding = ctx->encoding; + ctx->encoding = NULL; set_default_d_op(sb, &shmem_ci_dentry_ops); if (ctx->strict_encoding) sb->s_encoding_flags = SB_ENC_STRICT_MODE_FL; @@ -5095,6 +5097,9 @@ static void shmem_free_fc(struct fs_context *fc) struct shmem_options *ctx = fc->fs_private; if (ctx) { +#if IS_ENABLED(CONFIG_UNICODE) + utf8_unload(ctx->encoding); +#endif mpol_put(ctx->mpol); kfree(ctx); } From 455daf9352bf6dfd89dc4436ed61646664193471 Mon Sep 17 00:00:00 2001 From: Sang-Heon Jeon Date: Fri, 11 Sep 2026 01:30:48 +0900 Subject: [PATCH 1284/1352] bootconfig: remove redundant assignment in xbc_parse_array() After the loop, node is the last array value that xbc_add_child() added, and xbc_parse_array() sets node->child to 0. xbc_init_node() already set node->child to 0 when the value was added, and node->child is not modified after that, so the assignment is redundant. So remove the assignment. No functional change. Link: https://lore.kernel.org/20260910163051.1973775-1-ekffu200098@gmail.com Signed-off-by: Sang-Heon Jeon Signed-off-by: Andrew Morton Acked-by: Masami Hiramatsu --- lib/bootconfig.c | 1 - 1 file changed, 1 deletion(-) diff --git a/lib/bootconfig.c b/lib/bootconfig.c index 89c88e359179f0..f248b5770291cb 100644 --- a/lib/bootconfig.c +++ b/lib/bootconfig.c @@ -836,7 +836,6 @@ static int __init xbc_parse_array(char **__v) return -ENOMEM; *__v = next; } while (c == ','); - node->child = 0; return c; } From ae1b16da407046238ac4d36462d023b14ff4461c Mon Sep 17 00:00:00 2001 From: "Masami Hiramatsu (Google)" Date: Fri, 11 Sep 2026 23:13:06 +0900 Subject: [PATCH 1285/1352] bootconfig: reject unexpected data after null character Patch series "bootconfig: Reject unexpected data after null character and cleanups", v2. Make bootconfig reject unexpected config data after null character and implement other cleanups including tools/bootconfig to consolidate bootconfig initialization with errors, and to skip internal tree sanity checks in kernel. This patch (of 4): If a bootconfig buffer contains an intermediate null character in the middle of the configuration, xbc_parse_tree() stops at the null character because string delimiter searches (e.g. strpbrk()) stop at '\0', and cleanly breaks out of the loop without error. As a result, any configuration data following the intermediate null character is silently ignored, allowing unparsed or potentially malicious data to be hidden after an early termination. Fix this in xbc_parse_tree() by checking that no non-null data remains between the parser termination point and the end of the input buffer. Trailing null characters (such as alignment padding in initrd) continue to be accepted as valid. Also update apply_xbc() in tools/bootconfig/main.c to calculate the buffer size based on the loaded file size rather than strlen(), so that files with intermediate null characters are not truncated before validation. Link: https://lore.kernel.org/178913597653.248794.1237187523153227751.stgit@devnote2 Link: https://lore.kernel.org/178913598628.248794.1048773774471250986.stgit@devnote2 Signed-off-by: Masami Hiramatsu (Google) Signed-off-by: Andrew Morton Reviewed-by: Sang-Heon Jeon Assisted-by: Antigravity:gemini-3.8-flash --- lib/bootconfig.c | 7 +++++++ tools/bootconfig/main.c | 4 +++- tools/bootconfig/test-bootconfig.sh | 12 ++++++++++++ 3 files changed, 22 insertions(+), 1 deletion(-) diff --git a/lib/bootconfig.c b/lib/bootconfig.c index f248b5770291cb..b6adf8b274746c 100644 --- a/lib/bootconfig.c +++ b/lib/bootconfig.c @@ -1118,6 +1118,13 @@ static int __init xbc_parse_tree(void) } } while (!ret); + if (!ret) { + while (p < xbc_data + xbc_data_size - 1 && *p == '\0') + p++; + if (p < xbc_data + xbc_data_size - 1) + ret = xbc_parse_error("Unexpected data after null character", p); + } + return ret; } diff --git a/tools/bootconfig/main.c b/tools/bootconfig/main.c index 17d971d47f8791..aff169ba75b828 100644 --- a/tools/bootconfig/main.c +++ b/tools/bootconfig/main.c @@ -433,7 +433,9 @@ static int apply_xbc(const char *path, const char *xbc_path) pr_err("Failed to load %s : %d\n", xbc_path, ret); return ret; } - size = strlen(buf) + 1; + size = ret; + if (size == 0 || buf[size - 1] != '\0') + size++; csum = xbc_calc_checksum(buf, size); /* Backup the bootconfig data */ diff --git a/tools/bootconfig/test-bootconfig.sh b/tools/bootconfig/test-bootconfig.sh index fc69f815ce4af0..530ce7e28d634e 100755 --- a/tools/bootconfig/test-bootconfig.sh +++ b/tools/bootconfig/test-bootconfig.sh @@ -180,6 +180,18 @@ EOF $BOOTCONF -a $TEMPCONF $INITRD 2> $OUTFILE xpass grep -q "1:1" $OUTFILE +echo "Intermediate null character test" +printf "key = value\n\0extra = data\n" > $TEMPCONF +xfail $BOOTCONF -a $TEMPCONF $INITRD +$BOOTCONF -a $TEMPCONF $INITRD 2> $OUTFILE +xpass grep -q "Unexpected" $OUTFILE + +echo "Trailing null character test" +printf "key = value\n\0" > $TEMPCONF +xpass $BOOTCONF -a $TEMPCONF $INITRD +$BOOTCONF $INITRD > $OUTFILE +xpass grep -q "value" $OUTFILE + echo "=== expected failure cases ===" for i in samples/bad-* ; do xfail $BOOTCONF -a $i $INITRD From 41daaa21fc772b6fe9f676e50a1e693251d44a13 Mon Sep 17 00:00:00 2001 From: "Masami Hiramatsu (Google)" Date: Fri, 11 Sep 2026 23:13:15 +0900 Subject: [PATCH 1286/1352] tools/bootconfig: consolidate xbc_init() to error message wrapper Use init_xbc_with_error() for all bootconfig initialization in the bootconfig tool instead of showing errors in different way. This simplifies the code logic and make it easy to maintain. Link: https://lore.kernel.org/178913599508.248794.10388592925402087434.stgit@devnote2 Signed-off-by: Masami Hiramatsu (Google) Signed-off-by: Andrew Morton Reviewed-by: Sang-Heon Jeon --- tools/bootconfig/main.c | 100 +++++++++++++++++----------------------- 1 file changed, 43 insertions(+), 57 deletions(-) diff --git a/tools/bootconfig/main.c b/tools/bootconfig/main.c index aff169ba75b828..652e491b9c338f 100644 --- a/tools/bootconfig/main.c +++ b/tools/bootconfig/main.c @@ -21,6 +21,39 @@ #define BOOTCONFIG_FOOTER_SIZE \ (sizeof(uint32_t) * 2 + BOOTCONFIG_MAGIC_LEN) +static void show_xbc_error(const char *data, const char *msg, int pos) +{ + int lin = 1, col, i; + + if (pos < 0) { + pr_err("Error: %s.\n", msg); + return; + } + + /* Note that pos starts from 0 but lin and col should start from 1. */ + col = pos + 1; + for (i = 0; i < pos; i++) { + if (data[i] == '\n') { + lin++; + col = pos - i; + } + } + pr_err("Parse Error: %s at %d:%d\n", msg, lin, col); + +} + +static int init_xbc_with_error(char *buf, int len) +{ + const char *msg; + int ret, pos; + + ret = xbc_init(buf, len, &msg, &pos); + if (ret < 0) + show_xbc_error(buf, msg, pos); + + return ret; +} + static int xbc_show_value(struct xbc_node *node, bool semicolon) { const char *val, *eol; @@ -197,7 +230,6 @@ static int load_xbc_from_initrd(int fd, char **buf) int ret; uint32_t size = 0, csum = 0, rcsum; char magic[BOOTCONFIG_MAGIC_LEN]; - const char *msg; ret = fstat(fd, &stat); if (ret < 0) @@ -249,52 +281,9 @@ static int load_xbc_from_initrd(int fd, char **buf) return -EINVAL; } - ret = xbc_init(*buf, size, &msg, NULL); - /* Wrong data */ - if (ret < 0) { - pr_err("parse error: %s.\n", msg); - return ret; - } - - return size; -} - -static void show_xbc_error(const char *data, const char *msg, int pos) -{ - int lin = 1, col, i; - - if (pos < 0) { - pr_err("Error: %s.\n", msg); - return; - } - - /* Note that pos starts from 0 but lin and col should start from 1. */ - col = pos + 1; - for (i = 0; i < pos; i++) { - if (data[i] == '\n') { - lin++; - col = pos - i; - } - } - pr_err("Parse Error: %s at %d:%d\n", msg, lin, col); + ret = init_xbc_with_error(*buf, size); -} - -static int init_xbc_with_error(char *buf, int len) -{ - char *copy = strdup(buf); - const char *msg; - int ret, pos; - - if (!copy) - return -ENOMEM; - - ret = xbc_init(buf, len, &msg, &pos); - if (ret < 0) - show_xbc_error(copy, msg, pos); - free(copy); - - return ret; + return ret < 0 ? ret : size; } static int show_xbc_kernel_cmdline(void) @@ -423,9 +412,8 @@ static int apply_xbc(const char *path, const char *xbc_path) char *buf, *data; size_t total_size; struct stat stat; - const char *msg; uint32_t size, csum; - int pos, pad; + int pad; int ret, fd; ret = load_xbc_file(xbc_path, &buf); @@ -438,6 +426,13 @@ static int apply_xbc(const char *path, const char *xbc_path) size++; csum = xbc_calc_checksum(buf, size); + /* Verify the data format */ + ret = init_xbc_with_error(buf, size); + if (ret < 0) { + free(buf); + return ret; + } + /* Backup the bootconfig data */ data = calloc(size + BOOTCONFIG_ALIGN + BOOTCONFIG_FOOTER_SIZE, 1); if (!data) { @@ -446,15 +441,6 @@ static int apply_xbc(const char *path, const char *xbc_path) } memcpy(data, buf, size); - /* Check the data format */ - ret = xbc_init(buf, size, &msg, &pos); - if (ret < 0) { - show_xbc_error(data, msg, pos); - free(data); - free(buf); - - return ret; - } printf("Apply %s to %s\n", xbc_path, path); xbc_get_info(&ret, NULL); printf("\tNumber of nodes: %d\n", ret); From c2ce302b08ec1f9871391b91c2b0ff0d6f12c751 Mon Sep 17 00:00:00 2001 From: "Masami Hiramatsu (Google)" Date: Fri, 11 Sep 2026 23:13:24 +0900 Subject: [PATCH 1287/1352] bootconfig: skip internal tree sanity checks in kernel In xbc_verify_tree(), the loop iterating through all nodes to check that xbc_nodes[i].next < xbc_node_num and xbc_nodes[i].child < xbc_node_num is a defensive sanity check against implementation regressions (such an out-of-bounds index cannot be produced by malformed input). Running this check in the kernel adds unnecessary boot-time overhead. Split this check out into xbc_sanity_check_tree() for userspace, so that it continues to run during userspace bootconfig validation (e.g. when applying or testing bootconfig with tools/bootconfig), but is omitted in the kernel to speed up initialization. Link: https://lore.kernel.org/178913600409.248794.12941798680046327623.stgit@devnote2 Signed-off-by: Masami Hiramatsu (Google) Signed-off-by: Andrew Morton Reported-by: Sang-Heon Jeon Closes: https://lore.kernel.org/all/20260905141637.1547429-1-ekffu200098@gmail.com/ Reviewed-by: Sang-Heon Jeon --- lib/bootconfig.c | 38 ++++++++++++++++++++++++++------------ 1 file changed, 26 insertions(+), 12 deletions(-) diff --git a/lib/bootconfig.c b/lib/bootconfig.c index b6adf8b274746c..192a60a9f833cb 100644 --- a/lib/bootconfig.c +++ b/lib/bootconfig.c @@ -1002,9 +1002,30 @@ static int __init xbc_close_brace(char **k, char *n) return __xbc_close_brace(n - 1); } +#ifndef __KERNEL__ +/* Sanity check for regression: node indices must be within bounds */ +static int __init xbc_sanity_check_tree(void) +{ + int i; + + for (i = 0; i < xbc_node_num; i++) { + if (xbc_nodes[i].next >= xbc_node_num) { + return xbc_parse_error("No closing brace", + xbc_node_get_data(xbc_nodes + i)); + } + if (xbc_nodes[i].child >= xbc_node_num) { + return xbc_parse_error("Broken child node", + xbc_node_get_data(xbc_nodes + i)); + } + } + + return 0; +} +#endif + static int __init xbc_verify_tree(void) { - int i, depth; + int depth; size_t len, wlen; struct xbc_node *n, *m; @@ -1021,17 +1042,6 @@ static int __init xbc_verify_tree(void) return -ENOENT; } - for (i = 0; i < xbc_node_num; i++) { - if (xbc_nodes[i].next >= xbc_node_num) { - return xbc_parse_error("No closing brace", - xbc_node_get_data(xbc_nodes + i)); - } - if (xbc_nodes[i].child >= xbc_node_num) { - return xbc_parse_error("Broken child node", - xbc_node_get_data(xbc_nodes + i)); - } - } - /* Key tree limitation check */ n = &xbc_nodes[0]; depth = 1; @@ -1202,6 +1212,10 @@ int __init xbc_init(const char *data, size_t size, const char **emsg, int *epos) ret = xbc_parse_tree(); if (!ret) ret = xbc_verify_tree(); +#ifndef __KERNEL__ + if (!ret) + ret = xbc_sanity_check_tree(); +#endif if (ret < 0) { if (epos) From 6ab2f9361d146501565b0efa626b0cfd3be416c3 Mon Sep 17 00:00:00 2001 From: Hemanth Selam Date: Mon, 7 Sep 2026 14:43:24 +0530 Subject: [PATCH 1288/1352] scripts/spelling.txt: keep British spelling for "initalise" These are the only two entries in the file that map a British spelling to an American one: initalised||initialized initalise||initialize doc-guide/checkpatch.rst asks that British and American spellings be left alone, so correcting the missing "i" should not also change the variety of English. Following checkpatch here silently rewrites "initalised" to "initialized" in files that otherwise use British forms; arch/arm64 alone has 58 uses of "initialised" against 132 of "initialized", so both are clearly in use. Map them to the British forms instead, so the misspelling is still caught while the spelling variety is left to the author. Link: https://lore.kernel.org/20260907091324.32211-1-hemanth.selam@gmail.com Signed-off-by: Hemanth Selam Signed-off-by: Andrew Morton Suggested-by: Randy Dunlap Acked-by: Randy Dunlap Assisted-by: Cursor:claude-opus-5 Cc: Colin Ian King Cc: Joe Perches Cc: Jonathan Corbet --- scripts/spelling.txt | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/scripts/spelling.txt b/scripts/spelling.txt index 3372873cd7bb7d..9ec2529bc58472 100644 --- a/scripts/spelling.txt +++ b/scripts/spelling.txt @@ -872,8 +872,8 @@ infromation||information ingore||ignore inheritence||inheritance inital||initial -initalised||initialized -initalise||initialize +initalised||initialised +initalise||initialise initalized||initialized initalize||initialize initation||initiation From 1b84417f74c19fd0832c54f54ed39fe7b7d82530 Mon Sep 17 00:00:00 2001 From: Matt Turner Date: Sat, 12 Sep 2026 14:17:35 -0400 Subject: [PATCH 1289/1352] lib/decompress_bunzip2: fix off-by-one in run-length bounds check The run-length path rejects a block when dbufCount+t equals dbufSize, but the loop that follows writes exactly t bytes starting at dbufCount, so a block that fills the buffer exactly is legal. bzip2 allows it too: its decompressor bounds a block at 100000 * blockSize100k and checks that limit per byte appended. Use > instead of >=. bzip2's encoder stops filling a block 19 bytes early, so nothing it produces ever reaches the limit and the bug stays hidden. Compressors that use the full block size do reach it: an lbzip2 -9 image whose block ends on a run fails to decode, and a self-extracting kernel built that way does not boot. This code came from busybox, which fixed the same line in 2013 in commit 932e233a491b ("bunzip2: fix off-by-one check"). I ran into this because my system uses lbzip2 as /bin/bzip2 -- a common thing on Gentoo I believe. As far as I can tell, anyone using lbzip2 as their system bzip2 would run into this and it's only because it's very uncommon these days to compress a kernel with bzip2 that no one has noticed. I only noticed because I was adding support for various compression formats on alpha. Link: https://lore.kernel.org/20260912-b4-bunzip2-blocksize-fix-v1-1-c7384bbfc954@gmail.com Fixes: bc22c17e12c1 ("bzip2/lzma: library support for gzip, bzip2 and lzma decompression") Signed-off-by: Matt Turner Signed-off-by: Andrew Morton Cc: Alain Knaff Cc: "H. Peter Anvin" Cc: --- lib/decompress_bunzip2.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/lib/decompress_bunzip2.c b/lib/decompress_bunzip2.c index 1288f146661f1c..aaa75404250e2d 100644 --- a/lib/decompress_bunzip2.c +++ b/lib/decompress_bunzip2.c @@ -439,7 +439,7 @@ static int INIT get_next_block(struct bunzip_data *bd) array.) */ if (runPos) { runPos = 0; - if (dbufCount+t >= dbufSize) + if (dbufCount+t > dbufSize) return RETVAL_DATA_ERROR; uc = symToByte[mtfSymbol[0]]; From 7334defc90668e78d0f0e59265a0ce5aebc881cb Mon Sep 17 00:00:00 2001 From: Andrey Golovko Date: Sat, 12 Sep 2026 19:26:30 +0300 Subject: [PATCH 1290/1352] mailmap: update email address for Andrey Golovko Patches sent from andrey.golovko@gmail.com carry the authorship of my commits, but some of the tags I gave on other people's patches use andrey@golovko.me. Map the second address onto the first so the two do not look like two different contributors. Link: https://lore.kernel.org/20260912162630.7731-1-andrey.golovko@gmail.com Signed-off-by: Andrey Golovko Signed-off-by: Andrew Morton --- .mailmap | 1 + 1 file changed, 1 insertion(+) diff --git a/.mailmap b/.mailmap index b399a10744fc08..1a288972949802 100644 --- a/.mailmap +++ b/.mailmap @@ -92,6 +92,7 @@ Andrew Morton Andrew Murray Andrew Murray Andrew Vasquez +Andrey Golovko Andrey Konovalov Andrey Ryabinin Andrey Ryabinin From 18ba739a6e21753a9fb04212ff6cba032eecd579 Mon Sep 17 00:00:00 2001 From: James Kim Date: Mon, 14 Sep 2026 13:23:27 +0900 Subject: [PATCH 1291/1352] rapidio: mport_cdev: fix use-after-free in mport_mm_close() A use-after-free vulnerability was identified in mport_mm_close() in drivers/rapidio/devices/rio_mport_cdev.c, identical to the pattern previously addressed in dma_req_free(). This is observable from userspace when an application creates an mmap mapping via the RapidIO character device and subsequently unmaps it (or terminates, triggering exit_mmap()). During munmap, mport_mm_close() is invoked and drops the mapping reference via kref_put(). If kref_put() drops the last reference, mport_release_mapping() is called, which frees the underlying rio_mport_mapping structure. The subsequent mutex_unlock() then dereferences map->md to unlock buf_mutex, leading to a use-after-free: mport_mm_close() -> mutex_lock(&map->md->buf_mutex); ... -> kref_put(&map->ref, mport_release_mapping); /* map is freed */ -> mutex_unlock(&map->md->buf_mutex); /* UAF: map used */ Fix this by caching map->md before kref_put() and using the cached pointer for mutex unlocking, ensuring that freed memory is not accessed. Link: https://lore.kernel.org/20260914042327.49798-1-james010kim@gmail.com Fixes: e8de370188d0 ("rapidio: add mport char device driver") Signed-off-by: James Kim Signed-off-by: Andrew Morton Cc: Alexandre Bounine Cc: Dan Carpenter Cc: Greg Kroah-Hartman Cc: Matt Porter Cc: --- drivers/rapidio/devices/rio_mport_cdev.c | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/drivers/rapidio/devices/rio_mport_cdev.c b/drivers/rapidio/devices/rio_mport_cdev.c index ad82c2108a567d..47ba34b4afb296 100644 --- a/drivers/rapidio/devices/rio_mport_cdev.c +++ b/drivers/rapidio/devices/rio_mport_cdev.c @@ -2167,11 +2167,12 @@ static void mport_mm_open(struct vm_area_struct *vma) static void mport_mm_close(struct vm_area_struct *vma) { struct rio_mport_mapping *map = vma->vm_private_data; + struct mport_dev *md = map->md; rmcd_debug(MMAP, "%pad", &map->phys_addr); - mutex_lock(&map->md->buf_mutex); + mutex_lock(&md->buf_mutex); kref_put(&map->ref, mport_release_mapping); - mutex_unlock(&map->md->buf_mutex); + mutex_unlock(&md->buf_mutex); } static const struct vm_operations_struct vm_ops = { From 17b404e4101f3d097a128989e4922f7a35ceb632 Mon Sep 17 00:00:00 2001 From: Zhiling Zou Date: Sat, 12 Sep 2026 21:32:17 +0800 Subject: [PATCH 1292/1352] lib: validate in-memory LZ4 chunk length We found and validated an issue in lib/decompress_unlz4.c. The bug is reachable by a root user through kexec_file_load() with a crafted external initrd. unlz4() reads the compressed chunk length from an in-memory initrd and passes it to LZ4_decompress_safe(). It only checks the chunk length against the allocation size when the input is filled by a callback. Reject an in-memory chunk that extends past the remaining input before calling the LZ4 decoder. This prevents malformed initrds from making the decoder read past the mapped archive. Link: https://lore.kernel.org/59c6555c27aa7ba18ee227f9f02c78d15a37605f.1789219453.git.zhilinz@nebusec.ai Fixes: e76e1fdfa8f8 ("lib: add support for LZ4-compressed kernel") Signed-off-by: Zhiling Zou Signed-off-by: Andrew Morton Reported-by: VEGA Assisted-by: LLM Cc: Kyungsik Lee --- lib/decompress_unlz4.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/lib/decompress_unlz4.c b/lib/decompress_unlz4.c index c0dbb3cea915eb..86e9aaec04f6d3 100644 --- a/lib/decompress_unlz4.c +++ b/lib/decompress_unlz4.c @@ -139,6 +139,10 @@ STATIC inline int INIT unlz4(u8 *input, long in_len, if (!fill) { inp += 4; size -= 4; + if (chunksize > size) { + error("data corrupted"); + goto exit_2; + } } else { if (chunksize > LZ4_compressBound(uncomp_chunksize)) { error("chunk length is longer than allocated"); From 1e07e6fe28625bd9659f03724c03927575964ff6 Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Tue, 15 Sep 2026 18:06:23 +0000 Subject: [PATCH 1293/1352] taskstats: drop the unused CPU_DONT_CARE enum member CPU_DONT_CARE was added by the original listener cpumask code in 2006 (f9fd8914c1ac) and has never had a single user, not even in the commit that introduced it. It survived every refactor of the listener path since, including the recent handler folds. Kill it, twenty years is enough. Link: https://lore.kernel.org/20260915180623.22058-1-brads@mainlining.org Signed-off-by: Bradley Morgan Signed-off-by: Andrew Morton Reviewed-by: Balbir Singh --- kernel/taskstats.c | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/kernel/taskstats.c b/kernel/taskstats.c index 598d9cd8325018..3997dbb5aeadc1 100644 --- a/kernel/taskstats.c +++ b/kernel/taskstats.c @@ -59,8 +59,7 @@ static DEFINE_PER_CPU(struct listener_list, listener_array); enum actions { REGISTER, - DEREGISTER, - CPU_DONT_CARE + DEREGISTER }; static int prepare_reply(struct genl_info *info, u8 cmd, struct sk_buff **skbp, From 984a9286afcdb8dc8317604d34d71a2d198cb428 Mon Sep 17 00:00:00 2001 From: Jiaming Zhang Date: Mon, 14 Sep 2026 11:49:40 +0800 Subject: [PATCH 1294/1352] ocfs2: fix chunk number of the first chunk in a local quota file Mounting a crafted OCFS2 image can corrupt kernel memory. When the local quota file header of such an image claims zero chunks, the first quota entry allocated during the mount gets a chunk number taken from a kernel pointer, and releasing that entry clears a bit far outside the chunk bitmap. The write will corrupt the data stored at that address, and the corruption may lead to a system crash or damage unrelated data. Mounting requires CAP_SYS_ADMIN, so it takes an untrusted image, such as removable media or a loop mount of a file from elsewhere. Since the chunk number comes from a kernel pointer, the address of the bit that gets cleared differs from boot to boot, and the issue has been reported under several titles: - BUG: unable to handle kernel paging request in ocfs2_local_release_dquot - KASAN: use-after-free Write in ocfs2_local_release_dquot - KASAN: slab-out-of-bounds Write in ocfs2_local_release_dquot - KASAN: slab-use-after-free Write in ocfs2_local_release_dquot - KFENCE: use-after-free write in ocfs2_local_release_dquot Note that none of these is a use-after-free in the quota code: the chunk and its buffer head are alive. The bit that gets cleared lies far outside the bitmap. KASAN and KFENCE name each report based on the object occupying that address, which explains why the same issue is reported under so many different titles. In the report I sent, the address fell in a free page, which KASAN labels use-after-free. There are known reproducers. Both a syzkaller and a C reproducer are available from the Google Drive link [1] in my report thread. The issue was found by our modified syzkaller, not a theoretical thing found by an LLM. I have not seen it outside fuzzing, so it is not affecting me in the real world. The local quota file in OCFS2 is divided into chunks, and each chunk begins with a header block holding a bitmap of the quota entries that chunk has handed out. Chunks are numbered from zero, and that number is used to convert the file offset of an entry back into a bit position in the bitmap. ocfs2_local_quota_add_chunk() appends a new chunk to the in-memory list and numbers it one past the chunk that was last: list_add_tail(&chunk->qc_chunk, &oinfo->dqi_chunk); chunk->qc_num = list_entry(chunk->qc_chunk.prev, struct ocfs2_quota_chunk, qc_chunk)->qc_num + 1; The predecessor is looked up after the new chunk is added to the list, so if the list was empty, the prev pointer is the list head itself. The head is the dqi_chunk member of struct ocfs2_mem_dqinfo and is not a chunk, so reading qc_num through it lands 16 bytes past the start of the head, on the dqi_gqinode pointer that follows it. The first chunk of the file is then numbered with the lower half of a kernel pointer instead of 0. The list is empty when the local quota file header claims the file has no chunks. ocfs2_local_read_info() takes dqi_chunks from that header without validating it, so an image with dqi_chunks == 0 takes this path when the first quota entry is allocated. ocfs2_create_local_dquot() turns the bad number into a file offset with ol_dqblk_off(), which shifts a 32-bit block number left by the block size bits, so the top bits of such a large block number are lost. ocfs2_local_release_dquot() turns the offset back into a bit index with ol_dqblk_chunk_off(), using the full chunk number, so the lost bits push that index far outside the bitmap, and clearing it corrupts unrelated memory. Compute the chunk number before putting the chunk on the list, and use 0 when the list is empty. Link: https://lore.kernel.org/20260914034940.4070970-1-r772577952@gmail.com Link: https://drive.google.com/drive/folders/1-LzPTgOALEc3eOjmgM6oOfRKkYXJ-bnO?usp=drive_link [1] Fixes: 9e33d69f553a ("ocfs2: Implementation of local and global quota file handling") Signed-off-by: Jiaming Zhang Signed-off-by: Andrew Morton Closes: https://lore.kernel.org/lkml/CANypQFZ05tpth0Xc33gmP6jgPnkY4VuezyHcsajV-SqCsmN_gg@mail.gmail.com/ Reviewed-by: Joseph Qi Assisted-by: Claude Code:claude-opus-5 Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi Cc: Changwei Ge Cc: Jun Piao Cc: Heming Zhao Cc: --- fs/ocfs2/quota_local.c | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/fs/ocfs2/quota_local.c b/fs/ocfs2/quota_local.c index f55810c59b1b11..d351cda9211f85 100644 --- a/fs/ocfs2/quota_local.c +++ b/fs/ocfs2/quota_local.c @@ -1071,10 +1071,13 @@ static struct ocfs2_quota_chunk *ocfs2_local_quota_add_chunk( goto out; } + if (list_empty(&oinfo->dqi_chunk)) + chunk->qc_num = 0; + else + chunk->qc_num = list_entry(oinfo->dqi_chunk.prev, + struct ocfs2_quota_chunk, + qc_chunk)->qc_num + 1; list_add_tail(&chunk->qc_chunk, &oinfo->dqi_chunk); - chunk->qc_num = list_entry(chunk->qc_chunk.prev, - struct ocfs2_quota_chunk, - qc_chunk)->qc_num + 1; chunk->qc_headerbh = bh; *offset = 0; return chunk; From d784b07b2fac3acf8cbe967d1a9bdc8c9893ceb0 Mon Sep 17 00:00:00 2001 From: Andy Shevchenko Date: Tue, 15 Sep 2026 10:53:53 +0200 Subject: [PATCH 1295/1352] resource: replace open coded resource_overlaps() In iomem_map_sanity_check() a piece of code resembles the content of the resource_overlaps(). Replace open coded piece with the call to the existing helper. Link: https://lore.kernel.org/20260915085353.3518153-1-andriy.shevchenko@linux.intel.com Signed-off-by: Andy Shevchenko Signed-off-by: Andrew Morton Reviewed-by: Bradley Morgan Cc: Bjorn Helgaas --- kernel/resource.c | 14 ++++++-------- 1 file changed, 6 insertions(+), 8 deletions(-) diff --git a/kernel/resource.c b/kernel/resource.c index 54d7199695fbe9..cfc1a00e86aa03 100644 --- a/kernel/resource.c +++ b/kernel/resource.c @@ -1833,7 +1833,7 @@ __setup("reserve=", reserve_setup); */ int iomem_map_sanity_check(resource_size_t addr, unsigned long size) { - resource_size_t end = addr + size - 1; + struct resource mem = DEFINE_RES_MEM(addr, size); struct resource *p; int err = 0; @@ -1843,12 +1843,10 @@ int iomem_map_sanity_check(resource_size_t addr, unsigned long size) * We can probably skip the resources without * IORESOURCE_IO attribute? */ - if (p->start > end) + if (!resource_overlaps(p, &mem)) continue; - if (p->end < addr) - continue; - if (PFN_DOWN(p->start) <= PFN_DOWN(addr) && - PFN_DOWN(p->end) >= PFN_DOWN(end)) + if (PFN_DOWN(p->start) <= PFN_DOWN(mem.start) && + PFN_DOWN(p->end) >= PFN_DOWN(mem.end)) continue; /* * if a resource is "BUSY", it's not a hardware resource @@ -1859,8 +1857,8 @@ int iomem_map_sanity_check(resource_size_t addr, unsigned long size) if (p->flags & IORESOURCE_BUSY) continue; - pr_debug("resource sanity check: requesting [mem %pa-%pa], which spans more than %s %pR\n", - &addr, &end, p->name, p); + pr_debug("resource sanity check: requesting %pR, which spans more than %s %pR\n", + &mem, p->name, p); err = -1; break; } From ce67db232a955e027f0fe8316b014679e5924fc2 Mon Sep 17 00:00:00 2001 From: Andy Shevchenko Date: Tue, 15 Sep 2026 10:42:45 +0200 Subject: [PATCH 1296/1352] CREDITS: fix the ordering and other issues I assume that the ordering should be done in default (C) locale as other gives more lines shuffled. Fix the ordering accordingly. With that being said, clearly make at the top header how it should be sorted. Note, I took some empirical assumptions that people usually put their names in accordance with the First name(s) Last Name(s) split based on the records around (before and after the move). I hope I made no or little amount of mistakes. In any case it's just a dozen of names to update. Link: https://lore.kernel.org/20260915084248.3460260-1-andriy.shevchenko@linux.intel.com Signed-off-by: Andy Shevchenko Signed-off-by: Andrew Morton Cc: AngeloGiaocchino Del Regno Cc: Mathias Brugger --- CREDITS | 181 ++++++++++++++++++++++++++++---------------------------- 1 file changed, 91 insertions(+), 90 deletions(-) diff --git a/CREDITS b/CREDITS index a0bd941ca5dd26..3ae05ff1a0921b 100644 --- a/CREDITS +++ b/CREDITS @@ -1,6 +1,7 @@ This is at least a partial credits-file of people that have - contributed to the Linux project. It is sorted by name and - formatted to allow easy grepping and beautification by + contributed to the Linux project. It is sorted by family name, + assuming the used locale is default, e.g. by setting LC_ALL=C, + and formatted to allow easy grepping and beautification by scripts. The fields are: name (N), email (E), web-address (W), PGP key ID and fingerprint (P), description (D), and snail-mail address (S). @@ -344,6 +345,14 @@ D: Hardware spinlock (hwspinlock) subsystem D: OMAP hwspinlock driver D: OMAP remoteproc driver +N: Muli Ben-Yehuda +E: mulix@mulix.org +E: muli@il.ibm.com +W: http://www.mulix.org +D: trident OSS sound driver, x86-64 dma-ops and Calgary IOMMU, +D: KVM and Xen bits and other misc. hackery. +S: Haifa, Israel + N: Krzysztof Benedyczak E: golbi@mat.uni.torun.pl W: http://www.mat.uni.torun.pl/~golbi @@ -362,14 +371,6 @@ S: 2322 37th Ave SW S: Seattle, Washington 98126-2010 S: USA -N: Muli Ben-Yehuda -E: mulix@mulix.org -E: muli@il.ibm.com -W: http://www.mulix.org -D: trident OSS sound driver, x86-64 dma-ops and Calgary IOMMU, -D: KVM and Xen bits and other misc. hackery. -S: Haifa, Israel - N: Johannes Berg E: johannes@sipsolutions.net W: https://johannes.sipsolutions.net/ @@ -393,14 +394,14 @@ S: 1549 Hiironen Rd. S: Brimson, MN 55602 S: USA -N: Arnd Bergmann -D: Maintainer of Cell Broadband Engine Architecture - N: Hennus Bergman P: 1024/77D50909 76 99 FD 31 91 E1 96 1C 90 BB 22 80 62 F6 BD 63 D: Author and maintainer of the QIC-02 tape driver S: The Netherlands +N: Arnd Bergmann +D: Maintainer of Cell Broadband Engine Architecture + N: Tomas Berndtsson E: tomas@nocrew.org W: http://tomas.nocrew.org/ @@ -492,10 +493,6 @@ D: Various fixes (mostly networking) S: Montreal, Quebec S: Canada -N: Zoltán Böszörményi -E: zboszor@mail.externet.hu -D: MTRR emulation with Cyrix style ARR registers, Athlon MTRR support - N: John Boyd E: boyd@cis.ohio-state.edu D: Co-author of wd7000 SCSI driver @@ -589,9 +586,6 @@ N: Zach Brown E: zab@zabbo.net D: maestro pci sound -N: Zefan Li -D: Contribution to control group stuff - N: David Brownell D: Kernel engineer, mentor, and friend. Maintained USB EHCI and D: gadget layers, SPI subsystem, GPIO subsystem, and more than a few @@ -632,6 +626,10 @@ S: Ravenhorst 58 S: 2317 AK Leiden S: The Netherlands +N: Zoltán Böszörményi +E: zboszor@mail.externet.hu +D: MTRR emulation with Cyrix style ARR registers, Athlon MTRR support + N: Michael Callahan E: callahan@maths.ox.ac.uk D: PPP for Linux @@ -703,6 +701,10 @@ S: Tamsui town, Taipei county, S: Taiwan 251 S: Republic of China +N: Landen Chao +E: Landen.Chao@mediatek.com +D: MT7531 Ethernet switch support + N: Michael Elizabeth Chastain E: mec@shout.net D: Configure, Menuconfig, xconfig @@ -715,10 +717,6 @@ D: Media subsystem (V4L/DVB) drivers and core D: EDAC drivers and EDAC 3.0 core rework S: Brazil -N: Landen Chao -E: Landen.Chao@mediatek.com -D: MT7531 Ethernet switch support - N: Raymond Chen E: raymondc@microsoft.com D: Author of Configure script @@ -1447,6 +1445,10 @@ D: Made support for modules, ramdisk, generic-serial, etc. optional. D: Transformed old user space bdflush into 1st kernel thread - kflushd. D: Many other patches, documentation files, mini kernels, utilities, ... +N: Andy Gospodarek +E: andy@greyhouse.net +D: Maintenance and contributions to the network interface bonding driver. + N: Masanori GOTO E: gotom@debian.or.jp D: Workbit NinjaSCSI-32Bi/UDE driver @@ -1459,10 +1461,6 @@ S: 8124 Constitution Apt. 7 S: Sterling Heights, Michigan 48313 S: USA -N: Andy Gospodarek -E: andy@greyhouse.net -D: Maintenance and contributions to the network interface bonding driver. - N: Vivek Goyal E: vgoyal@redhat.com D: KDUMP, KEXEC, and VIRTIO FILE SYSTEM @@ -1535,6 +1533,14 @@ S: 44 St. Joseph Street, Suite 506 S: Toronto, Ontario, M4Y 2W4 S: Canada +N: Nitin Gupta +E: ngupta@vflare.org +D: zsmalloc memory allocator and zram block device driver + +N: Justin Guyett +E: jguyett@andrew.cmu.edu +D: via-rhine net driver hacking + N: Richard Günther E: rguenth@tat.physik.uni-tuebingen.de W: http://www.tat.physik.uni-tuebingen.de/~rguenth @@ -1543,14 +1549,6 @@ D: binfmt_misc S: 72074 Tübingen S: Germany -N: Justin Guyett -E: jguyett@andrew.cmu.edu -D: via-rhine net driver hacking - -N: Nitin Gupta -E: ngupta@vflare.org -D: zsmalloc memory allocator and zram block device driver - N: Danny ter Haar E: dth@cistron.nl D: /proc/cpuinfo, reboot on panic , kernel pre-patch tester ;) @@ -1945,16 +1943,6 @@ E: sjenning@redhat.com D: Creation and maintenance of zswap D: Creation and maintenace of the zbud allocator -N: Jeremy Kerr -D: Maintainer of SPU File System - -N: Michael Kerrisk -E: mtk.manpages@gmail.com -W: https://man7.org/ -P: 4096R/3A35CE5E E522 595B 52ED A4E6 BFCC CB5E 8561 9911 3A35 CE5E -D: Maintainer of the Linux man-pages project -D: Linux man pages online, at - N: Niels Kristian Bech Jensen E: nkbj1970@hotmail.com D: Miscellaneous kernel updates and fixes. @@ -2099,6 +2087,16 @@ S: Keplerstr. 6 S: 4050 Traun S: Austria +N: Jeremy Kerr +D: Maintainer of SPU File System + +N: Michael Kerrisk +E: mtk.manpages@gmail.com +W: https://man7.org/ +P: 4096R/3A35CE5E E522 595B 52ED A4E6 BFCC CB5E 8561 9911 3A35 CE5E +D: Maintainer of the Linux man-pages project +D: Linux man pages online, at + N: Karl Keyte E: karl@koft.com D: Disk usage statistics and modifications to line printer driver @@ -2483,6 +2481,9 @@ D: and much more. He was the maintainer of MD from 2016 to 2018. Shaohua D: passed away late 2018, he will be greatly missed. W: https://www.spinics.net/lists/raid/msg61993.html +N: Zefan Li +D: Contribution to control group stuff + N: Stephan Linz E: linz@mazet.de E: Stephan.Linz@gmx.de @@ -2542,11 +2543,6 @@ D: misc. kernel hacking and debugging S: Cambridge, MA 02139 S: USA -N: Martin von Löwis -E: loewis@informatik.hu-berlin.de -D: script binary format -D: NTFS driver - N: H.J. Lu E: hjl@gnu.ai.mit.edu D: GCC + libraries hacker @@ -2575,6 +2571,11 @@ S: Puistokaari 1 E 18 S: 00200 Helsinki S: Finland +N: Martin von Löwis +E: loewis@informatik.hu-berlin.de +D: script binary format +D: NTFS driver + N: Daniel J. Maas E: dmaas@dcine.com W: https://www.maasdigital.com @@ -2625,10 +2626,6 @@ S: PO BOX 220, HFX. CENTRAL S: Halifax, Nova Scotia S: Canada B3J 3C8 -N: Kai Mäkisara -E: Kai.Makisara@kolumbus.fi -D: SCSI Tape Driver - N: Asit Mallick E: asit.k.mallick@intel.com D: Linux/IA-64 @@ -2912,13 +2909,6 @@ S: 12725 SW Millikan Way, Suite 400 S: Beaverton, Oregon 97005 S: USA -N: Eberhard Mönkeberg -E: emoenke@gwdg.de -D: CDROM driver "sbpcd" (Matsushita/Panasonic/Soundblaster) -S: Ruhstrathöhe 2 b. -S: D-37085 Göttingen -S: Germany - N: Thomas Molina E: tmolina@cablespeed.com D: bug fixes, documentation, minor hackery @@ -3005,6 +2995,17 @@ S: Dragonvagen 1 A 13 S: FIN-00330 Helsingfors S: Finland +N: Kai Mäkisara +E: Kai.Makisara@kolumbus.fi +D: SCSI Tape Driver + +N: Eberhard Mönkeberg +E: emoenke@gwdg.de +D: CDROM driver "sbpcd" (Matsushita/Panasonic/Soundblaster) +S: Ruhstrathöhe 2 b. +S: D-37085 Göttingen +S: Germany + N: Matija Nalis E: mnalis@jagor.srce.hr E: mnalis@voyager.hr @@ -3200,13 +3201,6 @@ S: RR #5, 497 Pole Line Road S: Thunder Bay, Ontario S: CANADA P7C 5M9 -N: Inaky Perez-Gonzalez -E: inaky.perez-gonzalez@intel.com -E: linux-wimax@intel.com -E: inakypg@yahoo.com -D: WiMAX stack -D: Intel Wireless WiMAX Connection 2400 driver - N: Yuri Per E: yuri@pts.mipt.ru D: Some smbfs fixes @@ -3214,6 +3208,13 @@ S: Demonstratsii 8-382 S: Tula 300000 S: Russia +N: Inaky Perez-Gonzalez +E: inaky.perez-gonzalez@intel.com +E: linux-wimax@intel.com +E: inakypg@yahoo.com +D: WiMAX stack +D: Intel Wireless WiMAX Connection 2400 driver + N: Thomas Petazzoni E: thomas.petazzoni@bootlin.com D: Driver for the Marvell Armada 370/XP network unit. @@ -3328,14 +3329,14 @@ N: Frederic Potter E: fpotter@cirpack.com D: Some PCI kernel support -N: Rui Prior -E: rprior@inescn.pt -D: ATM device driver for NICStAR based cards - N: Roopa Prabhu E: roopa@nvidia.com D: Bridge co-maintainer, vxlan and networking contributor +N: Rui Prior +E: rprior@inescn.pt +D: ATM device driver for NICStAR based cards + N: Stefan Probst E: sp@caldera.de D: The Linux Support Team Erlangen, 1993-97 @@ -4023,12 +4024,6 @@ S: C/ Federico Garcia Lorca 1 10-A S: Sevilla 41005 S: Spain -N: Björn Töpel -E: bjorn@kernel.org -D: AF_XDP -S: Gothenburg -S: Sweden - N: Linus Torvalds E: torvalds@linux-foundation.org D: Original kernel hacker @@ -4082,17 +4077,6 @@ D: Linux-Workshop Köln (aka LUG Cologne, Germany), Installfests S: Tacitusstr. 6 S: D-50968 Köln -N: Tsu-Sheng Tsao -E: tsusheng@scf.usc.edu -D: IGMP (Internet Group Management Protocol) version 2 -S: 2F 14 ALY 31 LN 166 SEC 1 SHIH-PEI RD -S: Taipei -S: Taiwan 112 -S: Republic of China -S: 24335 Delta Drive -S: Diamond Bar, California 91765 -S: USA - N: Theodore Ts'o E: tytso@mit.edu D: Random Linux hacker @@ -4109,6 +4093,17 @@ S: 1 Amherst Street S: Cambridge, Massachusetts 02139 S: USA +N: Tsu-Sheng Tsao +E: tsusheng@scf.usc.edu +D: IGMP (Internet Group Management Protocol) version 2 +S: 2F 14 ALY 31 LN 166 SEC 1 SHIH-PEI RD +S: Taipei +S: Taiwan 112 +S: Republic of China +S: 24335 Delta Drive +S: Diamond Bar, California 91765 +S: USA + N: Luben Tuikov E: Luben Tuikov D: Maintainer of the DRM GPU Scheduler @@ -4135,6 +4130,12 @@ S: 44 Campbell Park Crescent S: Edinburgh EH13 0HT S: United Kingdom +N: Björn Töpel +E: bjorn@kernel.org +D: AF_XDP +S: Gothenburg +S: Sweden + N: Thomas Uhl E: uhl@sun1.rz.fh-heilbronn.de D: Application programmer From 2ab8423bab1f92adc8e5bc01f6531593d769ec81 Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Wed, 16 Sep 2026 18:29:52 +0000 Subject: [PATCH 1297/1352] panic: fix redirect CPU race in panic_try_force_cpu() Patch series "panic: fix panic_force_cpu= redirect races and NMI bypass", v7. The panic_force_cpu= parameter redirects a panic to a specific CPU so the crash kernel runs there. The redirect code in panic_try_force_cpu() had two races and an NMI bypass, all found by Sashiko. This series closes them and kills one more hack that the rework turned up. Panic from NMI still goes through the redirect path first, so the crash kernel ends up on the CPU the admin asked for. The redirect buffer is a static 1KB now, no initcall, no fallback message. This patch (of 6): The cmpxchg() in panic_try_force_cpu() makes sure that only one CPU tries to redirect panic() to the requested CPU. It is similar to the cmpxchg() in panic_try_start() which makes sure that only one CPU does the panic(). In both situations, only the winner of cmpxchg() should proceed further. Other CPUs should go offline. There is a bug because the cmpxchg loser returns false and falls through into vpanic(). Two non-target CPUs A and B panic, the requested CPU is C: cpu A cpu B ---------- ---------- panic() panic() vpanic() vpanic() panic_try_force_cpu() panic_try_force_cpu() cmpxchg wins cmpxchg fails redirect = A old_cpu = A IPI -> C return false <- BUG return true panic_try_start() wins panic_smp_self_stop() __crash_kexec() on B (A stops) (target C bypassed) The loser must stop, not fall through. It cannot just return true, though. A CPU that already won the redirect cmpxchg can reenter panic_try_force_cpu() on the same CPU, for example a nested NMI during the message formatting, before the IPI is sent: cpu A (1st) cpu A (nested) ---------- ---------- panic() vpanic() panic_try_force_cpu() cmpxchg wins (redirect = A) vsnprintf(msg) ... <-- NMI, nested panic --> panic() vpanic() panic_try_force_cpu() cmpxchg fails old_cpu == A (this CPU) return true <- would halt panic_smp_self_stop() (IPI never sent, panic abandoned) Check old_cpu against this_cpu so a second call from the same CPU returns false and falls through to panic_try_start() instead. Also fix the panic_in_progress() check. We must not redirect when panic_cpu is already assigned. Return true to stop when the panic is on another CPU, false to proceed when it is this one. Update the panic_try_force_cpu() doc comment for the new return value semantics. Link: https://lore.kernel.org/20260916182957.7788-1-brads@mainlining.org Link: https://lore.kernel.org/20260916182957.7788-2-brads@mainlining.org Fixes: 2e171ab29f91 ("panic: add panic_force_cpu= parameter to redirect panic to a specific CPU") Signed-off-by: Bradley Morgan Signed-off-by: Andrew Morton Reported-by: Sashiko Closes: https://sashiko.dev/#/patchset/20260705164123.18746-1-include@grrlz.net Closes: https://sashiko.dev/#/patchset/20260707172252.4842-1-include@grrlz.net Reviewed-by: Petr Mladek Cc: Guenetr Roeck Cc: Wang Jinchao Cc: Wim Van Sebroeck Cc: --- kernel/panic.c | 19 ++++++++++++------- 1 file changed, 12 insertions(+), 7 deletions(-) diff --git a/kernel/panic.c b/kernel/panic.c index 50715f14cf04ef..08072bfae42219 100644 --- a/kernel/panic.c +++ b/kernel/panic.c @@ -371,8 +371,9 @@ int __weak panic_smp_redirect_cpu(int target_cpu, void *msg) * for the crash kernel to function correctly. This function redirects * panic handling to the CPU specified via the panic_force_cpu= boot parameter. * - * Returns false if panic should proceed on current CPU. - * Returns true if panic was redirected. + * Returns true when this CPU must stop: the panic was redirected or is + * already running on another CPU. + * Returns false when panic() should proceed on this CPU. */ __printf(1, 0) static bool panic_try_force_cpu(const char *fmt, va_list args) @@ -396,16 +397,20 @@ static bool panic_try_force_cpu(const char *fmt, va_list args) return false; } - /* Another panic already in progress */ + /* + * Don't redirect when a panic is already in progress. Stop this + * CPU when it's another one, proceed when it's this one. + */ if (panic_in_progress()) - return false; + return panic_on_other_cpu(); /* - * Only one CPU can do the redirect. Use atomic cmpxchg to ensure - * we don't race with another CPU also trying to redirect. + * Only one CPU can do the redirection. Others should go offline. + * Continue with panic() when we already tried the redirection + * from this CPU before, for example via nmi_panic(). */ if (!atomic_try_cmpxchg(&panic_redirect_cpu, &old_cpu, this_cpu)) - return false; + return old_cpu != this_cpu; /* * Use dynamically allocated buffer if available, otherwise From 935878ee312115cf11da866c51879f0a0b5d6d3a Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Wed, 16 Sep 2026 18:29:53 +0000 Subject: [PATCH 1298/1352] panic: flatten nmi_panic control flow panic() is __noreturn, so the else after panic_try_start() is dead. Drop it so the force_cpu path can be added cleanly on top. Link: https://lore.kernel.org/20260916182957.7788-3-brads@mainlining.org Signed-off-by: Bradley Morgan Signed-off-by: Andrew Morton Reviewed-by: Petr Mladek Cc: Guenetr Roeck Cc: Sashiko Cc: Wang Jinchao Cc: Wim Van Sebroeck Cc: --- kernel/panic.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/kernel/panic.c b/kernel/panic.c index 08072bfae42219..5646d4fb82b7d4 100644 --- a/kernel/panic.c +++ b/kernel/panic.c @@ -517,7 +517,8 @@ void nmi_panic(struct pt_regs *regs, const char *msg) { if (panic_try_start()) panic("%s", msg); - else if (panic_on_other_cpu()) + + if (panic_on_other_cpu()) nmi_panic_self_stop(regs); } EXPORT_SYMBOL(nmi_panic); From 4e3b25d28479e89f089204aa71b8e7e24f2395e4 Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Wed, 16 Sep 2026 18:29:54 +0000 Subject: [PATCH 1299/1352] panic: fix va_list reuse in panic_try_force_cpu() vsnprintf() consumes the caller's va_list. When the redirect fails, vpanic() reuses it for the panic message, which is undefined behavior. Use va_copy(). Link: https://lore.kernel.org/20260916182957.7788-4-brads@mainlining.org Fixes: 2e171ab29f91 ("panic: add panic_force_cpu= parameter to redirect panic to a specific CPU") Signed-off-by: Bradley Morgan Signed-off-by: Andrew Morton Reviewed-by: Petr Mladek Cc: Guenetr Roeck Cc: Sashiko Cc: Wang Jinchao Cc: Wim Van Sebroeck Cc: --- kernel/panic.c | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/kernel/panic.c b/kernel/panic.c index 5646d4fb82b7d4..7388eb81a1c471 100644 --- a/kernel/panic.c +++ b/kernel/panic.c @@ -417,7 +417,12 @@ static bool panic_try_force_cpu(const char *fmt, va_list args) * fall back to static message for early boot panics or allocation failure. */ if (panic_force_buf) { - vsnprintf(panic_force_buf, PANIC_MSG_BUFSZ, fmt, args); + va_list ap; + + /* Do not consume args, the caller reuses it if we fail */ + va_copy(ap, args); + vsnprintf(panic_force_buf, PANIC_MSG_BUFSZ, fmt, ap); + va_end(ap); msg = panic_force_buf; } else { msg = "Redirected panic (buffer unavailable)"; From a41bd4512daa10f9b682efde6743daeeb68e30d1 Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Wed, 16 Sep 2026 18:29:55 +0000 Subject: [PATCH 1300/1352] panic: restore variable arguments to nmi_panic() nmi_panic() used to accept variable arguments until commit ebc41f20d77f ("panic: change nmi_panic from macro to function") flattened it to a final message string. vpanic() did not exist back then, so the function had to format through panic("%s", msg). Bring the variable arguments back and format with vpanic() directly. The next patch makes nmi_panic() try the panic_force_cpu= redirect before claiming panic_cpu. panic_try_force_cpu() needs to print the message but it is used also by panic() which accepts variable argument list. The string could be formatted only when a CPU gets assigned to process the redirection. Otherwise, there might be a race when writing to the `panic_force_buf`. No current caller passes a string with format specifiers. The closest one is hpwdt_pretimeout(), which builds panic_msg with hex_byte_pack() and has only two variants, both plain strings. But the new __printf() annotation on nmi_panic() would warn with -Wformat-security there because the buffer is passed directly as the format argument, so switch it to nmi_panic(regs, "%s", panic_msg). [pmladek@suse.com: changelog fix] Link: https://lore.kernel.org/arJxFybmtD7OBIcL@pathway.suse.cz Link: https://lore.kernel.org/20260916182957.7788-5-brads@mainlining.org Signed-off-by: Bradley Morgan Signed-off-by: Andrew Morton Suggested-by: Petr Mladek Reviewed-by: Petr Mladek Cc: Guenetr Roeck Cc: Sashiko Cc: Wang Jinchao Cc: Wim Van Sebroeck Cc: --- drivers/watchdog/hpwdt.c | 2 +- include/linux/panic.h | 3 ++- kernel/panic.c | 10 ++++++++-- 3 files changed, 11 insertions(+), 4 deletions(-) diff --git a/drivers/watchdog/hpwdt.c b/drivers/watchdog/hpwdt.c index 8af1fad2de0bd3..78227d200afe4b 100644 --- a/drivers/watchdog/hpwdt.c +++ b/drivers/watchdog/hpwdt.c @@ -199,7 +199,7 @@ static int hpwdt_pretimeout(unsigned int ulReason, struct pt_regs *regs) } hex_byte_pack(panic_msg, nmistat); - nmi_panic(regs, panic_msg); + nmi_panic(regs, "%s", panic_msg); return NMI_HANDLED; } diff --git a/include/linux/panic.h b/include/linux/panic.h index 98dd7dfd27de7a..17e61b61c45f8a 100644 --- a/include/linux/panic.h +++ b/include/linux/panic.h @@ -13,7 +13,8 @@ __printf(1, 2) void panic(const char *fmt, ...) __noreturn __cold; __printf(1, 0) void vpanic(const char *fmt, va_list args) __noreturn __cold; -void nmi_panic(struct pt_regs *regs, const char *msg); +__printf(2, 3) +void nmi_panic(struct pt_regs *regs, const char *fmt, ...); void check_panic_on_warn(const char *origin); extern void oops_enter(void); extern void oops_exit(void); diff --git a/kernel/panic.c b/kernel/panic.c index 7388eb81a1c471..e240ca06faab21 100644 --- a/kernel/panic.c +++ b/kernel/panic.c @@ -518,13 +518,19 @@ EXPORT_SYMBOL(panic_on_other_cpu); * nmi_panic_self_stop() which can provide architecture dependent code such * as saving register state for crash dump. */ -void nmi_panic(struct pt_regs *regs, const char *msg) +void nmi_panic(struct pt_regs *regs, const char *fmt, ...) { + va_list args; + + va_start(args, fmt); + if (panic_try_start()) - panic("%s", msg); + vpanic(fmt, args); if (panic_on_other_cpu()) nmi_panic_self_stop(regs); + + va_end(args); } EXPORT_SYMBOL(nmi_panic); From abab9edbcd3be278b143d87475be731fb80e9d8b Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Wed, 16 Sep 2026 18:29:56 +0000 Subject: [PATCH 1301/1352] panic: allow force_cpu redirect from an NMI nmi_panic() claims panic_cpu via panic_try_start() before calling panic(). When the panic later reaches panic_try_force_cpu(), the panic_in_progress() check sees panic_cpu set and refuses to redirect. The crash kernel runs on the CPU that took the NMI instead of the CPU requested with panic_force_cpu=: nmi_panic() panic_try_start() wins, panic_cpu = X panic("%s", msg) vpanic() panic_try_force_cpu() panic_in_progress() true, panic_cpu is X return false redirect bypassed panic_try_start() already won __crash_kexec() on X, not the requested CPU Try the redirect before claiming panic_cpu instead, as suggested by Petr Mladek. nmi_panic() now calls panic_try_force_cpu() first and claims panic_cpu only when no redirect happened. The requested CPU claims panic_cpu itself when it runs panic(), so panic_cpu does not need to be handed off. panic_try_force_cpu() copies the arguments before formatting (patch 3), so nmi_panic() can pass them to vpanic() again when no redirect happens. The redirect IPI is sent with smp_call_function_single_async(), which is not guaranteed to work from NMI context. Treat it as best effort. It is worth the risk because the redirection is only used when the crash kernel would not work on the panicking CPU anyway. Keep returning when the panic is already running on this CPU. A nested NMI, for example with unknown_nmi_panic while this CPU is inside panic(), must return and let the interrupted panic() continue instead of parking the CPU in nmi_panic_self_stop(). Mark the redirecting CPU offline before stopping it, like vpanic() does, so that panic_other_cpus_shutdown() on the target CPU does not wait for it. Link: https://lore.kernel.org/20260916182957.7788-6-brads@mainlining.org Fixes: 2e171ab29f91 ("panic: add panic_force_cpu= parameter to redirect panic to a specific CPU") Signed-off-by: Bradley Morgan Signed-off-by: Andrew Morton Reported-by: Sashiko Closes: https://sashiko.dev/#/patchset/20260708164312.19044-1-include@grrlz.net Reviewed-by: Petr Mladek Cc: Guenetr Roeck Cc: Wang Jinchao Cc: Wim Van Sebroeck Cc: --- kernel/panic.c | 19 +++++++++++++++---- 1 file changed, 15 insertions(+), 4 deletions(-) diff --git a/kernel/panic.c b/kernel/panic.c index e240ca06faab21..29c981926f5f34 100644 --- a/kernel/panic.c +++ b/kernel/panic.c @@ -513,10 +513,11 @@ bool panic_on_other_cpu(void) EXPORT_SYMBOL(panic_on_other_cpu); /* - * A variant of panic() called from NMI context. We return if we've already - * panicked on this CPU. If another CPU already panicked, loop in - * nmi_panic_self_stop() which can provide architecture dependent code such - * as saving register state for crash dump. + * A variant of panic() called from NMI context. The panic is first + * redirected to the CPU requested via panic_force_cpu=, when configured. + * We return if we've already panicked on this CPU. If another CPU already + * panicked, loop in nmi_panic_self_stop() which can provide architecture + * dependent code for saving register state for crash dump. */ void nmi_panic(struct pt_regs *regs, const char *fmt, ...) { @@ -524,6 +525,16 @@ void nmi_panic(struct pt_regs *regs, const char *fmt, ...) va_start(args, fmt); + /* Try to redirect to the requested CPU before claiming panic_cpu. */ + if (panic_try_force_cpu(fmt, args)) { + /* + * Mark ourselves offline so panic_other_cpus_shutdown() won't + * wait for us on architectures that check num_online_cpus(). + */ + set_cpu_online(raw_smp_processor_id(), false); + nmi_panic_self_stop(regs); + } + if (panic_try_start()) vpanic(fmt, args); From 8564b7589c43f62d7b74f1686b13e9ba3bf6d482 Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Wed, 16 Sep 2026 18:29:57 +0000 Subject: [PATCH 1302/1352] panic: kill the "buffer unavailable" redirect fallback The redirect buffer is a disgusting terrible hack. panic_force_buf is kmalloc'ed in a late_initcall, and until then the redirect delivers this as the panic message: Redirected panic (buffer unavailable) The whole point of the redirect is to hand the panic message to the target CPU, so the crash kernel boots knowing it panicked and not why. And that window is the entire boot, from the early_param to the late_initcall, which is exactly when you most want the message. Make it a static 1KB buffer and kill the initcall. The cost is 1KB of .bss in SMP crash dump builds, and it is only ever touched when panic_force_cpu= is set anyway. The local msg variable is gone too, the buffer goes directly to panic_smp_redirect_cpu(). Link: https://lore.kernel.org/20260916182957.7788-7-brads@mainlining.org Signed-off-by: Bradley Morgan Signed-off-by: Andrew Morton Suggested-by: Petr Mladek Reviewed-by: Petr Mladek Cc: Guenetr Roeck Cc: Sashiko Cc: Wang Jinchao Cc: Wim Van Sebroeck Cc: --- kernel/panic.c | 34 +++++++--------------------------- 1 file changed, 7 insertions(+), 27 deletions(-) diff --git a/kernel/panic.c b/kernel/panic.c index 29c981926f5f34..170744163fc2a8 100644 --- a/kernel/panic.c +++ b/kernel/panic.c @@ -307,7 +307,7 @@ atomic_t panic_cpu = ATOMIC_INIT(PANIC_CPU_INVALID); atomic_t panic_redirect_cpu = ATOMIC_INIT(PANIC_CPU_INVALID); #if defined(CONFIG_SMP) && defined(CONFIG_CRASH_DUMP) -static char *panic_force_buf; +static char panic_force_buf[PANIC_MSG_BUFSZ]; static int __init panic_force_cpu_setup(char *str) { @@ -326,17 +326,6 @@ static int __init panic_force_cpu_setup(char *str) } early_param("panic_force_cpu", panic_force_cpu_setup); -static int __init panic_force_cpu_late_init(void) -{ - if (panic_force_cpu < 0) - return 0; - - panic_force_buf = kmalloc(PANIC_MSG_BUFSZ, GFP_KERNEL); - - return 0; -} -late_initcall(panic_force_cpu_late_init); - static void do_panic_on_target_cpu(void *info) { panic("%s", (char *)info); @@ -380,7 +369,7 @@ static bool panic_try_force_cpu(const char *fmt, va_list args) { int this_cpu = raw_smp_processor_id(); int old_cpu = PANIC_CPU_INVALID; - const char *msg; + va_list ap; /* Feature not enabled via boot parameter */ if (panic_force_cpu < 0) @@ -413,20 +402,11 @@ static bool panic_try_force_cpu(const char *fmt, va_list args) return old_cpu != this_cpu; /* - * Use dynamically allocated buffer if available, otherwise - * fall back to static message for early boot panics or allocation failure. + * Do not consume args, the caller reuses them if we fail. */ - if (panic_force_buf) { - va_list ap; - - /* Do not consume args, the caller reuses it if we fail */ - va_copy(ap, args); - vsnprintf(panic_force_buf, PANIC_MSG_BUFSZ, fmt, ap); - va_end(ap); - msg = panic_force_buf; - } else { - msg = "Redirected panic (buffer unavailable)"; - } + va_copy(ap, args); + vsnprintf(panic_force_buf, PANIC_MSG_BUFSZ, fmt, ap); + va_end(ap); console_verbose(); bust_spinlocks(1); @@ -441,7 +421,7 @@ static bool panic_try_force_cpu(const char *fmt, va_list args) dump_stack(); } - if (panic_smp_redirect_cpu(panic_force_cpu, (void *)msg) != 0) { + if (panic_smp_redirect_cpu(panic_force_cpu, panic_force_buf) != 0) { atomic_set(&panic_redirect_cpu, PANIC_CPU_INVALID); pr_warn("panic: failed to redirect to CPU %d, continuing on CPU %d\n", panic_force_cpu, this_cpu); From 8024c1de3b4ee64d7bc07c634dee34d139636c53 Mon Sep 17 00:00:00 2001 From: Jonas Rebmann Date: Wed, 16 Sep 2026 19:38:06 +0200 Subject: [PATCH 1303/1352] lib/tests: string_helpers: check null terminator too Patch series "lib/string_helpers: fixes and test cases for string_unescape()". This series fixes two bugs in string_unescape() regarding the destination buffer length. Both fixes are accompanied with kunit tests which would fail without the fixes. To make this possible, preparatory patches 1 and 2 improve and clean up testing helpers and 3 introduces test_unescape_one() which allows for targeted testing of the string_unescape() function. This patch (of 5): string_unescape() returns the number of character written to dst, not counting the null terminator which is always written. Ensure string_unescape has included the null terminator by adding a separate check. While at it, improve output for failed tests by showing the memory dump even if the length differs. Link: https://lore.kernel.org/20260916-string_unescape-v1-0-7f8bd986fa33@pengutronix.de Link: https://lore.kernel.org/20260916-string_unescape-v1-1-7f8bd986fa33@pengutronix.de Signed-off-by: Jonas Rebmann Signed-off-by: Andrew Morton Cc: Andy Shevchenko Cc: Kees Cook Cc: Sascha Hauer --- lib/tests/string_helpers_kunit.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/lib/tests/string_helpers_kunit.c b/lib/tests/string_helpers_kunit.c index 9fbe91079c7ed0..9bc3acffaf2f10 100644 --- a/lib/tests/string_helpers_kunit.c +++ b/lib/tests/string_helpers_kunit.c @@ -22,7 +22,7 @@ static void test_string_check_buf(struct kunit *test, char *out_real, size_t q_real, char *out_test, size_t q_test) { - KUNIT_ASSERT_EQ_MSG(test, q_real, q_test, "name:%s", name); + KUNIT_EXPECT_EQ_MSG(test, q_real, q_test, "name:%s", name); KUNIT_EXPECT_MEMEQ_MSG(test, out_test, out_real, q_test, "name:%s", name); } @@ -103,6 +103,7 @@ static void test_string_unescape(struct kunit *test, test_string_check_buf(test, name, flags, in, p - 1, out_real, q_real, out_test, q_test); + KUNIT_EXPECT_EQ_MSG(test, out_real[q_real], '\0', "name:%s", name); } struct test_string_1 { From a7f53b6c1076c98b597f4edceaaa81cf504bf33e Mon Sep 17 00:00:00 2001 From: Jonas Rebmann Date: Wed, 16 Sep 2026 19:38:07 +0200 Subject: [PATCH 1304/1352] lib/tests: string_helpers: drop unused parameters Drop the unused parameters from test_string_check_buf() to improve readability. Link: https://lore.kernel.org/20260916-string_unescape-v1-2-7f8bd986fa33@pengutronix.de Signed-off-by: Jonas Rebmann Signed-off-by: Andrew Morton Cc: Andy Shevchenko Cc: Kees Cook Cc: Sascha Hauer --- lib/tests/string_helpers_kunit.c | 7 ++----- 1 file changed, 2 insertions(+), 5 deletions(-) diff --git a/lib/tests/string_helpers_kunit.c b/lib/tests/string_helpers_kunit.c index 9bc3acffaf2f10..1ed652f762d160 100644 --- a/lib/tests/string_helpers_kunit.c +++ b/lib/tests/string_helpers_kunit.c @@ -18,7 +18,6 @@ static void test_string_check_buf(struct kunit *test, const char *name, unsigned int flags, - char *in, size_t p, char *out_real, size_t q_real, char *out_test, size_t q_test) { @@ -101,8 +100,7 @@ static void test_string_unescape(struct kunit *test, q_real = string_unescape(in, out_real, q_real, flags); } - test_string_check_buf(test, name, flags, in, p - 1, out_real, q_real, - out_test, q_test); + test_string_check_buf(test, name, flags, out_real, q_real, out_test, q_test); KUNIT_EXPECT_EQ_MSG(test, out_real[q_real], '\0', "name:%s", name); } @@ -457,8 +455,7 @@ static void test_string_escape(struct kunit *test, const char *name, q_real = string_escape_mem(in, p, out_real, out_size, flags, esc); - test_string_check_buf(test, name, flags, in, p, out_real, q_real, out_test, - q_test); + test_string_check_buf(test, name, flags, out_real, q_real, out_test, q_test); test_string_escape_overflow(test, in, p, flags, esc, q_test, name); } From 24e5dcf72fa2de98993421deec6277535f452216 Mon Sep 17 00:00:00 2001 From: Jonas Rebmann Date: Wed, 16 Sep 2026 19:38:08 +0200 Subject: [PATCH 1305/1352] lib/tests: string_helpers: introduce test_string_unescape_one The existing test_string_unescape() function follows a complex procedure where it, given a set of UNESCAPE flags, appends multiple test fragments and predicts their unescape result for the chosen set of flags. Rename test_string_unescape() to a more descriptive test_string_unescape_combined In preparation to add simple regression tests, introduce test_string_unescape_one() which asserts on exactly one call to string_unescape. Add some tests for corner cases which already pass. Link: https://lore.kernel.org/20260916-string_unescape-v1-3-7f8bd986fa33@pengutronix.de Signed-off-by: Jonas Rebmann Signed-off-by: Andrew Morton Cc: Andy Shevchenko Cc: Kees Cook Cc: Sascha Hauer --- lib/tests/string_helpers_kunit.c | 31 +++++++++++++++++++++++++------ 1 file changed, 25 insertions(+), 6 deletions(-) diff --git a/lib/tests/string_helpers_kunit.c b/lib/tests/string_helpers_kunit.c index 1ed652f762d160..3c6fa7324965ce 100644 --- a/lib/tests/string_helpers_kunit.c +++ b/lib/tests/string_helpers_kunit.c @@ -55,9 +55,9 @@ static const struct test_string strings[] = { }, }; -static void test_string_unescape(struct kunit *test, - const char *name, unsigned int flags, - bool inplace) +static void test_string_unescape_combined(struct kunit *test, + const char *name, unsigned int flags, + bool inplace) { int q_real = 256; char *in = kunit_kzalloc(test, q_real, GFP_KERNEL); @@ -596,14 +596,33 @@ static void test_upper_lower(struct kunit *test) } } +static void test_string_unescape_one(struct kunit *test, + const char *name, unsigned int flags, + char *src, size_t len, + char *out_test, size_t q_test) +{ + char *out_real = kunit_kzalloc(test, len, GFP_KERNEL); + int q_real; + + q_real = string_unescape(src, out_real, len, flags); + test_string_check_buf(test, name, flags, out_real, q_real, out_test, q_test); +} + static void test_unescape(struct kunit *test) { unsigned int i; for (i = 0; i < UNESCAPE_ALL_MASK + 1; i++) - test_string_unescape(test, "unescape", i, false); - test_string_unescape(test, "unescape inplace", - get_random_u32_below(UNESCAPE_ALL_MASK + 1), true); + test_string_unescape_combined(test, "unescape", i, false); + test_string_unescape_combined(test, "unescape inplace", + get_random_u32_below(UNESCAPE_ALL_MASK + 1), true); + + test_string_unescape_one(test, "simple case", UNESCAPE_HEX | UNESCAPE_SPECIAL, "ABC", 6, "ABC", 3); + test_string_unescape_one(test, "single escape", UNESCAPE_HEX | UNESCAPE_SPECIAL, "A\\x42C", 6, "ABC", 3); + test_string_unescape_one(test, "escape before end", UNESCAPE_HEX, "B\\qX", 4, "B\\q", 3); + test_string_unescape_one(test, "escape at end", UNESCAPE_HEX, "a\\qX", 3, "a\\", 2); + test_string_unescape_one(test, "backslash before escape", UNESCAPE_HEX, "\\\\x41B", 12, "\\\\x41B", 6); + test_string_unescape_one(test, "backslash escape", UNESCAPE_HEX | UNESCAPE_SPECIAL, "\\\\x41B", 16, "\\x41B", 5); } static void test_escape(struct kunit *test) From 7f4d06cf459305ece458b7d254239e3942394251 Mon Sep 17 00:00:00 2001 From: Jonas Rebmann Date: Wed, 16 Sep 2026 19:38:09 +0200 Subject: [PATCH 1306/1352] lib/string_helpers: use full destination buffer in string_unescape() Although all of the available sequences expand to exactly one byte, the current implementation decrements the remaining bytes in the destination buffer twice, effectively shortening it by one byte per each unescaped character. The extra decrement is only needed in the one case where a single loop iteration produces two output bytes: when the sequence turns out not to be a valid escape sequence, the previously skipped backslash has to be emitted before the character is copied verbatim. Add a kunit regression-test that unescapes into a barely long enough 3 buffer. Link: https://lore.kernel.org/20260916-string_unescape-v1-4-7f8bd986fa33@pengutronix.de Fixes: 16c7fa05829e ("lib/string_helpers: introduce generic string_unescape") Signed-off-by: Jonas Rebmann Signed-off-by: Andrew Morton Cc: Andy Shevchenko Cc: Kees Cook Cc: Sascha Hauer --- lib/string_helpers.c | 2 +- lib/tests/string_helpers_kunit.c | 3 +++ 2 files changed, 4 insertions(+), 1 deletion(-) diff --git a/lib/string_helpers.c b/lib/string_helpers.c index 98d6ed0eaab7e9..cb41ef9d8c5bb9 100644 --- a/lib/string_helpers.c +++ b/lib/string_helpers.c @@ -331,7 +331,6 @@ int string_unescape(char *src, char *dst, size_t size, unsigned int flags) while (*src && --size) { if (src[0] == '\\' && src[1] != '\0' && size > 1) { src++; - size--; if (flags & UNESCAPE_SPACE && unescape_space(&src, &out)) @@ -350,6 +349,7 @@ int string_unescape(char *src, char *dst, size_t size, unsigned int flags) continue; *out++ = '\\'; + size--; } *out++ = *src++; } diff --git a/lib/tests/string_helpers_kunit.c b/lib/tests/string_helpers_kunit.c index 3c6fa7324965ce..2e02c680cbb21f 100644 --- a/lib/tests/string_helpers_kunit.c +++ b/lib/tests/string_helpers_kunit.c @@ -623,6 +623,9 @@ static void test_unescape(struct kunit *test) test_string_unescape_one(test, "escape at end", UNESCAPE_HEX, "a\\qX", 3, "a\\", 2); test_string_unescape_one(test, "backslash before escape", UNESCAPE_HEX, "\\\\x41B", 12, "\\\\x41B", 6); test_string_unescape_one(test, "backslash escape", UNESCAPE_HEX | UNESCAPE_SPECIAL, "\\\\x41B", 16, "\\x41B", 5); + + test_string_unescape_one(test, "short buffer", UNESCAPE_HEX, "\\x41\\x41B", 4, "AAB", 3); + test_string_unescape_one(test, "unrecognized escape at end", UNESCAPE_HEX, "B\\qX", 4, "B\\q", 3); } static void test_escape(struct kunit *test) From d934e57f6ffef77e4fee0f32d64a7da534f22af2 Mon Sep 17 00:00:00 2001 From: Jonas Rebmann Date: Wed, 16 Sep 2026 19:38:10 +0200 Subject: [PATCH 1307/1352] lib/string_helpers: fix counting of remaining bytes in string_unescape() All of the available sequences expand to exactly one byte, the size check in the loop condition is sufficient for the case of an escaped character too. Otherwise, an escape sequence that should be unescaped to the last character before terminating with null in the destination buffer will be output as backslash instead of the escaped character. The only exception is when encountering a backslash that turns out to not start a valid escape sequence and both the backslash and the character following are handled in one iteration. Move the check there. Add a kunit regression-test that unescapes a character to right in front of the null terminator of the destination buffer. Link: https://lore.kernel.org/20260916-string_unescape-v1-5-7f8bd986fa33@pengutronix.de Fixes: 16c7fa05829e ("lib/string_helpers: introduce generic string_unescape") Signed-off-by: Jonas Rebmann Signed-off-by: Andrew Morton Cc: Andy Shevchenko Cc: Kees Cook Cc: Sascha Hauer --- lib/string_helpers.c | 5 +++-- lib/tests/string_helpers_kunit.c | 2 ++ 2 files changed, 5 insertions(+), 2 deletions(-) diff --git a/lib/string_helpers.c b/lib/string_helpers.c index cb41ef9d8c5bb9..4a621f884bde3f 100644 --- a/lib/string_helpers.c +++ b/lib/string_helpers.c @@ -329,7 +329,7 @@ int string_unescape(char *src, char *dst, size_t size, unsigned int flags) size = SIZE_MAX; while (*src && --size) { - if (src[0] == '\\' && src[1] != '\0' && size > 1) { + if (src[0] == '\\' && src[1] != '\0') { src++; if (flags & UNESCAPE_SPACE && @@ -349,7 +349,8 @@ int string_unescape(char *src, char *dst, size_t size, unsigned int flags) continue; *out++ = '\\'; - size--; + if (!--size) + break; } *out++ = *src++; } diff --git a/lib/tests/string_helpers_kunit.c b/lib/tests/string_helpers_kunit.c index 2e02c680cbb21f..10763a01be83c2 100644 --- a/lib/tests/string_helpers_kunit.c +++ b/lib/tests/string_helpers_kunit.c @@ -626,6 +626,8 @@ static void test_unescape(struct kunit *test) test_string_unescape_one(test, "short buffer", UNESCAPE_HEX, "\\x41\\x41B", 4, "AAB", 3); test_string_unescape_one(test, "unrecognized escape at end", UNESCAPE_HEX, "B\\qX", 4, "B\\q", 3); + + test_string_unescape_one(test, "end of buffer", UNESCAPE_HEX, "B\\x41", 3, "BA", 2); } static void test_escape(struct kunit *test) From 4f0ed31e818ad184c2a0f04a9c886a66e4528b1b Mon Sep 17 00:00:00 2001 From: Aamir Ahmed Date: Wed, 16 Sep 2026 09:53:51 +0100 Subject: [PATCH 1308/1352] lib: decompress_bunzip2: fix integer overflow in run-length decoding The RUNA/RUNB decoder in get_next_block() accumulates a run length into the signed int t with no bound. runPos doubles per symbol, so t grows as at least 2^n-1 and reaches INT_MAX after 31 RUNA symbols. The guard that follows, dbufCount+t >= dbufSize, is itself a signed addition, and as the kernel builds with -fno-strict-overflow it wraps negative once dbufCount is nonzero, so the guard is skipped. while (t--) dbuf[dbufCount++] = uc then writes INT_MAX entries into a buffer holding at most 900000. One literal symbol ahead of the run is enough to make dbufCount nonzero. Bound t to the block size as it is accumulated. t only grows within a run, so the value tested never exceeds the final run length; any stream that decodes today satisfies dbufCount+t < dbufSize at the flush, hence t < dbufSize throughout, and no such stream is rejected. Because t grows as at least 2^n-1 the bound trips by the 20th symbol, leaving runPos at most 2^19, so the following runPos <<= 1 cannot overflow either. Link: https://lore.kernel.org/AS8P251MB00010FE5E38D253CF98A652AC8B92@AS8P251MB0001.EURP251.PROD.OUTLOOK.COM Fixes: bc22c17e12c1 ("bzip2/lzma: library support for gzip, bzip2 and lzma decompression") Signed-off-by: Aamir Ahmed Signed-off-by: Andrew Morton Assisted-by: LLM Cc: Alain Knaff Cc: "H. Peter Anvin" Cc: --- lib/decompress_bunzip2.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/lib/decompress_bunzip2.c b/lib/decompress_bunzip2.c index aaa75404250e2d..a9236d3e9678de 100644 --- a/lib/decompress_bunzip2.c +++ b/lib/decompress_bunzip2.c @@ -428,6 +428,10 @@ static int INIT get_next_block(struct bunzip_data *bd) t += (runPos << nextSym); /* +runPos if RUNA; +2*runPos if RUNB */ + /* Bound the run so t and runPos cannot overflow. */ + if (t >= dbufSize) + return RETVAL_DATA_ERROR; + runPos <<= 1; continue; } From 78fb84a4addd5c8860ac5d1e844b4208583577ed Mon Sep 17 00:00:00 2001 From: Su Yue Date: Wed, 16 Sep 2026 10:47:50 +0800 Subject: [PATCH 1309/1352] ocfs2: update xattr count before moving bucket entries Since commit 2f26f58df041 ("ocfs2: annotate flexible array members with __counted_by_le()"), the xh_entries array is annotated with __counted_by_le(xh_count), so FORTIFY uses xh_count to determine its bounds. When inserting an entry into a bucket, ocfs2_xa_bucket_add_entry() shifts existing entries before incrementing xh_count. The destination therefore extends one entry past the bounds described by the old count. With CONFIG_CC_HAS_COUNTED_BY and CONFIG_FORTIFY_SOURCE enabled, ocfs2-test: single_run-WIP.sh -t reflink triggers the following failure while adding an extended attribute: [ 150.156484] memmove: detected buffer overflow: 352 byte write of buffer size 336 [ 150.156487] WARNING: lib/string_helpers.c:1036 at __fortify_report+0x3d/0x50, CPU#15: reflink_test/2336 [ 150.160496] RIP: 0010:__fortify_report+0x40/0x50 [ 150.164031] Call Trace: [ 150.164141] [ 150.164232] __fortify_panic+0x9/0xb [ 150.164383] ocfs2_xa_bucket_add_entry.cold+0x17/0x28 [ocfs2] [ 150.164661] ocfs2_xa_set+0x8fb/0xf40 [ocfs2] Increment xh_count before memmove() so the destination bounds include the new entry. Keep the local count unchanged to calculate the insertion position and move length from the original number of entries. The caller has already checked that there is enough space for the new entry. Link: https://lore.kernel.org/20260916024750.9450-1-glass.su@suse.com Fixes: 2f26f58df041 ("ocfs2: annotate flexible array members with __counted_by_le()") Signed-off-by: Su Yue Signed-off-by: Andrew Morton Reviewed-by: Heming Zhao Reviewed-by: Joseph Qi Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi Cc: Changwei Ge Cc: Jun Piao Cc: --- fs/ocfs2/xattr.c | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/fs/ocfs2/xattr.c b/fs/ocfs2/xattr.c index a2d6c3d0f2e8d8..9292cf26e2769b 100644 --- a/fs/ocfs2/xattr.c +++ b/fs/ocfs2/xattr.c @@ -2146,12 +2146,17 @@ static void ocfs2_xa_bucket_add_entry(struct ocfs2_xa_loc *loc, u32 name_hash) } } + /* + * Increment xh_count before memmove() so __counted_by_le(xh_count) + * includes the new entry in the destination bounds. + */ + le16_add_cpu(&xh->xh_count, 1); + if (low != count) memmove(&xh->xh_entries[low + 1], &xh->xh_entries[low], ((count - low) * sizeof(struct ocfs2_xattr_entry))); - le16_add_cpu(&xh->xh_count, 1); loc->xl_entry = &xh->xh_entries[low]; memset(loc->xl_entry, 0, sizeof(struct ocfs2_xattr_entry)); } From f290fb63351bdf883e63211a7d4593c0c6932a0a Mon Sep 17 00:00:00 2001 From: Fang Xieyan Date: Thu, 17 Sep 2026 18:43:06 +0800 Subject: [PATCH 1310/1352] kcov: ignore an out-of-range comparison count in write_comp_data() write_comp_data() reads the comparison record count from area[0] and uses it to index the coverage buffer: area = (u64 *)t->kcov_area; max_pos = t->kcov_size * sizeof(unsigned long); count = READ_ONCE(area[0]); /* Every record is KCOV_WORDS_PER_CMP 64-bit words. */ start_index = 1 + count * KCOV_WORDS_PER_CMP; end_pos = (start_index + KCOV_WORDS_PER_CMP) * sizeof(u64); if (likely(end_pos <= max_pos)) { The buffer is mmap'd writable into the collecting process, so count is under its control and end_pos <= max_pos is its only bound. A count that wraps the u64 multiply leaves end_pos below max_pos, so the check passes while the record store lands 24 bytes before the buffer, in the unmapped vmalloc guard page, and faults: BUG: unable to handle page fault for address: ffa0000000b60fe8 #PF: supervisor write access in kernel mode #PF: error_code(0x0002) - not-present page Oops: 0002 [#1] SMP KASAN NOPTI RIP: 0010:write_comp_data+0x7e/0xa0 ... Kernel panic - not syncing: Fatal exception Bound count first: only max_pos / (sizeof(u64) * KCOV_WORDS_PER_CMP) records fit, so a larger count is not a valid index and is dropped. kcov_move_area() bounds the same untrusted count this way, and no count the end_pos <= max_pos check accepts reaches that limit, so no valid record is lost. kcov is a root-only debugfs file (debugfs_create_file_unsafe("kcov", 0600, ...)) and write_comp_data() exists only under CONFIG_KCOV_ENABLE_COMPARISONS, so this is a local, debug-kernel robustness fix: the process corrupts its own buffer and the kernel oopses. It crosses no privilege boundary. Additional details at [1] Link: https://lore.kernel.org/20260917104306.22145-1-fangxy@xiaopeng.com [1] Fixes: ded97d2c2b2c ("kcov: support comparison operands collection") Signed-off-by: Fang Xieyan Signed-off-by: Andrew Morton Reviewed-by: Alexander Potapenko Assisted-by: Hawkeye:GLM-5.3-flash Assisted-by: Qoder:Qwen3.8-Max Cc: Andrey Konovalov Cc: Dmitry Vyukov Cc: Marco Elver Cc: Victor Chibotaru Cc: --- kernel/kcov.c | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/kernel/kcov.c b/kernel/kcov.c index 35420f0ac524d4..54eaae97bfd67d 100644 --- a/kernel/kcov.c +++ b/kernel/kcov.c @@ -253,6 +253,14 @@ static void notrace write_comp_data(u64 type, u64 arg1, u64 arg2, u64 ip) count = READ_ONCE(area[0]); + /* + * area[0] is writable by the collecting process, so count cannot be + * trusted. Bound it to the records that fit, as kcov_move_area() + * does, so the end_pos multiply below cannot wrap past its check. + */ + if (count >= max_pos / (sizeof(u64) * KCOV_WORDS_PER_CMP)) + return; + /* Every record is KCOV_WORDS_PER_CMP 64-bit words. */ start_index = 1 + count * KCOV_WORDS_PER_CMP; end_pos = (start_index + KCOV_WORDS_PER_CMP) * sizeof(u64); From 9fc9fd79ff31a84740d5118681cd55163f50c3be Mon Sep 17 00:00:00 2001 From: Frank Li Date: Thu, 17 Sep 2026 16:50:12 -0400 Subject: [PATCH 1311/1352] rapidio: use dmaengine_get_dma_device() instead of chan->device->dev Replace direct dma_chan::device::dev access with the proper dmaengine_get_dma_device() for consumer API. chan->device->dev is not always the device used for DMA mapping. Some DMA engines support per-channel IOMMU mappings, so different channels may use different DMA devices. dmaengine_get_dma_device() returns the correct device for each channel. This also prepares for making the DMA engine provider data structures private. DMA consumers should not access DMA engine internals directly. Link: https://lore.kernel.org/20260917205016.1293317-1-Frank.Li@oss.nxp.com Signed-off-by: Frank Li Signed-off-by: Andrew Morton Assisted-by: LLM Cc: Alexandre Bounine Cc: Dan Carpenter Cc: Kees Cook Cc: Matt Porter Cc: Vinod Koul --- drivers/rapidio/devices/rio_mport_cdev.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/drivers/rapidio/devices/rio_mport_cdev.c b/drivers/rapidio/devices/rio_mport_cdev.c index 47ba34b4afb296..40d2953a9a7f9e 100644 --- a/drivers/rapidio/devices/rio_mport_cdev.c +++ b/drivers/rapidio/devices/rio_mport_cdev.c @@ -555,7 +555,7 @@ static void dma_req_free(struct kref *ref) refcount); struct mport_cdev_priv *priv = req->priv; - dma_unmap_sg(req->dmach->device->dev, + dma_unmap_sg(dmaengine_get_dma_device(req->dmach), req->sgt.sgl, req->sgt.nents, req->dir); sg_free_table(&req->sgt); if (req->page_list) { @@ -916,7 +916,7 @@ rio_dma_transfer(struct file *filp, u32 transfer_mode, xfer->offset, xfer->length); } - nents = dma_map_sg(chan->device->dev, + nents = dma_map_sg(dmaengine_get_dma_device(chan), req->sgt.sgl, req->sgt.nents, dir); if (nents == 0) { rmcd_error("Failed to map SG list"); From f7540cc3fcdec3bddf4c0ac0efe18c2f81c91062 Mon Sep 17 00:00:00 2001 From: "Jose A. Perez de Azpillaga" Date: Mon, 21 Sep 2026 23:55:04 +0200 Subject: [PATCH 1312/1352] percpu_counter: annotate lockless read in _limited_add() syzbot reports a data race on fbc->count between the lockless read in the fast path of __percpu_counter_limited_add() and the locked update in percpu_counter_add_batch(), reached via shmem's used_blocks counter: the alloc path reads it through percpu_counter_limited_add() while the free path updates it through percpu_counter_sub(). Annotate the read rather than change the logic: the lockless read is deliberate, only the annotation is missing. It is an approximation, not a conservative bound. A concurrent flush moves value between fbc->count and a per-cpu counter, and other CPUs' locked slow paths add to fbc->count too, so a stale low value can let the fast path proceed where the slow path would refuse. The error is bounded by the per-cpu slack (unknown = batch * num_online_cpus()). This is existing behavior, no functional change intended. Read it with data_race(READ_ONCE(fbc->count)): data_race() tells KCSAN the race is intended (READ_ONCE() alone still triggers the report, because KCSAN reports against watchpoints set up by plain accesses), while READ_ONCE() stops the compiler from refetching the value, as in percpu_counter_read_positive(). The lock-protected accesses stay plain, so future buggy lockless writes are still caught. Link: https://lore.kernel.org/20260921215508.141641-1-azpijr@gmail.com Signed-off-by: Jose A. Perez de Azpillaga Signed-off-by: Andrew Morton Reported-by: syzbot+a3c71b9db9c11c270f59@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=a3c71b9db9c11c270f59 Reviewed-by: Andrew Morton Cc: Dennis Zhou Cc: Tejun Heo Cc: Christoph Lameter Cc: Hugh Dickins Cc: Baolin Wang --- lib/percpu_counter.c | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/lib/percpu_counter.c b/lib/percpu_counter.c index 2891f94a11c654..87b261b87b0bc8 100644 --- a/lib/percpu_counter.c +++ b/lib/percpu_counter.c @@ -328,6 +328,7 @@ bool __percpu_counter_limited_add(struct percpu_counter *fbc, s64 limit, s64 amount, s32 batch) { s64 count; + s64 gcount; s64 unknown; unsigned long flags; bool good = false; @@ -338,11 +339,16 @@ bool __percpu_counter_limited_add(struct percpu_counter *fbc, local_irq_save(flags); unknown = batch * num_online_cpus(); count = __this_cpu_read(*fbc->counters); + /* + * Lockless on purpose: gcount may be stale, so this is only an + * approximation, bounded by the per-cpu slack ("unknown"). + */ + gcount = data_race(READ_ONCE(fbc->count)); /* Skip taking the lock when safe */ if (abs(count + amount) <= batch && - ((amount > 0 && fbc->count + unknown <= limit) || - (amount < 0 && fbc->count - unknown >= limit))) { + ((amount > 0 && gcount + unknown <= limit) || + (amount < 0 && gcount - unknown >= limit))) { this_cpu_add(*fbc->counters, amount); local_irq_restore(flags); return true; From 2ba9f46f7d41807175ced372a23cfb83054ad8c4 Mon Sep 17 00:00:00 2001 From: Sang-Heon Jeon Date: Tue, 22 Sep 2026 22:26:27 +0900 Subject: [PATCH 1313/1352] init/main.c: remove unreachable check in obsolete_checksetup() Since commit 1ecfea06386c ("init.h: remove long-dead __setup_null_param() macro"), .init.setup entries cannot have a NULL setup_func. So remove the NULL check. No functional change. Link: https://lore.kernel.org/20260922132630.177941-1-ekffu200098@gmail.com Signed-off-by: Sang-Heon Jeon Signed-off-by: Andrew Morton --- init/main.c | 4 ---- 1 file changed, 4 deletions(-) diff --git a/init/main.c b/init/main.c index d1cd8efb823860..008a0a137f511d 100644 --- a/init/main.c +++ b/init/main.c @@ -215,10 +215,6 @@ static bool __init obsolete_checksetup(char *line) * params and __setups of same names 8( */ if (line[n] == '\0' || line[n] == '=') had_early_param = true; - } else if (!p->setup_func) { - pr_warn("Parameter %s is obsolete, ignored\n", - p->str); - return true; } else if (p->setup_func(line + n)) return true; } From ef628913a136547596066115be6a265652a143f9 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Tue, 22 Sep 2026 17:42:52 +0800 Subject: [PATCH 1314/1352] ocfs2: restore -ERANGE check in ocfs2_xattr_tree_list_index_block() ocfs2_xattr_list_entry() returns -ERANGE when the buffer handed in by the caller is too small to hold the whole xattr name list. That is the normal listxattr(2) protocol, userspace simply retries with a larger buffer, so it is not an ocfs2 error and must not be logged as one. Commit a46fa684fcb700 ("ocfs2: Don't printk the error when listing too many xattrs.") already knew this and silenced two sites, quoting (27738,0):ocfs2_iterate_xattr_buckets:3158 ERROR: status = -34 (27738,0):ocfs2_xattr_tree_list_index_block:3264 ERROR: status = -34 The refactoring in commit 47bca4950bc40f ("ocfs2: Abstract ocfs2 xattr tree extend rec iteration process.") then rewrote ocfs2_xattr_tree_list_index_block() around the new ocfs2_iterate_xattr_index_block() helper and dropped that guard, so the second message has been emitted ever since: ocfs2_xattr_tree_list_index_block:4513 ERROR: status = -34 The exemption in ocfs2_iterate_xattr_buckets() survived because that function was left alone, and the new helper does check for -ERANGE itself, which is why only the outermost caller still logs it. Restore the missing check. Only the log line changes: -ERANGE is still returned to the VFS untouched, so listxattr(2) behaves exactly as before. Link: https://lore.kernel.org/20260922094252.971631-1-joseph.qi@linux.alibaba.com Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Cc: Heming Zhao Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi --- fs/ocfs2/xattr.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/fs/ocfs2/xattr.c b/fs/ocfs2/xattr.c index 9292cf26e2769b..9f620f6c600518 100644 --- a/fs/ocfs2/xattr.c +++ b/fs/ocfs2/xattr.c @@ -4502,7 +4502,8 @@ static int ocfs2_xattr_tree_list_index_block(struct inode *inode, ret = ocfs2_iterate_xattr_index_block(inode, blk_bh, ocfs2_list_xattr_tree_rec, &xl); if (ret) { - mlog_errno(ret); + if (ret != -ERANGE) + mlog_errno(ret); goto out; } From 5274469968895adf0738e813db9d696d36d290a3 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Tue, 22 Sep 2026 17:34:37 +0800 Subject: [PATCH 1315/1352] ocfs2: fix possible deadlock between nfs_sync_rwlock and fs_reclaim syzbot detected a circular locking dependency on &osb->nfs_sync_rwlock: CPU0 CPU1 ---- ---- lock(fs_reclaim); lock(&ocfs2_sysfile_lock_key[INODE_ALLOC_SYSTEM_INODE]); lock(fs_reclaim); rlock(&osb->nfs_sync_rwlock); *** DEADLOCK *** Chain exists of: &osb->nfs_sync_rwlock --> &ocfs2_sysfile_lock_key[INODE_ALLOC_SYSTEM_INODE] --> fs_reclaim CPU0 is kswapd. The dentry shrinker runs under fs_reclaim and reaches ocfs2_delete_inode() through ->evict_inode(): kswapd balance_pgdat shrink_node shrink_slab super_cache_scan prune_dcache_sb shrink_dentry_list dentry_kill ocfs2_dentry_iput evict ocfs2_evict_inode ocfs2_delete_inode ocfs2_nfs_sync_lock down_read(&osb->nfs_sync_rwlock) //C0: grabbing CPU1 is a task deleting an inode. It takes nfs_sync_rwlock first, then the orphan dir and inode alloc system inode i_rwsems, and allocates the metadata reservation with GFP_KERNEL while holding them: evict ocfs2_evict_inode ocfs2_delete_inode + ocfs2_nfs_sync_lock | down_read(&osb->nfs_sync_rwlock) //C1: hold + ocfs2_wipe_inode + inode_lock(orphan_dir_inode) //C1: hold + ocfs2_truncate_for_delete | ocfs2_commit_truncate | ocfs2_remove_btree_range | ocfs2_reserve_blocks_for_rec_trunc | ocfs2_reserve_new_metadata_blocks | kzalloc_obj() //C1: grabbing | // GFP_KERNEL -> might_alloc -> fs_reclaim_acquire + ocfs2_remove_inode inode_lock(inode_alloc_inode) jbd2 pins allocations to GFP_NOFS for the lifetime of a transaction handle, but these reservations are made before ocfs2_start_trans(), so fs_reclaim is acquired with ocfs2 locks still held and the cycle closes. The same cycle is reachable from ocfs2_get_dentry() and ocfs2_get_parent(), which hold nfs_sync_rwlock for write across ocfs2_test_inode_bit() and ocfs2_iget(). This is more than a lockdep artifact. A GFP_KERNEL allocation under the orphan dir i_rwsem enters direct reclaim, which runs the shrinkers, which can evict another ocfs2 inode and re-enter ocfs2_wipe_inode() on the same task; the second inode_lock() on the singleton orphan dir inode then self-deadlocks, since i_rwsem is not recursive. Establish a GFP_NOFS allocation context for the whole nfs_sync_rwlock critical section so that nothing holding it can recurse into filesystem reclaim. Do it in ocfs2_nfs_sync_lock()/ocfs2_nfs_sync_unlock() rather than at the call sites, so that future callers cannot forget it. Link: https://lore.kernel.org/20260922093437.954127-1-joseph.qi@linux.alibaba.com Fixes: 6ca497a83e59 ("ocfs2: fix rare stale inode errors when exporting via nfs") Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Reported-by: syzbot+a68ce48df87b8e36e915@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=a68ce48df87b8e36e915 Cc: Heming Zhao Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi --- fs/ocfs2/dlmglue.c | 29 +++++++++++++++++++++++++---- fs/ocfs2/dlmglue.h | 5 +++-- fs/ocfs2/export.c | 10 ++++++---- fs/ocfs2/inode.c | 5 +++-- 4 files changed, 37 insertions(+), 12 deletions(-) diff --git a/fs/ocfs2/dlmglue.c b/fs/ocfs2/dlmglue.c index cf3318b0d3a8c9..6bc4a21cf5f02f 100644 --- a/fs/ocfs2/dlmglue.c +++ b/fs/ocfs2/dlmglue.c @@ -18,6 +18,7 @@ #include #include #include +#include #include #include @@ -2861,21 +2862,34 @@ void ocfs2_rename_unlock(struct ocfs2_super *osb) ocfs2_cluster_unlock(osb, lockres, DLM_LOCK_EX); } -int ocfs2_nfs_sync_lock(struct ocfs2_super *osb, int ex) +int ocfs2_nfs_sync_lock(struct ocfs2_super *osb, int ex, unsigned int *nofs_flag) { int status; + unsigned int flags; struct ocfs2_lock_res *lockres = &osb->osb_nfs_sync_lockres; if (ocfs2_is_hard_readonly(osb)) return -EROFS; + /* + * ocfs2_delete_inode() takes this lock from ->evict_inode(), which the + * dentry shrinker reaches while holding fs_reclaim. Anything allocated + * under the lock must therefore stay out of filesystem reclaim, or + * reclaim recurses back into the shrinker and tries to take this lock + * again. Cover the whole critical section, including the cluster lock + * and the sysfile inode locks the callers take below it. + */ + flags = memalloc_nofs_save(); + if (ex) down_write(&osb->nfs_sync_rwlock); else down_read(&osb->nfs_sync_rwlock); - if (ocfs2_mount_local(osb)) + if (ocfs2_mount_local(osb)) { + *nofs_flag = flags; return 0; + } status = ocfs2_cluster_lock(osb, lockres, ex ? LKM_EXMODE : LKM_PRMODE, 0, 0); @@ -2886,12 +2900,17 @@ int ocfs2_nfs_sync_lock(struct ocfs2_super *osb, int ex) up_write(&osb->nfs_sync_rwlock); else up_read(&osb->nfs_sync_rwlock); + memalloc_nofs_restore(flags); + return status; } - return status; + *nofs_flag = flags; + + return 0; } -void ocfs2_nfs_sync_unlock(struct ocfs2_super *osb, int ex) +void ocfs2_nfs_sync_unlock(struct ocfs2_super *osb, int ex, + unsigned int nofs_flag) { struct ocfs2_lock_res *lockres = &osb->osb_nfs_sync_lockres; @@ -2902,6 +2921,8 @@ void ocfs2_nfs_sync_unlock(struct ocfs2_super *osb, int ex) up_write(&osb->nfs_sync_rwlock); else up_read(&osb->nfs_sync_rwlock); + + memalloc_nofs_restore(nofs_flag); } int ocfs2_trim_fs_lock(struct ocfs2_super *osb, diff --git a/fs/ocfs2/dlmglue.h b/fs/ocfs2/dlmglue.h index a3ebd7303ea20b..faa58ed8699b1c 100644 --- a/fs/ocfs2/dlmglue.h +++ b/fs/ocfs2/dlmglue.h @@ -161,8 +161,9 @@ void ocfs2_orphan_scan_unlock(struct ocfs2_super *osb, u32 seqno); int ocfs2_rename_lock(struct ocfs2_super *osb); void ocfs2_rename_unlock(struct ocfs2_super *osb); -int ocfs2_nfs_sync_lock(struct ocfs2_super *osb, int ex); -void ocfs2_nfs_sync_unlock(struct ocfs2_super *osb, int ex); +int ocfs2_nfs_sync_lock(struct ocfs2_super *osb, int ex, unsigned int *nofs_flag); +void ocfs2_nfs_sync_unlock(struct ocfs2_super *osb, int ex, + unsigned int nofs_flag); void ocfs2_trim_fs_lock_res_init(struct ocfs2_super *osb); void ocfs2_trim_fs_lock_res_uninit(struct ocfs2_super *osb); int ocfs2_trim_fs_lock(struct ocfs2_super *osb, diff --git a/fs/ocfs2/export.c b/fs/ocfs2/export.c index 9c2665dd24e218..90b9e510e700d5 100644 --- a/fs/ocfs2/export.c +++ b/fs/ocfs2/export.c @@ -37,6 +37,7 @@ static struct dentry *ocfs2_get_dentry(struct super_block *sb, struct inode *inode; struct ocfs2_super *osb = OCFS2_SB(sb); u64 blkno = handle->ih_blkno; + unsigned int nofs_flag = 0; int status, set; struct dentry *result; @@ -59,7 +60,7 @@ static struct dentry *ocfs2_get_dentry(struct super_block *sb, * This will synchronize us against ocfs2_delete_inode() on * all nodes */ - status = ocfs2_nfs_sync_lock(osb, 1); + status = ocfs2_nfs_sync_lock(osb, 1, &nofs_flag); if (status < 0) { mlog(ML_ERROR, "getting nfs sync lock(EX) failed %d\n", status); goto check_err; @@ -90,7 +91,7 @@ static struct dentry *ocfs2_get_dentry(struct super_block *sb, inode = ocfs2_iget(osb, blkno, 0, 0); unlock_nfs_sync: - ocfs2_nfs_sync_unlock(osb, 1); + ocfs2_nfs_sync_unlock(osb, 1, nofs_flag); check_err: if (status < 0) { @@ -133,12 +134,13 @@ static struct dentry *ocfs2_get_parent(struct dentry *child) u64 blkno; struct dentry *parent; struct inode *dir = d_inode(child); + unsigned int nofs_flag = 0; int set; trace_ocfs2_get_parent(child, child->d_name.len, child->d_name.name, (unsigned long long)OCFS2_I(dir)->ip_blkno); - status = ocfs2_nfs_sync_lock(OCFS2_SB(dir->i_sb), 1); + status = ocfs2_nfs_sync_lock(OCFS2_SB(dir->i_sb), 1, &nofs_flag); if (status < 0) { mlog(ML_ERROR, "getting nfs sync lock(EX) failed %d\n", status); parent = ERR_PTR(status); @@ -183,7 +185,7 @@ static struct dentry *ocfs2_get_parent(struct dentry *child) ocfs2_inode_unlock(dir, 0); unlock_nfs_sync: - ocfs2_nfs_sync_unlock(OCFS2_SB(dir->i_sb), 1); + ocfs2_nfs_sync_unlock(OCFS2_SB(dir->i_sb), 1, nofs_flag); bail: trace_ocfs2_get_parent_end(parent); diff --git a/fs/ocfs2/inode.c b/fs/ocfs2/inode.c index 92f3450010fbb0..80c36c60a8cef9 100644 --- a/fs/ocfs2/inode.c +++ b/fs/ocfs2/inode.c @@ -1109,6 +1109,7 @@ static void ocfs2_delete_inode(struct inode *inode) { int wipe, status; sigset_t oldset; + unsigned int nofs_flag = 0; struct buffer_head *di_bh = NULL; struct ocfs2_dinode *di = NULL; @@ -1143,7 +1144,7 @@ static void ocfs2_delete_inode(struct inode *inode) * shared mode so that all nodes can still concurrently * process deletes. */ - status = ocfs2_nfs_sync_lock(OCFS2_SB(inode->i_sb), 0); + status = ocfs2_nfs_sync_lock(OCFS2_SB(inode->i_sb), 0, &nofs_flag); if (status < 0) { mlog(ML_ERROR, "getting nfs sync lock(PR) failed %d\n", status); ocfs2_cleanup_delete_inode(inode, 0); @@ -1215,7 +1216,7 @@ static void ocfs2_delete_inode(struct inode *inode) brelse(di_bh); bail_unlock_nfs_sync: - ocfs2_nfs_sync_unlock(OCFS2_SB(inode->i_sb), 0); + ocfs2_nfs_sync_unlock(OCFS2_SB(inode->i_sb), 0, nofs_flag); bail_unblock: ocfs2_unblock_signals(&oldset); From 9967eb3c3528c4ad0961ced760858e4203a2939a Mon Sep 17 00:00:00 2001 From: Nguyen Ngoc Thang Date: Sun, 20 Sep 2026 21:07:10 +0700 Subject: [PATCH 1316/1352] ocfs2: validate chunk and block counts of the local quota file ocfs2_local_read_info() trusts dqi_chunks and dqi_blocks from the local quota file header. If dqi_blocks is too small, ocfs2_extend_local_quota_file() computes a bogus chunk length, extends the file, and maps logical block dqi_blocks, which is already in the inode's metadata cache (the header or a chunk header read at mount). sb_getblk() returns that cached buffer and ocfs2_set_new_buffer_uptodate() hits BUG_ON(ocfs2_buffer_cached()): kernel BUG at fs/ocfs2/uptodate.c:509! ocfs2_extend_local_quota_file+0x45c/0x1100 ocfs2_create_local_dquot+0x8ac/0xb50 ocfs2_acquire_dquot+0x614/0xae0 ocfs2_get_init_inode+0xe9/0x1b0 ocfs2_mkdir+0x174/0x430 ocfs2_local_quota_add_chunk() has the same exposure. Reject a header whose counts don't fit the chunk layout or exceed i_size. Reproduced with a crafted image (dqi_blocks = 1, chunk 0 dqc_free = 0); syzbot has no reproducer for this report. Link: https://lore.kernel.org/20260920140710.43938-1-ngocthang2710.1999@gmail.com Fixes: 9e33d69f553a ("ocfs2: Implementation of local and global quota file handling") Signed-off-by: Nguyen Ngoc Thang Signed-off-by: Andrew Morton Reported-by: syzbot+03aaa576f1daa1c1f2f0@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=03aaa576f1daa1c1f2f0 Reviewed-by: Joseph Qi Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi Cc: Heming Zhao --- fs/ocfs2/quota_local.c | 24 ++++++++++++++++++++++++ 1 file changed, 24 insertions(+) diff --git a/fs/ocfs2/quota_local.c b/fs/ocfs2/quota_local.c index d351cda9211f85..76e7dd5aecc86b 100644 --- a/fs/ocfs2/quota_local.c +++ b/fs/ocfs2/quota_local.c @@ -245,6 +245,25 @@ static void ocfs2_release_local_quota_bitmaps(struct list_head *head) } } +/* Check that the on-disk chunk and block counts match the file layout */ +static int ocfs2_check_local_quota_info(struct inode *inode, + unsigned int chunks, + unsigned int blocks) +{ + u64 chunk_len = ol_chunk_blocks(inode->i_sb) + 1; + u64 min_blocks = chunks ? 1 + chunk_len * (chunks - 1) + 1 : 1; + u64 max_blocks = 1 + chunk_len * chunks; + + if (blocks >= min_blocks && blocks <= max_blocks && + blocks <= i_size_read(inode) >> inode->i_sb->s_blocksize_bits) + return 0; + + return ocfs2_error(inode->i_sb, + "Quota file %llu has bad info: %u chunks, %u blocks\n", + (unsigned long long)OCFS2_I(inode)->ip_blkno, + chunks, blocks); +} + /* Load quota bitmaps into memory */ static int ocfs2_load_local_quota_bitmaps(struct inode *inode, struct ocfs2_local_disk_dqinfo *ldinfo, @@ -733,6 +752,11 @@ static int ocfs2_local_read_info(struct super_block *sb, int type) oinfo->dqi_blocks = le32_to_cpu(ldinfo->dqi_blocks); oinfo->dqi_libh = bh; + status = ocfs2_check_local_quota_info(lqinode, oinfo->dqi_chunks, + oinfo->dqi_blocks); + if (status < 0) + goto out_err; + /* We crashed when using local quota file? */ if (!(oinfo->dqi_flags & OLQF_CLEAN)) { rec = OCFS2_SB(sb)->quota_rec; From 7f8954edb8bed2f9f502eebc091d259b5fbe999b Mon Sep 17 00:00:00 2001 From: Giorgi Kobakhia Date: Tue, 22 Sep 2026 13:21:59 -0700 Subject: [PATCH 1317/1352] ocfs2: fix array out of bound access in __ocfs2_find_path() __ocfs2_find_path() walks down the extent tree and records path by calling find_path_ins(), which appends entry to path->p_node[]. It only has 5 spots. A corrupted ocfs2 image whose extent block is pointing to itself causes __ocfs2_find_path() descent endlessly, writing past the end of path->p_node[] array. UBSAN: array-index-out-of-bounds in fs/ocfs2/alloc.c:677:14 index 5 is out of range for type 'ocfs2_path_item [5]' Call Trace: find_path_ins (fs/ocfs2/alloc.c:677 fs/ocfs2/alloc.c:1914) __ocfs2_find_path.constprop.0 (fs/ocfs2/alloc.c:1882) ocfs2_commit_truncate (fs/ocfs2/alloc.c:1924 fs/ocfs2/alloc.c:7286) ocfs2_truncate_file (fs/ocfs2/file.c:515) ocfs2_setattr (fs/ocfs2/file.c:1224) notify_change (fs/attr.c:556) do_truncate (fs/open.c:68) do_ftruncate (fs/open.c:194 (discriminator 1)) ksys_ftruncate (fs/open.c:206) __x64_sys_ftruncate (fs/open.c:211 fs/open.c:209 fs/open.c:209) Commit a406aff8c051 ("ocfs2: validate l_tree_depth to avoid out-of-bounds access") already restricts el->l_tree_depth to be less than OCFS2_MAX_PATH_DEPTH, which is equal to 5. However, does not handle the infinite descent case. Check if the el->l_tree_depth decreases on each descent. Maximum descents are restricted to 4 and the path->p_node[] array does not overflow. Link: https://lore.kernel.org/20260922202159.2421642-1-gkobakhi@asu.edu Fixes: dcd0538ff4e8 ("ocfs2: sparse b-tree support") Signed-off-by: Giorgi Kobakhia Signed-off-by: Andrew Morton Tested-by: Xiang Mei Reviewed-by: Joseph Qi Assisted-by: LLM Cc: Mark Fasheh Cc: Joel Becker Cc: --- fs/ocfs2/alloc.c | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/fs/ocfs2/alloc.c b/fs/ocfs2/alloc.c index 2fdc5403b10b04..fc949e59462699 100644 --- a/fs/ocfs2/alloc.c +++ b/fs/ocfs2/alloc.c @@ -1844,6 +1844,7 @@ static int __ocfs2_find_path(struct ocfs2_caching_info *ci, int i, ret = 0; u32 range; u64 blkno; + u32 prev_depth = OCFS2_MAX_PATH_DEPTH; struct buffer_head *bh = NULL; struct ocfs2_extent_block *eb; struct ocfs2_extent_list *el; @@ -1851,14 +1852,16 @@ static int __ocfs2_find_path(struct ocfs2_caching_info *ci, el = root_el; while (el->l_tree_depth) { - if (unlikely(le16_to_cpu(el->l_tree_depth) >= OCFS2_MAX_PATH_DEPTH)) { + if (unlikely(le16_to_cpu(el->l_tree_depth) >= prev_depth)) { ocfs2_error(ocfs2_metadata_cache_get_super(ci), - "Owner %llu has invalid tree depth %u in extent list\n", + "Owner %llu has invalid tree depth %u in extent list (max %u)\n", (unsigned long long)ocfs2_metadata_cache_owner(ci), - le16_to_cpu(el->l_tree_depth)); + le16_to_cpu(el->l_tree_depth), prev_depth - 1); ret = -EROFS; goto out; } + prev_depth = le16_to_cpu(el->l_tree_depth); + if (!el->l_next_free_rec || !el->l_count) { ocfs2_error(ocfs2_metadata_cache_get_super(ci), "Owner %llu has empty extent list at depth %u\n" From cd340c54330c3207aad201d3832b9cd53335207d Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Wed, 23 Sep 2026 03:01:49 +0200 Subject: [PATCH 1318/1352] ocfs2/cluster: hold a reference on the heartbeat thread Since commit 688bc88e2046 ("ocfs2/cluster: keep heartbeat local node stable"), o2hb_thread() leaves its loop and returns when the local node changes, for example after "echo 0 > node//local". The thread was started with kthread_run() and nothing holds a reference to its task_struct, so the task is freed once it exits, while reg->hr_task still points to it. Reading the region's pid attribute then reads the freed task, and removing the region calls kthread_stop() on it: BUG: KASAN: slab-use-after-free in o2hb_region_pid_show+0xb3/0xc0 refcount_t: addition on 0; use-after-free. Oops: Oops: 0000 [#1] SMP KASAN NOPTI RIP: 0010:kthread_stop+0xb1/0x390 The thread could already return by itself before, when heartbeat start was aborted or on an unclean stop, but the local node change makes it reachable from userspace at any time. Create the thread, take a reference on it and only then wake it, so the reference cannot race with the thread exiting. Drop it with kthread_stop_put(). Link: https://lore.kernel.org/20260923010149.14391-1-kmehltretter@gmail.com Fixes: 688bc88e2046 ("ocfs2/cluster: keep heartbeat local node stable") Signed-off-by: Karl Mehltretter Signed-off-by: Andrew Morton Reported-by: Sashiko Closes: https://sashiko.dev/#/patchset/20260616074931.3774929-1-zzzccc427%40gmail.com Reviewed-by: Joseph Qi Assisted-by: LLM Cc: Mark Fasheh Cc: Joel Becker Cc: Cen Zhang Cc: Junxiao Bi Cc: Heming Zhao --- fs/ocfs2/cluster/heartbeat.c | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/fs/ocfs2/cluster/heartbeat.c b/fs/ocfs2/cluster/heartbeat.c index 1c3def99bb0765..a4c8ea695f5cf1 100644 --- a/fs/ocfs2/cluster/heartbeat.c +++ b/fs/ocfs2/cluster/heartbeat.c @@ -1966,18 +1966,22 @@ static ssize_t o2hb_region_dev_store(struct config_item *item, atomic_set(®->hr_unsteady_iterations, (live_threshold * 3)); o2hb_set_region_stopping(reg, false); - hb_task = kthread_run(o2hb_thread, reg, "o2hb-%s", - reg->hr_item.ci_name); + hb_task = kthread_create(o2hb_thread, reg, "o2hb-%s", + reg->hr_item.ci_name); if (IS_ERR(hb_task)) { ret = PTR_ERR(hb_task); mlog_errno(ret); goto out; } + /* The thread may exit on its own, so pin it before it can run. */ + get_task_struct(hb_task); spin_lock(&o2hb_live_lock); reg->hr_task = hb_task; spin_unlock(&o2hb_live_lock); + wake_up_process(hb_task); + ret = wait_event_interruptible(o2hb_steady_queue, atomic_read(®->hr_steady_iterations) == 0 || reg->hr_node_deleted); @@ -2022,7 +2026,7 @@ static ssize_t o2hb_region_dev_store(struct config_item *item, spin_unlock(&o2hb_live_lock); if (hb_task) - kthread_stop(hb_task); + kthread_stop_put(hb_task); o2hb_unmap_slot_data(reg); @@ -2208,7 +2212,7 @@ static void o2hb_heartbeat_group_drop_item(struct config_group *group, spin_unlock(&o2hb_live_lock); if (hb_task) - kthread_stop(hb_task); + kthread_stop_put(hb_task); if (o2hb_global_heartbeat_active()) { spin_lock(&o2hb_live_lock); From f7a1c270ff8e6c53fd08adaa1fc57be3f438b37e Mon Sep 17 00:00:00 2001 From: Guixin Liu Date: Thu, 24 Sep 2026 11:39:23 +0800 Subject: [PATCH 1319/1352] checkpatch: don't flag ACQUIRE_ERR() assignments in if conditions ACQUIRE_ERR() and its wrappers, PM_RUNTIME_ACQUIRE_ERR() and IIO_DEV_ACQUIRE_FAILED(), report whether a conditional cleanup.h guard was acquired, and drivers consume the result directly in an if condition: if ((rc = ACQUIRE_ERR(mutex_intr, &lock))) return rc; That combined form is the established style at the 49 in-tree call sites under drivers/cxl and drivers/pci/tsm.c, so ASSIGN_IN_IF fires there only as a false positive, and every patch touching those lines carries noise that reviewers have to wave off manually. Skip the check only when every assignment in the condition assigns the result of such a call, matched by the *_ACQUIRE_ERR() / *_ACQUIRE_FAILED() naming convention of its wrappers. Plain assignments, mixed conditions and near-miss identifiers still get flagged. Link: https://lore.kernel.org/20260924033923.4140210-1-kanie@linux.alibaba.com Signed-off-by: Guixin Liu Signed-off-by: Andrew Morton Suggested-by: Alison Schofield Acked-by: Joe Perches Assisted-by: LLM Cc: Andy Whitcroft Cc: Jonathan Cameron --- scripts/checkpatch.pl | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/scripts/checkpatch.pl b/scripts/checkpatch.pl index 3614cfe4dcbb45..598e3ed743bd9e 100755 --- a/scripts/checkpatch.pl +++ b/scripts/checkpatch.pl @@ -5787,7 +5787,21 @@ sub process { my ($s, $c) = ($stat, $cond); my $fixed_assign_in_if = 0; + # ACQUIRE_ERR() and its wrappers, e.g. PM_RUNTIME_ACQUIRE_ERR() + # and IIO_DEV_ACQUIRE_FAILED(), are meant to be evaluated in an + # if condition, with the error assigned in the condition: + # if ((rc = ACQUIRE_ERR(name, &lock))) + # Allow that only when every assignment in the condition assigns + # the result of such a call, so that a mixed condition keeps + # getting flagged: + # if ((rc = regular_function()) || (ret = ACQUIRE_ERR(name, &lock))) + my $assign_in_if = 0; if ($c =~ /\bif\s*\(.*[^<>!=]=[^=].*/s) { + my $has_assignment = $c =~ /\b$Lval\s*=\s*[^,)&|=]+/; + my $has_other_assignment = $c =~ /\b$Lval\s*=\s*(?!\s*\w*ACQUIRE_(?:ERR|FAILED)\s*\()[^,)&|=]+/; + $assign_in_if = !$has_assignment || $has_other_assignment; + } + if ($assign_in_if) { if (ERROR("ASSIGN_IN_IF", "do not use assignment in if condition\n" . $herecurr) && $fix && $perl_version_ok) { From f96a3fd61ce7be563413826b003f8bdf3e6e9e32 Mon Sep 17 00:00:00 2001 From: Andy Yan Date: Thu, 24 Sep 2026 18:50:33 +0800 Subject: [PATCH 1320/1352] mailmap: update entry for Andy Yan I will use andyshrk@163.com for future review and discussion. Link: https://lore.kernel.org/20260924105052.768760-1-andyshrk@163.com Signed-off-by: Andy Yan Signed-off-by: Andrew Morton Cc: Alexander Sverdlin Cc: Chuck Lever Cc: Jakub Kicinski Cc: Martin Kepplinger --- .mailmap | 1 + 1 file changed, 1 insertion(+) diff --git a/.mailmap b/.mailmap index 1a288972949802..3f9b79dac77fe8 100644 --- a/.mailmap +++ b/.mailmap @@ -103,6 +103,7 @@ Andy Chiu Andy Chiu Andy Shevchenko Andy Shevchenko +Andy Yan Anilkumar Kolli Anirudh Ghayal Antoine Tenart From 08597418a3b1d69f73a120c95173f21a10cb67d0 Mon Sep 17 00:00:00 2001 From: Danish Khateeb Date: Fri, 25 Sep 2026 15:55:02 -0500 Subject: [PATCH 1321/1352] kselftest/filelock: plan for the five ofdlocks tests ofdlocks reports five results but plans for four, so even when they all pass it ends with # Planned tests != run tests (4 != 5) and exits with KSFT_FAIL, which run_kselftest.sh reports as a failure. Link: https://lore.kernel.org/20260925205502.115327-1-danishkhateeb03@gmail.com Fixes: 33d5b13098fb ("kselftest/filelock: report each test in oftlocks separately") Signed-off-by: Danish Khateeb Signed-off-by: Andrew Morton Assisted-by: LLM Cc: Shuah Khan Cc: Mark Brown Cc: Jeff Layton --- tools/testing/selftests/filelock/ofdlocks.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tools/testing/selftests/filelock/ofdlocks.c b/tools/testing/selftests/filelock/ofdlocks.c index 68bac28b234b7d..0ab484cb075bc7 100644 --- a/tools/testing/selftests/filelock/ofdlocks.c +++ b/tools/testing/selftests/filelock/ofdlocks.c @@ -40,7 +40,7 @@ int main(void) int fd2 = open("/tmp/aa", O_RDONLY); ksft_print_header(); - ksft_set_plan(4); + ksft_set_plan(5); unlink("/tmp/aa"); assert(fd != -1); From 5817a47975f006731b464e8da2164b5780c9c014 Mon Sep 17 00:00:00 2001 From: Matthias Goergens Date: Mon, 28 Sep 2026 06:49:58 +0800 Subject: [PATCH 1322/1352] kdev_t: shift in dev_t in MKDEV() MKDEV(ma, mi) evaluates ((ma) << MINORBITS) in whatever type the caller passes, usually int. For majors >= 2048 (legal: majors go up to 4095) the result exceeds INT_MAX, which C11 leaves undefined for a signed left shift; shifting a negative ma is undefined regardless of the major's value. In practice, on the compilers and two's complement targets the kernel supports, both cases wrap to the intended bit pattern; this patch makes the macro well defined regardless by casting the major to dev_t before shifting. For major and minor numbers in range, the device number it produces is unchanged. isofs feeds the Rock Ridge 'PN' entry's dev_high and dev_low into MKDEV() while holding them in int variables (fs/isofs/rock.c), so a crafted image can drive this exact shift. Fixing the macro covers every caller instead of just that one. Link: https://lore.kernel.org/20260927224958.508964-1-matthias.goergens@gmail.com Signed-off-by: Matthias Goergens Signed-off-by: Andrew Morton Reviewed-by: Jan Kara Cc: Greg Kroah-Hartman --- include/linux/kdev_t.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/include/linux/kdev_t.h b/include/linux/kdev_t.h index 4856706fbfeb45..2dbbd47f1e68e1 100644 --- a/include/linux/kdev_t.h +++ b/include/linux/kdev_t.h @@ -9,7 +9,7 @@ #define MAJOR(dev) ((unsigned int) ((dev) >> MINORBITS)) #define MINOR(dev) ((unsigned int) ((dev) & MINORMASK)) -#define MKDEV(ma,mi) (((ma) << MINORBITS) | (mi)) +#define MKDEV(ma, mi) (((dev_t)(ma) << MINORBITS) | (mi)) #define print_dev_t(buffer, dev) \ sprintf((buffer), "%u:%u\n", MAJOR(dev), MINOR(dev)) From 3095e345ef24c28f97c035af6863bd30439380ed Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Sat, 26 Sep 2026 02:26:51 +0200 Subject: [PATCH 1323/1352] resource, kunit: stop selecting GET_FREE_REGION RESOURCE_KUNIT_TEST selects GET_FREE_REGION even when no other option needs it. Remove the selection to follow the dependency rule in Documentation/dev-tools/kunit/style.rst. Skip resource_test_region_intersects() when GET_FREE_REGION is disabled. GET_FREE_REGION has no prompt, so configurations without a production consumer cannot enable it. Most configurations will therefore skip this case; this is intentional. The union and intersection tests remain available. Since kunit_skip() does not return, the compiler drops the reference to the unavailable alloc_free_mem_region(), as in the CONFIG_OF_ADDRESS check in drivers/of/of_test.c. Link: https://lore.kernel.org/20260926002651.87267-1-kmehltretter@gmail.com Fixes: 99185c10d5d9 ("resource, kunit: add test case for region_intersects()") Signed-off-by: Karl Mehltretter Signed-off-by: Andrew Morton Reviewed-by: Bradley Morgan Tested-by: Bradley Morgan # Power10 Assisted-by: LLM Cc: Ying Huang Cc: Geert Uytterhoeven Cc: Brendan Higgins Cc: David Gow Cc: Rae Moar Cc: Andy Shevchenko --- kernel/resource_kunit.c | 3 +++ lib/Kconfig.debug | 1 - 2 files changed, 3 insertions(+), 1 deletion(-) diff --git a/kernel/resource_kunit.c b/kernel/resource_kunit.c index 42785796f1dbc5..9eeb2b1a85c01c 100644 --- a/kernel/resource_kunit.c +++ b/kernel/resource_kunit.c @@ -225,6 +225,9 @@ static void resource_test_region_intersects(struct kunit *test) struct resource *parent; resource_size_t start; + if (!IS_ENABLED(CONFIG_GET_FREE_REGION)) + kunit_skip(test, "CONFIG_GET_FREE_REGION is disabled"); + /* Find an iomem_resource hole to hold test resources */ parent = alloc_free_mem_region(&iomem_resource, RES_TEST_TOTAL_SIZE, SZ_1M, "test resources"); diff --git a/lib/Kconfig.debug b/lib/Kconfig.debug index c3f448f3b8f13c..56228eefdb4a9d 100644 --- a/lib/Kconfig.debug +++ b/lib/Kconfig.debug @@ -2776,7 +2776,6 @@ config RESOURCE_KUNIT_TEST tristate "KUnit test for resource API" if !KUNIT_ALL_TESTS depends on KUNIT default KUNIT_ALL_TESTS - select GET_FREE_REGION help This builds the resource API unit test. Tests the logic of API provided by resource.c and ioport.h. From d2be639adcf95f2a3a86d331b5e668549393f47d Mon Sep 17 00:00:00 2001 From: Jim Cromie Date: Tue, 29 Sep 2026 12:07:30 -0600 Subject: [PATCH 1324/1352] kallsyms: match compressed tokens on the fly during binary search Patch series "kallsyms: Accelerate symbol name lookups by ~7x", v7. In 2022, commit 60443c88f3a8 ("kallsyms: Improve the performance of kallsyms_lookup_name()") introduced kallsyms_seqs_of_names[] (+550 KiB .rodata), transforming an O(N) linear scan into an O(log N) binary search (5.2 ms -> ~7.2 us). While this was a major step forward, the binary search inner loop was left decompressing full candidate names and scanning across sparse 256:1 markers on every probe. Modern fleet observability, security daemons (e.g. CrowdStrike Falcon, Cilium, Datadog, Falco), and tracing tools resolve thousands of kernel functions by name at boot or service start. CrowdStrike recently hit this in production: commit 93e8fd1a565e ("ftrace: Use kallsyms binary search for single-symbol lookup") Attaching just 50 kprobe.session programs caused an 858 ms attach stall with 25% CPU burned in kallsyms. That commit routed single-symbol libbpf attach directly to kallsyms_lookup_name(). In larger workloads (such as the BPF selftest serial_test_kprobe_multi_bench_attach across 64,000 symbols), kallsyms_lookup_names() spends ~390 ms in raw CPU spin. This series accelerates kallsyms_lookup_names() by 7.0x (from 6,102 ns down to 866 ns per lookup), cutting 64k-symbol attach from ~390 ms to ~55 ms, by fixing two inner-loop bottlenecks: 0. Candidate symbols are fully decompressed into a 512-byte stack buffer before calling strcmp(), even though ~16 of the 17 search steps mismatch at the first differing character (0..N-1, heavily front-loaded toward 0-2). 1. Probes scan sequentially from 256:1 markers in kallsyms_names[], decoding an average of 127.5 symbols per probe (~2,170 hops across a 17-step search). The 3-patch progression: 0. Patch 1 introduces kallsyms_strcmp_symbol() to compare ASCII queries against compressed tokens on the fly, bailing out on first mismatch. Drops the 512-byte stack buffer and saves ~530 ns. 1. Patch 2 increases marker density from 256:1 to 16:1, cutting average scan distance from 127.5 to 7.5 hops and dropping lookup latency from 6,102 ns to 866 ns for +42.2 KiB of .rodata. 2. Patch 3 inlines and unrolls get_symbol_seq() 24-bit reconstruction. Results (CONFIG_KALLSYMS_SELFTEST across ~184k symbols): - Baseline (256:1): 6,102 ns - Patch 1 (strcmp): 5,572 ns (-530 ns) - Patch 2 (16:1): 866 ns (7.0x faster) Trade-offs: - .rodata footprint: +42.2 KiB (+10,782 u32 entries for ~184k symbols, ~0.1% of loaded kernel image). - Runtime overhead: 0 bytes dynamic RAM (no kmalloc/kvmalloc), 0 RCU, 0 new locks, and 0 new algorithms. - Kernel stack: -512 bytes freed in kallsyms_lookup_names(). - Build tooling: scripts/kallsyms.c includes ../kernel/kallsyms_internal.h (guarded by #ifdef __KERNEL__) so host build and kernel runtime share KALLSYMS_MARKER_SHIFT 4 as a single source of truth. - Build time: Unmeasurable delta (< 1 ms in scripts/kallsyms.c). What's Unchanged: - Symbol table layout in address order remains identical. - Streaming decompression for /proc/kallsyms and sprint_symbol() is untouched. - 0 new user-facing APIs, 0 new locking primitives, 0 Kconfig options. This patch (of 3): kallsyms_lookup_names() runs a binary search across ~184k tokenized (compressed) symbols. For each of the ~17 comparisons in the search, it currently decompresses the candidate symbol into a temporary buffer on the stack before calling strcmp(). Comparing tokenized symbols directly in compressed space is impossible. The BPE token table assigns values by frequency, not alphabetical order (e.g. token 0x05 might expand to "zebra" while 0x42 expands to "apple"), so comparing raw token values scrambles lexicographical order. Even sorting the token table alphabetically wouldn't help; "bpf_" and "bpf_foo_" do not *have* a determinative sorting order, because the suffixes following those tokens would matter. However, full string expansion at every step is equally wasteful: of the ~17 strcmps in the binary search, only the last needs to check all N chars in both strings, earlier steps will know +/- outcome at char 0,1,2..N-1. However, full string expansion at every step is equally wasteful: of the ~17 strcmp()s in the binary search, only the final matching step needs to test all characters. Earlier non-matching steps diverge at the first differing character (0..N-1), but the baseline expands every candidate symbol to the stack unconditionally, before comparing. So we introduce kallsyms_strcmp_symbol() to compare ASCII search_name against tokenized symbols on the fly. Like strcmp, it tests the strings char by char, but when it hits a token in the symbol-string, it continues the char-test against that token-string, which is in kallsyms_token_table[]. It returns +- on 1st mismatch. Measured across all ~184k symbols via CONFIG_KALLSYMS_SELFTEST, this shaves ~530 ns (~14%) off average kallsyms_lookup_name() latency (from ~3810 ns to ~3280 ns on the default 256:1 baseline) and drops the 512-byte namebuf buffer stack-alloc in kallsyms_lookup_names(). Link: https://lore.kernel.org/20260929-ksyms-tune-v7-0-be568ceef41e@gmail.com Link: https://lore.kernel.org/20260929-ksyms-tune-v7-1-be568ceef41e@gmail.com Signed-off-by: Jim Cromie Signed-off-by: Andrew Morton Reviewed-by: Kees Cook Cc: Petr Mladek Cc: Zhen Lei Cc: Luis Chamberlain Cc: Andrey Grodzovsky Cc: Steven Rostedt Cc: Lorenzo Stoakes Cc: David Laight Cc: Masahiro Yamada Cc: Jiri Olsa Cc: Geert Uytterhoeven --- kernel/kallsyms.c | 94 +++++++++++++++++++++++++++++------------------ 1 file changed, 59 insertions(+), 35 deletions(-) diff --git a/kernel/kallsyms.c b/kernel/kallsyms.c index aec2f06858afdb..d18d78e626db26 100644 --- a/kernel/kallsyms.c +++ b/kernel/kallsyms.c @@ -34,6 +34,21 @@ #include "kallsyms_internal.h" +/* + * Get the compressed symbol length and data pointer. + */ +static inline const u8 *get_symbol_data(unsigned int off, unsigned int *len) +{ + const u8 *p = &kallsyms_names[off]; + unsigned int l = *p++; + + if (unlikely(l & 0x80)) + l = (l & 0x7F) | (*p++ << 7); + *len = l; + + return p; +} + /* * Expand a compressed symbol data into the resulting uncompressed string, * if uncompressed string is too long (>= maxlen), it will be truncated, @@ -42,28 +57,12 @@ static unsigned int kallsyms_expand_symbol(unsigned int off, char *result, size_t maxlen) { - int len, skipped_first = 0; + int skipped_first = 0; const char *tptr; - const u8 *data; + unsigned int len; + const u8 *data = get_symbol_data(off, &len); - /* Get the compressed symbol length from the first symbol byte. */ - data = &kallsyms_names[off]; - len = *data; - data++; - off++; - - /* If MSB is 1, it is a "big" symbol, so needs an additional byte. */ - if ((len & 0x80) != 0) { - len = (len & 0x7F) | (*data << 7); - data++; - off++; - } - - /* - * Update the offset to return the offset for the next symbol on - * the compressed stream. - */ - off += len; + off = (data - kallsyms_names) + len; /* * For every byte on the compressed symbol data, copy the table @@ -101,14 +100,43 @@ static unsigned int kallsyms_expand_symbol(unsigned int off, */ static char kallsyms_get_symbol_type(unsigned int off) { - /* - * Get just the first code, look it up in the token table, - * and return the first char from this token. If MSB of length - * is 1, it is a "big" symbol, so needs an additional byte. - */ - if (kallsyms_names[off] & 0x80) - off++; - return kallsyms_token_table[kallsyms_token_index[kallsyms_names[off + 1]]]; + unsigned int len; + const u8 *data = get_symbol_data(off, &len); + + return kallsyms_token_table[kallsyms_token_index[*data]]; +} + +/* + * Compare an uncompressed ASCII string against a compressed symbol table entry. + * Returns negative if name < sym, positive if name > sym, 0 if equal. + * Exits immediately on the first mismatched character without decompressing + * the rest of the symbol name. + */ +static int kallsyms_strcmp_symbol(unsigned int off, const char *name) +{ + const char *tptr; + unsigned int len; + const u8 *data = get_symbol_data(off, &len); + + tptr = &kallsyms_token_table[kallsyms_token_index[*data++]] + 1; + while (*tptr) { + int diff = (unsigned char)*name++ - (unsigned char)*tptr++; + + if (diff) + return diff; + } + + while (--len) { + tptr = &kallsyms_token_table[kallsyms_token_index[*data++]]; + do { + int diff = (unsigned char)*name++ - (unsigned char)*tptr++; + + if (diff) + return diff; + } while (*tptr); + } + + return (unsigned char)*name; } @@ -174,7 +202,6 @@ static int kallsyms_lookup_names(const char *name, int ret; int low, mid, high; unsigned int seq, off; - char namebuf[KSYM_NAME_LEN]; low = 0; high = kallsyms_num_syms - 1; @@ -183,8 +210,7 @@ static int kallsyms_lookup_names(const char *name, mid = low + (high - low) / 2; seq = get_symbol_seq(mid); off = get_symbol_offset(seq); - kallsyms_expand_symbol(off, namebuf, ARRAY_SIZE(namebuf)); - ret = strcmp(name, namebuf); + ret = kallsyms_strcmp_symbol(off, name); if (ret > 0) low = mid + 1; else if (ret < 0) @@ -200,8 +226,7 @@ static int kallsyms_lookup_names(const char *name, while (low) { seq = get_symbol_seq(low - 1); off = get_symbol_offset(seq); - kallsyms_expand_symbol(off, namebuf, ARRAY_SIZE(namebuf)); - if (strcmp(name, namebuf)) + if (kallsyms_strcmp_symbol(off, name) != 0) break; low--; } @@ -212,8 +237,7 @@ static int kallsyms_lookup_names(const char *name, while (high < kallsyms_num_syms - 1) { seq = get_symbol_seq(high + 1); off = get_symbol_offset(seq); - kallsyms_expand_symbol(off, namebuf, ARRAY_SIZE(namebuf)); - if (strcmp(name, namebuf)) + if (kallsyms_strcmp_symbol(off, name) != 0) break; high++; } From 4fafd1165b33b42bef5059df9c76531996e63a96 Mon Sep 17 00:00:00 2001 From: Jim Cromie Date: Tue, 29 Sep 2026 12:07:31 -0600 Subject: [PATCH 1325/1352] kallsyms: increase marker density to 16:1 to accelerate lookups kallsyms stores symbols with remarkably efficient packing, and simple streaming unpacking, laid out sequentially in address order. That said, variable-length records make arbitrary access inherently linear. kallsyms_markers[] addressed this by marking stream offsets every 256 symbols, reducing the scan distance by 256x down to an average of 127.5 sequential steps. While 127.5 hops was negligible for rare, single-shot oops backtraces, both table size (~184k symbols) and lookup traffic have expanded substantially. In alphabetical binary search (kallsyms_lookup_names), each of the ~17 comparison probes must locate candidate symbols via get_symbol_offset(), compounding into ~2,170 sequential symbol hops per lookup. In bulk tracing workloads (such as BPF multi-kprobe attach), this penalty compounds into multi-second latency. Without altering the underlying storage layout, we can retune this trade-off directly by increasing marker density from 256:1 down to 16:1 (KALLSYMS_MARKER_SHIFT 4) in kernel/kallsyms_internal.h, shared between scripts/kallsyms.c and kernel/kallsyms.c. This caps the remainder scan at 15 symbols and cuts average scan distance from 127.5 down to 7.5 hops (a 17x reduction). Across a 17-step binary search, total hops collapse from ~2,170 down to ~127. For a kernel with ~184,000 symbols, this adds ~10,800 u32 marker entries (+42 KiB) to write-protected .rodata. In-tree CONFIG_KALLSYMS_SELFTEST measurements across all ~184k symbols show average lookup latency dropping from 6,102 ns down to 866 ns (a 7.0x speedup). Link: https://lore.kernel.org/20260929-ksyms-tune-v7-2-be568ceef41e@gmail.com Signed-off-by: Jim Cromie Signed-off-by: Andrew Morton Reviewed-by: Kees Cook Cc: Petr Mladek Cc: Zhen Lei Cc: Luis Chamberlain Cc: Andrey Grodzovsky Cc: Steven Rostedt Cc: Lorenzo Stoakes Cc: David Laight Cc: Masahiro Yamada Cc: Jiri Olsa Cc: Geert Uytterhoeven --- kernel/kallsyms.c | 8 ++++---- kernel/kallsyms_internal.h | 11 +++++++++++ scripts/kallsyms.c | 14 +++++++++----- 3 files changed, 24 insertions(+), 9 deletions(-) diff --git a/kernel/kallsyms.c b/kernel/kallsyms.c index d18d78e626db26..91ced7aa797eb7 100644 --- a/kernel/kallsyms.c +++ b/kernel/kallsyms.c @@ -150,10 +150,10 @@ static unsigned int get_symbol_offset(unsigned long pos) int i, len; /* - * Use the closest marker we have. We have markers every 256 positions, - * so that should be close enough. + * Use the closest marker we have. We have markers every + * (1 << KALLSYMS_MARKER_SHIFT) positions, so that should be close enough. */ - name = &kallsyms_names[kallsyms_markers[pos >> 8]]; + name = &kallsyms_names[kallsyms_markers[pos >> KALLSYMS_MARKER_SHIFT]]; /* * Sequentially scan all the symbols up to the point we're searching @@ -161,7 +161,7 @@ static unsigned int get_symbol_offset(unsigned long pos) * so we just need to add the len to the current pointer for every * symbol we wish to skip. */ - for (i = 0; i < (pos & 0xFF); i++) { + for (i = 0; i < (pos & KALLSYMS_MARKER_MASK); i++) { len = *name; /* diff --git a/kernel/kallsyms_internal.h b/kernel/kallsyms_internal.h index 81a867dbe57d48..6a781e4cc77feb 100644 --- a/kernel/kallsyms_internal.h +++ b/kernel/kallsyms_internal.h @@ -2,6 +2,16 @@ #ifndef LINUX_KALLSYMS_INTERNAL_H_ #define LINUX_KALLSYMS_INTERNAL_H_ +/* + * Provide compile-constants for scripts/kallsyms.c + * so it can build the corresponding kallsyms_marker[] table. + * and wrap the rest in __KERNEL__ + */ +#define KALLSYMS_MARKER_SHIFT 4 +#define KALLSYMS_MARKER_SIZE (1U << KALLSYMS_MARKER_SHIFT) +#define KALLSYMS_MARKER_MASK (KALLSYMS_MARKER_SIZE - 1U) + +#ifdef __KERNEL__ #include extern const int kallsyms_offsets[]; @@ -14,5 +24,6 @@ extern const u16 kallsyms_token_index[]; extern const unsigned int kallsyms_markers[]; extern const u8 kallsyms_seqs_of_names[]; +#endif /* __KERNEL__ */ #endif // LINUX_KALLSYMS_INTERNAL_H_ diff --git a/scripts/kallsyms.c b/scripts/kallsyms.c index 494852ade6d87a..be42a911135007 100644 --- a/scripts/kallsyms.c +++ b/scripts/kallsyms.c @@ -29,6 +29,8 @@ #include +#include "../kernel/kallsyms_internal.h" + #define ARRAY_SIZE(arr) (sizeof(arr) / sizeof(arr[0])) #define KSYM_NAME_LEN 512 @@ -349,16 +351,18 @@ static void write_src(void) printf("\t.long\t%u\n", table_cnt); printf("\n"); - /* table of offset markers, that give the offset in the compressed stream - * every 256 symbols */ - markers_cnt = (table_cnt + 255) / 256; + /* + * Table of offset markers, giving the offset in the compressed stream + * every (1 << KALLSYMS_MARKER_SHIFT) symbols. + */ + markers_cnt = (table_cnt + KALLSYMS_MARKER_MASK) >> KALLSYMS_MARKER_SHIFT; markers = xmalloc(sizeof(*markers) * markers_cnt); output_label("kallsyms_names"); off = 0; for (i = 0; i < table_cnt; i++) { - if ((i & 0xFF) == 0) - markers[i >> 8] = off; + if ((i & KALLSYMS_MARKER_MASK) == 0) + markers[i >> KALLSYMS_MARKER_SHIFT] = off; table[i]->seq = i; /* There cannot be any symbol of length zero. */ From e18d65bccfd1ecae5e836340fc83cc6077749dfa Mon Sep 17 00:00:00 2001 From: Jim Cromie Date: Tue, 29 Sep 2026 12:07:32 -0600 Subject: [PATCH 1326/1352] kallsyms: unroll 24-bit sequence reconstruction in get_symbol_seq() kallsyms_seqs_of_names[] stores 3-byte big-endian sequence indices that map alphabetical symbol positions to address-ordered symbol records. Currently, get_symbol_seq() reconstructs each 24-bit integer using a 3-iteration for-loop that shifts and bitwise-ORs each byte sequentially. During binary search in kallsyms_lookup_names() and duplicate boundary scans, this loop introduces branch and loop overhead on the hot lookup path. Mark get_symbol_seq() as static inline and unroll the 3-byte extraction into direct byte shifts: (p[0] << 16) | (p[1] << 8) | p[2]. This eliminates loop induction variable maintenance and allows the compiler to generate direct loads and constant shifts. Link: https://lore.kernel.org/20260929-ksyms-tune-v7-3-be568ceef41e@gmail.com Signed-off-by: Jim Cromie Signed-off-by: Andrew Morton Reviewed-by: Kees Cook Cc: Petr Mladek Cc: Zhen Lei Cc: Luis Chamberlain Cc: Andrey Grodzovsky Cc: Steven Rostedt Cc: Lorenzo Stoakes Cc: David Laight Cc: Masahiro Yamada Cc: Jiri Olsa Cc: Geert Uytterhoeven --- kernel/kallsyms.c | 9 +++------ 1 file changed, 3 insertions(+), 6 deletions(-) diff --git a/kernel/kallsyms.c b/kernel/kallsyms.c index 91ced7aa797eb7..52e41879c24bda 100644 --- a/kernel/kallsyms.c +++ b/kernel/kallsyms.c @@ -185,14 +185,11 @@ unsigned long kallsyms_sym_address(int idx) return (unsigned long)offset_to_ptr(kallsyms_offsets + idx); } -static unsigned int get_symbol_seq(int index) +static inline unsigned int get_symbol_seq(int index) { - unsigned int i, seq = 0; + const u8 *p = &kallsyms_seqs_of_names[3 * index]; - for (i = 0; i < 3; i++) - seq = (seq << 8) | kallsyms_seqs_of_names[3 * index + i]; - - return seq; + return (p[0] << 16) | (p[1] << 8) | p[2]; } static int kallsyms_lookup_names(const char *name, From 1eb2884295decb6fc9cdc934a9dc3b48043ee874 Mon Sep 17 00:00:00 2001 From: Yuntao Wang Date: Wed, 30 Sep 2026 09:57:48 +0800 Subject: [PATCH 1327/1352] init: simplify initramfs.o build rule CONFIG_BLK_DEV_INITRD is known to be enabled in the else branch, so there is no need to use obj-$(CONFIG_BLK_DEV_INITRD). Use obj-y directly to make the build rule clearer. Also, reorder the condition to make the code easier to follow. No functional change. Link: https://lore.kernel.org/20260930015748.311366-1-yuntao.wang@linux.dev Signed-off-by: Yuntao Wang Signed-off-by: Andrew Morton Reviewed-by: Nicolas Schier Cc: Nathan Chancellor --- init/Makefile | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/init/Makefile b/init/Makefile index d6f75d8907e098..9102fc34c0cc0f 100644 --- a/init/Makefile +++ b/init/Makefile @@ -6,10 +6,10 @@ ccflags-y := -fno-function-sections -fno-data-sections obj-y := main.o version.o mounts.o -ifneq ($(CONFIG_BLK_DEV_INITRD),y) -obj-y += noinitramfs.o +ifeq ($(CONFIG_BLK_DEV_INITRD),y) +obj-y += initramfs.o else -obj-$(CONFIG_BLK_DEV_INITRD) += initramfs.o +obj-y += noinitramfs.o endif obj-$(CONFIG_GENERIC_CALIBRATE_DELAY) += calibrate.o obj-$(CONFIG_INITRAMFS_TEST) += initramfs_test.o From 5e76e144b21113cb4823bef6d8cce44c52111dff Mon Sep 17 00:00:00 2001 From: Konstantin Khorenko Date: Thu, 1 Oct 2026 18:57:11 +0200 Subject: [PATCH 1328/1352] kbuild: use $(CFLAGS_GCOV) in the prefer-atomic try-run test Patch series "kbuild: GCOV cleanups following the prefer-atomic fix". Two cleanups Peter Oberparleiter suggested while reviewing commit 56cb9b7d96b2 ("gcov: use atomic counter updates to fix concurrent access crashes"): https://lore.kernel.org/e1b6e768-51af-468b-a729-0ed3fceef2f1@linux.ibm.com Patch 1 makes the prefer-atomic try-run test reuse $(CFLAGS_GCOV) instead of spelling the flags out again, so the test compiles with the exact GCOV flags, -fno-tree-loop-im included. Patch 2 moves the CFLAGS_GCOV block into scripts/Makefile.gcov, included via include-$(CONFIG_GCOV_KERNEL) like the other instrumentation makefiles. The try-run also stops running on builds without GCOV. This patch (of 2): Reference $(CFLAGS_GCOV) in the try-run test introduced by commit 56cb9b7d96b2 ("gcov: use atomic counter updates to fix concurrent access crashes"), otherwise it can miss potential future changes in the list of flags used for GCOV profiling like it misses -fno-tree-loop-im at the moment. The outcome of the test is unchanged: -fno-tree-loop-im does not affect the undefined symbols the test compares. Link: https://lore.kernel.org/20261001-b4-prep-gcov-v1-0-794792e06f24@virtuozzo.com Link: https://lore.kernel.org/20261001-b4-prep-gcov-v1-1-794792e06f24@virtuozzo.com Signed-off-by: Konstantin Khorenko Signed-off-by: Andrew Morton Suggested-by: Peter Oberparleiter Cc: Nathan Chancellor Cc: Nicolas Schier Cc: Masahiro Yamada Cc: Arnd Bergmann Cc: Mikhail Zaslonko --- Makefile | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/Makefile b/Makefile index 6b8b812f23b928..3b69b973f7598f 100644 --- a/Makefile +++ b/Makefile @@ -1184,11 +1184,11 @@ CFLAGS_GCOV += -fno-tree-loop-im # undefined symbols (e.g. libatomic calls that the kernel cannot link). CFLAGS_GCOV += $(call try-run,\ echo 'long long x; void f(void){x++;}' | \ - $(CC) $(KBUILD_CPPFLAGS) $(KBUILD_CFLAGS) -w -fprofile-arcs \ - -ftest-coverage -x c - -c -o "$$TMP.base" && \ + $(CC) $(KBUILD_CPPFLAGS) $(KBUILD_CFLAGS) -w $(CFLAGS_GCOV) \ + -x c - -c -o "$$TMP.base" && \ echo 'long long x; void f(void){x++;}' | \ - $(CC) $(KBUILD_CPPFLAGS) $(KBUILD_CFLAGS) -w -fprofile-arcs \ - -ftest-coverage -fprofile-update=prefer-atomic \ + $(CC) $(KBUILD_CPPFLAGS) $(KBUILD_CFLAGS) -w $(CFLAGS_GCOV) \ + -fprofile-update=prefer-atomic \ -x c - -c -o "$$TMP" && \ $(NM) "$$TMP.base" | grep ' U ' > "$$TMP.ubase" || true ; \ $(NM) "$$TMP" | grep ' U ' > "$$TMP.utest" || true ; \ From 20e59dfe656fbe716b984d3662fd5bf9bd50b331 Mon Sep 17 00:00:00 2001 From: Konstantin Khorenko Date: Thu, 1 Oct 2026 18:57:12 +0200 Subject: [PATCH 1329/1352] kbuild: move GCOV flags to scripts/Makefile.gcov * split the CFLAGS_GCOV block out of the top-level Makefile into scripts/Makefile.gcov * include it via include-$(CONFIG_GCOV_KERNEL) like neighbour KASAN/KCOV/UBSAN makefiles do * keep the include before scripts/Makefile.gcc-plugins: the try-run must not get -fplugin= options, plugins do not exist on clean builds CFLAGS_GCOV is no longer defined when CONFIG_GCOV_KERNEL is off: * its user in scripts/Makefile.lib is guarded by the CONFIG_GCOV_KERNEL * in i915/xe header an empty value is harmless The try-run compiler invocations also stop running on builds without GCOV. Link: https://lore.kernel.org/20261001-b4-prep-gcov-v1-2-794792e06f24@virtuozzo.com Signed-off-by: Konstantin Khorenko Signed-off-by: Andrew Morton Suggested-by: Peter Oberparleiter Cc: Nathan Chancellor Cc: Nicolas Schier Cc: Masahiro Yamada Cc: Arnd Bergmann Cc: Mikhail Zaslonko --- MAINTAINERS | 1 + Makefile | 22 +--------------------- scripts/Makefile.gcov | 21 +++++++++++++++++++++ 3 files changed, 23 insertions(+), 21 deletions(-) create mode 100644 scripts/Makefile.gcov diff --git a/MAINTAINERS b/MAINTAINERS index 67c42e55d60afd..72ba8a36bc2ab2 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -10898,6 +10898,7 @@ M: Peter Oberparleiter S: Maintained F: Documentation/dev-tools/gcov.rst F: kernel/gcov/ +F: scripts/Makefile.gcov GDB KERNEL DEBUGGING HELPER SCRIPTS M: Jan Kiszka diff --git a/Makefile b/Makefile index 3b69b973f7598f..f3ea94ceb1cece 100644 --- a/Makefile +++ b/Makefile @@ -1176,27 +1176,6 @@ endif # Ensure compilers do not transform certain loops into calls to wcslen() KBUILD_CFLAGS += -fno-builtin-wcslen -CFLAGS_GCOV := -fprofile-arcs -ftest-coverage -ifdef CONFIG_CC_IS_GCC -CFLAGS_GCOV += -fno-tree-loop-im -# Use atomic counter updates to avoid concurrent-access crashes in GCOV. -# Only enable if -fprofile-update=prefer-atomic does not introduce new -# undefined symbols (e.g. libatomic calls that the kernel cannot link). -CFLAGS_GCOV += $(call try-run,\ - echo 'long long x; void f(void){x++;}' | \ - $(CC) $(KBUILD_CPPFLAGS) $(KBUILD_CFLAGS) -w $(CFLAGS_GCOV) \ - -x c - -c -o "$$TMP.base" && \ - echo 'long long x; void f(void){x++;}' | \ - $(CC) $(KBUILD_CPPFLAGS) $(KBUILD_CFLAGS) -w $(CFLAGS_GCOV) \ - -fprofile-update=prefer-atomic \ - -x c - -c -o "$$TMP" && \ - $(NM) "$$TMP.base" | grep ' U ' > "$$TMP.ubase" || true ; \ - $(NM) "$$TMP" | grep ' U ' > "$$TMP.utest" || true ; \ - cmp -s "$$TMP.ubase" "$$TMP.utest",\ - -fprofile-update=prefer-atomic) -endif -export CFLAGS_GCOV - # change __FILE__ to the relative path to the source directory ifdef building_out_of_srctree KBUILD_CPPFLAGS += -fmacro-prefix-map=$(srcroot)/= @@ -1214,6 +1193,7 @@ include-$(CONFIG_KCSAN) += scripts/Makefile.kcsan include-$(CONFIG_KMSAN) += scripts/Makefile.kmsan include-$(CONFIG_UBSAN) += scripts/Makefile.ubsan include-$(CONFIG_KCOV) += scripts/Makefile.kcov +include-$(CONFIG_GCOV_KERNEL) += scripts/Makefile.gcov include-$(CONFIG_RANDSTRUCT) += scripts/Makefile.randstruct include-$(CONFIG_KSTACK_ERASE) += scripts/Makefile.kstack_erase include-$(CONFIG_AUTOFDO_CLANG) += scripts/Makefile.autofdo diff --git a/scripts/Makefile.gcov b/scripts/Makefile.gcov new file mode 100644 index 00000000000000..16c5b876440c24 --- /dev/null +++ b/scripts/Makefile.gcov @@ -0,0 +1,21 @@ +# SPDX-License-Identifier: GPL-2.0-only +CFLAGS_GCOV := -fprofile-arcs -ftest-coverage +ifdef CONFIG_CC_IS_GCC +CFLAGS_GCOV += -fno-tree-loop-im +# Use atomic counter updates to avoid concurrent-access crashes in GCOV. +# Only enable if -fprofile-update=prefer-atomic does not introduce new +# undefined symbols (e.g. libatomic calls that the kernel cannot link). +CFLAGS_GCOV += $(call try-run,\ + echo 'long long x; void f(void){x++;}' | \ + $(CC) $(KBUILD_CPPFLAGS) $(KBUILD_CFLAGS) -w $(CFLAGS_GCOV) \ + -x c - -c -o "$$TMP.base" && \ + echo 'long long x; void f(void){x++;}' | \ + $(CC) $(KBUILD_CPPFLAGS) $(KBUILD_CFLAGS) -w $(CFLAGS_GCOV) \ + -fprofile-update=prefer-atomic \ + -x c - -c -o "$$TMP" && \ + $(NM) "$$TMP.base" | grep ' U ' > "$$TMP.ubase" || true ; \ + $(NM) "$$TMP" | grep ' U ' > "$$TMP.utest" || true ; \ + cmp -s "$$TMP.ubase" "$$TMP.utest",\ + -fprofile-update=prefer-atomic) +endif +export CFLAGS_GCOV From cf60aac3d7acacaa54d634de81ea72aa4a655f74 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=EC=84=B1=EB=B3=91=EC=B0=AC?= Date: Wed, 30 Sep 2026 13:18:01 +0900 Subject: [PATCH 1330/1352] KEYS: Fix add_key() race with keyring restriction MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit __key_create_or_update() snapshots keyring->restrict_link before taking the destination keyring's semaphore. keyring_restrict() installs a restriction while holding that semaphore. This allows a writer to observe no restriction, wait for the keyring owner to install a reject-all restriction and return successfully, and then link a key using the stale NULL snapshot. The writer only needs write permission on the destination keyring. Move the restrict_link read after __key_link_lock() and __key_link_begin(). The read and the subsequent restriction check are then serialized with restriction installation by keyring->sem. The race was reproduced on v7.2.8 in 19 executions where restriction installation returned before the link completed. All 19 linked the key despite the reject-all restriction. With this change, 312 executions reached the same ordering and every add_key() call failed with -EPERM. The issue was found by manual concurrency analysis assisted by AI-based analysis and independently verified with a QEMU reproducer and kernel instrumentation. Fixes: 5ac7eace2d00 ("KEYS: Add a facility to restrict new links into a keyring") Cc: stable@vger.kernel.org Signed-off-by: 성병찬 Reviewed-by: Jarkko Sakkinen Link: https://lore.kernel.org/r/20260930041802.6114-1-tjdqudcks0424@naver.com Signed-off-by: Jarkko Sakkinen --- security/keys/key.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/security/keys/key.c b/security/keys/key.c index b34a64d81d47ab..a438c4508595d9 100644 --- a/security/keys/key.c +++ b/security/keys/key.c @@ -840,9 +840,6 @@ static key_ref_t __key_create_or_update(key_ref_t keyring_ref, key_check(keyring); - if (!(flags & KEY_ALLOC_BYPASS_RESTRICTION)) - restrict_link = keyring->restrict_link; - key_ref = ERR_PTR(-ENOTDIR); if (keyring->type != &key_type_keyring) goto error_put_type; @@ -880,6 +877,9 @@ static key_ref_t __key_create_or_update(key_ref_t keyring_ref, goto error_link_end; } + if (!(flags & KEY_ALLOC_BYPASS_RESTRICTION)) + restrict_link = keyring->restrict_link; + if (restrict_link && restrict_link->check) { ret = restrict_link->check(keyring, index_key.type, &prep.payload, restrict_link->key); From c1b88c88ff65353a5432250319dc8e200ba52e95 Mon Sep 17 00:00:00 2001 From: Arnd Bergmann Date: Fri, 2 Oct 2026 23:44:33 +0200 Subject: [PATCH 1331/1352] Merge tag 'v7.3-rc5' into for-next Linux 7.3-rc5 * tag 'v7.3-rc5': (1731 commits) Linux 7.3-rc5 workqueue: Fix NULL current_pwq deref in flush dependency check sched_ext: Add a size argument to scx_bpf_cid_topo() so struct scx_cid_topo can grow cgroup/cpuset: Return PERR_NOCPUS in remote_partition_enable() on subpartitions_cpus conflict KVM: SEV: Do cache maintenance on the source VM during intra-host migration KVM: SEV: Free have_run_cpus during VM destruction even if VM is no longer SEV ipe: protect the dm-verity root hash with RCU ipe: fix use-after-free when auditing a newly loaded policy netfs: Fix missing alloc tagging of direct mempool allocations bpf: fs/xattr: don't assume the inode is locked in path_unlink/path_rmdir autofs: fix sbi->pipe file reference leak in autofs_kill_sb() kprobes: Fix permanent hang when flushing the kprobe optimizer dcache: unpoison the inline name buffer in __d_alloc() ovl: fix UAF in ovl_do_mkdir() debug print MAINTAINERS: name the libata/linux for-next branch i2c: qcom-geni: Fix hardcoded clock index in SE_GENI_CLK_SEL tcp: prevent collapsing skbs across boundary in rtx queue vlan: ensure sufficient headroom in vlan_dev_hard_header() net/sched: sch_teql: fix shadowed err in __teql_resolve() bridge: check llc_mac_hdr_init() return value in br_send_bpdu() ... Signed-off-by: Arnd Bergmann From bdb122e397cb06a0235039f1b10560712526a137 Mon Sep 17 00:00:00 2001 From: Arnd Bergmann Date: Fri, 2 Oct 2026 23:46:07 +0200 Subject: [PATCH 1332/1352] soc: document merges Signed-off-by: Arnd Bergmann --- arch/arm/arm-soc-for-next-contents.txt | 23 +++++++++++++---------- 1 file changed, 13 insertions(+), 10 deletions(-) diff --git a/arch/arm/arm-soc-for-next-contents.txt b/arch/arm/arm-soc-for-next-contents.txt index 381938ff5b74bd..fc231be0c5ca1d 100644 --- a/arch/arm/arm-soc-for-next-contents.txt +++ b/arch/arm/arm-soc-for-next-contents.txt @@ -18,6 +18,8 @@ soc/arm soc/dt renesas/dt https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel tags/renesas-dts-for-v7.4-tag1 + contains renesas/fixes-2 + contains renesas/fixes rockchip/dt64 ssh://gitolite.kernel.org/pub/scm/linux/kernel/git/mmind/linux-rockchip tags/v7.4-rockchip-dts64-1 qcom/dt64 @@ -38,6 +40,7 @@ soc/drivers ssh://gitolite.kernel.org/pub/scm/linux/kernel/git/mediatek/linux tags/mtk-soc-for-v7.4 scmi/drivers https://git.kernel.org/pub/scm/linux/kernel/git/sudeep.holla/linux tags/scmi-updates-7.4 + contains scmi/fixes soc/defconfig renesas/defconfig @@ -48,14 +51,14 @@ soc/defconfig soc/late arm/fixes - (cfc1e9a543e3589ba200795b6e7fd8ef4314efdf) - git://git.kernel.org/pub/scm/linux/kernel/git/dinguyen/linux tags/socfpga_fix_for_v7.3 - socfpga/fix - git://git.kernel.org/pub/scm/linux/kernel/git/dinguyen/linux tags/socfpga_dts_fix_for_v7.3 - renesas/fixes - git://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel tags/renesas-fixes-for-v7.3-tag1 - scmi/fixes - git://git.kernel.org/pub/scm/linux/kernel/git/sudeep.holla/linux tags/scmi-ffa-fixes-7.3 - (406292fd75f95aa3010fec95b5beb5a8b7e3ba3a) - https://git.kernel.org/pub/scm/linux/kernel/git/amlogic/linux tags/amlogic-fixes-v7.3-rc + (55981579b27b6541aeeab81c768f248dba0491ac) + https://git.kernel.org/pub/scm/linux/kernel/git/qcom/linux tags/qcom-clk-fixes-for-7.3 + (35ce3c571297aee13c6da604d022929fd1aa9efb) + https://git.kernel.org/pub/scm/linux/kernel/git/qcom/linux tags/qcom-arm64-fixes-for-7.3 + (7ba5290db6bf365e46fb70bf13dc2a47048165ea) + https://git.kernel.org/pub/scm/linux/kernel/git/frank.li/linux tags/imx-fixes-7.3 + (3c0d2f1f431c51a11253f28763be723ac782fd7a) + git://git.kernel.org/pub/scm/linux/kernel/git/jenswi/linux-tee tags/optee-fixes-for-v7.3 + patch + MAINTAINERS: update my email address From f0406245cb9855e6318335a8a223551354291a46 Mon Sep 17 00:00:00 2001 From: Mark Brown Date: Sat, 3 Oct 2026 02:19:50 +0200 Subject: [PATCH 1333/1352] Add linux-next specific files for 20261002 Signed-off-by: Mark Brown --- Next/SHA1s | 433 + Next/Trees | 433 + Next/merge.log | 20065 ++++++++++++++++++++++++++++++++++++++++++++ localversion-next | 1 + 4 files changed, 20932 insertions(+) create mode 100644 Next/SHA1s create mode 100644 Next/Trees create mode 100644 Next/merge.log create mode 100644 localversion-next diff --git a/Next/SHA1s b/Next/SHA1s new file mode 100644 index 00000000000000..0e545c3619bdaf --- /dev/null +++ b/Next/SHA1s @@ -0,0 +1,433 @@ +Name SHA1 +---- ---- +origin ce1e0223d8ad4211275c82a17ed6d43ab81e13d9 +ext4-fixes 981fcc5674e67158d24d23e841523eccba19d0e7 +vfs-brauner-fixes b78b728e21c32ec4c330b299f657fb1eb02dffc2 +fscrypt-current cee9395acd8043be0644b25c34bfa86623f2b935 +fsverity-current cee9395acd8043be0644b25c34bfa86623f2b935 +btrfs-fixes a3dde27c5da2dce0b7d120342cafa19bf59f42f8 +vfs-fixes 49c5d168a3a8f4eb27d44a2a22b7e8a856ca601f +erofs-fixes 135d84c66f85426299db01a09d93a79a87af18ba +nfsd-fixes f76017a7663c4ce5e379f8a8d39f032bdb1fd865 +v9fs-fixes 028ef9c96e96197026887c0f092424679298aae8 +overlayfs-fixes 4549871118cf616eecdd2d939f78e3b9e1dddc48 +fscrypt a63883d2c3d47ce6d41918c2453a4513b921eaaa +btrfs 8147b3ee1381211a34df80119596791b964d1c54 +ceph dc173b37415e8f738fc4de477490056b479ddc9f +cifs 19465a9aeb664f1d710d0d22c07c31bfb7c85446 +configfs 620938e7da8943970ec25e1c8cdbf44ba1d44fa1 +configfs-rust 6dc9fb1516aa4930329ef561f078c7de9257abf2 +ecryptfs f81cb44f9a4b88d73ee5dec4a1ccdb0232fd2e3f +dlm ed9b6a1296f10e4881d93dfe6d76013fbbaeee87 +erofs a7d28aa0e9b2c983b915d091f9596f0e3b253355 +exfat 216426aff8f69379f03337857551ca0ed3190361 +ext3 2eda6a119a0024f5b94319947a0e4c4fe0ec1df4 +ext4 9091c97be34083587a75db174aab51551d8e8543 +f2fs 7566fec606d233f803e61f5ce37c54fbcdb00010 +fsverity b08e4ee274917074633a41b3598d5541cee2b914 +fuse b776cbe24cc1b79c7f5101762b0069b3f48b6a4b +gfs2 d0c4bce31a579216780b990a09250f15e00f38ff +jfs dad98c5b2a05ef744af4c884c97066a3c8cdad61 +ksmbd 276effe0501713f7ff73d2e2479af850b4679973 +nfs fd73f4a6659897191fa0d40695fe370925dd3780 +nfs-anna 9bafc322b7fca03717db78615b07bd401367b7c2 +nfsd ac04dab23b5ff28fc7e41957824c5c439ae99887 +ntfs 708f9d56cacae21aeee98d16bcdd50a66edc04a0 +ntfs3 f3b8ee6c05bed24a06192ea8e4cafcc06946c1bd +orangefs 2bc09cb0e9e44c7e622b956332ebfb9b638fd5f7 +overlayfs 1f6ee9be92f8df85a8c9a5a78c20fd39c0c21a95 +ubifs a5e0055eac837a1168c781653943d4a0d9920af3 +v9fs c60ae98c5aa64021751b38ab1313b19d620bf640 +v9fs-ericvh 028ef9c96e96197026887c0f092424679298aae8 +xfs b4787e7b9730d52f33cf8dd23b3c42f1237f9753 +zonefs 3a8389d42bdf4213730f4067f8bfa78bae6564ef +vfs-brauner 84086827932b58e7645d93d970bbc566c4ee408b +vfs 4dda01b67c8662c5d0c53034974cb8280c575a5d +mm-fixes c14d066c0108fb8788e1e980e04ff936bbd9ffa3 +fs-current e2aee709a9b357c58028515171187e7d36becef0 +kbuild-current fd73f4a6659897191fa0d40695fe370925dd3780 +clang-fixes-current df2908090cda368b01ff43709f51890076c56157 +arc-current f050c3e61d2a1aaece3170d459e4ce5fa2486c12 +arm-current 1039bffd6ae9c75b42b7d148d6c1106134107b66 +arm64-fixes e060d9069b49fe55f38057dfe592068014652671 +arm-soc-fixes 4dd1999783d7d12434006289338373e49492dc96 +davinci-current cee9395acd8043be0644b25c34bfa86623f2b935 +realtek-fixes dc59e4fea9d83f03bad6bddf3fa2e52491777482 +drivers-memory-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +sophgo-fixes 19272b37aa4f83ca52bdf9c16d5d81bdd1354494 +sophgo-soc-fixes 0af2f6be1b4281385b618cb86ad946eded089ac8 +m68k-current 2f8e3cad53b5c36ab0ed5d3195bfc55c59ea61a5 +powerpc-fixes 93f51579e7df248780214094418f205253383cc5 +s390-fixes 5b76268dac968612f7283d59b539036de955b7d9 +net 232d49dd4b40a666283de9e722899f088ed581b2 +bpf 7b12c538697f2d5014dfd378c73be193d700c7a9 +ipsec 86de3a1118a16dbbbe5fabe9aa20ffb93c072ed5 +netfilter e23a64eb244356ee47c0620f0722d51bd88db522 +ipvs a401a9d547c50ef34db1088cc1fb9a201a7af657 +bluetooth-fixes 86ef0f58bdecdedb3a1240971c56d71b7e4ce3fc +wireless d24e8ac715de2e16a53c144005b1863660a5fbea +ath 6f63e919fe1e335b8abcb3a28bfd4804a98d875a +iwlwifi 1e7e30fb650b8327534f65b2213a794c04975fa4 +wpan 2b4707a149a55e8fa75c9ef32b359d60f470a566 +rdma-fixes 72d3fcf802c45d00b300f25b848a93c3a2bd7c7e +sound-current e6229c0402ea4863ec74802d79d5b52b9398d66c +sound-asoc-fixes cad16d850e3759dd82cd869f739b56e9844a1fee +regmap-fixes 2e42cade8ff1ff579e77976d9869db4df1feaf74 +regulator-fixes 2f15518682c779b1014552cb2ccb4a2052f85f27 +spi-fixes 3d743adf090cd4c9a2120c1e02b0482e88aa0d2d +pci-current 97958feb6560b4f5eb57addeb9d3a214d55b154b +driver-core.current 72d3fcf802c45d00b300f25b848a93c3a2bd7c7e +tty.current 6c95ca52f27855dd2fb74131c8d4e8325d2af7de +usb.current a93862fa670a9c99fd9a24fb2ef69e4f948d9bd5 +usb-serial-fixes 4b1e4a071ff074d70272e552f71c30baf191573c +phy 93f51579e7df248780214094418f205253383cc5 +staging.current df2908090cda368b01ff43709f51890076c56157 +iio-fixes 9ee8306121495d2a25aa5d1bfd519f2748786b83 +watchdog-fixes 2686e13649a9e846f6c08cf4ada0256ca19c3475 +counter-current fd73f4a6659897191fa0d40695fe370925dd3780 +char-misc.current 8e242ada093af7d2ee82037f09c287fbc6f1e3f5 +soundwire-fixes 93f51579e7df248780214094418f205253383cc5 +thunderbolt-fixes 395e9f2967a7ac6898026e8f136c369805dc2fd9 +input-current a215720959c5450b3699e243904b39c3f8f68888 +crypto-current 10396a2d6d41d594975b6ece712278570c3c970c +libcrypto-fixes 572af6872e520c7102e58362ef7cd7c2ed4342be +vfio-fixes e242e974e812e7a47e3088860c80d9492fac314f +kselftest-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +dmaengine-fixes 93f51579e7df248780214094418f205253383cc5 +backlight-fixes dc59e4fea9d83f03bad6bddf3fa2e52491777482 +mtd-fixes 1c1a342aceec79528b3ba51f376eca2be5928fe7 +mfd-fixes d5d2d7a8d8be18681a0864f58e3875f1c639e11c +v4l-dvb-fixes 2579cbe68005f46fc7f8f95364f6b101b07b9d1c +reset-fixes 71827776667f4e4677a4fa806bcfb24d4b8dd9d7 +mips-fixes 93f51579e7df248780214094418f205253383cc5 +at91-fixes afb1ecfeda3fc31d6300a176f4854cc62e289bf3 +omap-fixes 2fabd2f406d0cf787be47b3bbeea99b30004033d +tegra-fixes 3f60097e76e4dcf45b041803e3b2ccad845f8ec2 +kvm-fixes f9bfc323e76120c2cd1fdea93bbf315eec5eb934 +kvms390-fixes f47190b08b71e8482072978373ee88cb2dfbdaf4 +kvm-arm-fixes 71cc2c67fb8f8d5aa8154eb482e9846d2214f11b +hwmon-fixes 61406e9cac695b19b979b002d790d346b9f07887 +nvdimm-fixes a8aec14230322ed8f1e8042b6d656c1631d41163 +cxl-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +dma-mapping-fixes 057e5e07420c753f248f9ce148040ab6dcf8359f +drivers-x86-fixes d144a494d81fcf2d1c5cf58b01c655bb8bafc701 +samsung-krzk-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +pinctrl-samsung-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +pinctrl-qcom-fixes 19fc4240358be2a25ce8e0a49a2819588ed61c5d +devicetree-fixes 30724547b221e0e670ddfef140fc7d816b2aaec6 +dt-krzk-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +scsi-fixes 42d1221d321e55afc7bba9109a77aaf5a817c8a3 +drm-fixes 26fe67ec0dca018767b038fe65c9fc9da058ac19 +drm-intel-fixes c034e8a46e4cb703018fe7e10fe5a3a974d6a7c3 +mmc-fixes c96ca94425b076dc398d929366e378d1e22ec794 +rtc-fixes 055ef5ce9f67a6a3a1363fa663007f8196cc0fb8 +gnss-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +hyperv-fixes ca039df94983c441fdee0e2488263fc744e8ecb6 +risc-v-fixes 247ac82ac82d82cfae97da2a76cd30a071950d30 +riscv-dt-fixes 0f70fd6f4c1d6c1f2ea298d8dd07c4a4d5dff9af +riscv-soc-fixes dc59e4fea9d83f03bad6bddf3fa2e52491777482 +fpga-fixes 19272b37aa4f83ca52bdf9c16d5d81bdd1354494 +spdx cee9395acd8043be0644b25c34bfa86623f2b935 +gpio-brgl-fixes ff82fc3a4a1d417dee1229681cc5285986edb1fa +gpio-intel-fixes 8d37f20173a2760718bb02b58bafc5a7cac93d67 +pinctrl-intel-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +auxdisplay-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +kunit-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +renesas-fixes 8dc2615d5702059b2b71fca6f93c0d7d10ae54cb +perf-current aadea57f532882d8bab444646863c7ef8a778ff1 +efi-fixes d8809f6931065cbbf3554647a50a65a471ab5983 +battery-fixes a58cbc8b36ec0cd00d2ba3de7d003de818a27523 +iommufd-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +rust-fixes ce1e0223d8ad4211275c82a17ed6d43ab81e13d9 +w1-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +pmdomain-fixes 4c66c423593e26273ea0d644906ed007630c6f6d +i2c-andi-fixes c98bf3b86609a18ab16d067280257326e557c636 +i2c-rust-fixes 4eb422482ca5d924d7212ad2ca1cb7ea6f5b524d +sparc-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +clk-fixes 81493c1dd1b1ebecb2843a7815973a5bd9a37e5a +thead-clk-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +tenstorrent-clk-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +fustini-config-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +pwrseq-fixes 58a00330248154461c52a6f3bd5c01e5b67f9560 +thead-dt-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +ftrace-fixes 1650a1b6cb1ae6cb99bb4fce21b30ebdf9fc238e +ring-buffer-fixes 057caace5214da3b457bbd295e1a2ad34d3685ea +trace-fixes d860c67c051685abb0460b593b193f0f45f4fa92 +tracefs-fixes 07004a8c4b572171934390148ee48c4175c77eed +spacemit-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +tip-fixes 254979e0c666240a3afc2f8092a329ca0a161a50 +kexec-fixes a901b0778ae82d46084b55237e65d3a7bb5f0c86 +liveupdate-fixes 3a0b8fa2eb36afc88b62a95f33f0c77c71fa5ded +drm-msm-fixes a15fac810c76397ec9f62a6fc26c4d7ab6e238a7 +uml-fixes af421e9aed3920c7ac88c24daa48606c7112feca +fwctl-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +devsec-tsm-fixes c3fd16c3b98ed726294feab2f94f876290bf7b61 +drm-rust-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +tenstorrent-dt-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +nfc-fixes b61732f47316d45f27706db7812950145d3327b5 +mm-nonmm-hotfixes-stable a243ede718463c7b481878656f1ff32a0ce0fd54 +mm-nonmm-hotfixes-unstable a3e3527ac09d0f3ba2f96f5f9c2d0c7b0c428b67 +drm-misc-fixes bca45af5998a05f34b13a2ef11e639bac9c62643 +rust 948440b3f429f84680db58b4bf529b1ed2a20e15 +rust-interop 05f7e89ab9731565d8a62e3b5d1ec206485eeb0b +rust-alloc 6e8339118040f24ebcc21db22c9c1b7398b7e39e +rust-io 86731a2a651e58953fc949573895f2fa6d456841 +rust-pin-init 499b8f8a5d0401a6ab83b346a307f923640a024a +rust-timekeeping 2ea0119f72dba597aa8a98cbdb72c564bfc5cb38 +rust-xarray c455f19bbe6104debd980bb15515faf716bd81b8 +rust-analyzer 5f45afb8ab04d934fc0601a202f95267ebc20059 +mm 3bea2ede9c01b6e8cf956b85c8cc0d4c200dfdb7 +mm-nonmm-stable a243ede718463c7b481878656f1ff32a0ce0fd54 +mm-nonmm-unstable 20e59dfe656fbe716b984d3662fd5bf9bd50b331 +kbuild 2eaa399a85efad571f4043ec8903e076b73a3dbb +clang-fixes 0bb666d5f5a2339a5692afb312c2161df9620503 +clang-format 8f0b4cce4481fb22653697cced8d0d04027cb1e8 +perf 2ed38aa8a52d3e245a9f71b98825ad7f77956d50 +compiler-attributes 8f0b4cce4481fb22653697cced8d0d04027cb1e8 +dma-mapping 57a57ae077f0f167056fc08f1666678f78e5b147 +asm-generic adbbd9714f8058730f93c8df5c5bf1679456424b +alpha d58041d2c63e09a1c9083e0e9f4151e487c4e16a +arm 1a89abc009cb5035d0cb4e5ba48d32b94f30e8e4 +arm64 e4e0f90fbf1bd7fc380f18a76a07295317ab3628 +arm-perf cee9395acd8043be0644b25c34bfa86623f2b935 +arm-soc bdb122e397cb06a0235039f1b10560712526a137 +amlogic 8610d31587e139380db9cb8708f4263833b482d0 +asahi-soc 9378cd5ddcebc06fdd803373487ed1a376532391 +at91 bd49a1c72abbdddbc7b3d691d926623e33a7c1a0 +bmc cd7d1ef7d74ed5f5a1790a40389bc38382924556 +broadcom c0f2033a9ce4b843ca3e947536a0607ecf982be1 +cix a0cffbd8878c55b12dc4555f883adf3372cecd91 +davinci cee9395acd8043be0644b25c34bfa86623f2b935 +drivers-memory a22355280361d2a376a2020059a2bae11e6cea10 +fsl df0fd0f5af4ddd84a76e21da8e6b489928c5e300 +imx-mxs 31f9ec9f38735f42e1f0ba57347c420e56b99824 +mediatek 4bfec57314396c8c816ee78f4803987882b668a7 +mvebu 73b0de63d7f7d9e504c8d1169b573cdded24e1b8 +omap 7cdd46c9d6c5a7f4800195af82bb718081e7c84d +qcom 705529dab1fa872d8b53b09c95e1b03a95573376 +realtek 3c778f0c9fa36b6b9a26bd2146df24ed4e830ba4 +renesas fb050251fd4fe3b4d0231495651901626c2a1464 +reset d373605cd514837d8a6de3d00c786d4bae6dbaf8 +rockchip f4d0e3d48d590e93de51c4562d9c005daa012945 +samsung-krzk 08df370772f32b2bce3f88eed253b6a1b0035e40 +scmi e917780f1c75d8128ead37c2ea82ea2b5bf92a59 +sophgo 76acfee87c74dc0dc7a68a00b70b9c39d1c8428f +sophgo-soc c8754c7deab4cbfa947fa2d656cbaf83771828ef +spacemit 44034f5791c106ae588a7f063f88e2ec98f84b31 +stm32 f831584128ac2a36fb4fa62fe106793d08c2dbf1 +sunxi afb622bddaa6fc7b6d205fd080b9c0d3ec853431 +tee b0adf8e5e1ebf416abfe055475f08ae6a72fc2f7 +tegra 494422afb7e186888ffec89c3e30afa805f9aca6 +tenstorrent-dt cee9395acd8043be0644b25c34bfa86623f2b935 +fustini-config cee9395acd8043be0644b25c34bfa86623f2b935 +thead-dt 2414ca8f5a0cafdfdb2b5899fd173adc09b5adf1 +ti 9e2716807275dd86dbbc86d1b0a2ca957d9a37c1 +xilinx bf126d9aa20f2d8bb1c71f5b7f7ba216178fd80d +socfpga a2a94b6a6f0187b5d3b298b4f12a031335f5698f +clk 9832cce7217fc417dda4c328069729bb5ed502e1 +clk-imx 39ec460b56b26319d1f31b459e8be7ea34ae9c67 +clk-renesas 21cbd7ec29930f8e7e156e18b81736f099f03849 +thead-clk cee9395acd8043be0644b25c34bfa86623f2b935 +tenstorrent-clk c64b54ddb692c30b8fb139a2ead32eb4db7faef5 +csky abb81e5ce7d995baa41556b8125fa59e28ba3be8 +loongarch a2628ce4ddb6873e35380a42396d17a66e704a1a +m68k af32c3cb72529b14bfb2685bbdaa2b1514496b58 +m68knommu 29036c5910df0f0a4c824d7da5b6254c7d4ce500 +microblaze 6b125c73aaa0dab9ce625437c63b176afe453776 +mips bfe97f12a0a64c3682be21a02404238e3b91e945 +openrisc 6620f5e8c11c4f7e41222a86f5c97150cc5f84a5 +parisc-hd a5f6df25261b31eb75f61047a890515671d95a29 +powerpc 12d238d584ba485a99e02e033746100fc7099184 +risc-v 4735883c0d4bc3dae51b92218f00b2faff7d49b8 +riscv-dt 2186ec5f59c23fcf0bf896ca0543b6a1135d7bb5 +riscv-soc b7516f2f64fd5ac461eae470ca050fc0a5eea18f +s390 c57ae907d4f9ae822bdcd7b16f2d0dc56fb7973b +sh be5a19d95030ffa7a37d2981d0ab4eac8a6c89fa +sparc cee9395acd8043be0644b25c34bfa86623f2b935 +uml 2f88f5689de1a039764d00f466209acf0c010eaf +xtensa 28722ed2527aa63ac1454013defe718c56032886 +fs-next 7e86d255ddfb4516c9bce453194bd4f0a220e98e +printk 3dd0c9c7109dec0e5544d7e85756194e6c02783b +pci a0818fd5eaf507619e77b6727dccc394205dd13c +pstore 7c756181175d50200be487affa905e973f6cbaf5 +hid 145c2b2e9a5c0f794fb4009bcb072ab19f8ccfcd +i2c 8cd9520d35a6c38db6567e97dd93b1f11f185dc6 +i2c-andi 91c65ba57f3d8ee1ee5b4738682307c9e410c10b +i2c-rust 61ddec70c9bcc3d4b0471c8683e57d3b03cecf0a +i3c b0cfe49a2e3b984f94f86dd0888b7102f69d98da +dmi 1afafbaf749d8e8ec53f8e38efdc731131902b5b +hwmon-staging 4781ca52761e666cf18b591e6bb0478396c90320 +jc_docs 2d72a4c09867a8f9131afc2e705a728f9fba2417 +v4l-dvb 89a3d2a2237e017cdbefcb036db71ce74e67faf5 +v4l-dvb-next adc218676eef25575469234709c2d87185ca223a +pm 1729fbce1c728686abb2f08350996b45cec7830c +cpufreq-arm d82d896f00e7bf697b61d5df0555941afa8d0657 +cpupower cee9395acd8043be0644b25c34bfa86623f2b935 +devfreq 9a222650d9e70d0027126e4df5b00b1b5b678a97 +pmdomain d1a93cf1d3bed4667de58a561afb9682430e4f50 +opp 706081c6b9327dd611dd72b20130a77f0f059b6e +thermal e856ca3013b3074fd07cb81c90b1de332f69153c +rdma 485effd117d0310a7f589defb4f442fa3bb52ae2 +net-next cfb7793d1bc0f7d90571611979654cf1b3886b29 +bpf-next 2b5440b31cafad2fb4b4a83efb780a6d7430a385 +ipsec-next 014d795c73837ea2339a4ea8e8f82c6e959b845d +mlx5-next 00290b9ff59ef47b990d26a6b1a8b2ea6dedf5e9 +netfilter-next 87b80c2f6b05cad9f0ff9136709c62a0f59923e3 +ipvs-next 6ebcf5074cff0402730c6981d2397139fee6322d +bluetooth 036d4119079a757c9d1b4d35205c7c467f0018fb +wireless-next f49defea7668d8c68ec19fa085ef3da6075561c7 +ath-next 21b4248bfa0f410fa22202fc2c82f4808d672cda +iwlwifi-next f57fea3e69a06c3377092834e74d9841e1af3210 +wpan-next a6bfdfcc6711d1d5a92e98644359dedc67c0c858 +wpan-staging a6bfdfcc6711d1d5a92e98644359dedc67c0c858 +mtd 112a666bd82e96e4a01e0cd8a0fb9c88dd0bce38 +nand 55c5b6d5f59f59a8c98a7effc3195226cc20175c +spi-nor 68d7115d77a873c28d8c7d2f239c3d9ca83bf39f +crypto 4c3ca7c8c152278332590bfd3df1b0f086364759 +libcrypto da4d5933e30340766925480f9dd9f250c4355e24 +drm 845c5cc3697c5520fe3d7fc8597d393d172cc5fd +drm-exynos 3a8660878839faadb4f1a6dd72c3179c1df56787 +drm-misc 54e61ea6c492d69a1717018826d8340810c812eb +amdgpu 41505ac433cc4c5179deee8860d197b04f6d1c1d +drm-intel 2c1bb96681bea17c2ffbd134816961615292a159 +drm-msm 5e4a3f7b262060d70cfdc7889cfe0f4aba9ea3fc +drm-msm-lumag 5e4a3f7b262060d70cfdc7889cfe0f4aba9ea3fc +drm-xe cf4171a20d13918e07ed01a1e01a6b546825c601 +drm-rust df718311a84188e26d499301c3921e4c392d557e +drm-nova 93296e9d9528f0d87f2cf3fee494599060a0f14a +etnaviv 6bde14ba5f7ef59e103ac317df6cc5ac4291ff4a +fbdev 9f4c6043f33c95db7e46d6ca198344ce0ca2e6c5 +regmap 117e6a5fd98fb7002e70ba39c3ba393498399982 +sound cb8e306fd99c70cc6f57d1a3ec39659dd4c5a6e6 +ieee1394 a6e7c3836b81235df08814bd409f7a23efcfb34d +sound-asoc 0d6ef8b530db8dbad6b06309cf2683f1032f8013 +modules ff7360c5c731356a59171eef732a26a72381a8ba +input 5b051ea055b319fb9a4f1d09f6f181ca7017bdd8 +block 67cd826ded183c9144deee5af268907dfd8df0f0 +device-mapper c0df022cb0fa2bbee74bd9b90c66790bcd4e3050 +libata cfce1dc635041d6b9dd5fad5367a331faaf644fe +pcmcia b3c26ea81ccc522e77ed0b1707add61fc9206216 +mmc fd9af27f6c2319713bb23832066585f8a59a1771 +mfd 319633b06ff2bfc2a7a60d2cfcab2e681a343b27 +backlight 5e1631df673f3d5355acd06263a884e00d8d007b +battery 4fc88ba435dadbc05990951e3f3fbd8ccd2df140 +regulator 6722016006986d96906dae1a2880875a9c414759 +security 881f19c2ffbc73f351b257e09c71e6059939b770 +apparmor cee9395acd8043be0644b25c34bfa86623f2b935 +integrity 1dc317438308c79cf90a6f14d3e4116347ea3716 +selinux 7ab68f08381f2edb0e9b33256f3204d7e5f93f3d +smack fedc88e38ce979a720cd2de042578cb5df3dc8de +tomoyo 72d3fcf802c45d00b300f25b848a93c3a2bd7c7e +tpmdd-tpm 015fb29a748342a186f37b602ac70017098c8251 +tpmdd-keys cf60aac3d7acacaa54d634de81ea72aa4a655f74 +watchdog 8b5a9f09037e3c372c4e9f2fbda85fcfbf5ea5f0 +iommu cec2dc7663e7182c311162163d71b1bbdfdbf09b +audit 88bd5e852addf09db73a38d93f7911863ef5a104 +devicetree bb91f8668c2478dbea41e3b08f214732c5f5870a +dt-krzk dfcea1641506d7ba096a9e6ead5d15f5573206ec +mailbox 14af7a96afa39f4f3c1972705489b9ba15c01857 +spi ed5f8f8b115db9cb3fc986687eaf957b29551e1d +tip 8f511d67b4fb0463b64c6b27390a0f3480b3ec36 +kexec 9d0b028715b007188a61ce70b9656ead84a2b3ef +liveupdate 5221d141653a83a974595b4e5eedd9d6983ab270 +clockevents 1b8b356b4b06e3a28feb852c04e06caeac07bc87 +edac 898e6a2c5ce0e4aba55d126bbe610f0f77e71226 +ftrace ca3c36c6c4ded6a82af65023a632f259e4a24eb6 +rcu 5826bf4c6281129ea9aa7f22270f98f64ae5fde5 +paulmck afb36d3025c845d8ae32f4ca83bb0ae31240d790 +kvm d4b7fb647204f0c81dfeae2d1a708e4d858e0c94 +kvm-arm 73ed60746b21c1f6dac360ec0299421a39d0fde4 +kvms390 044ae0767d8cc1fc2a53930361f05c9ed2c3c081 +kvm-ppc 93f51579e7df248780214094418f205253383cc5 +kvm-riscv 41e81f7e3ef96594fb840445343c0ee7723aa550 +kvm-x86 6bd2905303c58679e303941b7d8ae8c074cd95ce +xen-tip 93f51579e7df248780214094418f205253383cc5 +percpu 8f0b4cce4481fb22653697cced8d0d04027cb1e8 +workqueues fee1265a725725dac0ff7459f912d69558798b36 +sched-ext 6a119bf5e64d3a6fb0f643e41afa6e79cdb2a553 +drivers-x86 fe5030c8cc7156223f48530e9b49aa87c0305bcd +chrome-platform 317e7abcaf60b79765dcf35c014edac8b9d37a6e +chrome-platform-firmware f2e05ce9763feb1dd7da15afd2cc61df227ee85a +hsi e81250ec6b69248b00d38c523dc6a13efaf38aab +leds-lj 05b4738b0078f7d6f154f68068a11c8a0635e9df +ipmi 921fcdb737891cb527e0cc4786713dc00cf32409 +driver-core f1850e443b0e4f2429ddf42a8d5033ea54ae8a90 +usb 639df5d23876a548b86fe6526bed8b97edf64d96 +thunderbolt a93a8e3200002f0c345fec4c340390e9a60aabc7 +usb-serial 6583f9741341b98ade67aa764794a02fea1eb7db +tty c44a4925cdac02af781ebfae96df68a7bf3580b0 +char-misc 04434a1d0f311d76b1f15fa987918b471a0b8c6a +coresight a3281934612608db9e1d24aaf913c846e913de89 +fastrpc ef071c4906eb45d16b60f09154cf0bc6ec8f5435 +fpga 093da48782df3d50953c520626b345ca70af8cbd +icc 3fd56a2833c22f62eb84297d0dbd47739632f1a0 +iio a3b3580713f3ac5a32dc2874ee546828977a1d68 +nfc fd73f4a6659897191fa0d40695fe370925dd3780 +phy-next c7f2322431cb6d108b18fb4154606b49e2bc50f7 +soundwire 90b63b309fd6c0f192337659e73e3047a67df99f +extcon 8d3ae59288f1e7d58d76558a6ee96d533bc5019f +gnss cee9395acd8043be0644b25c34bfa86623f2b935 +vfio b30b52c2fb82e6d38ee2845a4d85e8254b44f3d8 +w1 813a5b9a3c9f6cee830b1e23343fbd2359683557 +spmi 8cdeaa50eae8dad34885515f62559ee83e7e8dda +staging dbc2fa996f44602af72ea56efa68c043407eaaac +counter-next edac5cf35699492027fb54e348dc6234b2661617 +mux ac7bde3c53166656d80e3aacc7d274d3c60a6de4 +dmaengine 0a8dda0a15d3926422d286567f945a05328a4ac6 +cgroup 7965da4a7fc0a2b2e7d599021f7e6157c80f8704 +scsi 6147f16c23efb58a98fdfc71b794b063dc01c767 +scsi-mkp f09d2c7485b32adb82336d0d748935c8237a649e +vhost 8f2c2fb94a01320e5136c6e89a47fb12355ad27c +rpmsg 5f7775e2a346257aa13128feb2f640cb5a2d6e5b +gpio-brgl c4e74a7058b573617f142e07b4b5bc9eee04f4c8 +gpio-intel 0fc424b6a8ef447450f6087db42abce3a6feb328 +pinctrl 3b9c0102861ab5658d79c147a778702bd01dad73 +pinctrl-intel c016587866e573fa8dff50c3bdae9734c4418099 +pinctrl-renesas 0cd4a7b3a4883970df3ae8814d9aabdecf65a81e +pinctrl-samsung cee9395acd8043be0644b25c34bfa86623f2b935 +pinctrl-qcom 4d7c9430a26aea6a69521af3aa2d78dd5bc8d3ed +pwm e74b9a7ee50071aad25d3989bf985292f90b0c6e +ktest 932cdaf3e273a2727e77af97f79f12577174c5a0 +kselftest 30af56a227e27b892d34bc44c23298b5274913e3 +kunit cee9395acd8043be0644b25c34bfa86623f2b935 +kunit-next e38f53f0482468efd04397f66bda4648b70ac9fa +livepatching 5d791d3396ca4c9e5fd9d20b49386c20d8179eb8 +rtc 8ef8e9839a23ca7f5dda6a69b670633de5a23716 +nvdimm 855a46681e0ba601952c346e5b62d867e51ecb23 +at24 dc59e4fea9d83f03bad6bddf3fa2e52491777482 +ntb dc59e4fea9d83f03bad6bddf3fa2e52491777482 +seccomp 832b9b176be06a20747a267db6bc0010a24b74f4 +slimbus 4350d705466992dce4a21c382985553bfb568f65 +nvmem 41f42ff4ae134eba8dbe3424cf2ca1824c652edd +hyperv be0cfab740e58b70047ef6e7e3d578f00ed5d258 +auxdisplay f63dc0eea92d07c5bb799a406571b555cd2bbd97 +kgdb fdbdd0ccb30af18d3b29e714ac8d5ab6163279e0 +hmm cee9395acd8043be0644b25c34bfa86623f2b935 +cfi dc59e4fea9d83f03bad6bddf3fa2e52491777482 +mhi 710bf7329abf86e37db529f19c8f394099238892 +cxl 71392a644e88c6bb921cdf12cb2b00adec070fe4 +zstd 65d1f5507ed2c78c64fce40e44e5574a9419eb09 +efi 7eef16311a234b5d77c1493e19c6538a3aa1a203 +unicode a511442085c140da9cdbe60e3f0fab1c71480801 +random eb13a1ff271b0d180687eebb30611b0b276d5193 +landlock 02619311dbfc4351e78e5dc6652532cbe7603661 +sysctl 4991c8b72b564cd16cb48124f880617fd49f1013 +execve df2908090cda368b01ff43709f51890076c56157 +bitmap 452a6d5b5e0556848a8c28428c8f86874ea4ee02 +hte 30167fadbf87fa901a2fee49c30a5c5a506a43e8 +kspp 760f96b7f54beb8dc6f85d97ac7854fddba86835 +nolibc 5f81ffdf26072fd262093c4127e656276564395f +iommufd 54dadb030c7e2350957855d3995de05ae02c2e66 +turbostat ccdfcb7e7ab84573046061ece61bfd2368577f1e +pwrseq 09baa2f4caab013160eed975db47235e6c930af4 +capabilities-next 507adb6448378155ec89495fed8c3c09d8c83920 +ipe bcaa4d1d69368b6034b1ceb226ffd0a203a5fbc7 +kcsan a8488ecbd7ba44d65b912dfe88a73f438eba2447 +crc cee9395acd8043be0644b25c34bfa86623f2b935 +keys-next 965e9a2cf23b066d8bdeb690dff9cd7089c5f667 +fwctl cee9395acd8043be0644b25c34bfa86623f2b935 +devsec-tsm 3177779ae17db4c66c851f799505fb95c7530c03 +hisilicon a5db65458a911daa6f8927264366dab501715dc0 +device-id 995832b2cebe6969d1b42635db698803ee31294d +kthread fa39ec4f89f2637ed1cdbcde3656825951787668 +pagemap-headers e02cb91d4644dfff593146f89e70d2008cc6ac16 diff --git a/Next/Trees b/Next/Trees new file mode 100644 index 00000000000000..32b5b670b60a62 --- /dev/null +++ b/Next/Trees @@ -0,0 +1,433 @@ +Trees included into this release: + +Name Url +---- --- +origin https://git.kernel.org/pub/scm/linux/kernel/git/torvalds/linux.git#master +ext4-fixes https://git.kernel.org/pub/scm/linux/kernel/git/tytso/ext4.git#fixes +vfs-brauner-fixes https://git.kernel.org/pub/scm/linux/kernel/git/vfs/vfs.git#vfs.fixes +fscrypt-current https://git.kernel.org/pub/scm/fs/fscrypt/linux.git#for-current +fsverity-current https://git.kernel.org/pub/scm/fs/fsverity/linux.git#for-current +btrfs-fixes https://git.kernel.org/pub/scm/linux/kernel/git/kdave/linux.git#next-fixes +vfs-fixes https://git.kernel.org/pub/scm/linux/kernel/git/viro/vfs.git#fixes +erofs-fixes https://git.kernel.org/pub/scm/linux/kernel/git/xiang/erofs.git#fixes +nfsd-fixes https://git.kernel.org/pub/scm/linux/kernel/git/cel/linux#nfsd-fixes +v9fs-fixes https://git.kernel.org/pub/scm/linux/kernel/git/ericvh/v9fs.git#fixes/next +overlayfs-fixes https://git.kernel.org/pub/scm/linux/kernel/git/overlayfs/vfs.git#ovl-fixes +fscrypt https://git.kernel.org/pub/scm/fs/fscrypt/linux.git#for-next +btrfs https://git.kernel.org/pub/scm/linux/kernel/git/kdave/linux.git#for-next +ceph https://github.com/ceph/ceph-client.git#master +cifs https://git.manguebit.org/linux.git#cifs-next +configfs https://git.kernel.org/pub/scm/linux/kernel/git/leitao/linux.git#configfs-next +configfs-rust https://git.kernel.org/pub/scm/linux/kernel/git/a.hindborg/linux.git#configfs-next +ecryptfs https://git.kernel.org/pub/scm/linux/kernel/git/tyhicks/ecryptfs.git#next +dlm https://git.kernel.org/pub/scm/linux/kernel/git/teigland/linux-dlm.git#next +erofs https://git.kernel.org/pub/scm/linux/kernel/git/xiang/erofs.git#dev +exfat https://git.kernel.org/pub/scm/linux/kernel/git/linkinjeon/exfat.git#dev +ext3 https://git.kernel.org/pub/scm/linux/kernel/git/jack/linux-fs.git#for_next +ext4 https://git.kernel.org/pub/scm/linux/kernel/git/tytso/ext4.git#dev +f2fs https://git.kernel.org/pub/scm/linux/kernel/git/jaegeuk/f2fs.git#dev +fsverity https://git.kernel.org/pub/scm/fs/fsverity/linux.git#for-next +fuse https://git.kernel.org/pub/scm/linux/kernel/git/mszeredi/fuse.git#for-next +gfs2 https://git.kernel.org/pub/scm/linux/kernel/git/gfs2/linux-gfs2.git#for-next +jfs https://github.com/kleikamp/linux-shaggy.git#jfs-next +ksmbd https://git.kernel.org/pub/scm/linux/kernel/git/linkinjeon/smb.git#ksmbd-for-next +nfs git://git.linux-nfs.org/projects/trondmy/nfs-2.6.git#linux-next +nfs-anna git://git.linux-nfs.org/projects/anna/linux-nfs.git#linux-next +nfsd https://git.kernel.org/pub/scm/linux/kernel/git/cel/linux#nfsd-next +ntfs https://git.kernel.org/pub/scm/linux/kernel/git/linkinjeon/ntfs.git#ntfs-next +ntfs3 https://github.com/Paragon-Software-Group/linux-ntfs3.git#master +orangefs https://git.kernel.org/pub/scm/linux/kernel/git/hubcap/linux.git#for-next +overlayfs https://git.kernel.org/pub/scm/linux/kernel/git/overlayfs/vfs.git#overlayfs-next +ubifs https://git.kernel.org/pub/scm/linux/kernel/git/rw/ubifs.git#next +v9fs https://github.com/martinetd/linux#9p-next +v9fs-ericvh https://git.kernel.org/pub/scm/linux/kernel/git/ericvh/v9fs.git#ericvh/for-next +xfs https://git.kernel.org/pub/scm/fs/xfs/xfs-linux.git#for-next +zonefs https://git.kernel.org/pub/scm/linux/kernel/git/dlemoal/zonefs.git#for-next +vfs-brauner https://git.kernel.org/pub/scm/linux/kernel/git/vfs/vfs.git#vfs.all +vfs https://git.kernel.org/pub/scm/linux/kernel/git/viro/vfs.git#for-next +mm-fixes https://git.kernel.org/pub/scm/linux/kernel/git/mm/linux.git#for-next-fixes +kbuild-current https://git.kernel.org/pub/scm/linux/kernel/git/kbuild/linux.git#kbuild-fixes-for-next +clang-fixes-current https://git.kernel.org/pub/scm/linux/kernel/git/nathan/linux.git#clang-fixes-for-current +arc-current https://git.kernel.org/pub/scm/linux/kernel/git/vgupta/arc.git#for-curr +arm-current https://git.kernel.org/pub/scm/linux/kernel/git/rmk/linux.git#fixes +arm64-fixes https://git.kernel.org/pub/scm/linux/kernel/git/arm64/linux#for-next/fixes +arm-soc-fixes https://git.kernel.org/pub/scm/linux/kernel/git/soc/soc.git#arm/fixes +davinci-current https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#davinci/for-current +realtek-fixes https://git.kernel.org/pub/scm/linux/kernel/git/yu_chun/linux.git#fixes +drivers-memory-fixes https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-mem-ctrl.git#fixes +sophgo-fixes https://github.com/sophgo/linux.git#fixes +sophgo-soc-fixes https://github.com/sophgo/linux.git#soc-fixes +m68k-current https://git.kernel.org/pub/scm/linux/kernel/git/geert/linux-m68k.git#for-linus +powerpc-fixes https://git.kernel.org/pub/scm/linux/kernel/git/powerpc/linux.git#fixes +s390-fixes https://git.kernel.org/pub/scm/linux/kernel/git/s390/linux.git#fixes +net https://git.kernel.org/pub/scm/linux/kernel/git/netdev/net.git#main +bpf https://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf.git/#master +ipsec https://git.kernel.org/pub/scm/linux/kernel/git/klassert/ipsec.git#master +netfilter https://git.kernel.org/pub/scm/linux/kernel/git/netfilter/nf.git#main +ipvs https://git.kernel.org/pub/scm/linux/kernel/git/horms/ipvs.git#main +bluetooth-fixes https://git.kernel.org/pub/scm/linux/kernel/git/bluetooth/bluetooth.git#master +wireless https://git.kernel.org/pub/scm/linux/kernel/git/wireless/wireless.git#for-next +ath https://git.kernel.org/pub/scm/linux/kernel/git/ath/ath.git#for-current +iwlwifi https://git.kernel.org/pub/scm/linux/kernel/git/iwlwifi/iwlwifi-next.git#fixes +wpan https://git.kernel.org/pub/scm/linux/kernel/git/wpan/wpan.git#master +rdma-fixes https://git.kernel.org/pub/scm/linux/kernel/git/rdma/rdma.git#for-rc +sound-current https://git.kernel.org/pub/scm/linux/kernel/git/tiwai/sound.git#for-linus +sound-asoc-fixes https://git.kernel.org/pub/scm/linux/kernel/git/broonie/sound.git#for-linus +regmap-fixes https://git.kernel.org/pub/scm/linux/kernel/git/broonie/regmap.git#for-linus +regulator-fixes https://git.kernel.org/pub/scm/linux/kernel/git/broonie/regulator.git#for-linus +spi-fixes https://git.kernel.org/pub/scm/linux/kernel/git/broonie/spi.git#for-linus +pci-current https://git.kernel.org/pub/scm/linux/kernel/git/pci/pci.git#for-linus +driver-core.current https://git.kernel.org/pub/scm/linux/kernel/git/driver-core/driver-core.git#driver-core-linus +tty.current https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/tty.git#tty-linus +usb.current https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/usb.git#usb-linus +usb-serial-fixes https://git.kernel.org/pub/scm/linux/kernel/git/johan/usb-serial.git#usb-linus +phy https://git.kernel.org/pub/scm/linux/kernel/git/phy/linux-phy.git#fixes +staging.current https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/staging.git#staging-linus +iio-fixes https://git.kernel.org/pub/scm/linux/kernel/git/jic23/iio.git#fixes-togreg +watchdog-fixes https://git.kernel.org/pub/scm/linux/kernel/git/groeck/linux-staging.git#watchdog +counter-current https://git.kernel.org/pub/scm/linux/kernel/git/wbg/counter.git#counter-current +char-misc.current https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/char-misc.git#char-misc-linus +soundwire-fixes https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/soundwire.git#fixes +thunderbolt-fixes https://git.kernel.org/pub/scm/linux/kernel/git/westeri/thunderbolt.git#fixes +input-current https://git.kernel.org/pub/scm/linux/kernel/git/dtor/input.git#for-linus +crypto-current https://git.kernel.org/pub/scm/linux/kernel/git/herbert/crypto-2.6.git#master +libcrypto-fixes https://git.kernel.org/pub/scm/linux/kernel/git/ebiggers/linux.git#libcrypto-fixes +vfio-fixes https://github.com/awilliam/linux-vfio.git#for-linus +kselftest-fixes https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git#fixes +dmaengine-fixes https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/dmaengine.git#fixes +backlight-fixes https://git.kernel.org/pub/scm/linux/kernel/git/lee/backlight.git#for-backlight-fixes +mtd-fixes https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git#mtd/fixes +mfd-fixes https://git.kernel.org/pub/scm/linux/kernel/git/lee/mfd.git#for-mfd-fixes +v4l-dvb-fixes git://linuxtv.org/media-ci/media-pending.git#fixes +reset-fixes https://git.kernel.org/pub/scm/linux/kernel/git/pza/linux#reset/fixes +mips-fixes https://git.kernel.org/pub/scm/linux/kernel/git/mips/linux.git#mips-fixes +at91-fixes https://git.kernel.org/pub/scm/linux/kernel/git/at91/linux.git#at91-fixes +omap-fixes https://git.kernel.org/pub/scm/linux/kernel/git/khilman/linux-omap.git#fixes +tegra-fixes https://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux.git#fixes +kvm-fixes git://git.kernel.org/pub/scm/virt/kvm/kvm.git#master +kvms390-fixes https://git.kernel.org/pub/scm/linux/kernel/git/kvms390/linux.git#master +kvm-arm-fixes https://git.kernel.org/pub/scm/linux/kernel/git/kvmarm/kvmarm.git#fixes +hwmon-fixes https://git.kernel.org/pub/scm/linux/kernel/git/groeck/linux-staging.git#hwmon +nvdimm-fixes https://git.kernel.org/pub/scm/linux/kernel/git/nvdimm/nvdimm.git#libnvdimm-fixes +cxl-fixes https://git.kernel.org/pub/scm/linux/kernel/git/cxl/cxl.git#fixes +dma-mapping-fixes https://git.kernel.org/pub/scm/linux/kernel/git/mszyprowski/linux.git#dma-mapping-fixes +drivers-x86-fixes https://git.kernel.org/pub/scm/linux/kernel/git/pdx86/platform-drivers-x86.git#fixes +samsung-krzk-fixes https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux.git#fixes +pinctrl-samsung-fixes https://git.kernel.org/pub/scm/linux/kernel/git/pinctrl/samsung.git#fixes +pinctrl-qcom-fixes https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#pinctrl-qcom/for-current +devicetree-fixes https://git.kernel.org/pub/scm/linux/kernel/git/robh/linux.git#dt/linus +dt-krzk-fixes https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-dt.git#fixes +scsi-fixes https://git.kernel.org/pub/scm/linux/kernel/git/mkp/scsi.git#fixes +drm-fixes https://gitlab.freedesktop.org/drm/kernel.git#drm-fixes +drm-intel-fixes https://gitlab.freedesktop.org/drm/i915/kernel.git#for-linux-next-fixes +mmc-fixes https://git.kernel.org/pub/scm/linux/kernel/git/ulfh/mmc.git#fixes +rtc-fixes https://git.kernel.org/pub/scm/linux/kernel/git/abelloni/linux.git#rtc-fixes +gnss-fixes https://git.kernel.org/pub/scm/linux/kernel/git/johan/gnss.git#gnss-linus +hyperv-fixes https://git.kernel.org/pub/scm/linux/kernel/git/hyperv/linux.git#hyperv-fixes +risc-v-fixes https://git.kernel.org/pub/scm/linux/kernel/git/riscv/linux.git#fixes +riscv-dt-fixes https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux.git#riscv-dt-fixes +riscv-soc-fixes https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux.git#riscv-soc-fixes +fpga-fixes https://git.kernel.org/pub/scm/linux/kernel/git/fpga/linux-fpga.git#fixes +spdx https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/spdx.git#spdx-linus +gpio-brgl-fixes https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#gpio/for-current +gpio-intel-fixes https://git.kernel.org/pub/scm/linux/kernel/git/andy/linux-gpio-intel.git#fixes +pinctrl-intel-fixes https://git.kernel.org/pub/scm/linux/kernel/git/pinctrl/intel.git#fixes +auxdisplay-fixes https://git.kernel.org/pub/scm/linux/kernel/git/andy/linux-auxdisplay.git#fixes +kunit-fixes https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git#kunit-fixes +renesas-fixes https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel.git#fixes +perf-current https://git.kernel.org/pub/scm/linux/kernel/git/perf/perf-tools.git#perf-tools +efi-fixes https://git.kernel.org/pub/scm/linux/kernel/git/efi/efi.git#urgent +battery-fixes https://git.kernel.org/pub/scm/linux/kernel/git/sre/linux-power-supply.git#fixes +iommufd-fixes https://git.kernel.org/pub/scm/linux/kernel/git/jgg/iommufd.git#for-rc +rust-fixes https://github.com/Rust-for-Linux/linux.git#rust-fixes +w1-fixes https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-w1.git#fixes +pmdomain-fixes https://git.kernel.org/pub/scm/linux/kernel/git/ulfh/linux-pm.git#fixes +i2c-andi-fixes https://git.kernel.org/pub/scm/linux/kernel/git/andi.shyti/linux.git#i2c/i2c-fixes +i2c-rust-fixes https://github.com/ikrtn/rust-for-linux#rust-i2c-fixes +sparc-fixes https://git.kernel.org/pub/scm/linux/kernel/git/alarsson/linux-sparc.git#for-linus +clk-fixes https://git.kernel.org/pub/scm/linux/kernel/git/clk/linux.git#clk-fixes +thead-clk-fixes https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git#thead-clk-fixes +tenstorrent-clk-fixes https://git.kernel.org/pub/scm/linux/kernel/git/tenstorrent/linux.git#tenstorrent-clk-fixes +fustini-config-fixes https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git#riscv-config-fixes +pwrseq-fixes https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#pwrseq/for-current +thead-dt-fixes https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git#thead-dt-fixes +ftrace-fixes https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git#ftrace/fixes +ring-buffer-fixes https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git#ring-buffer/fixes +trace-fixes https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git#trace/fixes +tracefs-fixes https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git#tracefs/fixes +spacemit-fixes https://git.kernel.org/pub/scm/linux/kernel/git/spacemit/linux#fixes +tip-fixes https://git.kernel.org/pub/scm/linux/kernel/git/tip/tip.git#tip/urgent +kexec-fixes https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git#kexec-fixes +liveupdate-fixes https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git#fixes +drm-msm-fixes https://gitlab.freedesktop.org/drm/msm.git#msm-fixes +uml-fixes https://git.kernel.org/pub/scm/linux/kernel/git/uml/linux.git#fixes +fwctl-fixes https://git.kernel.org/pub/scm/linux/kernel/git/fwctl/fwctl.git#for-rc +devsec-tsm-fixes https://git.kernel.org/pub/scm/linux/kernel/git/devsec/tsm.git#fixes +drm-rust-fixes https://gitlab.freedesktop.org/drm/rust/kernel.git#for-linux-next-fixes +tenstorrent-dt-fixes https://git.kernel.org/pub/scm/linux/kernel/git/tenstorrent/linux.git#tenstorrent-dt-fixes +nfc-fixes https://codeberg.org/linux-nfc/linux.git#for-linus +mm-nonmm-hotfixes-stable https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm#mm-nonmm-hotfixes-stable +mm-nonmm-hotfixes-unstable https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm#mm-nonmm-hotfixes-unstable +drm-misc-fixes https://gitlab.freedesktop.org/drm/misc/kernel.git#for-linux-next-fixes +rust https://github.com/Rust-for-Linux/linux.git#rust-next +rust-interop https://github.com/Rust-for-Linux/linux.git#interop-next +rust-alloc https://github.com/Rust-for-Linux/linux.git#alloc-next +rust-io https://github.com/Rust-for-Linux/linux.git#io-next +rust-pin-init https://github.com/Rust-for-Linux/linux.git#pin-init-next +rust-timekeeping https://github.com/Rust-for-Linux/linux.git#timekeeping-next +rust-xarray https://github.com/Rust-for-Linux/linux.git#xarray-next +rust-analyzer https://github.com/Rust-for-Linux/linux.git#rust-analyzer-next +mm https://git.kernel.org/pub/scm/linux/kernel/git/mm/linux.git#for-next +mm-nonmm-stable https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm#mm-nonmm-stable +mm-nonmm-unstable https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm#mm-nonmm-unstable +kbuild https://git.kernel.org/pub/scm/linux/kernel/git/kbuild/linux.git#kbuild-for-next +clang-fixes https://git.kernel.org/pub/scm/linux/kernel/git/nathan/linux.git#clang-fixes-for-next +clang-format https://github.com/ojeda/linux.git#clang-format +perf https://git.kernel.org/pub/scm/linux/kernel/git/perf/perf-tools-next.git#perf-tools-next +compiler-attributes https://github.com/ojeda/linux.git#compiler-attributes +dma-mapping https://git.kernel.org/pub/scm/linux/kernel/git/mszyprowski/linux.git#dma-mapping-for-next +asm-generic https://git.kernel.org/pub/scm/linux/kernel/git/arnd/asm-generic#master +alpha https://git.kernel.org/pub/scm/linux/kernel/git/mattst88/alpha.git#alpha-next +arm https://git.kernel.org/pub/scm/linux/kernel/git/rmk/linux.git#for-next +arm64 https://git.kernel.org/pub/scm/linux/kernel/git/arm64/linux#for-next/core +arm-perf https://git.kernel.org/pub/scm/linux/kernel/git/will/linux.git#for-next/perf +arm-soc https://git.kernel.org/pub/scm/linux/kernel/git/soc/soc.git#for-next +amlogic https://git.kernel.org/pub/scm/linux/kernel/git/amlogic/linux.git#for-next +asahi-soc https://github.com/AsahiLinux/linux.git#asahi-soc/for-next +at91 https://git.kernel.org/pub/scm/linux/kernel/git/at91/linux.git#at91-next +bmc https://git.kernel.org/pub/scm/linux/kernel/git/bmc/linux.git#for-next +broadcom https://github.com/Broadcom/stblinux.git#next +cix https://git.kernel.org/pub/scm/linux/kernel/git/peter.chen/cix.git#for-next +davinci https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#davinci/for-next +drivers-memory https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-mem-ctrl.git#for-next +fsl https://git.kernel.org/pub/scm/linux/kernel/git/chleroy/linux.git#soc_fsl +imx-mxs https://git.kernel.org/pub/scm/linux/kernel/git/frank.li/linux.git#for-next +mediatek https://git.kernel.org/pub/scm/linux/kernel/git/mediatek/linux.git#for-next +mvebu https://git.kernel.org/pub/scm/linux/kernel/git/gclement/mvebu.git#for-next +omap https://git.kernel.org/pub/scm/linux/kernel/git/khilman/linux-omap.git#for-next +qcom https://git.kernel.org/pub/scm/linux/kernel/git/qcom/linux.git#for-next +realtek https://git.kernel.org/pub/scm/linux/kernel/git/yu_chun/linux.git#for-next +renesas https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel.git#next +reset https://git.kernel.org/pub/scm/linux/kernel/git/pza/linux#reset/next +rockchip https://git.kernel.org/pub/scm/linux/kernel/git/mmind/linux-rockchip.git#for-next +samsung-krzk https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux.git#for-next +scmi https://git.kernel.org/pub/scm/linux/kernel/git/sudeep.holla/linux.git#for-linux-next +sophgo https://github.com/sophgo/linux.git#for-next +sophgo-soc https://github.com/sophgo/linux.git#soc-for-next +spacemit https://git.kernel.org/pub/scm/linux/kernel/git/spacemit/linux#for-next +stm32 https://git.kernel.org/pub/scm/linux/kernel/git/atorgue/stm32.git#stm32-next +sunxi https://git.kernel.org/pub/scm/linux/kernel/git/sunxi/linux.git#sunxi/for-next +tee https://git.kernel.org/pub/scm/linux/kernel/git/jenswi/linux-tee.git#next +tegra https://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux.git#for-next +tenstorrent-dt https://git.kernel.org/pub/scm/linux/kernel/git/tenstorrent/linux.git#tenstorrent-dt-for-next +fustini-config https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git#riscv-config-for-next +thead-dt https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git#thead-dt-for-next +ti https://git.kernel.org/pub/scm/linux/kernel/git/ti/linux.git#ti-next +xilinx https://github.com/Xilinx/linux-xlnx.git#for-next +socfpga https://git.kernel.org/pub/scm/linux/kernel/git/dinguyen/linux.git#for-next +clk https://git.kernel.org/pub/scm/linux/kernel/git/clk/linux.git#clk-next +clk-imx https://git.kernel.org/pub/scm/linux/kernel/git/abelvesa/linux.git#for-next +clk-renesas https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-drivers.git#renesas-clk +thead-clk https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git#thead-clk-for-next +tenstorrent-clk https://git.kernel.org/pub/scm/linux/kernel/git/tenstorrent/linux.git#tenstorrent-clk-for-next +csky https://github.com/c-sky/csky-linux.git#linux-next +loongarch https://git.kernel.org/pub/scm/linux/kernel/git/chenhuacai/linux-loongson.git#loongarch-next +m68k https://git.kernel.org/pub/scm/linux/kernel/git/geert/linux-m68k.git#for-next +m68knommu https://git.kernel.org/pub/scm/linux/kernel/git/gerg/m68knommu.git#for-next +microblaze git://git.monstr.eu/linux-2.6-microblaze.git#next +mips https://git.kernel.org/pub/scm/linux/kernel/git/mips/linux.git#mips-next +openrisc https://github.com/openrisc/linux.git#for-next +parisc-hd https://git.kernel.org/pub/scm/linux/kernel/git/deller/parisc-linux.git#for-next +powerpc https://git.kernel.org/pub/scm/linux/kernel/git/powerpc/linux.git#next +risc-v https://git.kernel.org/pub/scm/linux/kernel/git/riscv/linux.git#for-next +riscv-dt https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux.git#riscv-dt-for-next +riscv-soc https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux.git#riscv-soc-for-next +s390 https://git.kernel.org/pub/scm/linux/kernel/git/s390/linux.git#for-next +sh https://git.kernel.org/pub/scm/linux/kernel/git/glaubitz/sh-linux.git#for-next +sparc https://git.kernel.org/pub/scm/linux/kernel/git/alarsson/linux-sparc.git#for-next +uml https://git.kernel.org/pub/scm/linux/kernel/git/uml/linux.git#next +xtensa https://github.com/jcmvbkbc/linux-xtensa.git#xtensa-for-next +printk https://git.kernel.org/pub/scm/linux/kernel/git/printk/linux.git#for-next +pci https://git.kernel.org/pub/scm/linux/kernel/git/pci/pci.git#next +pstore https://git.kernel.org/pub/scm/linux/kernel/git/kees/linux.git#for-next/pstore +hid https://git.kernel.org/pub/scm/linux/kernel/git/hid/hid.git#for-next +i2c https://git.kernel.org/pub/scm/linux/kernel/git/wsa/linux.git#i2c/for-next +i2c-andi https://git.kernel.org/pub/scm/linux/kernel/git/andi.shyti/linux.git#i2c/i2c-next +i2c-rust https://github.com/ikrtn/rust-for-linux#rust-i2c-next +i3c https://git.kernel.org/pub/scm/linux/kernel/git/i3c/linux.git#i3c/next +dmi https://git.kernel.org/pub/scm/linux/kernel/git/jdelvare/staging.git#dmi-for-next +hwmon-staging https://git.kernel.org/pub/scm/linux/kernel/git/groeck/linux-staging.git#hwmon-next +jc_docs git://git.lwn.net/linux.git#docs-next +v4l-dvb git://linuxtv.org/media-ci/media-pending.git#next +v4l-dvb-next git://linuxtv.org/mchehab/media-next.git#master +pm https://git.kernel.org/pub/scm/linux/kernel/git/rafael/linux-pm.git#linux-next +cpufreq-arm https://git.kernel.org/pub/scm/linux/kernel/git/vireshk/pm.git#cpufreq/arm/linux-next +cpupower https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux.git#cpupower +devfreq https://git.kernel.org/pub/scm/linux/kernel/git/chanwoo/linux.git#devfreq-next +pmdomain https://git.kernel.org/pub/scm/linux/kernel/git/ulfh/linux-pm.git#next +opp https://git.kernel.org/pub/scm/linux/kernel/git/vireshk/pm.git#opp/linux-next +thermal https://git.kernel.org/pub/scm/linux/kernel/git/thermal/linux.git#thermal/linux-next +rdma https://git.kernel.org/pub/scm/linux/kernel/git/rdma/rdma.git#for-next +net-next https://git.kernel.org/pub/scm/linux/kernel/git/netdev/net-next.git#main +bpf-next https://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf-next.git#for-next +ipsec-next https://git.kernel.org/pub/scm/linux/kernel/git/klassert/ipsec-next.git#master +mlx5-next https://git.kernel.org/pub/scm/linux/kernel/git/mellanox/linux.git#mlx5-next +netfilter-next https://git.kernel.org/pub/scm/linux/kernel/git/netfilter/nf-next.git#main +ipvs-next https://git.kernel.org/pub/scm/linux/kernel/git/horms/ipvs-next.git#main +bluetooth https://git.kernel.org/pub/scm/linux/kernel/git/bluetooth/bluetooth-next.git#master +wireless-next https://git.kernel.org/pub/scm/linux/kernel/git/wireless/wireless-next.git#for-next +ath-next https://git.kernel.org/pub/scm/linux/kernel/git/ath/ath.git#for-next +iwlwifi-next https://git.kernel.org/pub/scm/linux/kernel/git/iwlwifi/iwlwifi-next.git#next +wpan-next https://git.kernel.org/pub/scm/linux/kernel/git/wpan/wpan-next.git#master +wpan-staging https://git.kernel.org/pub/scm/linux/kernel/git/wpan/wpan-next.git#staging +mtd https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git#mtd/next +nand https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git#nand/next +spi-nor https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git#spi-nor/next +crypto https://git.kernel.org/pub/scm/linux/kernel/git/herbert/cryptodev-2.6.git#master +libcrypto https://git.kernel.org/pub/scm/linux/kernel/git/ebiggers/linux.git#libcrypto-next +drm https://gitlab.freedesktop.org/drm/kernel.git#drm-next +drm-exynos https://git.kernel.org/pub/scm/linux/kernel/git/daeinki/drm-exynos.git#for-linux-next +drm-misc https://gitlab.freedesktop.org/drm/misc/kernel.git#for-linux-next +amdgpu https://gitlab.freedesktop.org/agd5f/linux.git#drm-next +drm-intel https://gitlab.freedesktop.org/drm/i915/kernel.git#for-linux-next +drm-msm https://gitlab.freedesktop.org/drm/msm.git#msm-next +drm-msm-lumag https://gitlab.freedesktop.org/lumag/msm.git#msm-next-lumag +drm-xe https://gitlab.freedesktop.org/drm/xe/kernel.git#drm-xe-next +drm-rust https://gitlab.freedesktop.org/drm/rust/kernel.git#for-linux-next +drm-nova https://gitlab.freedesktop.org/drm/nova.git#nova-next +etnaviv https://git.pengutronix.de/git/lst/linux#etnaviv/next +fbdev https://git.kernel.org/pub/scm/linux/kernel/git/deller/linux-fbdev.git#for-next +regmap https://git.kernel.org/pub/scm/linux/kernel/git/broonie/regmap.git#for-next +sound https://git.kernel.org/pub/scm/linux/kernel/git/tiwai/sound.git#for-next +ieee1394 https://git.kernel.org/pub/scm/linux/kernel/git/ieee1394/linux1394.git#for-next +sound-asoc https://git.kernel.org/pub/scm/linux/kernel/git/broonie/sound.git#for-next +modules https://git.kernel.org/pub/scm/linux/kernel/git/modules/linux.git#modules-next +input https://git.kernel.org/pub/scm/linux/kernel/git/dtor/input.git#next +block https://git.kernel.org/pub/scm/linux/kernel/git/axboe/linux.git#for-next +device-mapper https://git.kernel.org/pub/scm/linux/kernel/git/device-mapper/linux-dm.git#for-next +libata https://git.kernel.org/pub/scm/linux/kernel/git/libata/linux#for-next +pcmcia https://git.kernel.org/pub/scm/linux/kernel/git/brodo/linux.git#pcmcia-next +mmc https://git.kernel.org/pub/scm/linux/kernel/git/ulfh/mmc.git#next +mfd https://git.kernel.org/pub/scm/linux/kernel/git/lee/mfd.git#for-mfd-next +backlight https://git.kernel.org/pub/scm/linux/kernel/git/lee/backlight.git#for-backlight-next +battery https://git.kernel.org/pub/scm/linux/kernel/git/sre/linux-power-supply.git#for-next +regulator https://git.kernel.org/pub/scm/linux/kernel/git/broonie/regulator.git#for-next +security https://git.kernel.org/pub/scm/linux/kernel/git/pcmoore/lsm.git#next +apparmor https://git.kernel.org/pub/scm/linux/kernel/git/jj/linux-apparmor#apparmor-next +integrity https://git.kernel.org/pub/scm/linux/kernel/git/zohar/linux-integrity#next-integrity +selinux https://git.kernel.org/pub/scm/linux/kernel/git/pcmoore/selinux.git#next +smack https://github.com/cschaufler/smack-next#next +tomoyo git://git.code.sf.net/p/tomoyo/tomoyo.git#master +tpmdd-tpm https://git.kernel.org/pub/scm/linux/kernel/git/jarkko/linux-tpmdd.git#for-next-tpm +tpmdd-keys https://git.kernel.org/pub/scm/linux/kernel/git/jarkko/linux-tpmdd.git#for-next-keys +watchdog https://git.kernel.org/pub/scm/linux/kernel/git/groeck/linux-staging.git#watchdog-next +iommu https://git.kernel.org/pub/scm/linux/kernel/git/iommu/linux.git#next +audit https://git.kernel.org/pub/scm/linux/kernel/git/pcmoore/audit.git#next +devicetree https://git.kernel.org/pub/scm/linux/kernel/git/robh/linux.git#for-next +dt-krzk https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-dt.git#for-next +mailbox https://git.kernel.org/pub/scm/linux/kernel/git/jassibrar/mailbox.git#for-next +spi https://git.kernel.org/pub/scm/linux/kernel/git/broonie/spi.git#for-next +tip https://git.kernel.org/pub/scm/linux/kernel/git/tip/tip.git#master +kexec https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git#kexec-next +liveupdate https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git#next +clockevents https://git.kernel.org/pub/scm/linux/kernel/git/daniel.lezcano/linux.git#timers/drivers/next +edac https://git.kernel.org/pub/scm/linux/kernel/git/ras/ras.git#edac-for-next +ftrace https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git#for-next +rcu https://git.kernel.org/pub/scm/linux/kernel/git/rcu/linux#next +paulmck https://git.kernel.org/pub/scm/linux/kernel/git/paulmck/linux-rcu.git#non-rcu/next +kvm git://git.kernel.org/pub/scm/virt/kvm/kvm.git#next +kvm-arm https://git.kernel.org/pub/scm/linux/kernel/git/kvmarm/kvmarm.git#next +kvms390 https://git.kernel.org/pub/scm/linux/kernel/git/kvms390/linux.git#next +kvm-ppc https://git.kernel.org/pub/scm/linux/kernel/git/powerpc/linux.git#topic/ppc-kvm +kvm-riscv https://github.com/kvm-riscv/linux.git#riscv_kvm_next +kvm-x86 https://github.com/kvm-x86/linux.git#next +xen-tip https://git.kernel.org/pub/scm/linux/kernel/git/xen/tip.git#linux-next +percpu https://git.kernel.org/pub/scm/linux/kernel/git/dennis/percpu.git#for-next +workqueues https://git.kernel.org/pub/scm/linux/kernel/git/tj/wq.git#for-next +sched-ext https://git.kernel.org/pub/scm/linux/kernel/git/tj/sched_ext.git#for-next +drivers-x86 https://git.kernel.org/pub/scm/linux/kernel/git/pdx86/platform-drivers-x86.git#for-next +chrome-platform https://git.kernel.org/pub/scm/linux/kernel/git/chrome-platform/linux.git#for-next +chrome-platform-firmware https://git.kernel.org/pub/scm/linux/kernel/git/chrome-platform/linux.git#for-firmware-next +hsi https://git.kernel.org/pub/scm/linux/kernel/git/sre/linux-hsi.git#for-next +leds-lj https://git.kernel.org/pub/scm/linux/kernel/git/lee/leds.git#for-leds-next +ipmi https://github.com/cminyard/linux-ipmi.git#for-next +driver-core https://git.kernel.org/pub/scm/linux/kernel/git/driver-core/driver-core.git#driver-core-next +usb https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/usb.git#usb-next +thunderbolt https://git.kernel.org/pub/scm/linux/kernel/git/westeri/thunderbolt.git#next +usb-serial https://git.kernel.org/pub/scm/linux/kernel/git/johan/usb-serial.git#usb-next +tty https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/tty.git#tty-next +char-misc https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/char-misc.git#char-misc-next +coresight https://git.kernel.org/pub/scm/linux/kernel/git/coresight/linux.git#next +fastrpc https://git.kernel.org/pub/scm/linux/kernel/git/srini/fastrpc.git#for-next +fpga https://git.kernel.org/pub/scm/linux/kernel/git/fpga/linux-fpga.git#for-next +icc https://git.kernel.org/pub/scm/linux/kernel/git/djakov/icc.git#icc-next +iio https://git.kernel.org/pub/scm/linux/kernel/git/jic23/iio.git#togreg +nfc https://codeberg.org/linux-nfc/linux.git#for-next +phy-next https://git.kernel.org/pub/scm/linux/kernel/git/phy/linux-phy.git#next +soundwire https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/soundwire.git#next +extcon https://git.kernel.org/pub/scm/linux/kernel/git/chanwoo/extcon.git#extcon-next +gnss https://git.kernel.org/pub/scm/linux/kernel/git/johan/gnss.git#gnss-next +vfio https://github.com/awilliam/linux-vfio.git#next +w1 https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-w1.git#for-next +spmi https://git.kernel.org/pub/scm/linux/kernel/git/sboyd/spmi.git#spmi-next +staging https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/staging.git#staging-next +counter-next https://git.kernel.org/pub/scm/linux/kernel/git/wbg/counter.git#counter-next +mux https://gitlab.com/peda-linux/mux.git#for-next +dmaengine https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/dmaengine.git#next +cgroup https://git.kernel.org/pub/scm/linux/kernel/git/tj/cgroup.git#for-next +scsi https://git.kernel.org/pub/scm/linux/kernel/git/jejb/scsi.git#for-next +scsi-mkp https://git.kernel.org/pub/scm/linux/kernel/git/mkp/scsi.git#for-next +vhost https://git.kernel.org/pub/scm/linux/kernel/git/mst/vhost.git#linux-next +rpmsg https://git.kernel.org/pub/scm/linux/kernel/git/remoteproc/linux.git#for-next +gpio-brgl https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#gpio/for-next +gpio-intel https://git.kernel.org/pub/scm/linux/kernel/git/andy/linux-gpio-intel.git#for-next +pinctrl https://git.kernel.org/pub/scm/linux/kernel/git/linusw/linux-pinctrl.git#for-next +pinctrl-intel https://git.kernel.org/pub/scm/linux/kernel/git/pinctrl/intel.git#for-next +pinctrl-renesas https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-drivers.git#renesas-pinctrl +pinctrl-samsung https://git.kernel.org/pub/scm/linux/kernel/git/pinctrl/samsung.git#for-next +pinctrl-qcom https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#pinctrl-qcom/for-next +pwm https://git.kernel.org/pub/scm/linux/kernel/git/ukleinek/linux.git#pwm/for-next +ktest https://git.kernel.org/pub/scm/linux/kernel/git/rostedt/linux-ktest.git#for-next +kselftest https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git#next +kunit https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git#test +kunit-next https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git#kunit +livepatching https://git.kernel.org/pub/scm/linux/kernel/git/livepatching/livepatching.git#for-next +rtc https://git.kernel.org/pub/scm/linux/kernel/git/abelloni/linux.git#rtc-next +nvdimm https://git.kernel.org/pub/scm/linux/kernel/git/nvdimm/nvdimm.git#libnvdimm-for-next +at24 https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#at24/for-next +ntb https://github.com/jonmason/ntb.git#ntb-next +seccomp https://git.kernel.org/pub/scm/linux/kernel/git/kees/linux.git#for-next/seccomp +slimbus https://git.kernel.org/pub/scm/linux/kernel/git/srini/slimbus.git#for-next +nvmem https://git.kernel.org/pub/scm/linux/kernel/git/srini/nvmem.git#for-next +hyperv https://git.kernel.org/pub/scm/linux/kernel/git/hyperv/linux.git#hyperv-next +auxdisplay https://git.kernel.org/pub/scm/linux/kernel/git/andy/linux-auxdisplay.git#for-next +kgdb https://git.kernel.org/pub/scm/linux/kernel/git/danielt/linux.git#kgdb/for-next +hmm https://git.kernel.org/pub/scm/linux/kernel/git/rdma/rdma.git#hmm +cfi https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git#cfi/next +mhi https://git.kernel.org/pub/scm/linux/kernel/git/mani/mhi.git#mhi-next +cxl https://git.kernel.org/pub/scm/linux/kernel/git/cxl/cxl.git#next +zstd https://github.com/terrelln/linux.git#zstd-next +efi https://git.kernel.org/pub/scm/linux/kernel/git/efi/efi.git#next +unicode https://git.kernel.org/pub/scm/linux/kernel/git/krisman/unicode.git#for-next +random https://git.kernel.org/pub/scm/linux/kernel/git/crng/random.git#master +landlock https://git.kernel.org/pub/scm/linux/kernel/git/mic/linux.git#next +sysctl https://git.kernel.org/pub/scm/linux/kernel/git/sysctl/sysctl.git#sysctl-next +execve https://git.kernel.org/pub/scm/linux/kernel/git/kees/linux.git#for-next/execve +bitmap https://github.com/norov/linux.git#bitmap-for-next +hte https://git.kernel.org/pub/scm/linux/kernel/git/pateldipen1984/linux.git#for-next +kspp https://git.kernel.org/pub/scm/linux/kernel/git/kees/linux.git#for-next/kspp +nolibc https://git.kernel.org/pub/scm/linux/kernel/git/nolibc/linux-nolibc.git#for-next +iommufd https://git.kernel.org/pub/scm/linux/kernel/git/jgg/iommufd.git#for-next +turbostat https://git.kernel.org/pub/scm/linux/kernel/git/lenb/linux.git#next +pwrseq https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#pwrseq/for-next +capabilities-next https://git.kernel.org/pub/scm/linux/kernel/git/sergeh/linux.git#caps-next +ipe https://git.kernel.org/pub/scm/linux/kernel/git/wufan/ipe.git#next +kcsan https://git.kernel.org/pub/scm/linux/kernel/git/melver/linux.git#next +crc https://git.kernel.org/pub/scm/linux/kernel/git/ebiggers/linux.git#crc-next +keys-next https://git.kernel.org/pub/scm/linux/kernel/git/dhowells/linux-fs.git#keys-next +fwctl https://git.kernel.org/pub/scm/linux/kernel/git/fwctl/fwctl.git#for-next +devsec-tsm https://git.kernel.org/pub/scm/linux/kernel/git/devsec/tsm.git#next +hisilicon https://github.com/hisilicon/linux-hisi.git#for-next +device-id https://git.kernel.org/pub/scm/linux/kernel/git/ukleinek/linux.git#device-id-rework +kthread https://git.kernel.org/pub/scm/linux/kernel/git/frederic/linux-dynticks.git#for-next +pagemap-headers git://git.infradead.org/users/willy/pagecache.git#headers diff --git a/Next/merge.log b/Next/merge.log new file mode 100644 index 00000000000000..4e7f182eefe5a3 --- /dev/null +++ b/Next/merge.log @@ -0,0 +1,20065 @@ +$ date -R +Fri, 02 Oct 2026 11:46:12 +0100 +$ git checkout master +Already on 'master' +$ git reset --hard stable +Updating files: 41% (5065/12270) Updating files: 42% (5154/12270) Updating files: 43% (5277/12270) Updating files: 44% (5399/12270) Updating files: 45% (5522/12270) Updating files: 46% (5645/12270) Updating files: 47% (5767/12270) Updating files: 48% (5890/12270) Updating files: 49% (6013/12270) Updating files: 50% (6135/12270) Updating files: 51% (6258/12270) Updating files: 52% (6381/12270) Updating files: 53% (6504/12270) Updating files: 54% (6626/12270) Updating files: 55% (6749/12270) Updating files: 56% (6872/12270) Updating files: 57% (6994/12270) Updating files: 58% (7117/12270) Updating files: 59% (7240/12270) Updating files: 60% (7362/12270) Updating files: 61% (7485/12270) Updating files: 62% (7608/12270) Updating files: 63% (7731/12270) Updating files: 64% (7853/12270) Updating files: 65% (7976/12270) Updating files: 66% (8099/12270) Updating files: 67% (8221/12270) Updating files: 68% (8344/12270) Updating files: 69% (8467/12270) Updating files: 69% (8559/12270) Updating files: 70% (8589/12270) Updating files: 71% (8712/12270) Updating files: 72% (8835/12270) Updating files: 73% (8958/12270) Updating files: 74% (9080/12270) Updating files: 75% (9203/12270) Updating files: 76% (9326/12270) Updating files: 77% (9448/12270) Updating files: 78% (9571/12270) Updating files: 79% (9694/12270) Updating files: 80% (9816/12270) Updating files: 81% (9939/12270) Updating files: 82% (10062/12270) Updating files: 83% (10185/12270) Updating files: 84% (10307/12270) Updating files: 85% (10430/12270) Updating files: 86% (10553/12270) Updating files: 87% (10675/12270) Updating files: 88% (10798/12270) Updating files: 89% (10921/12270) Updating files: 90% (11043/12270) Updating files: 91% (11166/12270) Updating files: 92% (11289/12270) Updating files: 93% (11412/12270) Updating files: 94% (11534/12270) Updating files: 95% (11657/12270) Updating files: 96% (11780/12270) Updating files: 97% (11902/12270) Updating files: 98% (12025/12270) Updating files: 99% (12148/12270) Updating files: 100% (12270/12270) Updating files: 100% (12270/12270), done. +HEAD is now at 551c722f40809 Merge tag 'rtc-7.3-fixes' of git://git.kernel.org/pub/scm/linux/kernel/git/abelloni/linux +Merging origin/master (ce1e0223d8ad4 Merge tag 'devicetree-fixes-for-7.3-2' of git://git.kernel.org/pub/scm/linux/kernel/git/robh/linux) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/torvalds/linux.git origin/master +Updating 551c722f40809..ce1e0223d8ad4 +Fast-forward (no commit created; -m option ignored) + .mailmap | 2 + + MAINTAINERS | 9 +- + arch/arm64/kvm/emulate-nested.c | 3 - + arch/arm64/kvm/hyp/include/nvhe/pkvm.h | 9 + + arch/arm64/kvm/hyp/nvhe/hyp-main.c | 7 +- + arch/arm64/kvm/hyp/nvhe/pkvm.c | 25 +- + arch/arm64/kvm/mmu.c | 47 +-- + arch/arm64/kvm/vgic/vgic-init.c | 12 +- + arch/arm64/kvm/vgic/vgic-its.c | 12 +- + arch/arm64/kvm/vgic/vgic-v2.c | 6 +- + arch/arm64/kvm/vgic/vgic-v3.c | 6 +- + arch/arm64/kvm/vgic/vgic.c | 21 +- + arch/powerpc/kvm/book3s_hv.c | 2 - + arch/riscv/kvm/aia_device.c | 2 +- + arch/s390/kvm/s390/s390.c | 5 +- + arch/x86/crypto/aesni-intel_glue.c | 2 +- + arch/x86/kvm/mmu/mmu.c | 6 + + arch/x86/kvm/svm/sev.c | 57 ++-- + arch/x86/kvm/svm/svm.c | 43 ++- + arch/x86/kvm/svm/svm.h | 1 + + arch/x86/kvm/vmx/nested.c | 6 + + arch/x86/kvm/vmx/tdx.c | 5 - + crypto/aes.c | 51 ++- + drivers/bluetooth/btintel.c | 15 +- + drivers/bluetooth/btintel_pcie.c | 12 +- + drivers/cpufreq/cpufreq.c | 4 +- + drivers/cpufreq/intel_pstate.c | 10 + + drivers/net/ethernet/broadcom/bcmsysport.c | 8 +- + drivers/net/ethernet/broadcom/genet/bcmgenet.c | 34 +- + drivers/net/ethernet/broadcom/genet/bcmgenet.h | 2 + + drivers/net/ethernet/google/gve/gve_tx_dqo.c | 14 +- + drivers/net/ethernet/marvell/mvneta.c | 2 +- + drivers/net/ethernet/marvell/octeontx2/af/mcs.c | 4 +- + drivers/net/ethernet/marvell/octeontx2/nic/cn20k.c | 11 +- + .../ethernet/marvell/octeontx2/nic/otx2_common.c | 18 +- + .../ethernet/marvell/octeontx2/nic/otx2_common.h | 9 + + drivers/net/ethernet/mediatek/mtk_eth_soc.c | 38 ++- + drivers/net/ethernet/mellanox/mlx5/core/en/xdp.c | 3 + + drivers/net/ethernet/mellanox/mlx5/core/lag/lag.c | 10 +- + .../net/ethernet/microchip/sparx5/sparx5_main.c | 1 + + .../net/ethernet/microchip/sparx5/sparx5_main.h | 1 + + .../net/ethernet/microchip/sparx5/sparx5_netdev.c | 1 + + drivers/net/ethernet/microchip/sparx5/sparx5_ptp.c | 3 + + drivers/net/ethernet/microchip/vcap/vcap_api.c | 2 +- + drivers/net/ethernet/realtek/r8169_main.c | 6 + + drivers/net/ethernet/stmicro/stmmac/stmmac_main.c | 83 +++-- + drivers/net/pcs/pcs-rzn1-miic.c | 3 + + drivers/net/pfcp.c | 3 + + drivers/net/phy/aquantia/aquantia_main.c | 5 +- + drivers/net/phy/qcom/at803x.c | 8 +- + drivers/net/usb/qmi_wwan.c | 1 + + drivers/net/wireless/ath/ath11k/mac.c | 1 + + drivers/net/wireless/ath/ath9k/hif_usb.c | 21 +- + drivers/net/wireless/ath/ath9k/htc.h | 1 - + drivers/net/wireless/ath/ath9k/htc_drv_init.c | 7 +- + drivers/net/wireless/ath/ath9k/htc_drv_txrx.c | 8 +- + drivers/net/wireless/ath/ath9k/wmi.c | 28 +- + drivers/net/wireless/ath/ath9k/wmi.h | 1 + + drivers/net/wireless/intel/iwlegacy/3945-mac.c | 1 + + drivers/net/wireless/intersil/p54/fwio.c | 42 ++- + drivers/net/wireless/silabs/wfx/hif_tx.c | 4 +- + drivers/net/wireless/st/cw1200/bh.c | 8 +- + drivers/net/wireless/st/cw1200/txrx.c | 14 + + drivers/net/wireless/ti/wlcore/main.c | 4 +- + drivers/of/base.c | 10 +- + drivers/of/irq.c | 2 + + drivers/of/overlay.c | 17 +- + include/linux/cpufreq.h | 3 + + include/linux/kvm_host.h | 1 - + include/linux/netdevice.h | 3 + + include/linux/netfilter_netdev.h | 2 +- + include/linux/rtnetlink.h | 7 +- + include/linux/skbuff.h | 2 +- + include/net/bluetooth/hci_core.h | 5 +- + include/net/gso.h | 8 + + include/net/netfilter/nf_flow_table.h | 2 +- + include/net/sch_generic.h | 9 + + include/net/xdp.h | 5 + + kernel/audit_tree.c | 36 +- + kernel/sysctl.c | 2 +- + kernel/time/jiffies.c | 2 + + net/bluetooth/hci_conn.c | 18 +- + net/bluetooth/hci_core.c | 54 ++- + net/bluetooth/hci_debugfs.c | 2 +- + net/bluetooth/hci_sync.c | 19 +- + net/bluetooth/rfcomm/tty.c | 7 +- + net/bluetooth/smp.c | 2 + + net/core/dev.c | 78 +++-- + net/core/dev_ioctl.c | 3 +- + net/core/gso.c | 1 + + net/core/page_pool.c | 2 +- + net/core/skbuff.c | 8 +- + net/core/sock.c | 4 +- + net/ethtool/tsconfig.c | 24 +- + net/ipv4/af_inet.c | 3 + + net/ipv4/tcp_input.c | 6 + + net/ipv6/addrconf.c | 2 +- + net/ipv6/ip6_offload.c | 4 + + net/ipv6/seg6_iptunnel.c | 2 +- + net/ipv6/seg6_local.c | 140 ++++++-- + net/mac80211/agg-rx.c | 2 +- + net/mac80211/chan.c | 50 ++- + net/mac80211/drop.h | 1 + + net/mac80211/fils_aead.c | 8 + + net/mac80211/ieee80211_i.h | 2 +- + net/mac80211/iface.c | 11 +- + net/mac80211/mesh_pathtbl.c | 30 +- + net/mac80211/mlme.c | 4 +- + net/mac80211/rc80211_minstrel_ht.c | 89 ++++- + net/mac80211/rx.c | 6 + + net/mac80211/spectmgmt.c | 4 +- + net/mac80211/sta_info.c | 6 +- + net/mac80211/status.c | 70 ++++ + net/mac80211/tx.c | 45 ++- + net/mctp/device.c | 6 +- + net/netfilter/ipset/ip_set_bitmap_gen.h | 2 +- + net/netfilter/ipvs/ip_vs_conn.c | 3 + + net/netfilter/ipvs/ip_vs_lblc.c | 4 + + net/netfilter/ipvs/ip_vs_lblcr.c | 3 + + net/netfilter/ipvs/ip_vs_sync.c | 33 +- + net/netfilter/nf_flow_table_core.c | 7 +- + net/netfilter/nf_flow_table_offload.c | 14 +- + net/netfilter/nf_flow_table_path.c | 3 + + net/netfilter/nf_nat_bpf.c | 3 + + net/netfilter/nf_nat_core.c | 5 +- + net/netfilter/nft_flow_offload.c | 7 +- + net/netfilter/nft_set_rbtree.c | 2 + + net/packet/af_packet.c | 3 + + net/sched/act_ct.c | 2 +- + net/sched/act_skbedit.c | 5 +- + net/sched/cls_api.c | 15 +- + net/sched/sch_codel.c | 2 +- + net/sched/sch_fq_codel.c | 4 +- + net/sctp/socket.c | 2 +- + net/tipc/crypto.c | 20 +- + net/wireless/nl80211.c | 11 +- + net/wireless/scan.c | 22 +- + rust/kernel/net/netlink.rs | 9 + + tools/testing/selftests/kvm/Makefile.kvm | 2 + + .../testing/selftests/kvm/arm64/hidden_features.c | 184 ++++++++++ + .../testing/selftests/kvm/x86/nested_x2apic_test.c | 222 ++++++++++++ + tools/testing/selftests/net/.gitignore | 1 + + tools/testing/selftests/net/Makefile | 1 + + .../selftests/net/packetdrill/tcp_old_ack_ts.pkt | 22 ++ + tools/testing/selftests/net/udp_splice_checksum.c | 378 +++++++++++++++++++++ + virt/kvm/kvm_main.c | 35 +- + 146 files changed, 2112 insertions(+), 533 deletions(-) + create mode 100644 tools/testing/selftests/kvm/arm64/hidden_features.c + create mode 100644 tools/testing/selftests/kvm/x86/nested_x2apic_test.c + create mode 100644 tools/testing/selftests/net/packetdrill/tcp_old_ack_ts.pkt + create mode 100644 tools/testing/selftests/net/udp_splice_checksum.c +Merging ext4-fixes/fixes (981fcc5674e67 jbd2: fix deadlock in jbd2_journal_cancel_revoke()) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/tytso/ext4.git ext4-fixes/fixes +Already up to date. +Merging vfs-brauner-fixes/vfs.fixes (b78b728e21c32 netfs: Fix missing alloc tagging of direct mempool allocations) +$ git merge -m Merge branch 'vfs.fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/vfs/vfs.git vfs-brauner-fixes/vfs.fixes +Already up to date. +Merging fscrypt-current/for-current (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-current' of https://git.kernel.org/pub/scm/fs/fscrypt/linux.git fscrypt-current/for-current +Already up to date. +Merging fsverity-current/for-current (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-current' of https://git.kernel.org/pub/scm/fs/fsverity/linux.git fsverity-current/for-current +Already up to date. +Merging btrfs-fixes/next-fixes (a3dde27c5da2d Merge branch 'misc-7.3' into next-fixes) +$ git merge -m Merge branch 'next-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/kdave/linux.git btrfs-fixes/next-fixes +Merge made by the 'ort' strategy. + fs/btrfs/ctree.c | 4 ++-- + fs/btrfs/inode.c | 11 +++++------ + fs/btrfs/ioctl.c | 31 ++++++++++++++++--------------- + fs/btrfs/scrub.c | 4 +++- + fs/btrfs/transaction.c | 1 + + fs/btrfs/tree-log.c | 4 ++-- + fs/btrfs/xattr.c | 15 +++++++++++---- + 7 files changed, 40 insertions(+), 30 deletions(-) +Merging vfs-fixes/fixes (49c5d168a3a8f udf: fix nls leak on udf_fill_super() failure) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/viro/vfs.git vfs-fixes/fixes +Auto-merging fs/udf/super.c +Merge made by the 'ort' strategy. +Merging erofs-fixes/fixes (135d84c66f854 erofs: add missing buf->off in erofs_bread()) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/xiang/erofs.git erofs-fixes/fixes +Already up to date. +Merging nfsd-fixes/nfsd-fixes (f76017a7663c4 nfsd: fix handling of NFSEXP_PNFS in the netlink codepath) +$ git merge -m Merge branch 'nfsd-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/cel/linux nfsd-fixes/nfsd-fixes +Already up to date. +Merging v9fs-fixes/fixes/next (028ef9c96e961 Linux 7.0) +$ git merge -m Merge branch 'fixes/next' of https://git.kernel.org/pub/scm/linux/kernel/git/ericvh/v9fs.git v9fs-fixes/fixes/next +Already up to date. +Merging overlayfs-fixes/ovl-fixes (4549871118cf6 Linux 7.1-rc7) +$ git merge -m Merge branch 'ovl-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/overlayfs/vfs.git overlayfs-fixes/ovl-fixes +Already up to date. +Merging fscrypt/for-next (a63883d2c3d47 MAINTAINERS: List the fscrypt branches) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/fs/fscrypt/linux.git fscrypt/for-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 3 ++- + fs/crypto/Kconfig | 2 +- + fs/crypto/crypto.c | 5 +++++ + fs/crypto/fscrypt_private.h | 27 --------------------------- + fs/crypto/hooks.c | 9 +++++++++ + fs/crypto/keyring.c | 13 +++++++++++++ + fs/crypto/keysetup_v1.c | 5 ++--- + 7 files changed, 32 insertions(+), 32 deletions(-) +Merging btrfs/for-next (8147b3ee13812 Merge branch 'for-next-next-v7.3-20260930' into for-next-20260930) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/kdave/linux.git btrfs/for-next +Merge made by the 'ort' strategy. + fs/btrfs/Kconfig | 1 + + fs/btrfs/backref.c | 2 +- + fs/btrfs/bio.c | 163 ++--- + fs/btrfs/bio.h | 9 +- + fs/btrfs/block-group.c | 557 +++------------- + fs/btrfs/block-group.h | 16 +- + fs/btrfs/btrfs_inode.h | 28 +- + fs/btrfs/compression.c | 61 +- + fs/btrfs/delalloc-space.c | 13 +- + fs/btrfs/delayed-inode.c | 115 +++- + fs/btrfs/delayed-inode.h | 22 +- + fs/btrfs/dev-replace.c | 10 +- + fs/btrfs/dir-item.c | 50 +- + fs/btrfs/dir-item.h | 5 +- + fs/btrfs/direct-io.c | 7 + + fs/btrfs/disk-io.c | 132 +--- + fs/btrfs/extent-io-tree.c | 2 +- + fs/btrfs/extent-tree.c | 10 + + fs/btrfs/extent_io.c | 247 ++++--- + fs/btrfs/extent_map.c | 8 + + fs/btrfs/file-item.c | 46 +- + fs/btrfs/free-space-cache.c | 1344 +------------------------------------- + fs/btrfs/free-space-cache.h | 29 +- + fs/btrfs/fs.c | 2 +- + fs/btrfs/fs.h | 4 +- + fs/btrfs/inode.c | 331 ++++------ + fs/btrfs/ioctl.c | 14 +- + fs/btrfs/ordered-data.c | 27 +- + fs/btrfs/qgroup.c | 128 ++-- + fs/btrfs/qgroup.h | 16 +- + fs/btrfs/raid56.c | 57 +- + fs/btrfs/reflink.c | 51 +- + fs/btrfs/relocation.c | 2 +- + fs/btrfs/send.c | 2 +- + fs/btrfs/space-info.c | 2 - + fs/btrfs/space-info.h | 4 - + fs/btrfs/super.c | 48 +- + fs/btrfs/sysfs.c | 134 +--- + fs/btrfs/tests/btrfs-tests.c | 9 - + fs/btrfs/tests/btrfs-tests.h | 2 - + fs/btrfs/tests/extent-io-tests.c | 129 ++-- + fs/btrfs/transaction.c | 153 +++-- + fs/btrfs/transaction.h | 58 +- + fs/btrfs/tree-checker.c | 83 ++- + fs/btrfs/tree-log.c | 7 +- + fs/btrfs/tree-mod-log.c | 13 +- + fs/btrfs/verity.c | 31 +- + fs/btrfs/volumes.c | 67 +- + fs/btrfs/volumes.h | 1 + + fs/btrfs/zlib.c | 10 +- + fs/btrfs/zoned.c | 29 +- + fs/btrfs/zstd.c | 73 ++- + include/uapi/linux/btrfs_tree.h | 25 +- + 53 files changed, 1287 insertions(+), 3102 deletions(-) +Merging ceph/master (dc173b37415e8 ceph: apply nearfull_sync option on remount) +$ git merge -m Merge branch 'master' of https://github.com/ceph/ceph-client.git ceph/master +Already up to date. +Merging cifs/cifs-next (19465a9aeb664 smb: client: split cifsFileInfo bitfields to avoid shared-byte RMW races) +$ git merge -m Merge branch 'cifs-next' of https://git.manguebit.org/linux.git cifs/cifs-next +Merge made by the 'ort' strategy. + Documentation/filesystems/netfs_library.rst | 26 ++++++ + fs/netfs/buffered_read.c | 7 ++ + fs/netfs/buffered_write.c | 9 ++ + fs/netfs/direct_write.c | 9 ++ + fs/netfs/internal.h | 2 + + fs/netfs/misc.c | 99 +++++++++++++++++++++ + fs/netfs/read_collect.c | 24 +++++ + fs/smb/client/cifsfs.c | 41 +++++++-- + fs/smb/client/cifsfs.h | 5 +- + fs/smb/client/cifsglob.h | 10 +-- + fs/smb/client/file.c | 5 +- + fs/smb/client/inode.c | 24 +++-- + fs/smb/client/smb2ops.c | 130 +++++++++++++++++++++------- + fs/smb/client/smb2pdu.c | 10 ++- + include/linux/netfs.h | 2 + + 15 files changed, 347 insertions(+), 56 deletions(-) +Merging configfs/configfs-next (620938e7da894 samples: configfs: constify the configfs_attribute structures) +$ git merge -m Merge branch 'configfs-next' of https://git.kernel.org/pub/scm/linux/kernel/git/leitao/linux.git configfs/configfs-next +Auto-merging MAINTAINERS +Auto-merging tools/testing/selftests/Makefile +Merge made by the 'ort' strategy. + MAINTAINERS | 1 + + drivers/acpi/acpi_configfs.c | 2 +- + drivers/gpu/drm/xe/xe_configfs.c | 4 +- + drivers/virt/coco/guest/report.c | 6 +- + fs/configfs/configfs_internal.h | 10 +- + fs/configfs/dir.c | 10 +- + fs/configfs/file.c | 6 +- + fs/configfs/inode.c | 2 +- + include/linux/configfs.h | 73 ++-- + rust/kernel/configfs.rs | 20 +- + samples/configfs/configfs_sample.c | 137 +++++- + tools/testing/selftests/Makefile | 1 + + .../selftests/filesystems/configfs/.gitignore | 2 + + .../selftests/filesystems/configfs/Makefile | 8 + + .../testing/selftests/filesystems/configfs/config | 5 + + .../selftests/filesystems/configfs/configfs_test.c | 481 +++++++++++++++++++++ + 16 files changed, 697 insertions(+), 71 deletions(-) + create mode 100644 tools/testing/selftests/filesystems/configfs/.gitignore + create mode 100644 tools/testing/selftests/filesystems/configfs/Makefile + create mode 100644 tools/testing/selftests/filesystems/configfs/config + create mode 100644 tools/testing/selftests/filesystems/configfs/configfs_test.c +Merging configfs-rust/configfs-next (6dc9fb1516aa4 rust: configfs: fix data offset calculation for subsystem callbacks) +$ git merge -m Merge branch 'configfs-next' of https://git.kernel.org/pub/scm/linux/kernel/git/a.hindborg/linux.git configfs-rust/configfs-next +Auto-merging rust/kernel/configfs.rs +Merge made by the 'ort' strategy. + rust/kernel/configfs.rs | 107 ++++++++++++++++++++++++++++++++---------------- + 1 file changed, 71 insertions(+), 36 deletions(-) +Merging ecryptfs/next (f81cb44f9a4b8 ecryptfs: ecryptfs_kernel.h: clean up kernel-doc comments) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/tyhicks/ecryptfs.git ecryptfs/next +Already up to date. +Merging dlm/next (ed9b6a1296f10 dlm: wait for outstanding SRCU callbacks to complete in exit paths) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/teigland/linux-dlm.git dlm/next +Merge made by the 'ort' strategy. + fs/dlm/config.c | 30 ++++++++++++++++++++++++++++-- + fs/dlm/dlm_internal.h | 5 +++++ + fs/dlm/lock.c | 18 +++++++++++++----- + fs/dlm/lock.h | 4 ++-- + fs/dlm/lowcomms.c | 1 + + fs/dlm/midcomms.c | 1 + + fs/dlm/plock.c | 14 +++++++++++++- + fs/dlm/user.c | 23 +++++++++++++++++++++++ + 8 files changed, 86 insertions(+), 10 deletions(-) +Merging erofs/dev (a7d28aa0e9b2c erofs: simplify z_erofs_gbuf_growsize()) +$ git merge -m Merge branch 'dev' of https://git.kernel.org/pub/scm/linux/kernel/git/xiang/erofs.git erofs/dev +Already up to date. +Merging exfat/dev (216426aff8f69 MAINTAINERS: add exFAT documentation) +$ git merge -m Merge branch 'dev' of https://git.kernel.org/pub/scm/linux/kernel/git/linkinjeon/exfat.git exfat/dev +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + Documentation/filesystems/exfat.rst | 117 ++++++++++++++++++++++++++++++++++++ + Documentation/filesystems/index.rst | 1 + + MAINTAINERS | 1 + + fs/exfat/dir.c | 52 ++++++++++++++-- + fs/exfat/exfat_fs.h | 1 + + fs/exfat/fatent.c | 19 +++--- + fs/exfat/iomap.c | 28 ++++++++- + fs/exfat/namei.c | 7 ++- + fs/exfat/super.c | 30 +++++++-- + 9 files changed, 234 insertions(+), 22 deletions(-) + create mode 100644 Documentation/filesystems/exfat.rst +Merging ext3/for_next (2eda6a119a002 Pull isofs name buffer cleanups and continuation entry fixes.) +$ git merge -m Merge branch 'for_next' of https://git.kernel.org/pub/scm/linux/kernel/git/jack/linux-fs.git ext3/for_next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 1 + + fs/ext2/Makefile | 2 + + fs/ext2/balloc.c | 4 ++ + fs/ext2/ext2.h | 19 ++++++---- + fs/ext2/file.c | 7 ++++ + fs/ext2/inode.c | 7 ++++ + fs/ext2/super.c | 33 +++++++++------- + fs/isofs/compress.c | 77 +++++++++++++++++++++----------------- + fs/isofs/dir.c | 14 +++++-- + fs/isofs/inode.c | 70 ++++++++++++++-------------------- + fs/isofs/isofs.h | 16 +++++++- + fs/isofs/joliet.c | 27 +++++++++---- + fs/isofs/namei.c | 8 ++-- + fs/isofs/rock.c | 19 ++++++++-- + fs/notify/fanotify/fanotify_user.c | 2 +- + fs/ocfs2/quota_local.c | 4 ++ + fs/quota/dquot.c | 16 ++++++++ + fs/udf/inode.c | 59 ++++++++++++++++++++++++++--- + fs/udf/misc.c | 15 ++++---- + fs/udf/super.c | 5 ++- + include/linux/once_lite.h | 8 ++-- + mm/shmem_quota.c | 5 +++ + 22 files changed, 284 insertions(+), 134 deletions(-) +Merging ext4/dev (9091c97be3408 ext4: fix estimate extent index blocks in ext4_ext_index_trans_blocks()) +$ git merge -m Merge branch 'dev' of https://git.kernel.org/pub/scm/linux/kernel/git/tytso/ext4.git ext4/dev +Already up to date. +Merging f2fs/dev (7566fec606d23 f2fs: reject device aliasing without a multi-device configuration) +$ git merge -m Merge branch 'dev' of https://git.kernel.org/pub/scm/linux/kernel/git/jaegeuk/f2fs.git f2fs/dev +Merge made by the 'ort' strategy. + Documentation/ABI/testing/sysfs-fs-f2fs | 7 + + fs/f2fs/Makefile | 2 +- + fs/f2fs/acl.c | 26 +- + fs/f2fs/acl.h | 8 +- + fs/f2fs/cache.c | 720 +++++++++++++++++ + fs/f2fs/cache.h | 240 ++++++ + fs/f2fs/checkpoint.c | 558 +++++++------- + fs/f2fs/compress.c | 183 ++--- + fs/f2fs/data.c | 648 +++++++++++----- + fs/f2fs/debug.c | 94 ++- + fs/f2fs/dir.c | 221 +++--- + fs/f2fs/extent_cache.c | 25 +- + fs/f2fs/f2fs.h | 508 +++++++----- + fs/f2fs/file.c | 181 ++--- + fs/f2fs/gc.c | 222 +++--- + fs/f2fs/inline.c | 293 +++---- + fs/f2fs/inode.c | 242 +++--- + fs/f2fs/iostat.h | 11 + + fs/f2fs/namei.c | 114 +-- + fs/f2fs/node.c | 1277 +++++++++++++++---------------- + fs/f2fs/node.h | 191 +++-- + fs/f2fs/recovery.c | 255 +++--- + fs/f2fs/segment.c | 429 ++++++----- + fs/f2fs/segment.h | 87 ++- + fs/f2fs/shrinker.c | 16 +- + fs/f2fs/super.c | 288 +++---- + fs/f2fs/sysfs.c | 32 +- + fs/f2fs/verity.c | 6 +- + fs/f2fs/xattr.c | 133 ++-- + fs/f2fs/xattr.h | 23 +- + include/linux/f2fs_fs.h | 140 ++-- + include/trace/events/f2fs.h | 71 ++ + 32 files changed, 4353 insertions(+), 2898 deletions(-) + create mode 100644 fs/f2fs/cache.c + create mode 100644 fs/f2fs/cache.h +Merging fsverity/for-next (b08e4ee274917 MAINTAINERS: List the fsverity branches) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/fs/fsverity/linux.git fsverity/for-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 3 ++- + 1 file changed, 2 insertions(+), 1 deletion(-) +Merging fuse/for-next (b776cbe24cc1b fuse: make fuse_backing_get() static) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mszeredi/fuse.git fuse/for-next +Auto-merging fs/fuse/file.c +Merge made by the 'ort' strategy. + fs/fuse/Kconfig | 8 +- + fs/fuse/Makefile | 2 +- + fs/fuse/backing.c | 2 +- + fs/fuse/dax.c | 215 +++++++++++---------- + fs/fuse/dev.c | 50 +++-- + fs/fuse/dev.h | 2 +- + fs/fuse/dev_uring.c | 6 + + fs/fuse/dir.c | 4 +- + fs/fuse/file.c | 129 ++++++++----- + fs/fuse/fuse_i.h | 96 +++++---- + fs/fuse/inode.c | 82 +++++--- + fs/fuse/ioctl.c | 3 - + fs/fuse/iomode.c | 4 +- + fs/fuse/notify.c | 4 +- + fs/fuse/passthrough.c | 15 +- + fs/fuse/req.c | 7 +- + fs/fuse/virtio_fs.c | 28 +-- + include/uapi/linux/fuse.h | 12 +- + .../testing/selftests/filesystems/fuse/.gitignore | 1 + + tools/testing/selftests/filesystems/fuse/Makefile | 2 +- + 20 files changed, 380 insertions(+), 292 deletions(-) +Merging gfs2/for-next (d0c4bce31a579 gfs2: gfs2_create_inode excl cleanup) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/gfs2/linux-gfs2.git gfs2/for-next +Merge made by the 'ort' strategy. + fs/gfs2/acl.c | 11 +++-- + fs/gfs2/aops.c | 17 +++---- + fs/gfs2/bmap.c | 93 ++++++++++++++++++++---------------- + fs/gfs2/dentry.c | 6 +-- + fs/gfs2/dir.c | 61 +++++++++++++++--------- + fs/gfs2/export.c | 7 +-- + fs/gfs2/file.c | 72 ++++++++++++++++------------ + fs/gfs2/glock.c | 60 +++++++++++++++-------- + fs/gfs2/glops.c | 13 +++-- + fs/gfs2/incore.h | 8 ++-- + fs/gfs2/inode.c | 132 ++++++++++++++++++++++++++++----------------------- + fs/gfs2/lock_dlm.c | 4 +- + fs/gfs2/log.c | 3 +- + fs/gfs2/lops.c | 18 ++++--- + fs/gfs2/meta_io.c | 7 +-- + fs/gfs2/ops_fstype.c | 26 +++++----- + fs/gfs2/quota.c | 57 ++++++++++++---------- + fs/gfs2/recovery.c | 19 ++++---- + fs/gfs2/rgrp.c | 129 +++++++++++++++++++++++++------------------------ + fs/gfs2/super.c | 98 +++++++++++++++++++++++--------------- + fs/gfs2/trace_gfs2.h | 6 +-- + fs/gfs2/util.c | 8 ++-- + fs/gfs2/xattr.c | 69 ++++++++++++++++----------- + 23 files changed, 528 insertions(+), 396 deletions(-) +Merging jfs/jfs-next (dad98c5b2a05e jfs: avoid -Wtautological-constant-out-of-range-compare warning again) +$ git merge -m Merge branch 'jfs-next' of https://github.com/kleikamp/linux-shaggy.git jfs/jfs-next +Already up to date. +Merging ksmbd/ksmbd-for-next (276effe050171 ksmbd: wait for negotiation before receiving another PDU) +$ git merge -m Merge branch 'ksmbd-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/linkinjeon/smb.git ksmbd/ksmbd-for-next +Merge made by the 'ort' strategy. + Documentation/filesystems/smb/ksmbd.rst | 14 +- + fs/smb/common/compress/compress.c | 16 +- + fs/smb/common/fscc.h | 5 +- + fs/smb/common/smbglob.h | 1 + + fs/smb/server/Kconfig | 2 + + fs/smb/server/Makefile | 1 + + fs/smb/server/compress.c | 1 + + fs/smb/server/connection.c | 3 +- + fs/smb/server/connection.h | 3 + + fs/smb/server/ksmbd_work.c | 3 - + fs/smb/server/ksmbd_work.h | 5 +- + fs/smb/server/mgmt/user_session.c | 4 +- + fs/smb/server/server.c | 21 + + fs/smb/server/smb2ops.c | 33 ++ + fs/smb/server/smb2pdu.c | 750 +++++++++++++++----------------- + fs/smb/server/smb2pdu.h | 23 +- + fs/smb/server/smb_common.c | 4 +- + fs/smb/server/smbacl.c | 7 +- + fs/smb/server/tests/Kconfig | 15 + + fs/smb/server/tests/Makefile | 4 + + fs/smb/server/tests/smbacl_kunit.c | 301 +++++++++++++ + fs/smb/server/transport_tcp.c | 8 +- + fs/smb/server/vfs.c | 12 +- + fs/smb/server/vfs.h | 5 + + fs/smb/server/vfs_cache.c | 66 +-- + fs/smb/server/vfs_cache.h | 6 - + 26 files changed, 820 insertions(+), 493 deletions(-) + create mode 100644 fs/smb/server/tests/Kconfig + create mode 100644 fs/smb/server/tests/Makefile + create mode 100644 fs/smb/server/tests/smbacl_kunit.c +$ git am -3 ../patches/0001-ntfs3-Fix-up-merge-with-Linus.patch +Applying: ntfs3: Fix up merge with Linus +Using index info to reconstruct a base tree... +M fs/ntfs3/file.c +Falling back to patching base and 3-way merge... +Auto-merging fs/ntfs3/file.c +No changes -- Patch already applied. +Merging nfs/linux-next (fd73f4a665989 Linux 7.3-rc3) +$ git merge -m Merge branch 'linux-next' of git://git.linux-nfs.org/projects/trondmy/nfs-2.6.git nfs/linux-next +Already up to date. +Merging nfs-anna/linux-next (9bafc322b7fca nfs: split up block layout and SCSI layout support) +$ git merge -m Merge branch 'linux-next' of git://git.linux-nfs.org/projects/anna/linux-nfs.git nfs-anna/linux-next +Merge made by the 'ort' strategy. + fs/nfs/Kconfig | 17 +- + fs/nfs/blocklayout/Makefile | 3 +- + fs/nfs/blocklayout/blocklayout.c | 119 ++++++--- + fs/nfs/blocklayout/dev.c | 32 ++- + fs/nfs/callback_proc.c | 23 +- + fs/nfs/callback_xdr.c | 9 +- + fs/nfs/client.c | 6 +- + fs/nfs/direct.c | 139 +++++----- + fs/nfs/filelayout/filelayout.c | 17 +- + fs/nfs/filelayout/filelayoutdev.c | 19 +- + fs/nfs/flexfilelayout/flexfilelayout.c | 413 +++++++++++++++++++----------- + fs/nfs/flexfilelayout/flexfilelayout.h | 48 ++-- + fs/nfs/flexfilelayout/flexfilelayoutdev.c | 212 +++++++++------ + fs/nfs/internal.h | 3 +- + fs/nfs/netns.h | 5 +- + fs/nfs/nfs3client.c | 9 +- + fs/nfs/nfs4_fs.h | 3 + + fs/nfs/nfs4client.c | 8 +- + fs/nfs/nfs4proc.c | 136 ++++++++++ + fs/nfs/nfs4state.c | 7 +- + fs/nfs/nfs4xdr.c | 2 +- + fs/nfs/pagelist.c | 66 ++++- + fs/nfs/pnfs.c | 376 +++++++++++++++++++++++++-- + fs/nfs/pnfs.h | 102 +++++++- + fs/nfs/pnfs_dev.c | 30 ++- + fs/nfs/pnfs_nfs.c | 108 ++++++-- + fs/nfs/read.c | 2 +- + fs/nfs/write.c | 2 +- + include/linux/nfs_fs_sb.h | 8 + + include/linux/nfs_page.h | 8 +- + include/linux/nfs_xdr.h | 2 + + net/sunrpc/xprtsock.c | 69 ++--- + 32 files changed, 1513 insertions(+), 490 deletions(-) +Merging nfsd/nfsd-next (ac04dab23b5ff NFSD: Return NFSERR_ISDIR for NFSv2 READ and WRITE on a non-regular file) +$ git merge -m Merge branch 'nfsd-next' of https://git.kernel.org/pub/scm/linux/kernel/git/cel/linux nfsd/nfsd-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + Documentation/admin-guide/nfs/pnfs-scsi-server.rst | 2 +- + MAINTAINERS | 3 +- + fs/lockd/svc.c | 1 - + fs/lockd/svclock.c | 33 +- + fs/lockd/trace.h | 1 - + fs/lockd/xdr.h | 2 +- + fs/namei.c | 2 + + fs/nfs/nfs4file.c | 1 + + fs/nfs/super.c | 25 - + fs/nfs_common/nfs_ssc.c | 126 +++-- + fs/nfsd/blocklayout.c | 1 + + fs/nfsd/blocklayoutxdr.c | 11 + + fs/nfsd/export.c | 8 +- + fs/nfsd/export.h | 3 +- + fs/nfsd/filecache.c | 1 + + fs/nfsd/flexfilelayout.c | 24 +- + fs/nfsd/flexfilelayoutxdr.c | 9 +- + fs/nfsd/flexfilelayoutxdr.h | 8 +- + fs/nfsd/localio.c | 10 +- + fs/nfsd/lockd.c | 5 +- + fs/nfsd/netns.h | 17 +- + fs/nfsd/nfs2acl.c | 48 +- + fs/nfsd/nfs3acl.c | 1 + + fs/nfsd/nfs3proc.c | 79 ++- + fs/nfsd/nfs3xdr.c | 4 + + fs/nfsd/nfs4acl.c | 1 + + fs/nfsd/nfs4callback.c | 8 +- + fs/nfsd/nfs4ctl.h | 83 +++ + fs/nfsd/nfs4idmap.c | 1 + + fs/nfsd/nfs4layouts.c | 1 + + fs/nfsd/nfs4proc.c | 465 +++++++++------- + fs/nfsd/nfs4recover.c | 1 + + fs/nfsd/nfs4state.c | 587 ++++++++++++++++++--- + fs/nfsd/nfs4xdr.c | 188 +++++-- + fs/nfsd/nfscache.c | 1 + + fs/nfsd/nfsctl.c | 36 +- + fs/nfsd/nfsd.h | 237 +-------- + fs/nfsd/nfserr.h | 159 ++++++ + fs/nfsd/nfsfh.c | 73 +-- + fs/nfsd/nfsfh.h | 28 +- + fs/nfsd/nfsproc.c | 39 +- + fs/nfsd/nfssvc.c | 8 + + fs/nfsd/nfsxdr.c | 1 + + fs/nfsd/state.h | 46 +- + fs/nfsd/trace.h | 45 +- + fs/nfsd/vfs.c | 298 +++++------ + fs/nfsd/vfs.h | 39 +- + fs/nfsd/xdr.h | 6 +- + fs/nfsd/xdr3.h | 2 +- + fs/nfsd/xdr4.h | 156 +----- + fs/nfsd/xdr4cb.h | 20 +- + include/linux/nfs.h | 55 +- + include/linux/nfs3.h | 43 ++ + include/linux/nfs4.h | 6 + + include/linux/nfs_fh.h | 63 +++ + include/linux/nfs_ssc.h | 69 +-- + include/linux/nfsd_ssc.h | 38 ++ + include/linux/nfslocalio.h | 11 +- + include/linux/sunrpc/svc_xprt.h | 5 +- + include/trace/misc/nfs.h | 13 +- + net/sunrpc/svc_xprt.c | 36 +- + net/sunrpc/svcauth_unix.c | 9 + + net/sunrpc/svcsock.c | 490 ++++++++++------- + net/sunrpc/xprtrdma/svc_rdma_recvfrom.c | 18 +- + net/sunrpc/xprtrdma/svc_rdma_transport.c | 3 +- + 65 files changed, 2431 insertions(+), 1382 deletions(-) + create mode 100644 fs/nfsd/nfs4ctl.h + create mode 100644 fs/nfsd/nfserr.h + create mode 100644 include/linux/nfs_fh.h + create mode 100644 include/linux/nfsd_ssc.h +$ git am -3 ../patches/0001-Revert-smb-client-implement-fileattr_get-to-support-.patch +Applying: Revert "smb: client: implement fileattr_get to support FS_IOC_GETFLAGS" +Using index info to reconstruct a base tree... +M fs/smb/client/cifsfs.c +M fs/smb/client/cifsfs.h +M fs/smb/client/inode.c +Falling back to patching base and 3-way merge... +Auto-merging fs/smb/client/inode.c +Auto-merging fs/smb/client/cifsfs.h +Auto-merging fs/smb/client/cifsfs.c +No changes -- Patch already applied. +Merging ntfs/ntfs-next (708f9d56cacae ntfs: reject non-resident attributes whose sizes exceed their allocation) +$ git merge -m Merge branch 'ntfs-next' of https://git.kernel.org/pub/scm/linux/kernel/git/linkinjeon/ntfs.git ntfs/ntfs-next +Merge made by the 'ort' strategy. + fs/ntfs/attrib.c | 76 ++++- + fs/ntfs/attrlist.c | 14 +- + fs/ntfs/bitmap.c | 19 +- + fs/ntfs/collate.c | 8 +- + fs/ntfs/dir.c | 23 +- + fs/ntfs/file.c | 56 ++-- + fs/ntfs/index.c | 2 +- + fs/ntfs/index.h | 2 +- + fs/ntfs/inode.c | 131 ++++++-- + fs/ntfs/iomap.c | 16 +- + fs/ntfs/layout.h | 19 +- + fs/ntfs/mft.c | 921 +++++++++++++++++++++++++++++++++++------------------ + fs/ntfs/mft.h | 19 +- + fs/ntfs/mst.c | 4 +- + fs/ntfs/namei.c | 31 +- + fs/ntfs/ntfs.h | 7 +- + fs/ntfs/reparse.c | 2 +- + fs/ntfs/runlist.c | 5 +- + fs/ntfs/super.c | 391 +++++++++++++++++++---- + fs/ntfs/unistr.c | 29 +- + fs/ntfs/volume.h | 15 + + 21 files changed, 1266 insertions(+), 524 deletions(-) +Merging ntfs3/master (f3b8ee6c05bed fs/ntfs3: return -ERANGE for short xattr buffers) +$ git merge -m Merge branch 'master' of https://github.com/Paragon-Software-Group/linux-ntfs3.git ntfs3/master +Auto-merging fs/ntfs3/inode.c +Merge made by the 'ort' strategy. + Documentation/filesystems/ntfs3.rst | 23 ++++++++++++++++++ + fs/ntfs3/attrib.c | 47 ++++++++++++++++++++++++++++++------- + fs/ntfs3/dir.c | 3 +++ + fs/ntfs3/file.c | 3 +++ + fs/ntfs3/frecord.c | 5 ++++ + fs/ntfs3/fslog.c | 4 +++- + fs/ntfs3/fsntfs.c | 3 +++ + fs/ntfs3/index.c | 13 ++++++++-- + fs/ntfs3/inode.c | 19 ++++++++++----- + fs/ntfs3/record.c | 16 ++++++++++++- + fs/ntfs3/super.c | 2 +- + fs/ntfs3/xattr.c | 14 +++++++---- + 12 files changed, 127 insertions(+), 25 deletions(-) +Merging orangefs/for-next (2bc09cb0e9e44 orangefs: don't continue on to gpf if client dies on write.) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/hubcap/linux.git orangefs/for-next +Merge made by the 'ort' strategy. + fs/orangefs/inode.c | 21 ++++++++++++++++++++- + fs/orangefs/orangefs-debugfs.c | 10 ++++++---- + fs/orangefs/super.c | 18 ++++++++++++++++++ + 3 files changed, 44 insertions(+), 5 deletions(-) +Merging overlayfs/overlayfs-next (1f6ee9be92f8d ovl: make fsync after metadata copy-up opt-in mount option) +$ git merge -m Merge branch 'overlayfs-next' of https://git.kernel.org/pub/scm/linux/kernel/git/overlayfs/vfs.git overlayfs/overlayfs-next +Already up to date. +Merging ubifs/next (a5e0055eac837 UBI: support per-device wear-leveling threshold) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/rw/ubifs.git ubifs/next +Already up to date. +Merging v9fs/9p-next (c60ae98c5aa64 9p: Fix v9fs_issue_write() to update i_size and remote_i_size) +$ git merge -m Merge branch '9p-next' of https://github.com/martinetd/linux v9fs/9p-next +Already up to date. +Merging v9fs-ericvh/ericvh/for-next (028ef9c96e961 Linux 7.0) +$ git merge -m Merge branch 'ericvh/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/ericvh/v9fs.git v9fs-ericvh/ericvh/for-next +Already up to date. +Merging xfs/for-next (b4787e7b9730d Merge remote-tracking branch 'xfs-linux/xfs-7.3-fixes' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/fs/xfs/xfs-linux.git xfs/for-next +Merge made by the 'ort' strategy. + .../filesystems/xfs/xfs-online-fsck-design.rst | 13 +- + fs/xfs/libxfs/xfs_attr.c | 14 +- + fs/xfs/libxfs/xfs_bmap.c | 5 +- + fs/xfs/libxfs/xfs_bmap.h | 2 +- + fs/xfs/libxfs/xfs_btree.c | 10 +- + fs/xfs/libxfs/xfs_btree_mem.c | 6 +- + fs/xfs/libxfs/xfs_dquot_buf.c | 13 +- + fs/xfs/libxfs/xfs_exchmaps.c | 1 - + fs/xfs/libxfs/xfs_exchmaps.h | 3 +- + fs/xfs/libxfs/xfs_ialloc.c | 2 +- + fs/xfs/libxfs/xfs_ialloc_btree.c | 1 - + fs/xfs/libxfs/xfs_ialloc_btree.h | 5 +- + fs/xfs/libxfs/xfs_inode_fork.c | 4 +- + fs/xfs/libxfs/xfs_metadir.c | 32 +- + fs/xfs/libxfs/xfs_metadir.h | 5 +- + fs/xfs/libxfs/xfs_parent.h | 1 - + fs/xfs/libxfs/xfs_refcount_btree.c | 16 +- + fs/xfs/libxfs/xfs_refcount_btree.h | 26 ++ + fs/xfs/libxfs/xfs_rmap.c | 13 +- + fs/xfs/libxfs/xfs_rmap_btree.c | 16 +- + fs/xfs/libxfs/xfs_rmap_btree.h | 26 ++ + fs/xfs/libxfs/xfs_rtrefcount_btree.c | 121 +----- + fs/xfs/libxfs/xfs_rtrefcount_btree.h | 5 +- + fs/xfs/libxfs/xfs_rtrmap_btree.c | 236 +----------- + fs/xfs/libxfs/xfs_rtrmap_btree.h | 3 +- + fs/xfs/libxfs/xfs_sb.c | 40 -- + fs/xfs/libxfs/xfs_sb.h | 1 - + fs/xfs/libxfs/xfs_symlink_remote.c | 8 +- + fs/xfs/libxfs/xfs_symlink_remote.h | 2 +- + fs/xfs/libxfs/xfs_trans_resv.c | 45 +-- + fs/xfs/libxfs/xfs_trans_space.c | 6 +- + fs/xfs/libxfs/xfs_types.c | 7 +- + fs/xfs/libxfs/xfs_types.h | 7 +- + fs/xfs/scrub/alloc_repair.c | 15 +- + fs/xfs/scrub/bmap.c | 11 +- + fs/xfs/scrub/bmap_repair.c | 4 +- + fs/xfs/scrub/inode_repair.c | 32 +- + fs/xfs/scrub/orphanage.c | 2 +- + fs/xfs/scrub/quota.c | 2 +- + fs/xfs/scrub/quota_repair.c | 9 +- + fs/xfs/scrub/quotacheck_repair.c | 54 ++- + fs/xfs/scrub/rtrefcount_repair.c | 3 +- + fs/xfs/scrub/rtrmap_repair.c | 2 +- + fs/xfs/scrub/xfarray.c | 201 ++++------ + fs/xfs/scrub/xfarray.h | 9 +- + fs/xfs/xfs_bmap_item.c | 2 +- + fs/xfs/xfs_dquot.c | 20 +- + fs/xfs/xfs_dquot.h | 2 +- + fs/xfs/xfs_exchmaps_item.c | 4 +- + fs/xfs/xfs_exchrange.c | 2 +- + fs/xfs/xfs_file.c | 22 +- + fs/xfs/xfs_handle.c | 2 +- + fs/xfs/xfs_inode.c | 20 +- + fs/xfs/xfs_ioctl.c | 421 +++++++++++++-------- + fs/xfs/xfs_ioctl.h | 4 +- + fs/xfs/xfs_ioctl32.c | 188 +++++---- + fs/xfs/xfs_qm.c | 6 +- + fs/xfs/xfs_rmap_item.c | 2 +- + fs/xfs/xfs_rtalloc.h | 49 ++- + fs/xfs/xfs_super.c | 2 +- + fs/xfs/xfs_symlink.c | 4 +- + fs/xfs/xfs_trace.h | 28 +- + fs/xfs/xfs_trans_dquot.c | 6 +- + 63 files changed, 881 insertions(+), 942 deletions(-) +Merging zonefs/for-next (3a8389d42bdf4 zonefs: handle integer overflow in zonefs_fname_to_fno) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/dlemoal/zonefs.git zonefs/for-next +Already up to date. +Merging vfs-brauner/vfs.all (84086827932b5 Merge branch 'vfs-7.4.sync_close' into vfs.all) +$ git merge -m Merge branch 'vfs.all' of https://git.kernel.org/pub/scm/linux/kernel/git/vfs/vfs.git vfs-brauner/vfs.all +Auto-merging Documentation/filesystems/index.rst +Auto-merging Documentation/filesystems/netfs_library.rst +Auto-merging MAINTAINERS +Auto-merging drivers/gpio/gpiolib-cdev.c +Auto-merging fs/9p/vfs_addr.c +Auto-merging fs/btrfs/btrfs_inode.h +Auto-merging fs/btrfs/inode.c +Auto-merging fs/btrfs/ioctl.c +Auto-merging fs/btrfs/xattr.c +Auto-merging fs/configfs/configfs_internal.h +Auto-merging fs/configfs/dir.c +Auto-merging fs/configfs/inode.c +Auto-merging fs/dax.c +Auto-merging fs/debugfs/inode.c +Auto-merging fs/exec.c +Auto-merging fs/exfat/exfat_fs.h +Auto-merging fs/exfat/namei.c +Auto-merging fs/ext2/ext2.h +Auto-merging fs/ext2/inode.c +Auto-merging fs/f2fs/acl.c +Auto-merging fs/f2fs/acl.h +Auto-merging fs/f2fs/f2fs.h +Auto-merging fs/f2fs/file.c +Auto-merging fs/f2fs/namei.c +Auto-merging fs/f2fs/xattr.c +Auto-merging fs/fuse/dir.c +Auto-merging fs/fuse/file.c +Auto-merging fs/fuse/fuse_i.h +Auto-merging fs/fuse/ioctl.c +Auto-merging fs/fuse/req.c +Auto-merging fs/gfs2/acl.c +Auto-merging fs/gfs2/file.c +Auto-merging fs/gfs2/inode.c +Auto-merging fs/gfs2/log.c +Auto-merging fs/gfs2/lops.c +Auto-merging fs/gfs2/xattr.c +Auto-merging fs/namei.c +Auto-merging fs/netfs/buffered_read.c +Auto-merging fs/netfs/buffered_write.c +Auto-merging fs/netfs/direct_write.c +Auto-merging fs/netfs/internal.h +Auto-merging fs/netfs/misc.c +Auto-merging fs/netfs/read_collect.c +Auto-merging fs/nfs/Kconfig +Auto-merging fs/nfs/internal.h +Auto-merging fs/nfs/nfs4proc.c +Auto-merging fs/nfsd/vfs.c +Auto-merging fs/ntfs/file.c +Auto-merging fs/ntfs/namei.c +Auto-merging fs/ntfs3/file.c +Auto-merging fs/ntfs3/inode.c +Auto-merging fs/ntfs3/xattr.c +Auto-merging fs/ocfs2/namei.c +Auto-merging fs/ocfs2/xattr.c +Auto-merging fs/orangefs/inode.c +Auto-merging fs/quota/dquot.c +Auto-merging fs/smb/client/cifsfs.c +Auto-merging fs/smb/client/cifsfs.h +Auto-merging fs/smb/client/dir.c +Auto-merging fs/smb/client/inode.c +Auto-merging fs/smb/client/transport.c +Auto-merging fs/smb/server/smb2pdu.c +CONFLICT (content): Merge conflict in fs/smb/server/smb2pdu.c +Auto-merging fs/smb/server/smb_common.c +Auto-merging fs/smb/server/smbacl.c +Auto-merging fs/smb/server/vfs.c +CONFLICT (content): Merge conflict in fs/smb/server/vfs.c +Auto-merging fs/smb/server/vfs.h +CONFLICT (content): Merge conflict in fs/smb/server/vfs.h +Auto-merging fs/super.c +Auto-merging fs/xfs/libxfs/xfs_errortag.h +Auto-merging fs/xfs/xfs_file.c +Auto-merging fs/xfs/xfs_handle.c +Auto-merging fs/xfs/xfs_inode.c +Auto-merging fs/xfs/xfs_ioctl.c +Auto-merging fs/xfs/xfs_ioctl.h +Auto-merging fs/xfs/xfs_super.c +Auto-merging fs/xfs/xfs_symlink.c +Auto-merging fs/xfs/xfs_trace.h +Auto-merging fs/xfs/xfs_zone_alloc.c +Auto-merging include/linux/netfs.h +Auto-merging include/linux/sched.h +Auto-merging kernel/exit.c +Auto-merging kernel/fork.c +Auto-merging kernel/signal.c +Auto-merging mm/shmem.c +Auto-merging security/selinux/hooks.c +Auto-merging tools/testing/selftests/Makefile +Resolved 'fs/smb/server/smb2pdu.c' using previous resolution. +Resolved 'fs/smb/server/vfs.c' using previous resolution. +Resolved 'fs/smb/server/vfs.h' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[fs-next ea7ec0b29e4ec] Merge branch 'vfs.all' of https://git.kernel.org/pub/scm/linux/kernel/git/vfs/vfs.git +$ git diff -M --stat --summary HEAD^.. + CREDITS | 2 +- + Documentation/admin-guide/binfmt-misc.rst | 3 + + Documentation/filesystems/befs.rst | 4 +- + Documentation/filesystems/bfs.rst | 60 - + Documentation/filesystems/index.rst | 1 - + Documentation/filesystems/locking.rst | 24 +- + Documentation/filesystems/netfs_library.rst | 7 +- + Documentation/filesystems/porting.rst | 25 +- + Documentation/filesystems/proc.rst | 4 + + Documentation/filesystems/sharedsubtree.rst | 20 +- + Documentation/filesystems/squashfs.rst | 6 +- + Documentation/filesystems/vfs.rst | 30 +- + MAINTAINERS | 8 +- + arch/mips/configs/malta_defconfig | 1 - + arch/mips/configs/malta_kvm_defconfig | 1 - + arch/mips/configs/maltaup_xpa_defconfig | 1 - + arch/mips/configs/rm200_defconfig | 1 - + arch/powerpc/Kconfig | 1 - + arch/powerpc/configs/fsl-emb-nonhw.config | 1 - + arch/powerpc/configs/ppc6xx_defconfig | 1 - + arch/powerpc/include/asm/elf.h | 6 - + arch/powerpc/include/asm/spu.h | 3 - + arch/powerpc/platforms/cell/Kconfig | 1 - + arch/powerpc/platforms/cell/spu_syscalls.c | 20 - + arch/powerpc/platforms/cell/spufs/Makefile | 1 - + arch/powerpc/platforms/cell/spufs/coredump.c | 183 -- + arch/powerpc/platforms/cell/spufs/file.c | 114 -- + arch/powerpc/platforms/cell/spufs/inode.c | 14 +- + arch/powerpc/platforms/cell/spufs/spufs.h | 12 - + arch/powerpc/platforms/cell/spufs/syscalls.c | 4 - + block/bio-integrity-fs.c | 15 +- + block/bio-integrity.c | 1 + + block/bio.c | 219 +-- + block/blk-map.c | 2 +- + block/blk-settings.c | 6 + + block/fops.c | 3 +- + block/partitions/efi.h | 8 +- + drivers/android/binder/rust_binderfs.c | 2 +- + drivers/android/binderfs.c | 2 +- + drivers/base/devtmpfs.c | 108 +- + drivers/gpio/gpiolib-cdev.c | 18 +- + drivers/gpu/drm/msm/msm_perfcntr.c | 4 +- + drivers/media/mc/mc-request.c | 8 +- + drivers/misc/ntsync.c | 6 +- + fs/9p/acl.c | 4 +- + fs/9p/acl.h | 4 +- + fs/9p/v9fs.h | 2 +- + fs/9p/v9fs_vfs.h | 2 +- + fs/9p/vfs_addr.c | 1 - + fs/9p/vfs_inode.c | 14 +- + fs/9p/vfs_inode_dotl.c | 14 +- + fs/9p/xattr.c | 2 +- + fs/Kconfig | 9 +- + fs/Makefile | 1 - + fs/adfs/adfs.h | 2 +- + fs/adfs/dir.c | 2 +- + fs/adfs/inode.c | 2 +- + fs/affs/affs.h | 10 +- + fs/affs/inode.c | 2 +- + fs/affs/namei.c | 8 +- + fs/afs/dir.c | 16 +- + fs/afs/file.c | 8 +- + fs/afs/inode.c | 4 +- + fs/afs/internal.h | 8 +- + fs/afs/security.c | 2 +- + fs/afs/xattr.c | 4 +- + fs/aio.c | 11 +- + fs/anon_inodes.c | 4 +- + fs/attr.c | 16 +- + fs/autofs/root.c | 12 +- + fs/backing-file.c | 2 +- + fs/bad_inode.c | 20 +- + fs/bfs/Kconfig | 21 - + fs/bfs/Makefile | 8 - + fs/bfs/bfs.h | 69 - + fs/bfs/dir.c | 4 +- + fs/bfs/file.c | 203 -- + fs/bfs/inode.c | 538 ------ + fs/binfmt_elf.c | 18 +- + fs/binfmt_elf_fdpic.c | 14 +- + fs/binfmt_misc.c | 12 +- + fs/bpf_fs_kfuncs.c | 4 +- + fs/btrfs/acl.c | 2 +- + fs/btrfs/acl.h | 2 +- + fs/btrfs/btrfs_inode.h | 2 +- + fs/btrfs/inode.c | 24 +- + fs/btrfs/ioctl.c | 16 +- + fs/btrfs/ioctl.h | 2 +- + fs/btrfs/xattr.c | 6 +- + fs/buffer.c | 131 +- + fs/cachefiles/Kconfig | 2 +- + fs/cachefiles/interface.c | 96 +- + fs/cachefiles/internal.h | 18 +- + fs/cachefiles/io.c | 472 +++-- + fs/cachefiles/namei.c | 34 +- + fs/cachefiles/xattr.c | 82 +- + fs/ceph/Kconfig | 1 + + fs/ceph/acl.c | 2 +- + fs/ceph/addr.c | 4 +- + fs/ceph/dir.c | 10 +- + fs/ceph/file.c | 2 +- + fs/ceph/inode.c | 10 +- + fs/ceph/mds_client.h | 2 +- + fs/ceph/super.h | 10 +- + fs/ceph/xattr.c | 2 +- + fs/char_dev.c | 4 +- + fs/coda/coda_linux.h | 6 +- + fs/coda/dir.c | 10 +- + fs/coda/inode.c | 4 +- + fs/coda/pioctl.c | 4 +- + fs/configfs/configfs_internal.h | 4 +- + fs/configfs/dir.c | 2 +- + fs/configfs/inode.c | 2 +- + fs/configfs/symlink.c | 2 +- + fs/coredump.c | 558 ++++-- + fs/dax.c | 26 +- + fs/dcache.c | 206 +- + fs/debugfs/inode.c | 2 +- + fs/devpts/inode.c | 2 + + fs/ecryptfs/inode.c | 26 +- + fs/efivarfs/inode.c | 6 +- + fs/erofs/inode.c | 2 +- + fs/erofs/internal.h | 2 +- + fs/eventfd.c | 4 +- + fs/eventpoll.c | 6 +- + fs/exec.c | 28 +- + fs/exfat/exfat_fs.h | 4 +- + fs/exfat/file.c | 6 +- + fs/exfat/misc.c | 2 +- + fs/exfat/namei.c | 6 +- + fs/ext2/acl.c | 2 +- + fs/ext2/acl.h | 2 +- + fs/ext2/ext2.h | 6 +- + fs/ext2/inode.c | 4 +- + fs/ext2/ioctl.c | 2 +- + fs/ext2/namei.c | 12 +- + fs/ext2/xattr.c | 2 +- + fs/ext2/xattr_security.c | 2 +- + fs/ext2/xattr_trusted.c | 2 +- + fs/ext2/xattr_user.c | 2 +- + fs/ext4/acl.c | 2 +- + fs/ext4/acl.h | 2 +- + fs/ext4/ext4.h | 10 +- + fs/ext4/ext4_jbd2.c | 2 +- + fs/ext4/ialloc.c | 2 +- + fs/ext4/inode.c | 6 +- + fs/ext4/ioctl.c | 6 +- + fs/ext4/mmp.c | 2 +- + fs/ext4/namei.c | 16 +- + fs/ext4/symlink.c | 2 +- + fs/ext4/xattr_hurd.c | 2 +- + fs/ext4/xattr_security.c | 2 +- + fs/ext4/xattr_trusted.c | 2 +- + fs/ext4/xattr_user.c | 2 +- + fs/f2fs/acl.c | 6 +- + fs/f2fs/acl.h | 2 +- + fs/f2fs/f2fs.h | 8 +- + fs/f2fs/file.c | 14 +- + fs/f2fs/namei.c | 24 +- + fs/f2fs/xattr.c | 4 +- + fs/failfs.c | 4 +- + fs/fat/fat.h | 4 +- + fs/fat/file.c | 6 +- + fs/fat/misc.c | 2 +- + fs/fat/namei_msdos.c | 6 +- + fs/fat/namei_vfat.c | 6 +- + fs/fhandle.c | 2 +- + fs/file.c | 321 +++- + fs/file_attr.c | 6 +- + fs/fs-writeback.c | 23 +- + fs/fs_pin.c | 9 +- + fs/fuse/acl.c | 4 +- + fs/fuse/dir.c | 51 +- + fs/fuse/file.c | 2 +- + fs/fuse/fuse_i.h | 12 +- + fs/fuse/ioctl.c | 2 +- + fs/fuse/req.c | 8 +- + fs/fuse/xattr.c | 2 +- + fs/gfs2/acl.c | 2 +- + fs/gfs2/acl.h | 2 +- + fs/gfs2/file.c | 2 +- + fs/gfs2/inode.c | 16 +- + fs/gfs2/inode.h | 4 +- + fs/gfs2/log.c | 4 +- + fs/gfs2/lops.c | 4 +- + fs/gfs2/xattr.c | 2 +- + fs/hfs/attr.c | 2 +- + fs/hfs/dir.c | 6 +- + fs/hfs/hfs_fs.h | 2 +- + fs/hfs/inode.c | 2 +- + fs/hfsplus/dir.c | 10 +- + fs/hfsplus/hfsplus_fs.h | 4 +- + fs/hfsplus/inode.c | 6 +- + fs/hfsplus/xattr.c | 2 +- + fs/hfsplus/xattr_security.c | 2 +- + fs/hfsplus/xattr_trusted.c | 2 +- + fs/hfsplus/xattr_user.c | 2 +- + fs/hostfs/hostfs_kern.c | 14 +- + fs/hpfs/hpfs_fn.h | 2 +- + fs/hpfs/inode.c | 2 +- + fs/hpfs/namei.c | 10 +- + fs/hugetlbfs/inode.c | 14 +- + fs/inode.c | 14 +- + fs/internal.h | 30 +- + fs/iomap/bio.c | 4 +- + fs/iomap/buffered-io.c | 8 +- + fs/iomap/direct-io.c | 46 +- + fs/iomap/ioend.c | 122 +- + fs/jbd2/commit.c | 22 +- + fs/jbd2/journal.c | 31 +- + fs/jbd2/transaction.c | 2 +- + fs/jffs2/acl.c | 2 +- + fs/jffs2/acl.h | 2 +- + fs/jffs2/dir.c | 20 +- + fs/jffs2/fs.c | 2 +- + fs/jffs2/os-linux.h | 2 +- + fs/jffs2/security.c | 2 +- + fs/jffs2/xattr_trusted.c | 2 +- + fs/jffs2/xattr_user.c | 2 +- + fs/jfs/acl.c | 2 +- + fs/jfs/file.c | 2 +- + fs/jfs/ioctl.c | 2 +- + fs/jfs/jfs_acl.h | 2 +- + fs/jfs/jfs_inode.h | 4 +- + fs/jfs/namei.c | 10 +- + fs/jfs/xattr.c | 4 +- + fs/kernfs/dir.c | 231 ++- + fs/kernfs/file.c | 47 +- + fs/kernfs/inode.c | 10 +- + fs/kernfs/kernfs-internal.h | 24 +- + fs/kernfs/mount.c | 34 +- + fs/kernfs/symlink.c | 17 +- + fs/libfs.c | 8 +- + fs/minix/file.c | 2 +- + fs/minix/inode.c | 2 +- + fs/minix/minix.h | 2 +- + fs/minix/namei.c | 12 +- + fs/mnt_idmapping.c | 33 +- + fs/mount.h | 3 +- + fs/namei.c | 153 +- + fs/namespace.c | 75 +- + fs/netfs/Kconfig | 3 + + fs/netfs/Makefile | 2 +- + fs/netfs/buffered_read.c | 217 ++- + fs/netfs/buffered_write.c | 69 +- + fs/netfs/direct_read.c | 14 +- + fs/netfs/direct_write.c | 8 +- + fs/netfs/fscache_cookie.c | 8 +- + fs/netfs/fscache_internal.h | 14 - + fs/netfs/fscache_io.c | 10 +- + fs/netfs/internal.h | 70 +- + fs/netfs/iterator.c | 2 +- + fs/netfs/main.c | 1 - + fs/netfs/misc.c | 10 +- + fs/netfs/objects.c | 7 +- + fs/netfs/read_collect.c | 14 +- + fs/netfs/read_pgpriv2.c | 13 +- + fs/netfs/read_retry.c | 7 +- + fs/netfs/read_single.c | 50 +- + fs/netfs/stats.c | 4 +- + fs/netfs/write_collect.c | 157 +- + fs/netfs/write_issue.c | 141 +- + fs/netfs/write_retry.c | 8 +- + fs/nfs/Kconfig | 1 + + fs/nfs/dir.c | 12 +- + fs/nfs/inode.c | 4 +- + fs/nfs/internal.h | 10 +- + fs/nfs/namespace.c | 4 +- + fs/nfs/nfs3_fs.h | 2 +- + fs/nfs/nfs3acl.c | 2 +- + fs/nfs/nfs4proc.c | 10 +- + fs/nfs/unlink.c | 3 + + fs/nfsd/vfs.c | 13 +- + fs/nilfs2/inode.c | 4 +- + fs/nilfs2/ioctl.c | 2 +- + fs/nilfs2/namei.c | 10 +- + fs/nilfs2/nilfs.h | 6 +- + fs/nls/nls_iso8859-14.c | 26 +- + fs/nsfs.c | 4 +- + fs/ntfs/ea.c | 10 +- + fs/ntfs/ea.h | 6 +- + fs/ntfs/file.c | 4 +- + fs/ntfs/inode.h | 4 +- + fs/ntfs/namei.c | 12 +- + fs/ntfs3/file.c | 6 +- + fs/ntfs3/inode.c | 2 +- + fs/ntfs3/namei.c | 10 +- + fs/ntfs3/ntfs_fs.h | 16 +- + fs/ntfs3/xattr.c | 12 +- + fs/ocfs2/acl.c | 2 +- + fs/ocfs2/acl.h | 2 +- + fs/ocfs2/buffer_head_io.c | 12 +- + fs/ocfs2/dlmfs/dlmfs.c | 6 +- + fs/ocfs2/file.c | 6 +- + fs/ocfs2/file.h | 6 +- + fs/ocfs2/ioctl.c | 2 +- + fs/ocfs2/ioctl.h | 2 +- + fs/ocfs2/journal.c | 25 +- + fs/ocfs2/namei.c | 10 +- + fs/ocfs2/xattr.c | 6 +- + fs/omfs/dir.c | 6 +- + fs/omfs/file.c | 2 +- + fs/omfs/inode.c | 4 +- + fs/open.c | 50 +- + fs/orangefs/acl.c | 2 +- + fs/orangefs/inode.c | 8 +- + fs/orangefs/namei.c | 8 +- + fs/orangefs/orangefs-kernel.h | 8 +- + fs/orangefs/xattr.c | 2 +- + fs/overlayfs/dir.c | 14 +- + fs/overlayfs/file.c | 2 +- + fs/overlayfs/inode.c | 16 +- + fs/overlayfs/overlayfs.h | 14 +- + fs/overlayfs/ovl_entry.h | 2 +- + fs/overlayfs/util.c | 4 +- + fs/overlayfs/xattrs.c | 4 +- + fs/pidfs.c | 6 +- + fs/pipe.c | 2 +- + fs/pnode.c | 108 +- + fs/posix_acl.c | 26 +- + fs/proc/base.c | 53 +- + fs/proc/fd.c | 6 +- + fs/proc/fd.h | 2 +- + fs/proc/generic.c | 4 +- + fs/proc/internal.h | 4 +- + fs/proc/proc_net.c | 2 +- + fs/proc/proc_sysctl.c | 6 +- + fs/proc/root.c | 2 +- + fs/proc/vmcore.c | 20 + + fs/quota/dquot.c | 2 +- + fs/ramfs/file-nommu.c | 4 +- + fs/ramfs/inode.c | 10 +- + fs/read_write.c | 8 +- + fs/remap_range.c | 2 +- + fs/smb/client/cifsacl.c | 4 +- + fs/smb/client/cifsfs.c | 2 +- + fs/smb/client/cifsfs.h | 16 +- + fs/smb/client/cifsproto.h | 4 +- + fs/smb/client/dir.c | 6 +- + fs/smb/client/inode.c | 8 +- + fs/smb/client/link.c | 2 +- + fs/smb/client/transport.c | 13 +- + fs/smb/client/xattr.c | 2 +- + fs/smb/server/ndr.c | 2 +- + fs/smb/server/ndr.h | 2 +- + fs/smb/server/oplock.c | 2 +- + fs/smb/server/smb2pdu.c | 22 +- + fs/smb/server/smb_common.c | 2 +- + fs/smb/server/smbacl.c | 20 +- + fs/smb/server/smbacl.h | 8 +- + fs/smb/server/vfs.c | 44 +- + fs/smb/server/vfs.h | 32 +- + fs/splice.c | 70 +- + fs/stat.c | 4 +- + fs/super.c | 19 +- + fs/tests/.kunitconfig | 2 + + fs/tests/fdtable_kunit.c | 72 + + fs/tracefs/event_inode.c | 2 +- + fs/tracefs/inode.c | 8 +- + fs/ubifs/dir.c | 14 +- + fs/ubifs/file.c | 4 +- + fs/ubifs/ioctl.c | 2 +- + fs/ubifs/ubifs.h | 6 +- + fs/ubifs/xattr.c | 2 +- + fs/udf/file.c | 2 +- + fs/udf/namei.c | 12 +- + fs/udf/symlink.c | 2 +- + fs/ufs/dir.c | 2 +- + fs/ufs/inode.c | 2 +- + fs/ufs/namei.c | 10 +- + fs/ufs/ufs.h | 2 +- + fs/vboxsf/dir.c | 8 +- + fs/vboxsf/utils.c | 4 +- + fs/vboxsf/vfsmod.h | 4 +- + fs/xattr.c | 30 +- + fs/xfs/libxfs/xfs_errortag.h | 6 +- + fs/xfs/libxfs/xfs_inode_util.h | 2 +- + fs/xfs/xfs_acl.c | 2 +- + fs/xfs/xfs_acl.h | 2 +- + fs/xfs/xfs_aops.c | 13 +- + fs/xfs/xfs_buf.c | 11 + + fs/xfs/xfs_file.c | 11 +- + fs/xfs/xfs_handle.c | 6 +- + fs/xfs/xfs_inode.c | 4 +- + fs/xfs/xfs_inode.h | 2 +- + fs/xfs/xfs_ioctl.c | 2 +- + fs/xfs/xfs_ioctl.h | 2 +- + fs/xfs/xfs_ioend.c | 137 +- + fs/xfs/xfs_ioend.h | 2 + + fs/xfs/xfs_iops.c | 24 +- + fs/xfs/xfs_iops.h | 2 +- + fs/xfs/xfs_itable.c | 2 +- + fs/xfs/xfs_itable.h | 2 +- + fs/xfs/xfs_mount.h | 8 + + fs/xfs/xfs_super.c | 1 + + fs/xfs/xfs_symlink.c | 2 +- + fs/xfs/xfs_symlink.h | 2 +- + fs/xfs/xfs_sysfs.c | 78 +- + fs/xfs/xfs_trace.h | 1 + + fs/xfs/xfs_xattr.c | 2 +- + fs/xfs/xfs_zone_alloc.c | 4 + + fs/zonefs/super.c | 2 +- + include/linux/binfmts.h | 3 +- + include/linux/bio-integrity.h | 3 +- + include/linux/bio.h | 11 +- + include/linux/blkdev.h | 8 +- + include/linux/buffer_head.h | 86 +- + include/linux/capability.h | 8 +- + include/linux/cleanup.h | 7 - + include/linux/coredump.h | 37 +- + include/linux/dax.h | 12 - + include/linux/dcache.h | 38 +- + include/linux/fdtable.h | 15 +- + include/linux/file.h | 130 +- + include/linux/fileattr.h | 2 +- + include/linux/fs.h | 127 +- + include/linux/fs_context.h | 4 + + include/linux/fscache-cache.h | 2 +- + include/linux/fscache.h | 53 +- + include/linux/iomap.h | 38 +- + include/linux/lsm_hook_defs.h | 25 +- + include/linux/mnt_idmapping.h | 24 +- + include/linux/mount.h | 4 +- + include/linux/namei.h | 19 +- + include/linux/netfs.h | 112 +- + include/linux/nfs_fs.h | 6 +- + include/linux/posix_acl.h | 24 +- + include/linux/quotaops.h | 6 +- + include/linux/sched.h | 2 +- + include/linux/sched/signal.h | 33 +- + include/linux/security.h | 61 +- + include/linux/splice.h | 4 +- + include/linux/uidgid.h | 12 +- + include/linux/user_namespace.h | 11 +- + include/linux/wait_bit.h | 26 + + include/linux/xattr.h | 20 +- + include/trace/events/cachefiles.h | 81 +- + include/trace/events/fscache.h | 10 +- + include/trace/events/netfs.h | 145 +- + include/uapi/linux/close_range.h | 31 +- + include/uapi/linux/coredump.h | 149 +- + include/uapi/linux/fs.h | 2 +- + init/Kconfig | 11 + + init/initramfs.c | 11 +- + io_uring/io-wq.c | 2 + + io_uring/mock_file.c | 8 +- + ipc/mqueue.c | 9 +- + kernel/Makefile | 1 + + kernel/bpf/bpf_iter.c | 6 +- + kernel/bpf/inode.c | 6 +- + kernel/bpf/token.c | 6 +- + kernel/capability.c | 4 +- + kernel/exit.c | 15 +- + kernel/fork.c | 68 +- + kernel/kthread.c | 2 +- + kernel/pid_namespace.c | 3 +- + kernel/ptrace.c | 6 + + kernel/signal.c | 14 + + kernel/tests/.kunitconfig | 4 + + kernel/tests/user_ns_map_kunit.c | 98 + + kernel/user_namespace.c | 60 +- + kernel/utsname.c | 1 - + mm/secretmem.c | 2 +- + mm/shmem.c | 30 +- + mm/userfaultfd.c | 6 +- + net/core/scm.c | 8 +- + net/handshake/netlink.c | 8 +- + net/kcm/kcmsock.c | 6 +- + net/socket.c | 6 +- + net/unix/af_unix.c | 2 +- + security/apparmor/apparmorfs.c | 2 +- + security/apparmor/lsm.c | 4 +- + security/commoncap.c | 10 +- + security/integrity/evm/evm_main.c | 26 +- + security/integrity/ima/ima.h | 10 +- + security/integrity/ima/ima_api.c | 2 +- + security/integrity/ima/ima_appraise.c | 12 +- + security/integrity/ima/ima_main.c | 6 +- + security/integrity/ima/ima_policy.c | 4 +- + security/security.c | 49 +- + security/selinux/hooks.c | 45 +- + security/selinux/selinuxfs.c | 2 +- + security/smack/smack_lsm.c | 14 +- + tools/include/uapi/linux/coredump.h | 149 +- + tools/testing/selftests/Makefile | 3 + + .../selftests/clone3/clone3_clear_sighand.c | 6 +- + tools/testing/selftests/core/close_range_test.c | 951 ++++++++++ + tools/testing/selftests/coredump/.gitignore | 2 + + tools/testing/selftests/coredump/Makefile | 11 +- + .../selftests/coredump/coredump_notify_signal.h | 29 + + .../coredump/coredump_notify_signal_helper.c | 46 + + .../coredump/coredump_notify_signal_test.c | 245 +++ + .../selftests/coredump/coredump_signal_test.c | 238 +++ + .../coredump/coredump_socket_protocol_test.c | 1983 +++++++++++++++----- + tools/testing/selftests/coredump/coredump_test.h | 32 +- + .../selftests/coredump/coredump_test_helpers.c | 1742 ++++++++++++++++- + .../selftests/coredump/coredump_test_helpers.h | 79 + + .../selftests/coredump/coredump_worker_test.c | 447 +++++ + tools/testing/selftests/exec/Makefile | 4 + + tools/testing/selftests/exec/binfmt_misc_delim.c | 127 ++ + tools/testing/selftests/filesystems/.gitignore | 1 - + tools/testing/selftests/filesystems/Makefile | 2 +- + tools/testing/selftests/filesystems/config | 8 + + .../selftests/filesystems/file_stressor/.gitignore | 2 + + .../selftests/filesystems/file_stressor/Makefile | 6 + + .../{ => file_stressor}/file_stressor.c | 0 + .../selftests/filesystems/file_stressor/settings | 3 + + .../selftests/filesystems/fscontext_ns/.gitignore | 2 + + tools/testing/selftests/filesystems/kernfs_test.c | 1203 +++++++++++- + .../filesystems/mntns_unbindable/Makefile | 6 + + .../mntns_unbindable/mntns_unbindable_test.c | 227 +++ + .../selftests/filesystems/openat2/openat2_test.c | 10 +- + .../selftests/filesystems/openat2/resolve_test.c | 7 +- + .../filesystems/statmount/statmount_test.c | 2 +- + .../filesystems/umount_propagation/Makefile | 6 + + .../umount_propagation/umount_propagation_test.c | 226 +++ + .../move_mount_set_group_test.c | 74 +- + tools/testing/selftests/pidfd/pidfd_open_test.c | 2 +- + virt/kvm/guest_memfd.c | 2 +- + 519 files changed, 12646 insertions(+), 5165 deletions(-) + delete mode 100644 Documentation/filesystems/bfs.rst + delete mode 100644 arch/powerpc/platforms/cell/spufs/coredump.c + delete mode 100644 fs/bfs/Kconfig + delete mode 100644 fs/bfs/Makefile + delete mode 100644 fs/bfs/bfs.h + delete mode 100644 fs/bfs/file.c + delete mode 100644 fs/bfs/inode.c + delete mode 100644 fs/netfs/fscache_internal.h + create mode 100644 fs/tests/.kunitconfig + create mode 100644 fs/tests/fdtable_kunit.c + create mode 100644 kernel/tests/.kunitconfig + create mode 100644 kernel/tests/user_ns_map_kunit.c + create mode 100644 tools/testing/selftests/coredump/coredump_notify_signal.h + create mode 100644 tools/testing/selftests/coredump/coredump_notify_signal_helper.c + create mode 100644 tools/testing/selftests/coredump/coredump_notify_signal_test.c + create mode 100644 tools/testing/selftests/coredump/coredump_signal_test.c + create mode 100644 tools/testing/selftests/coredump/coredump_test_helpers.h + create mode 100644 tools/testing/selftests/coredump/coredump_worker_test.c + create mode 100644 tools/testing/selftests/exec/binfmt_misc_delim.c + create mode 100644 tools/testing/selftests/filesystems/config + create mode 100644 tools/testing/selftests/filesystems/file_stressor/.gitignore + create mode 100644 tools/testing/selftests/filesystems/file_stressor/Makefile + rename tools/testing/selftests/filesystems/{ => file_stressor}/file_stressor.c (100%) + create mode 100644 tools/testing/selftests/filesystems/file_stressor/settings + create mode 100644 tools/testing/selftests/filesystems/fscontext_ns/.gitignore + create mode 100644 tools/testing/selftests/filesystems/mntns_unbindable/Makefile + create mode 100644 tools/testing/selftests/filesystems/mntns_unbindable/mntns_unbindable_test.c + create mode 100644 tools/testing/selftests/filesystems/umount_propagation/Makefile + create mode 100644 tools/testing/selftests/filesystems/umount_propagation/umount_propagation_test.c +$ git am -3 ../patches/0001-ksmbd-Fix-removal-of-type-parameter-from-vfs_path_pa.patch +Applying: ksmbd: Fix removal of type parameter from vfs_path_parent_lookup() +Using index info to reconstruct a base tree... +M fs/smb/server/vfs.c +Falling back to patching base and 3-way merge... +Auto-merging fs/smb/server/vfs.c +No changes -- Patch already applied. +Merging vfs/for-next (4dda01b67c866 Merge branches 'work.dcache', 'work.dcache-d_add' and 'work.configfs' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/viro/vfs.git vfs/for-next +Merge made by the 'ort' strategy. +Merging mm-fixes/for-next-fixes (c14d066c0108f Merge https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm.git mm-hotfixes-unstable into for-next-fixes) +$ git merge -m Merge branch 'for-next-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/mm/linux.git mm-fixes/for-next-fixes +Updating ce1e0223d8ad4..c14d066c0108f +Fast-forward (no commit created; -m option ignored) + .mailmap | 3 +- + MAINTAINERS | 2 +- + drivers/char/mem.c | 2 +- + kernel/taskstats.c | 3 + + lib/assoc_array.c | 36 +++++---- + lib/test_xarray.c | 70 ++++++++++++++++++ + lib/xarray.c | 8 +- + mm/hugetlb.c | 25 +++++-- + mm/mremap.c | 70 +++++++++++++----- + mm/page_alloc.c | 106 +++++++++++++++++++-------- + mm/pgtable-generic.c | 20 ++++- + mm/slub.c | 7 +- + mm/userfaultfd.c | 1 + + mm/vma.c | 21 +++++- + mm/vma.h | 7 +- + mm/vmalloc.c | 79 +++++++++++++------- + tools/testing/selftests/mm/hugetlb-mmap.c | 4 +- + tools/testing/selftests/mm/uffd-unit-tests.c | 84 +++++++++++++++++++++ + tools/testing/vma/tests/vma.c | 10 +-- + 19 files changed, 439 insertions(+), 119 deletions(-) +Merging fs-current (e2aee709a9b35 Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/viro/vfs.git) +$ git merge -m Merge branch 'fs-current' of linux-next fs-current +Merge made by the 'ort' strategy. + fs/btrfs/ctree.c | 4 ++-- + fs/btrfs/inode.c | 11 +++++------ + fs/btrfs/ioctl.c | 31 ++++++++++++++++--------------- + fs/btrfs/scrub.c | 4 +++- + fs/btrfs/transaction.c | 1 + + fs/btrfs/tree-log.c | 4 ++-- + fs/btrfs/xattr.c | 15 +++++++++++---- + 7 files changed, 40 insertions(+), 30 deletions(-) +Merging kbuild-current/kbuild-fixes-for-next (fd73f4a665989 Linux 7.3-rc3) +$ git merge -m Merge branch 'kbuild-fixes-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/kbuild/linux.git kbuild-current/kbuild-fixes-for-next +Already up to date. +Merging clang-fixes-current/clang-fixes-for-current (df2908090cda3 Linux 7.3-rc2) +$ git merge -m Merge branch 'clang-fixes-for-current' of https://git.kernel.org/pub/scm/linux/kernel/git/nathan/linux.git clang-fixes-current/clang-fixes-for-current +Already up to date. +Merging arc-current/for-curr (f050c3e61d2a1 ARC: arch_cmpxchg_relaxed to use size of pointed type not pointer) +$ git merge -m Merge branch 'for-curr' of https://git.kernel.org/pub/scm/linux/kernel/git/vgupta/arc.git arc-current/for-curr +Already up to date. +Merging arm-current/fixes (1039bffd6ae9c ARM: 9485/1: mm: acquire mmap write lock around show_pte() for user faults) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/rmk/linux.git arm-current/fixes +Already up to date. +Merging arm64-fixes/for-next/fixes (e060d9069b49f arm64/fpsimd: signal: Forbid non-streaming SVE payload on SME-only systems) +$ git merge -m Merge branch 'for-next/fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/arm64/linux arm64-fixes/for-next/fixes +Merge made by the 'ort' strategy. + Documentation/arch/arm64/silicon-errata.rst | 2 ++ + arch/arm64/Kconfig | 30 +++++++++++++++++++++++++ + arch/arm64/kernel/cpu_errata.c | 24 +++++++++++++++----- + arch/arm64/kernel/cpufeature.c | 2 +- + arch/arm64/kernel/signal.c | 34 +++++++++++++++++------------ + arch/arm64/kernel/topology.c | 22 +++++++++---------- + arch/arm64/mm/mmu.c | 2 +- + arch/arm64/tools/cpucaps | 2 +- + 8 files changed, 84 insertions(+), 34 deletions(-) +Merging arm-soc-fixes/arm/fixes (4dd1999783d7d soc: samsung: exynos-pmu: fix use-after-free of interrupt generator node) +$ git merge -m Merge branch 'arm/fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/soc/soc.git arm-soc-fixes/arm/fixes +Already up to date. +Merging davinci-current/davinci/for-current (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'davinci/for-current' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git davinci-current/davinci/for-current +Already up to date. +Merging realtek-fixes/fixes (dc59e4fea9d83 Linux 7.2-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/yu_chun/linux.git realtek-fixes/fixes +Already up to date. +Merging drivers-memory-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-mem-ctrl.git drivers-memory-fixes/fixes +Already up to date. +Merging sophgo-fixes/fixes (19272b37aa4f8 Linux 6.16-rc1) +$ git merge -m Merge branch 'fixes' of https://github.com/sophgo/linux.git sophgo-fixes/fixes +Already up to date. +Merging sophgo-soc-fixes/soc-fixes (0af2f6be1b428 Linux 6.15-rc1) +$ git merge -m Merge branch 'soc-fixes' of https://github.com/sophgo/linux.git sophgo-soc-fixes/soc-fixes +Already up to date. +Merging m68k-current/for-linus (2f8e3cad53b5c m68k: nfcon: Do not call console_is_registered() in nfcon_device()) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/geert/linux-m68k.git m68k-current/for-linus +Already up to date. +Merging powerpc-fixes/fixes (93f51579e7df2 Linux 7.3-rc4) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/powerpc/linux.git powerpc-fixes/fixes +Already up to date. +Merging s390-fixes/fixes (5b76268dac968 s390/cio: Fix NULL pointer dereference in ccw_device_get_util_str()) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/s390/linux.git s390-fixes/fixes +Already up to date. +Merging net/main (232d49dd4b40a net: stmmac: propagate PTP addend and system time programming errors) +$ git merge -m Merge branch 'main' of https://git.kernel.org/pub/scm/linux/kernel/git/netdev/net.git net/main +Merge made by the 'ort' strategy. + drivers/net/amt.c | 48 +++----- + .../chelsio/inline_crypto/ch_ktls/chcr_ktls.c | 4 +- + drivers/net/ethernet/ibm/ibmveth.c | 22 ++-- + drivers/net/ethernet/stmicro/stmmac/stmmac_main.c | 106 ++++++++++++----- + drivers/net/ethernet/stmicro/stmmac/stmmac_ptp.c | 10 +- + include/net/amt.h | 4 - + lib/Kconfig.debug | 15 +++ + lib/dim/Makefile | 2 + + lib/dim/dim.c | 10 +- + lib/dim/dim_kunit.c | 126 +++++++++++++++++++++ + net/packet/af_packet.c | 3 +- + net/sched/em_text.c | 3 + + net/tipc/node.c | 10 +- + net/tipc/topsrv.c | 80 ++++++++++--- + tools/testing/selftests/net/amt.sh | 29 +++++ + 15 files changed, 371 insertions(+), 101 deletions(-) + create mode 100644 lib/dim/dim_kunit.c +Merging bpf/master (7b12c538697f2 Merge branch 'bpf-fix-objects-stuck-in-free_by_rcu_ttrace') +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf.git/ bpf/master +Merge made by the 'ort' strategy. + arch/arm64/net/bpf_jit_comp.c | 37 ++-- + arch/loongarch/net/bpf_jit.c | 60 ++++--- + arch/powerpc/net/bpf_jit_comp.c | 45 ++--- + arch/riscv/net/bpf_jit_comp64.c | 56 ++++-- + arch/s390/net/bpf_jit_comp.c | 46 +++-- + arch/x86/net/bpf_jit_comp.c | 26 +-- + include/linux/bpf.h | 45 ++++- + kernel/bpf/core.c | 10 +- + kernel/bpf/hashtab.c | 17 +- + kernel/bpf/memalloc.c | 57 +++++-- + kernel/bpf/syscall.c | 19 ++- + kernel/bpf/trampoline.c | 85 ++++++++-- + kernel/bpf/verifier.c | 10 +- + .../selftests/bpf/prog_tests/bpf_ma_ttrace.c | 60 +++++++ + .../selftests/bpf/prog_tests/bpf_mod_race.c | 34 +--- + .../selftests/bpf/prog_tests/tramp_prog_detach.c | 188 +++++++++++++++++++++ + tools/testing/selftests/bpf/progs/bpf_ma_ttrace.c | 50 ++++++ + .../selftests/bpf/progs/tramp_prog_detach.c | 56 ++++++ + tools/testing/selftests/bpf/progs/verifier_align.c | 6 +- + .../bpf/progs/verifier_direct_packet_access.c | 178 +++++++++++++++++++ + .../selftests/bpf/progs/verifier_meta_access.c | 30 ++++ + tools/testing/selftests/bpf/testing_helpers.c | 28 +++ + tools/testing/selftests/bpf/testing_helpers.h | 2 + + 23 files changed, 973 insertions(+), 172 deletions(-) + create mode 100644 tools/testing/selftests/bpf/prog_tests/bpf_ma_ttrace.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/tramp_prog_detach.c + create mode 100644 tools/testing/selftests/bpf/progs/bpf_ma_ttrace.c + create mode 100644 tools/testing/selftests/bpf/progs/tramp_prog_detach.c +Merging ipsec/master (86de3a1118a16 xfrm: interface: validate the IP header on xmit) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/klassert/ipsec.git ipsec/master +Merge made by the 'ort' strategy. + net/xfrm/xfrm_interface_core.c | 8 ++++++++ + net/xfrm/xfrm_policy.c | 11 +++++++++-- + 2 files changed, 17 insertions(+), 2 deletions(-) +Merging netfilter/main (e23a64eb24435 Merge tag 'nf-26-09-30' of git://git.kernel.org/pub/scm/linux/kernel/git/netfilter/nf) +$ git merge -m Merge branch 'main' of https://git.kernel.org/pub/scm/linux/kernel/git/netfilter/nf.git netfilter/main +Already up to date. +Merging ipvs/main (a401a9d547c50 Merge branch 'net-macb-fix-two-probe-path-leaks') +$ git merge -m Merge branch 'main' of https://git.kernel.org/pub/scm/linux/kernel/git/horms/ipvs.git ipvs/main +Already up to date. +Merging bluetooth-fixes/master (86ef0f58bdecd Bluetooth: MGMT: Fix status of pending commands flushed on power off) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/bluetooth/bluetooth.git bluetooth-fixes/master +Merge made by the 'ort' strategy. + drivers/bluetooth/btintel_pcie.c | 9 ++- + include/net/bluetooth/rfcomm.h | 1 + + net/bluetooth/mgmt.c | 2 +- + net/bluetooth/rfcomm/core.c | 134 +++++++++++++++++++++++++++------------ + net/bluetooth/rfcomm/sock.c | 5 +- + net/bluetooth/sco.c | 87 ++++++++++++++++++------- + 6 files changed, 174 insertions(+), 64 deletions(-) +Merging wireless/for-next (d24e8ac715de2 Merge tag 'net-7.3-rc6' of git://git.kernel.org/pub/scm/linux/kernel/git/netdev/net) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/wireless/wireless.git wireless/for-next +Already up to date. +Merging ath/for-current (6f63e919fe1e3 wifi: mac80211: fix slab-out-of-bounds read in ieee80211_monitor_select_queue()) +$ git merge -m Merge branch 'for-current' of https://git.kernel.org/pub/scm/linux/kernel/git/ath/ath.git ath/for-current +Already up to date. +Merging iwlwifi/fixes (1e7e30fb650b8 wifi: iwlwifi: mvm: fix VHT NSS reporting on pre-MQ devices) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/iwlwifi/iwlwifi-next.git iwlwifi/fixes +Merge made by the 'ort' strategy. + drivers/net/wireless/intel/iwlwifi/iwl-nvm-parse.c | 6 +++--- + drivers/net/wireless/intel/iwlwifi/iwl-nvm-parse.h | 7 +++++-- + drivers/net/wireless/intel/iwlwifi/iwl-trans.c | 1 - + drivers/net/wireless/intel/iwlwifi/mvm/debugfs.c | 5 ++++- + drivers/net/wireless/intel/iwlwifi/mvm/fw.c | 3 +++ + drivers/net/wireless/intel/iwlwifi/mvm/rx.c | 4 ++-- + .../net/wireless/intel/iwlwifi/tests/nvm_parse.c | 24 ++++++++++++++++++---- + 7 files changed, 37 insertions(+), 13 deletions(-) +Merging wpan/master (2b4707a149a55 net: mctp: i3c: serialize probe with bus removal) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/wpan/wpan.git wpan/master +Already up to date. +Merging rdma-fixes/for-rc (72d3fcf802c45 Linux 7.3-rc5) +$ git merge -m Merge branch 'for-rc' of https://git.kernel.org/pub/scm/linux/kernel/git/rdma/rdma.git rdma-fixes/for-rc +Already up to date. +Merging sound-current/for-linus (e6229c0402ea4 ALSA: ice1712: Fix the error handling via auto-cleanup at probe) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/tiwai/sound.git sound-current/for-linus +Merge made by the 'ort' strategy. + include/sound/core.h | 9 ++++++ + sound/core/seq/seq_compat.c | 4 ++- + sound/hda/codecs/realtek/alc269.c | 62 +++++++++++++++++++++++++++++++++++--- + sound/hda/codecs/realtek/realtek.c | 6 ++++ + sound/hda/controllers/tegra.c | 3 +- + sound/pci/ctxfi/cthw20k2.c | 2 ++ + sound/pci/ice1712/ice1712.c | 2 +- + sound/usb/caiaq/device.c | 3 ++ + sound/usb/line6/playback.c | 7 +++++ + sound/usb/misc/ua101.c | 9 ++++++ + sound/usb/mixer_maps.c | 4 +++ + sound/usb/quirks.c | 16 +++++++--- + 12 files changed, 116 insertions(+), 11 deletions(-) +Merging sound-asoc-fixes/for-linus (cad16d850e375 spi: cs42l43: Workaround for wrong speaker ID on Dell XPS 13 DX13260) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/broonie/sound.git sound-asoc-fixes/for-linus +Merge made by the 'ort' strategy. + drivers/spi/spi-cs42l43.c | 97 ++++++++++++++++++++++ + sound/soc/amd/acp-config.c | 14 ++++ + sound/soc/amd/acp/Kconfig | 1 + + sound/soc/amd/acp/amd-acp70-acpi-match.c | 20 +++++ + sound/soc/amd/yc/acp6x-mach.c | 14 ++++ + sound/soc/codecs/Kconfig | 4 +- + sound/soc/codecs/lpass-wsa-macro.c | 3 + + sound/soc/codecs/max98363.c | 2 +- + sound/soc/codecs/rt1017-sdca-sdw.c | 27 ++---- + sound/soc/codecs/rt274.c | 4 +- + sound/soc/codecs/rt286.c | 2 +- + sound/soc/codecs/wm8903.c | 2 +- + sound/soc/intel/boards/bytcr_rt5651.c | 17 ++++ + sound/soc/intel/common/soc-acpi-intel-lnl-match.c | 1 + + .../soc/intel/common/soc-acpi-intel-sdca-quirks.c | 16 ++++ + .../soc/intel/common/soc-acpi-intel-sdca-quirks.h | 1 + + sound/soc/qcom/lpass-cpu.c | 4 +- + sound/soc/sdca/sdca_class_function.c | 19 +++-- + sound/soc/tegra/tegra186_asrc.c | 2 +- + 19 files changed, 215 insertions(+), 35 deletions(-) +Merging regmap-fixes/for-linus (2e42cade8ff1f regmap: irq: Free the irqdomain we create) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/broonie/regmap.git regmap-fixes/for-linus +Merge made by the 'ort' strategy. + drivers/base/regmap/regmap-irq.c | 2 +- + 1 file changed, 1 insertion(+), 1 deletion(-) +Merging regulator-fixes/for-linus (2f15518682c77 regulator: tps6594-regulator: Fix n_linear_ranges for TPS65224 BUCK2-4) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/broonie/regulator.git regulator-fixes/for-linus +Merge made by the 'ort' strategy. + drivers/regulator/tps6594-regulator.c | 6 +++--- + 1 file changed, 3 insertions(+), 3 deletions(-) +Merging spi-fixes/for-linus (3d743adf090cd spi: fsl-qspi: Reprogram the clock rate when the operation frequency changes) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/broonie/spi.git spi-fixes/for-linus +Already up to date. +Merging pci-current/for-linus (97958feb6560b PCI: Accept AtomicOps already enabled by the hypervisor) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/pci/pci.git pci-current/for-linus +Merge made by the 'ort' strategy. + drivers/pci/pci.c | 12 +++++++++++- + 1 file changed, 11 insertions(+), 1 deletion(-) +Merging driver-core.current/driver-core-linus (72d3fcf802c45 Linux 7.3-rc5) +$ git merge -m Merge branch 'driver-core-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/driver-core/driver-core.git driver-core.current/driver-core-linus +Already up to date. +Merging tty.current/tty-linus (6c95ca52f2785 tty: add missing driver flag kernel-doc colon) +$ git merge -m Merge branch 'tty-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/tty.git tty.current/tty-linus +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 5 + + drivers/staging/greybus/uart.c | 3 +- + drivers/tty/amiserial.c | 9 +- + drivers/tty/n_gsm.c | 47 ++----- + drivers/tty/serial/8250/8250_bcm7271.c | 2 +- + drivers/tty/serial/8250/8250_omap.c | 15 ++- + drivers/tty/serial/8250/8250_port.c | 2 +- + drivers/tty/serial/kgdboc.c | 9 ++ + drivers/tty/serial/ma35d1_serial.c | 6 +- + drivers/tty/serial/max3100.c | 13 +- + drivers/tty/serial/mpc52xx_uart.c | 6 +- + drivers/tty/serial/qcom_geni_serial.c | 229 ++++++++++++++++++--------------- + drivers/tty/serial/sc16is7xx.c | 53 ++++++-- + drivers/tty/serial/serial-tegra.c | 18 ++- + drivers/tty/serial/serial_core.c | 76 ++++++++--- + drivers/tty/serial/vt8500_serial.c | 1 + + drivers/tty/tty_io.c | 54 +++++--- + drivers/tty/tty_port.c | 91 ++++++++----- + drivers/tty/vcc.c | 5 +- + drivers/tty/vt/selection.c | 4 +- + drivers/tty/vt/vc_screen.c | 23 ++-- + drivers/tty/vt/vt.c | 5 +- + drivers/usb/class/cdc-acm.c | 2 +- + drivers/usb/serial/usb-serial.c | 3 +- + include/linux/soc/qcom/geni-se.h | 15 +-- + include/linux/tty.h | 2 + + include/linux/tty_driver.h | 6 + + include/linux/tty_port.h | 22 +--- + 28 files changed, 442 insertions(+), 284 deletions(-) +Merging usb.current/usb-linus (a93862fa670a9 Merge tag 'thunderbolt-for-v7.3-rc6' of ssh://gitolite.kernel.org/pub/scm/linux/kernel/git/westeri/thunderbolt into usb-linus) +$ git merge -m Merge branch 'usb-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/usb.git usb.current/usb-linus +Auto-merging drivers/usb/class/cdc-acm.c +Auto-merging drivers/usb/serial/usb-serial.c +Merge made by the 'ort' strategy. + drivers/thunderbolt/ctl.c | 63 +++++++------- + drivers/thunderbolt/domain.c | 26 +----- + drivers/thunderbolt/nhi.c | 31 ++----- + drivers/thunderbolt/nhi.h | 21 +---- + drivers/thunderbolt/nhi_regs.h | 4 - + drivers/thunderbolt/pci.c | 23 ----- + drivers/thunderbolt/quirks.c | 3 + + drivers/thunderbolt/stream.c | 3 + + drivers/thunderbolt/switch.c | 9 +- + drivers/thunderbolt/tb.c | 37 +++++--- + drivers/thunderbolt/test.c | 58 ++++++++++--- + drivers/thunderbolt/tunnel.c | 73 ++++++++++------ + drivers/thunderbolt/tunnel.h | 8 +- + drivers/thunderbolt/xdomain.c | 5 +- + drivers/usb/cdns3/cdns3-gadget.c | 4 + + drivers/usb/cdns3/cdns3-pci-wrap.c | 5 ++ + drivers/usb/chipidea/ci_hdrc_tegra.c | 4 +- + drivers/usb/class/cdc-acm.c | 39 +++++---- + drivers/usb/class/cdc-acm.h | 1 - + drivers/usb/core/hub.c | 7 +- + drivers/usb/core/message.c | 8 +- + drivers/usb/dwc2/hcd.c | 3 +- + drivers/usb/dwc3/core.h | 19 +++++ + drivers/usb/dwc3/dwc3-am62.c | 1 + + drivers/usb/dwc3/ep0.c | 9 ++ + drivers/usb/dwc3/gadget.c | 108 +++++++++++++++++++++++- + drivers/usb/gadget/function/f_fs.c | 15 +++- + drivers/usb/gadget/function/f_midi2.c | 4 +- + drivers/usb/gadget/function/f_uac1_legacy.c | 6 ++ + drivers/usb/gadget/udc/aspeed-vhub/core.c | 4 + + drivers/usb/gadget/udc/aspeed-vhub/dev.c | 2 +- + drivers/usb/gadget/udc/aspeed-vhub/hub.c | 5 +- + drivers/usb/gadget/udc/aspeed-vhub/vhub.h | 1 + + drivers/usb/gadget/udc/dummy_hcd.c | 4 + + drivers/usb/host/octeon-hcd.c | 54 ++++++++++-- + drivers/usb/host/ohci-da8xx.c | 2 + + drivers/usb/host/ohci-s3c2410.c | 1 + + drivers/usb/host/ohci-spear.c | 1 + + drivers/usb/host/ohci-st.c | 1 + + drivers/usb/serial/bus.c | 6 +- + drivers/usb/serial/cp210x.c | 1 + + drivers/usb/serial/generic.c | 10 ++- + drivers/usb/serial/option.c | 14 ++++ + drivers/usb/serial/quatech2.c | 19 ++++- + drivers/usb/serial/ssu100.c | 18 +++- + drivers/usb/serial/usb-serial.c | 126 ++++++++++++++++++++++------ + drivers/usb/serial/xr_serial.c | 10 +-- + drivers/usb/storage/sierra_ms.c | 7 ++ + drivers/usb/typec/anx7411.c | 5 +- + drivers/usb/typec/port-mapper.c | 21 ++++- + drivers/usb/typec/tcpm/tcpm.c | 6 +- + drivers/usb/typec/tipd/core.c | 11 +-- + drivers/usb/typec/ucsi/displayport.c | 9 +- + drivers/usb/typec/ucsi/ucsi.c | 17 +++- + drivers/usb/typec/ucsi/ucsi_acpi.c | 34 ++++++++ + drivers/usb/typec/ucsi/ucsi_glink.c | 15 +++- + include/linux/thunderbolt.h | 3 + + include/linux/usb/serial.h | 1 + + 58 files changed, 726 insertions(+), 279 deletions(-) +Merging usb-serial-fixes/usb-linus (4b1e4a071ff07 USB: serial: option: add Rolling Wireless RN947R) +$ git merge -m Merge branch 'usb-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/johan/usb-serial.git usb-serial-fixes/usb-linus +Merge made by the 'ort' strategy. + drivers/usb/serial/option.c | 4 ++++ + 1 file changed, 4 insertions(+) +Merging phy/fixes (93f51579e7df2 Linux 7.3-rc4) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/phy/linux-phy.git phy/fixes +Already up to date. +Merging staging.current/staging-linus (df2908090cda3 Linux 7.3-rc2) +$ git merge -m Merge branch 'staging-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/staging.git staging.current/staging-linus +Already up to date. +Merging iio-fixes/fixes-togreg (9ee8306121495 iio: adc: ad_sigma_delta: fix use-after-free on unbind) +$ git merge -m Merge branch 'fixes-togreg' of https://git.kernel.org/pub/scm/linux/kernel/git/jic23/iio.git iio-fixes/fixes-togreg +Merge made by the 'ort' strategy. + .../bindings/iio/adc/rockchip-saradc.yaml | 18 +++---- + drivers/android/binder.c | 3 +- + drivers/android/binder/node.rs | 20 +++++-- + drivers/android/binder/node/wrapper.rs | 34 +++++++++++- + drivers/android/binder/rust_binderfs.c | 8 +++ + drivers/android/binder/thread.rs | 8 ++- + drivers/android/binderfs.c | 3 +- + drivers/iio/accel/kionix-kx022a.c | 30 ++++++++--- + drivers/iio/accel/kxcjk-1013.c | 2 +- + drivers/iio/accel/sca3000.c | 2 +- + drivers/iio/adc/ad4030.c | 11 +++- + drivers/iio/adc/ad7173.c | 4 +- + drivers/iio/adc/ad_sigma_delta.c | 31 ++++++----- + drivers/iio/adc/ade9000.c | 36 +++++++------ + drivers/iio/adc/adi-axi-adc.c | 4 ++ + drivers/iio/adc/aspeed_adc.c | 4 +- + drivers/iio/adc/axp288_adc.c | 8 +++ + drivers/iio/adc/max1363.c | 8 +++ + drivers/iio/adc/pac1934.c | 4 ++ + drivers/iio/adc/rohm-bd79124.c | 20 ++++--- + drivers/iio/adc/stm32-adc.c | 47 +++++++++------- + drivers/iio/adc/sun4i-gpadc-iio.c | 10 ++-- + drivers/iio/adc/xilinx-xadc-core.c | 9 ++-- + drivers/iio/buffer/industrialio-buffer-dmaengine.c | 16 ++++-- + drivers/iio/cdc/ad7150.c | 8 +-- + .../iio/common/inv_sensors/inv_sensors_timestamp.c | 11 ++-- + drivers/iio/dac/mcp47a1.c | 4 +- + drivers/iio/dac/rohm-bd79703.c | 3 ++ + drivers/iio/frequency/adf4377.c | 4 +- + drivers/iio/frequency/admv1013.c | 9 ++-- + drivers/iio/gyro/adis16136.c | 2 +- + drivers/iio/health/max30102.c | 12 +++-- + drivers/iio/imu/adis16400.c | 2 +- + drivers/iio/imu/adis16480.c | 4 +- + drivers/iio/imu/inv_icm42607/inv_icm42607_core.c | 38 +++++++++---- + drivers/iio/industrialio-buffer.c | 7 ++- + drivers/iio/industrialio-trigger.c | 2 + + drivers/iio/light/gp2ap020a00f.c | 2 + + drivers/iio/light/rohm-bu27034.c | 62 +++++++++++++--------- + drivers/iio/pressure/bmp280-core.c | 2 +- + drivers/iio/pressure/rohm-bm1390.c | 2 +- + drivers/iio/proximity/aw96103.c | 30 ++++++++--- + drivers/iio/proximity/isl29501.c | 2 +- + drivers/iio/proximity/pulsedlight-lidar-lite-v2.c | 4 +- + drivers/iio/proximity/sx9324.c | 2 +- + drivers/iio/proximity/vcnl3020.c | 24 ++++++--- + drivers/iio/proximity/vl53l0x-i2c.c | 3 ++ + drivers/virt/nitro_enclaves/ne_misc_dev.c | 1 + + 48 files changed, 407 insertions(+), 173 deletions(-) +Merging watchdog-fixes/watchdog (2686e13649a9e watchdog: mediatek: Apply the driver's mode to a watchdog left running) +$ git merge -m Merge branch 'watchdog' of https://git.kernel.org/pub/scm/linux/kernel/git/groeck/linux-staging.git watchdog-fixes/watchdog +Merge made by the 'ort' strategy. + drivers/watchdog/mtk_wdt.c | 43 ++++++++++++++++++++++++------------------- + 1 file changed, 24 insertions(+), 19 deletions(-) +Merging counter-current/counter-current (fd73f4a665989 Linux 7.3-rc3) +$ git merge -m Merge branch 'counter-current' of https://git.kernel.org/pub/scm/linux/kernel/git/wbg/counter.git counter-current/counter-current +Already up to date. +Merging char-misc.current/char-misc-linus (8e242ada093af Merge tag 'iio-fixes-late-7.3' of ssh://gitolite.kernel.org/pub/scm/linux/kernel/git/jic23/iio into char-misc-linus) +$ git merge -m Merge branch 'char-misc-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/char-misc.git char-misc.current/char-misc-linus +Merge made by the 'ort' strategy. + drivers/interconnect/qcom/x1e80100.c | 485 ----------------------------------- + 1 file changed, 485 deletions(-) +Merging soundwire-fixes/fixes (93f51579e7df2 Linux 7.3-rc4) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/soundwire.git soundwire-fixes/fixes +Already up to date. +Merging thunderbolt-fixes/fixes (395e9f2967a7a thunderbolt: Disable CL states for the Anker Prime TB5 dock) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/westeri/thunderbolt.git thunderbolt-fixes/fixes +Already up to date. +Merging input-current/for-linus (a215720959c54 Input: s6sy761 - fix resume ordering and restore sensing) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/dtor/input.git input-current/for-linus +Merge made by the 'ort' strategy. + drivers/input/mouse/synaptics.c | 7 +++++-- + drivers/input/serio/i8042-acpipnpio.h | 9 +++++++++ + drivers/input/touchscreen/s6sy761.c | 14 +++++++++++++- + 3 files changed, 27 insertions(+), 3 deletions(-) +Merging crypto-current/master (10396a2d6d41d crypto: s390/hmac - Generate intermediate CV for API partial block handling) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/herbert/crypto-2.6.git crypto-current/master +Already up to date. +Merging libcrypto-fixes/libcrypto-fixes (572af6872e520 crypto: aes - Fix undesired override of some optimized AES modes) +$ git merge -m Merge branch 'libcrypto-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/ebiggers/linux.git libcrypto-fixes/libcrypto-fixes +Already up to date. +Merging vfio-fixes/for-linus (e242e974e812e vfio: selftests: Add luuid to libvfio.mk's list of libraries, not to the Makefile) +$ git merge -m Merge branch 'for-linus' of https://github.com/awilliam/linux-vfio.git vfio-fixes/for-linus +Already up to date. +Merging kselftest-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git kselftest-fixes/fixes +Already up to date. +Merging dmaengine-fixes/fixes (93f51579e7df2 Linux 7.3-rc4) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/dmaengine.git dmaengine-fixes/fixes +Already up to date. +Merging backlight-fixes/for-backlight-fixes (dc59e4fea9d83 Linux 7.2-rc1) +$ git merge -m Merge branch 'for-backlight-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/lee/backlight.git backlight-fixes/for-backlight-fixes +Already up to date. +Merging mtd-fixes/mtd/fixes (1c1a342aceec7 mtd: spinand: Do not update the QE bit on devices without one) +$ git merge -m Merge branch 'mtd/fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git mtd-fixes/mtd/fixes +Already up to date. +Merging mfd-fixes/for-mfd-fixes (d5d2d7a8d8be1 MAINTAINERS: Add a mailing list entry to MFD) +$ git merge -m Merge branch 'for-mfd-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/lee/mfd.git mfd-fixes/for-mfd-fixes +Already up to date. +Merging v4l-dvb-fixes/fixes (2579cbe68005f media: em28xx: use video_unregister_device for radio_dev) +$ git merge -m Merge branch 'fixes' of git://linuxtv.org/media-ci/media-pending.git v4l-dvb-fixes/fixes +Merge made by the 'ort' strategy. + drivers/media/usb/em28xx/em28xx-video.c | 4 ++-- + 1 file changed, 2 insertions(+), 2 deletions(-) +Merging reset-fixes/reset/fixes (71827776667f4 reset: imx7: Correct polarity of MIPI CSI resets on i.MX8MQ) +$ git merge -m Merge branch 'reset/fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/pza/linux reset-fixes/reset/fixes +Already up to date. +Merging mips-fixes/mips-fixes (93f51579e7df2 Linux 7.3-rc4) +$ git merge -m Merge branch 'mips-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/mips/linux.git mips-fixes/mips-fixes +Already up to date. +Merging at91-fixes/at91-fixes (afb1ecfeda3fc ARM: configs: sama5: enable current Microchip KSZ DSA symbols) +$ git merge -m Merge branch 'at91-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/at91/linux.git at91-fixes/at91-fixes +Merge made by the 'ort' strategy. + arch/arm/configs/sama5_defconfig | 4 ++-- + 1 file changed, 2 insertions(+), 2 deletions(-) +Merging omap-fixes/fixes (2fabd2f406d0c ARM: dts: ti/omap: dra7: fix PCIe PHY clock divider definition) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/khilman/linux-omap.git omap-fixes/fixes +Merge made by the 'ort' strategy. + arch/arm/boot/dts/ti/omap/dra7xx-clocks.dtsi | 2 +- + 1 file changed, 1 insertion(+), 1 deletion(-) +Merging tegra-fixes/fixes (3f60097e76e4d Merge branch for-7.3/arm64/dt into fixes) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux.git tegra-fixes/fixes +Merge made by the 'ort' strategy. + .../boot/dts/nvidia/tegra264-p4071-0000+p3834.dtsi | 13 ++ + arch/arm64/boot/dts/nvidia/tegra264.dtsi | 180 ++++++++++++++++++--- + drivers/soc/tegra/fuse/tegra-apbmisc.c | 2 +- + drivers/soc/tegra/pmc.c | 1 + + 4 files changed, 171 insertions(+), 25 deletions(-) +Merging kvm-fixes/master (f9bfc323e7612 Merge tag 'kvmarm-fixes-7.3-2' of https://git.kernel.org/pub/scm/linux/kernel/git/kvmarm/kvmarm into HEAD) +$ git merge -m Merge branch 'master' of git://git.kernel.org/pub/scm/virt/kvm/kvm.git kvm-fixes/master +Already up to date. +Merging kvms390-fixes/master (f47190b08b71e s390/uv: Prevent potential out-of-bounds read) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/kvms390/linux.git kvms390-fixes/master +Already up to date. +Merging kvm-arm-fixes/fixes (71cc2c67fb8f8 KVM: arm64: Use stage-1 leaf size for VM_PFNMAP) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/kvmarm/kvmarm.git kvm-arm-fixes/fixes +Already up to date. +Merging hwmon-fixes/hwmon (61406e9cac695 hwmon: (sht4x) Fix jiffies wraparound in heater-ready check) +$ git merge -m Merge branch 'hwmon' of https://git.kernel.org/pub/scm/linux/kernel/git/groeck/linux-staging.git hwmon-fixes/hwmon +Merge made by the 'ort' strategy. + Documentation/hwmon/k10temp.rst | 8 +++----- + drivers/hwmon/coretemp.c | 5 +++++ + drivers/hwmon/k10temp.c | 5 ++--- + drivers/hwmon/lm70.c | 8 ++++---- + drivers/hwmon/lm95245.c | 6 +++--- + drivers/hwmon/sht4x.c | 18 +++++++++--------- + drivers/hwmon/tmp102.c | 8 ++++---- + drivers/hwmon/tmp108.c | 8 ++++---- + 8 files changed, 34 insertions(+), 32 deletions(-) +Merging nvdimm-fixes/libnvdimm-fixes (a8aec14230322 nvdimm/bus: Fix potential use after free in asynchronous initialization) +$ git merge -m Merge branch 'libnvdimm-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/nvdimm/nvdimm.git nvdimm-fixes/libnvdimm-fixes +Already up to date. +Merging cxl-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/cxl/cxl.git cxl-fixes/fixes +Already up to date. +Merging dma-mapping-fixes/dma-mapping-fixes (057e5e07420c7 iommu/dma: skip swiotlb bounce for DMA_ATTR_MMIO in iommu_dma_map_phys) +$ git merge -m Merge branch 'dma-mapping-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/mszyprowski/linux.git dma-mapping-fixes/dma-mapping-fixes +Merge made by the 'ort' strategy. + drivers/iommu/dma-iommu.c | 4 ++-- + 1 file changed, 2 insertions(+), 2 deletions(-) +Merging drivers-x86-fixes/fixes (d144a494d81fc MAINTAINERS: fix sysfs-platform-ayaneo-ec documentation path) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/pdx86/platform-drivers-x86.git drivers-x86-fixes/fixes +Already up to date. +Merging samsung-krzk-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux.git samsung-krzk-fixes/fixes +Already up to date. +Merging pinctrl-samsung-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/pinctrl/samsung.git pinctrl-samsung-fixes/fixes +Already up to date. +Merging pinctrl-qcom-fixes/pinctrl-qcom/for-current (19fc4240358be pinctrl: qcom: spmi-gpio: make direction changes exclusive) +$ git merge -m Merge branch 'pinctrl-qcom/for-current' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git pinctrl-qcom-fixes/pinctrl-qcom/for-current +Merge made by the 'ort' strategy. + drivers/pinctrl/qcom/pinctrl-hawi.c | 1 + + drivers/pinctrl/qcom/pinctrl-ipq5018.c | 4 ++-- + drivers/pinctrl/qcom/pinctrl-maili.c | 1 + + drivers/pinctrl/qcom/pinctrl-nord.c | 1 + + drivers/pinctrl/qcom/pinctrl-qcs8300.c | 1 + + drivers/pinctrl/qcom/pinctrl-spmi-gpio.c | 37 +++++++++++++++++++++++++------- + 6 files changed, 35 insertions(+), 10 deletions(-) +Merging devicetree-fixes/dt/linus (30724547b221e of/irq: Fix remaining refcount leaks in of_irq_init()) +$ git merge -m Merge branch 'dt/linus' of https://git.kernel.org/pub/scm/linux/kernel/git/robh/linux.git devicetree-fixes/dt/linus +Already up to date. +Merging dt-krzk-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-dt.git dt-krzk-fixes/fixes +Already up to date. +Merging scsi-fixes/fixes (42d1221d321e5 scsi: megaraid_sas: Protect megasas_get_ctrl_info() in megasas_resume()) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/mkp/scsi.git scsi-fixes/fixes +Already up to date. +Merging drm-fixes/drm-fixes (26fe67ec0dca0 Merge tag 'drm-intel-fixes-2026-10-01' of https://gitlab.freedesktop.org/drm/i915/kernel into drm-fixes) +$ git merge -m Merge branch 'drm-fixes' of https://gitlab.freedesktop.org/drm/kernel.git drm-fixes/drm-fixes +Merge made by the 'ort' strategy. + .../gpu/drm/i915/display/intel_display_params.c | 4 +++ + .../gpu/drm/i915/display/intel_display_params.h | 1 + + drivers/gpu/drm/i915/display/intel_vrr.c | 4 ++- + drivers/gpu/drm/xe/xe_pci_sriov.c | 30 +++++++++++++++++----- + 4 files changed, 31 insertions(+), 8 deletions(-) +Merging drm-intel-fixes/for-linux-next-fixes (c034e8a46e4cb drm/i915/vrr: Disable DC balance by default) +$ git merge -m Merge branch 'for-linux-next-fixes' of https://gitlab.freedesktop.org/drm/i915/kernel.git drm-intel-fixes/for-linux-next-fixes +Already up to date. +Merging mmc-fixes/fixes (c96ca94425b07 memstick: rtsx_usb_ms: complete requests after eject instead of dropping them) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/ulfh/mmc.git mmc-fixes/fixes +Merge made by the 'ort' strategy. + drivers/memstick/core/memstick.c | 8 ++------ + drivers/memstick/host/rtsx_usb_ms.c | 31 +++++++++++-------------------- + drivers/mmc/host/cavium-octeon.c | 5 ++++- + drivers/mmc/host/cavium-thunderx.c | 5 ++++- + drivers/mmc/host/mtk-sd.c | 1 + + drivers/mmc/host/sdhci-sprd.c | 4 ++++ + 6 files changed, 26 insertions(+), 28 deletions(-) +Merging rtc-fixes/rtc-fixes (055ef5ce9f67a rtc: spear: initialize IRQ state before requesting alarm IRQ) +$ git merge -m Merge branch 'rtc-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/abelloni/linux.git rtc-fixes/rtc-fixes +Already up to date. +Merging gnss-fixes/gnss-linus (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'gnss-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/johan/gnss.git gnss-fixes/gnss-linus +Already up to date. +Merging hyperv-fixes/hyperv-fixes (ca039df94983c PCI: hv: Warn when wait_for_response() waits indefinitely) +$ git merge -m Merge branch 'hyperv-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/hyperv/linux.git hyperv-fixes/hyperv-fixes +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 2 -- + arch/x86/hyperv/hv_init.c | 7 +++--- + drivers/hv/channel.c | 2 +- + drivers/hv/channel_mgmt.c | 12 ++++++--- + drivers/hv/connection.c | 1 - + drivers/hv/mshv_vtl_main.c | 3 +++ + drivers/hv/vmbus_drv.c | 19 ++++++-------- + drivers/pci/controller/pci-hyperv.c | 49 +++++++++++++++++++++++++++++-------- + tools/hv/hv_get_dns_info.sh | 2 +- + 9 files changed, 64 insertions(+), 33 deletions(-) +Merging risc-v-fixes/fixes (247ac82ac82d8 RISC-V: Clear HSTATUS.HU on CPU initialization) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/riscv/linux.git risc-v-fixes/fixes +Merge made by the 'ort' strategy. + arch/riscv/Kconfig.errata | 16 +++++++++++++++- + arch/riscv/Kconfig.socs | 1 + + arch/riscv/errata/sifive/errata.c | 20 ++++++++++++++++++++ + arch/riscv/include/asm/cpufeature.h | 2 ++ + arch/riscv/include/asm/errata_list_vendors.h | 3 ++- + arch/riscv/include/asm/insn-def.h | 16 +++++++++++++--- + arch/riscv/include/asm/vector.h | 8 ++++---- + arch/riscv/include/uapi/asm/vendor/mips.h | 4 +++- + arch/riscv/kernel/cpufeature.c | 12 ++++++++++++ + arch/riscv/kernel/setup.c | 2 ++ + arch/riscv/kernel/smpboot.c | 2 ++ + arch/riscv/kernel/suspend.c | 2 ++ + 12 files changed, 78 insertions(+), 10 deletions(-) +Merging riscv-dt-fixes/riscv-dt-fixes (0f70fd6f4c1d6 riscv: dts: microchip: beaglev-fire: remove double definition of gpio interrupts) +$ git merge -m Merge branch 'riscv-dt-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux.git riscv-dt-fixes/riscv-dt-fixes +Merge made by the 'ort' strategy. + arch/riscv/boot/dts/microchip/mpfs-beaglev-fire.dts | 18 ------------------ + 1 file changed, 18 deletions(-) +Merging riscv-soc-fixes/riscv-soc-fixes (dc59e4fea9d83 Linux 7.2-rc1) +$ git merge -m Merge branch 'riscv-soc-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux.git riscv-soc-fixes/riscv-soc-fixes +Already up to date. +Merging fpga-fixes/fixes (19272b37aa4f8 Linux 6.16-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/fpga/linux-fpga.git fpga-fixes/fixes +Already up to date. +Merging spdx/spdx-linus (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'spdx-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/spdx.git spdx/spdx-linus +Already up to date. +Merging gpio-brgl-fixes/gpio/for-current (ff82fc3a4a1d4 gpio: xilinx: fix runtime PM leak on request error path) +$ git merge -m Merge branch 'gpio/for-current' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git gpio-brgl-fixes/gpio/for-current +Merge made by the 'ort' strategy. + drivers/gpio/gpio-xilinx.c | 9 +-------- + 1 file changed, 1 insertion(+), 8 deletions(-) +Merging gpio-intel-fixes/fixes (8d37f20173a27 gpiolib: acpi: Add quirk for Lenovo IdeaPad Slim 3 15ABR8) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/andy/linux-gpio-intel.git gpio-intel-fixes/fixes +Merge made by the 'ort' strategy. + drivers/gpio/gpio-crystalcove.c | 1 + + drivers/gpio/gpiolib-acpi-quirks.c | 13 +++++++++++++ + 2 files changed, 14 insertions(+) +Merging pinctrl-intel-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/pinctrl/intel.git pinctrl-intel-fixes/fixes +Already up to date. +Merging auxdisplay-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/andy/linux-auxdisplay.git auxdisplay-fixes/fixes +Already up to date. +Merging kunit-fixes/kunit-fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'kunit-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git kunit-fixes/kunit-fixes +Already up to date. +Merging renesas-fixes/fixes (8dc2615d57020 arm64: dts: renesas: r8a779f0: Set UFS lane count) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel.git renesas-fixes/fixes +Already up to date. +Merging perf-current/perf-tools (aadea57f53288 perf powerpc-vpadtl: Fix raw_size of DTL samples) +$ git merge -m Merge branch 'perf-tools' of https://git.kernel.org/pub/scm/linux/kernel/git/perf/perf-tools.git perf-current/perf-tools +Already up to date. +Merging efi-fixes/urgent (d8809f6931065 efi: sysfb_efi: Extend quirk to cover IdeaPad Duet 3 10IGL5-LTE) +$ git merge -m Merge branch 'urgent' of https://git.kernel.org/pub/scm/linux/kernel/git/efi/efi.git efi-fixes/urgent +Already up to date. +Merging battery-fixes/fixes (a58cbc8b36ec0 power: supply: ds2780: Fix ds2780_get_capacity() returning raw value) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/sre/linux-power-supply.git battery-fixes/fixes +Merge made by the 'ort' strategy. + drivers/power/supply/ds2780_battery.c | 2 +- + drivers/power/supply/max17042_battery.c | 2 +- + include/linux/power_supply.h | 6 +++--- + 3 files changed, 5 insertions(+), 5 deletions(-) +Merging iommufd-fixes/for-rc (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-rc' of https://git.kernel.org/pub/scm/linux/kernel/git/jgg/iommufd.git iommufd-fixes/for-rc +Already up to date. +Merging rust-fixes/rust-fixes (ce1e0223d8ad4 Merge tag 'devicetree-fixes-for-7.3-2' of git://git.kernel.org/pub/scm/linux/kernel/git/robh/linux) +$ git merge -m Merge branch 'rust-fixes' of https://github.com/Rust-for-Linux/linux.git rust-fixes/rust-fixes +Already up to date. +Merging w1-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-w1.git w1-fixes/fixes +Already up to date. +Merging pmdomain-fixes/fixes (4c66c423593e2 pmdomain: rockchip: don't ignore clock lookup errors on attach) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/ulfh/linux-pm.git pmdomain-fixes/fixes +Merge made by the 'ort' strategy. + drivers/pmdomain/imx/imx8m-blk-ctrl.c | 15 +++++++++++++++ + drivers/pmdomain/rockchip/pm-domains.c | 20 +++++++++++++++++--- + 2 files changed, 32 insertions(+), 3 deletions(-) +Merging i2c-andi-fixes/i2c/i2c-fixes (c98bf3b86609a i2c: at91: release DMA channels when probe defers) +$ git merge -m Merge branch 'i2c/i2c-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/andi.shyti/linux.git i2c-andi-fixes/i2c/i2c-fixes +Merge made by the 'ort' strategy. + drivers/i2c/busses/i2c-at91-master.c | 8 ++++- + drivers/i2c/busses/i2c-xiic.c | 61 +++++++++++++++++++++++++++--------- + 2 files changed, 53 insertions(+), 16 deletions(-) +Merging i2c-rust-fixes/rust-i2c-fixes (4eb422482ca5d rust: i2c: fix I2cAdapter refcounts double increment) +$ git merge -m Merge branch 'rust-i2c-fixes' of https://github.com/ikrtn/rust-for-linux i2c-rust-fixes/rust-i2c-fixes +Already up to date. +Merging sparc-fixes/for-linus (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/alarsson/linux-sparc.git sparc-fixes/for-linus +Already up to date. +Merging clk-fixes/clk-fixes (81493c1dd1b1e Merge tag 'tags/spacemit-clk-fixes-for-7.3-1' into clk-fixes) +$ git merge -m Merge branch 'clk-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/clk/linux.git clk-fixes/clk-fixes +Merge made by the 'ort' strategy. + drivers/clk/spacemit/ccu-k3.c | 92 +++++++++++++++++++++++++++++++++++++++++++ + drivers/clk/ti/composite.c | 26 ++++++------ + 2 files changed, 104 insertions(+), 14 deletions(-) +Merging thead-clk-fixes/thead-clk-fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'thead-clk-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git thead-clk-fixes/thead-clk-fixes +Already up to date. +Merging tenstorrent-clk-fixes/tenstorrent-clk-fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'tenstorrent-clk-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/tenstorrent/linux.git tenstorrent-clk-fixes/tenstorrent-clk-fixes +Already up to date. +Merging fustini-config-fixes/riscv-config-fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'riscv-config-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git fustini-config-fixes/riscv-config-fixes +Already up to date. +Merging pwrseq-fixes/pwrseq/for-current (58a0033024815 power: sequencing: pcie-m2: Add Lenovo ThinkPad T14s gen6 WCN7850 subsystem PCI ids) +$ git merge -m Merge branch 'pwrseq/for-current' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git pwrseq-fixes/pwrseq/for-current +Merge made by the 'ort' strategy. + drivers/power/sequencing/pwrseq-pcie-m2.c | 2 ++ + 1 file changed, 2 insertions(+) +Merging thead-dt-fixes/thead-dt-fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'thead-dt-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git thead-dt-fixes/thead-dt-fixes +Already up to date. +Merging ftrace-fixes/ftrace/fixes (1650a1b6cb1ae fgraph: Check ftrace_pids_enabled on registration for early filtering) +$ git merge -m Merge branch 'ftrace/fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git ftrace-fixes/ftrace/fixes +Already up to date. +Merging ring-buffer-fixes/ring-buffer/fixes (057caace5214d tracing: Create output file from cmd_check_undefined) +$ git merge -m Merge branch 'ring-buffer/fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git ring-buffer-fixes/ring-buffer/fixes +Already up to date. +Merging trace-fixes/trace/fixes (d860c67c05168 ring-buffer: Check resize_disabled before publishing the new subbuf order) +$ git merge -m Merge branch 'trace/fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git trace-fixes/trace/fixes +Already up to date. +Merging tracefs-fixes/tracefs/fixes (07004a8c4b572 eventfs: Hold eventfs_mutex and SRCU when remount walks events) +$ git merge -m Merge branch 'tracefs/fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git tracefs-fixes/tracefs/fixes +Already up to date. +Merging spacemit-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/spacemit/linux spacemit-fixes/fixes +Already up to date. +Merging tip-fixes/tip/urgent (254979e0c6662 Merge branch into tip/master: 'x86/mm') +$ git merge -m Merge branch 'tip/urgent' of https://git.kernel.org/pub/scm/linux/kernel/git/tip/tip.git tip-fixes/tip/urgent +Merge made by the 'ort' strategy. + arch/x86/include/asm/pgtable.h | 6 +++++- + arch/x86/include/uapi/asm/ptrace-abi.h | 4 ++-- + arch/x86/kernel/sys_x86_64.c | 15 ++++++++++---- + arch/x86/mm/init_64.c | 3 ++- + arch/x86/mm/pat/set_memory.c | 23 ++++++++++++++++---- + arch/x86/mm/pgtable.c | 38 +++++++++++++--------------------- + arch/x86/um/asm/ptrace.h | 4 +--- + arch/x86/um/ptrace_32.c | 1 + + include/linux/hrtimer_rearm.h | 2 +- + 9 files changed, 56 insertions(+), 40 deletions(-) +Merging kexec-fixes/kexec-fixes (a901b0778ae82 Merge patch series "kexec: fix probe error codes and error propagation") +$ git merge -m Merge branch 'kexec-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git kexec-fixes/kexec-fixes +Merge made by the 'ort' strategy. + arch/arm64/kernel/kexec_image.c | 4 ++-- + arch/loongarch/kernel/kexec_efi.c | 4 ++-- + arch/riscv/kernel/kexec_image.c | 4 ++-- + kernel/kexec_elf.c | 4 ++-- + kernel/kexec_file.c | 12 +++++++----- + 5 files changed, 15 insertions(+), 13 deletions(-) +Merging liveupdate-fixes/fixes (3a0b8fa2eb36a kho: fix size calculation in kho_preserved_memory_reserve()) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git liveupdate-fixes/fixes +Already up to date. +Merging drm-msm-fixes/msm-fixes (a15fac810c763 dt-bindings: display/msm: Use consistent indentation in the example) +$ git merge -m Merge branch 'msm-fixes' of https://gitlab.freedesktop.org/drm/msm.git drm-msm-fixes/msm-fixes +Already up to date. +Merging uml-fixes/fixes (af421e9aed392 um: vector: fix use-after-free in vector_mmsg_rx()) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/uml/linux.git uml-fixes/fixes +Already up to date. +Merging fwctl-fixes/for-rc (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-rc' of https://git.kernel.org/pub/scm/linux/kernel/git/fwctl/fwctl.git fwctl-fixes/for-rc +Already up to date. +Merging devsec-tsm-fixes/fixes (c3fd16c3b98ed virt: tdx-guest: Fix handling of host controlled 'quote' buffer length) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/devsec/tsm.git devsec-tsm-fixes/fixes +Already up to date. +Merging drm-rust-fixes/for-linux-next-fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-linux-next-fixes' of https://gitlab.freedesktop.org/drm/rust/kernel.git drm-rust-fixes/for-linux-next-fixes +Already up to date. +Merging tenstorrent-dt-fixes/tenstorrent-dt-fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'tenstorrent-dt-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/tenstorrent/linux.git tenstorrent-dt-fixes/tenstorrent-dt-fixes +Already up to date. +Merging nfc-fixes/for-linus (b61732f47316d nfc: pn533: fix OOB read in pn533_acr122_is_rx_frame_valid()) +$ git merge -m Merge branch 'for-linus' of https://codeberg.org/linux-nfc/linux.git nfc-fixes/for-linus +Already up to date. +Merging mm-nonmm-hotfixes-stable/mm-nonmm-hotfixes-stable (a243ede718463 Merge tag 'mtd/fixes-for-7.3-rc6' of git://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux) +$ git merge -m Merge branch 'mm-nonmm-hotfixes-stable' of https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm mm-nonmm-hotfixes-stable/mm-nonmm-hotfixes-stable +Already up to date. +Merging mm-nonmm-hotfixes-unstable/mm-nonmm-hotfixes-unstable (a3e3527ac09d0 fault-inject: fix dentry leak) +$ git merge -m Merge branch 'mm-nonmm-hotfixes-unstable' of https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm mm-nonmm-hotfixes-unstable/mm-nonmm-hotfixes-unstable +Merge made by the 'ort' strategy. + drivers/infiniband/hw/hfi1/fault.c | 1 - + fs/fat/fat.h | 2 +- + fs/fat/fatent.c | 21 ++++++++++++++++++--- + fs/fat/file.c | 3 ++- + fs/fat/misc.c | 6 ++---- + include/linux/fault-inject.h | 10 ++++++++-- + kernel/resource.c | 2 +- + lib/fault-inject.c | 7 +++++-- + 8 files changed, 37 insertions(+), 15 deletions(-) +Merging drm-misc-fixes/for-linux-next-fixes (bca45af5998a0 drm/vc4: Use kvmalloc_objs() for the BO cache size list) +$ git merge -m Merge branch 'for-linux-next-fixes' of https://gitlab.freedesktop.org/drm/misc/kernel.git drm-misc-fixes/for-linux-next-fixes +Merge made by the 'ort' strategy. + drivers/gpu/drm/aspeed/aspeed_gfx_drv.c | 5 ++--- + drivers/gpu/drm/bridge/th1520-dw-hdmi.c | 8 +++---- + drivers/gpu/drm/display/drm_dp_helper.c | 4 +++- + drivers/gpu/drm/drm_bridge.c | 8 +++++-- + drivers/gpu/drm/drm_fb_helper.c | 6 +++--- + drivers/gpu/drm/nouveau/nvkm/engine/device/user.c | 2 +- + drivers/gpu/drm/panthor/panthor_sched.c | 3 +++ + drivers/gpu/drm/vc4/vc4_bo.c | 5 +++-- + drivers/gpu/drm/vc4/vc4_drv.c | 5 ++++- + drivers/gpu/drm/vc4/vc4_v3d.c | 8 +++++++ + drivers/gpu/drm/vmwgfx/vmwgfx_cursor_plane.c | 26 ++++++++++++++++++----- + drivers/gpu/drm/vmwgfx/vmwgfx_kms.c | 8 +++++++ + 12 files changed, 66 insertions(+), 22 deletions(-) +Merging rust/rust-next (948440b3f429f rust: mem: add `Align` type) +$ git merge -m Merge branch 'rust-next' of https://github.com/Rust-for-Linux/linux.git rust/rust-next +Merge made by the 'ort' strategy. + rust/bindgen_parameters | 2 +- + rust/bindings/bindings_helper.h | 2 +- + rust/kernel/Kconfig.test | 11 +++++++ + rust/kernel/bitfield.rs | 1 - + rust/kernel/bug.rs | 33 ++++++++++++++++++- + rust/kernel/lib.rs | 1 + + rust/kernel/mem.rs | 70 +++++++++++++++++++++++++++++++++++++++++ + rust/macros/lib.rs | 2 +- + rust/macros/module.rs | 2 +- + scripts/rustdoc_test_builder.rs | 7 +++++ + 10 files changed, 125 insertions(+), 6 deletions(-) + create mode 100644 rust/kernel/mem.rs +Merging rust-interop/interop-next (05f7e89ab9731 Linux 6.19) +$ git merge -m Merge branch 'interop-next' of https://github.com/Rust-for-Linux/linux.git rust-interop/interop-next +Already up to date. +Merging rust-alloc/alloc-next (6e8339118040f rust: alloc: add `NumaNode::id()` accessor) +$ git merge -m Merge branch 'alloc-next' of https://github.com/Rust-for-Linux/linux.git rust-alloc/alloc-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 1 + + rust/kernel/alloc.rs | 6 ++++++ + rust/kernel/alloc/kbox.rs | 7 ++++--- + 3 files changed, 11 insertions(+), 3 deletions(-) +Merging rust-io/io-next (86731a2a651e5 Linux 6.16-rc3) +$ git merge -m Merge branch 'io-next' of https://github.com/Rust-for-Linux/linux.git rust-io/io-next +Already up to date. +Merging rust-pin-init/pin-init-next (499b8f8a5d040 rust: proc-macro2: enable `proc_macro_span` feature) +$ git merge -m Merge branch 'pin-init-next' of https://github.com/Rust-for-Linux/linux.git rust-pin-init/pin-init-next +Merge made by the 'ort' strategy. + rust/Makefile | 1 + + rust/pin-init/internal/src/diagnostics.rs | 169 ++++++-- + rust/pin-init/internal/src/init.rs | 636 ++++++++++++++++++++++++------ + rust/pin-init/internal/src/lib.rs | 40 +- + rust/pin-init/internal/src/pin_data.rs | 164 ++++---- + rust/pin-init/internal/src/util.rs | 54 +++ + rust/pin-init/src/__internal.rs | 64 +-- + rust/pin-init/src/lib.rs | 82 +++- + rust/proc-macro2/README.md | 3 +- + rust/proc-macro2/probe/proc_macro_span.rs | 9 - + 10 files changed, 916 insertions(+), 306 deletions(-) + create mode 100644 rust/pin-init/internal/src/util.rs +Merging rust-timekeeping/timekeeping-next (2ea0119f72dba rust: hrtimer: document handle based design rationale) +$ git merge -m Merge branch 'timekeeping-next' of https://github.com/Rust-for-Linux/linux.git rust-timekeeping/timekeeping-next +Auto-merging rust/bindings/lib.rs +Auto-merging rust/kernel/Kconfig.test +CONFLICT (content): Merge conflict in rust/kernel/Kconfig.test +Resolved 'rust/kernel/Kconfig.test' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 4daa031172ed6] Merge branch 'timekeeping-next' of https://github.com/Rust-for-Linux/linux.git +$ git diff -M --stat --summary HEAD^.. + rust/bindings/lib.rs | 5 ++ + rust/helpers/helpers.c | 1 + + rust/helpers/math.c | 8 +++ + rust/kernel/Kconfig.test | 6 ++ + rust/kernel/time.rs | 139 ++++++++++++++++++++++++++++++++++++++------ + rust/kernel/time/delay.rs | 17 ++++-- + rust/kernel/time/hrtimer.rs | 93 ++++++++++++++++------------- + 7 files changed, 206 insertions(+), 63 deletions(-) + create mode 100644 rust/helpers/math.c +Merging rust-xarray/xarray-next (c455f19bbe610 rust: xarray: add __rust_helper to helpers) +$ git merge -m Merge branch 'xarray-next' of https://github.com/Rust-for-Linux/linux.git rust-xarray/xarray-next +Already up to date. +Merging rust-analyzer/rust-analyzer-next (5f45afb8ab04d scripts: generate_rust_analyzer.py: pass cfg to macros crate) +$ git merge -m Merge branch 'rust-analyzer-next' of https://github.com/Rust-for-Linux/linux.git rust-analyzer/rust-analyzer-next +Auto-merging scripts/generate_rust_analyzer.py +Merge made by the 'ort' strategy. + scripts/generate_rust_analyzer.py | 1 + + 1 file changed, 1 insertion(+) +Merging mm/for-next (3bea2ede9c01b Merge https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm.git mm-unstable into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mm/linux.git mm/for-next +Auto-merging MAINTAINERS +Auto-merging arch/arm64/Kconfig +Auto-merging arch/arm64/mm/mmu.c +Auto-merging arch/x86/include/asm/pgtable.h +Auto-merging arch/x86/mm/pat/set_memory.c +Auto-merging drivers/usb/cdns3/cdns3-gadget.c +Merge made by the 'ort' strategy. + .clang-format | 2 +- + Documentation/ABI/testing/sysfs-kernel-mm-damon | 35 +- + Documentation/admin-guide/cgroup-v2.rst | 4 + + Documentation/admin-guide/kdump/vmcoreinfo.rst | 2 +- + Documentation/admin-guide/kernel-parameters.txt | 3 + + Documentation/admin-guide/mm/damon/usage.rst | 41 +- + Documentation/admin-guide/mm/ksm.rst | 8 +- + Documentation/arch/powerpc/vmemmap_dedup.rst | 88 +- + Documentation/core-api/memory-allocation.rst | 32 +- + Documentation/core-api/mm-api.rst | 9 + + Documentation/core-api/pin_user_pages.rst | 9 - + Documentation/dev-tools/kmemleak.rst | 24 +- + Documentation/filesystems/mmap_prepare.rst | 81 ++ + Documentation/filesystems/vfs.rst | 6 +- + Documentation/mm/damon/design.rst | 64 +- + Documentation/mm/damon/maintainer-profile.rst | 19 +- + Documentation/mm/index.rst | 1 + + Documentation/mm/kernel-page-tables.rst | 410 ++++++++++ + Documentation/mm/ksm.rst | 2 +- + Documentation/mm/page_owner.rst | 8 +- + Documentation/mm/physical_memory.rst | 6 + + Documentation/mm/process_addrs.rst | 11 + + Documentation/mm/vmemmap_dedup.rst | 32 +- + MAINTAINERS | 12 +- + arch/Kconfig | 8 - + arch/alpha/Kconfig | 1 - + arch/alpha/include/asm/pgtable.h | 7 - + arch/arc/include/asm/pgalloc.h | 6 +- + arch/arc/include/asm/pgtable-levels.h | 11 - + arch/arm/Kconfig | 2 - + arch/arm/Kconfig.debug | 2 +- + arch/arm/configs/aspeed_g4_defconfig | 2 +- + arch/arm/configs/aspeed_g5_defconfig | 2 +- + arch/arm/configs/shmobile_defconfig | 2 +- + arch/arm/include/asm/pgtable.h | 7 - + arch/arm/include/asm/ptdump.h | 6 +- + arch/arm/kernel/traps.c | 17 - + arch/arm/mm/init.c | 2 +- + arch/arm64/Kconfig | 4 +- + arch/arm64/include/asm/mte.h | 2 - + arch/arm64/include/asm/pgtable.h | 64 +- + arch/arm64/include/asm/set_memory.h | 5 +- + arch/arm64/kvm/mmu.c | 2 +- + arch/arm64/mm/fault.c | 16 +- + arch/arm64/mm/mmu.c | 48 ++ + arch/arm64/mm/mteswap.c | 32 +- + arch/arm64/mm/pageattr.c | 24 +- + arch/csky/include/asm/pgtable.h | 4 - + arch/hexagon/include/asm/pgtable.h | 3 - + arch/loongarch/Kconfig | 2 - + arch/loongarch/include/asm/pgtable.h | 14 +- + arch/loongarch/include/asm/set_memory.h | 5 +- + arch/loongarch/mm/init.c | 4 +- + arch/loongarch/mm/pageattr.c | 27 +- + arch/m68k/Kconfig | 1 + + arch/m68k/include/asm/mcf_pgalloc.h | 5 +- + arch/m68k/include/asm/mcf_pgtable.h | 6 - + arch/m68k/include/asm/motorola_pgalloc.h | 9 +- + arch/m68k/include/asm/motorola_pgtable.h | 8 - + arch/m68k/include/asm/sun3_pgtable.h | 7 - + arch/m68k/mm/motorola.c | 119 ++- + arch/microblaze/include/asm/pgalloc.h | 2 +- + arch/microblaze/include/asm/pgtable.h | 7 - + arch/mips/Kconfig | 1 - + arch/mips/include/asm/pgtable-32.h | 10 - + arch/mips/include/asm/pgtable-64.h | 13 - + arch/nios2/include/asm/pgtable.h | 7 - + arch/openrisc/include/asm/pgtable.h | 7 - + arch/parisc/Kconfig | 1 - + arch/parisc/include/asm/pgtable.h | 9 - + arch/parisc/kernel/pci-dma.c | 6 +- + arch/powerpc/Kconfig | 3 +- + arch/powerpc/configs/ppc64_defconfig | 2 +- + arch/powerpc/include/asm/book3s/32/pgtable.h | 2 - + arch/powerpc/include/asm/book3s/64/pgtable.h | 7 - + arch/powerpc/include/asm/nohash/32/pgtable.h | 2 - + arch/powerpc/include/asm/nohash/64/pgtable-4k.h | 3 - + arch/powerpc/include/asm/nohash/64/pgtable.h | 5 - + arch/powerpc/mm/book3s64/radix_pgtable.c | 135 +--- + arch/powerpc/mm/book3s64/radix_tlb.c | 6 +- + arch/powerpc/mm/nohash/e500_hugetlbpage.c | 2 +- + arch/powerpc/mm/nohash/tlb.c | 2 +- + arch/powerpc/mm/ptdump/ptdump.c | 2 +- + arch/powerpc/platforms/powernv/Kconfig | 1 - + arch/powerpc/platforms/pseries/Kconfig | 1 - + arch/riscv/Kconfig | 4 +- + arch/riscv/include/asm/page.h | 6 - + arch/riscv/include/asm/pgtable-64.h | 9 - + arch/riscv/include/asm/pgtable.h | 4 - + arch/riscv/include/asm/set_memory.h | 5 +- + arch/riscv/kvm/mmu.c | 2 +- + arch/riscv/mm/init.c | 1 + + arch/riscv/mm/pageattr.c | 23 +- + arch/riscv/mm/tlbflush.c | 2 +- + arch/s390/Kconfig | 4 +- + arch/s390/configs/debug_defconfig | 2 +- + arch/s390/configs/defconfig | 2 +- + arch/s390/include/asm/pgtable.h | 11 - + arch/s390/include/asm/set_memory.h | 5 +- + arch/s390/mm/dump_pagetables.c | 2 +- + arch/s390/mm/gmap_helpers.c | 6 +- + arch/s390/mm/pageattr.c | 20 +- + arch/sh/Kconfig | 1 + + arch/sh/include/asm/pgalloc.h | 6 +- + arch/sh/include/asm/pgtable-3level.h | 3 - + arch/sh/include/asm/pgtable_32.h | 13 - + arch/sh/mm/init.c | 16 +- + arch/sh/mm/pgtable.c | 20 + + arch/sparc/Kconfig | 4 +- + arch/sparc/include/asm/pgalloc_32.h | 7 +- + arch/sparc/include/asm/pgalloc_64.h | 8 - + arch/sparc/include/asm/pgtable_32.h | 3 - + arch/sparc/include/asm/pgtable_64.h | 10 - + arch/sparc/include/asm/tlb_64.h | 2 - + arch/sparc/lib/bitext.c | 14 +- + arch/sparc/mm/init_64.c | 2 +- + arch/sparc/mm/srmmu.c | 32 +- + arch/um/Kconfig | 1 - + arch/um/include/asm/pgtable-2level.h | 7 - + arch/um/include/asm/pgtable-4level.h | 13 - + arch/x86/Kconfig | 6 +- + arch/x86/configs/x86_64_defconfig | 2 +- + arch/x86/entry/vdso/vdso32/fake_32bit_build.h | 2 +- + arch/x86/events/intel/bts.c | 3 - + arch/x86/events/intel/pt.c | 6 +- + arch/x86/include/asm/pgtable-2level.h | 5 - + arch/x86/include/asm/pgtable-3level.h | 11 - + arch/x86/include/asm/pgtable.h | 6 +- + arch/x86/include/asm/pgtable_64.h | 18 - + arch/x86/include/asm/set_memory.h | 5 +- + arch/x86/include/asm/string_64.h | 85 +- + arch/x86/kernel/uprobes.c | 2 +- + arch/x86/mm/pat/set_memory.c | 20 +- + arch/x86/mm/pti.c | 2 +- + arch/xtensa/include/asm/pgtable.h | 4 - + arch/xtensa/include/asm/tlb.h | 2 +- + drivers/android/binder/page_range.rs | 19 +- + drivers/android/binder_alloc.c | 63 +- + drivers/base/memory.c | 9 +- + drivers/block/zram/zcomp.c | 23 +- + drivers/block/zram/zcomp.h | 4 +- + drivers/block/zram/zram_drv.c | 192 ++--- + drivers/char/Makefile | 2 +- + drivers/gpu/drm/drm_gpusvm.c | 5 +- + drivers/gpu/drm/sti/sti_cursor.c | 4 +- + drivers/gpu/drm/sti/sti_hqvdp.c | 2 +- + drivers/gpu/ipu-v3/ipu-image-convert.c | 2 +- + drivers/hid/hid-core.c | 4 +- + drivers/hsi/clients/cmt_speech.c | 35 +- + drivers/infiniband/hw/hfi1/file_ops.c | 84 +- + drivers/md/md-bitmap.c | 18 +- + drivers/media/platform/nxp/imx7-media-csi.c | 2 +- + .../media/platform/nxp/imx8-isi/imx8-isi-video.c | 2 +- + drivers/mtd/nand/raw/gpmi-nand/gpmi-nand.c | 2 +- + drivers/nvdimm/pmem.c | 2 +- + drivers/nvdimm/pmem.h | 12 - + drivers/scsi/sg.c | 115 ++- + drivers/spi/spi-atmel.c | 4 +- + drivers/spi/spi-ti-qspi.c | 2 +- + drivers/staging/media/imx/imx-media-utils.c | 2 +- + drivers/usb/cdns3/cdns3-gadget.c | 2 +- + drivers/usb/gadget/udc/cdns2/cdns2-gadget.c | 2 +- + drivers/usb/gadget/udc/lpc32xx_udc.c | 2 +- + drivers/usb/mon/mon_bin.c | 98 ++- + drivers/video/fbdev/core/fb_defio.c | 6 +- + drivers/video/fbdev/fsl-diu-fb.c | 2 +- + drivers/video/fbdev/ssd1307fb.c | 2 + + drivers/xen/grant-table.c | 11 +- + fs/Kconfig | 2 +- + fs/buffer.c | 8 - + fs/ceph/addr.c | 8 +- + fs/coredump.c | 6 +- + fs/crypto/crypto.c | 2 - + fs/erofs/data.c | 16 +- + fs/erofs/zdata.c | 13 +- + fs/f2fs/compress.c | 35 +- + fs/f2fs/data.c | 2 +- + fs/f2fs/f2fs.h | 99 +-- + fs/f2fs/segment.c | 2 +- + fs/fuse/dax.c | 2 +- + fs/hugetlbfs/inode.c | 10 +- + fs/iomap/buffered-io.c | 3 +- + fs/nfs/file.c | 4 +- + fs/nfs/write.c | 2 - + fs/proc/base.c | 3 +- + fs/proc/internal.h | 2 - + fs/proc/page.c | 8 +- + fs/proc/task_mmu.c | 448 ++++------- + fs/proc/vmcore.c | 27 +- + fs/ubifs/file.c | 8 +- + fs/xfs/libxfs/xfs_btree.c | 18 +- + fs/xfs/xfs_platform.h | 4 - + include/asm-generic/pgtable-nop4d.h | 1 - + include/asm-generic/pgtable-nopmd.h | 1 - + include/asm-generic/pgtable-nopud.h | 1 - + include/asm-generic/tlb.h | 70 +- + include/linux/buffer_head.h | 8 +- + include/linux/cache.h | 1 + + include/linux/cgroup.h | 3 + + include/linux/compaction.h | 11 +- + include/linux/compiler.h | 5 + + include/linux/damon.h | 102 ++- + include/linux/gfp_types.h | 2 +- + include/linux/huge_mm.h | 9 - + include/linux/hugetlb.h | 50 +- + include/linux/hugetlb_inline.h | 28 - + include/linux/idr.h | 5 +- + include/linux/kernel-page-flags.h | 1 - + include/linux/maple_tree.h | 4 +- + include/linux/memblock.h | 1 - + include/linux/memcontrol.h | 464 ++++++----- + include/linux/mm.h | 378 +++++++-- + include/linux/mm_inline.h | 115 ++- + include/linux/mm_types.h | 66 +- + include/linux/mmap_lock.h | 106 ++- + include/linux/mmzone.h | 224 +++--- + include/linux/page-flags.h | 64 +- + include/linux/page_counter.h | 3 +- + include/linux/pagemap.h | 68 +- + include/linux/pgtable.h | 24 +- + include/linux/ptdump.h | 4 +- + include/linux/rmap.h | 2 +- + include/linux/sched.h | 4 +- + include/linux/set_memory.h | 120 ++- + include/linux/shmem_fs.h | 12 +- + include/linux/string.h | 13 + + include/linux/swap.h | 59 +- + include/linux/swapops.h | 9 - + include/linux/userfaultfd_k.h | 1 - + include/linux/vmalloc.h | 4 + + include/linux/vmemmap-optimization.h | 115 +++ + include/linux/zsmalloc.h | 4 - + include/linux/zswap.h | 8 +- + include/net/mana/mana.h | 4 +- + include/trace/events/huge_memory.h | 18 +- + include/trace/events/mmflags.h | 45 +- + include/trace/events/pagemap.h | 2 +- + include/trace/events/vmscan.h | 14 - + init/main.c | 2 +- + kernel/bpf/arena.c | 3 +- + kernel/bpf/stackmap.c | 17 +- + kernel/bpf/task_iter.c | 2 +- + kernel/cgroup/cgroup-internal.h | 1 - + kernel/configs/debug.config | 2 +- + kernel/events/core.c | 2 +- + kernel/events/ring_buffer.c | 7 +- + kernel/events/uprobes.c | 4 +- + kernel/fork.c | 2 - + kernel/power/snapshot.c | 4 +- + kernel/sched/fair.c | 3 +- + kernel/vmcore_info.c | 1 - + lib/interval_tree_test.c | 2 +- + lib/region_alloc_benchmark.c | 8 +- + lib/test_hmm.c | 2 + + mm/Kconfig | 82 +- + mm/Kconfig.debug | 40 - + mm/Makefile | 3 +- + mm/alloc_tag.c | 2 +- + drivers/char/mem.c => mm/char-mem.c | 21 +- + mm/cma.h | 23 +- + mm/collapse.h | 162 ++++ + mm/compaction.c | 11 +- + mm/damon/core.c | 423 +++++++--- + mm/damon/lru_sort.c | 17 +- + mm/damon/ops-common.c | 195 ++++- + mm/damon/ops-common.h | 12 + + mm/damon/paddr.c | 83 +- + mm/damon/reclaim.c | 13 +- + mm/damon/sysfs-schemes.c | 81 +- + mm/damon/sysfs.c | 364 ++++++++- + mm/damon/tests/core-kunit.h | 886 ++++++++++++++++++++- + mm/damon/vaddr.c | 299 +++++-- + mm/debug.c | 4 - + mm/execmem.c | 148 ++-- + mm/filemap.c | 34 +- + mm/folio.c | 21 +- + mm/gup.c | 222 +++--- + mm/gup_test.c | 11 +- + mm/hmm.c | 3 +- + mm/huge_memory.c | 757 +++++++++--------- + mm/hugetlb.c | 303 ++++--- + mm/hugetlb_cgroup.c | 33 + + mm/hugetlb_sysfs.c | 10 +- + mm/hugetlb_vmemmap.c | 130 +-- + mm/hugetlb_vmemmap.h | 15 +- + mm/init-mm.c | 2 - + mm/internal.h | 140 ++-- + mm/interval_tree.c | 101 +++ + mm/kfence/core.c | 2 +- + mm/khugepaged.c | 690 ++++++++-------- + mm/kmemleak.c | 72 +- + mm/ksm.c | 27 +- + mm/list_lru.c | 13 +- + mm/madvise.c | 231 +++++- + mm/memblock.c | 27 +- + mm/memcontrol-v1.c | 413 +--------- + mm/memcontrol-v1.h | 31 +- + mm/memcontrol.c | 648 ++++++++++----- + mm/memfd.c | 31 +- + mm/memory-failure.c | 31 +- + mm/memory-tiers.c | 7 +- + mm/memory.c | 342 ++++---- + mm/memory_hotplug.c | 155 ++-- + mm/mempolicy.c | 191 +++-- + mm/migrate.c | 21 +- + mm/migrate_device.c | 18 +- + mm/mincore.c | 31 +- + mm/mlock.c | 61 +- + mm/mm_init.c | 248 +++--- + mm/mm_init.h | 7 +- + mm/mmap.c | 8 +- + mm/mmap_lock.c | 61 +- + mm/mmu_gather.c | 32 +- + mm/mprotect.c | 11 +- + mm/mremap.c | 21 +- + mm/nommu.c | 6 +- + mm/oom_kill.c | 15 +- + mm/page-writeback.c | 2 +- + mm/page_alloc.c | 121 ++- + mm/page_alloc.h | 2 +- + mm/page_counter.c | 41 +- + mm/page_io.c | 59 +- + mm/page_isolation.c | 40 +- + mm/page_owner.c | 6 +- + mm/page_table_check.c | 6 +- + mm/page_vma_mapped.c | 11 +- + mm/pagewalk.c | 4 +- + mm/percpu.c | 9 +- + mm/pgtable-generic.c | 37 +- + mm/rmap.c | 37 +- + mm/secretmem.c | 8 +- + mm/shmem.c | 418 +++++++--- + mm/slab.h | 28 +- + mm/slab_common.c | 2 +- + mm/slub.c | 330 +++++--- + mm/sparse-vmemmap.c | 523 +++++------- + mm/sparse.c | 236 ++---- + mm/sparse.h | 32 +- + mm/swap.h | 17 +- + mm/swap_state.c | 82 +- + mm/swapfile.c | 151 ++-- + mm/truncate.c | 131 +-- + mm/userfaultfd.c | 120 +-- + mm/util.c | 31 +- + mm/vma.c | 466 +++++++---- + mm/vma.h | 65 +- + mm/vma_internal.h | 1 - + mm/vmalloc.c | 231 +++--- + mm/vmalloc.h | 2 +- + mm/vmpressure.c | 3 - + mm/vmscan.c | 629 +++++++++------ + mm/vmstat.c | 17 +- + mm/workingset.c | 7 +- + mm/zpdesc.h | 2 +- + mm/zsmalloc.c | 101 +-- + mm/zswap.c | 363 +++++---- + net/ipv4/tcp.c | 31 +- + rust/kernel/mm.rs | 58 +- + samples/damon/mtier.c | 10 +- + samples/damon/prcl.c | 5 +- + samples/damon/wsse.c | 5 +- + scripts/gdb/linux/mm.py | 22 +- + security/selinux/selinuxfs.c | 11 +- + sound/core/pcm_native.c | 48 +- + sound/mips/snd-n64.c | 2 +- + tools/cgroup/memcg_shrinker.py | 2 +- + tools/include/linux/compiler.h | 5 + + tools/include/linux/mm.h | 6 +- + tools/lib/mm/file_utils.c | 106 +++ + tools/lib/mm/file_utils.h | 13 + + .../selftests => lib}/mm/hugepage_settings.c | 186 ++++- + .../selftests => lib}/mm/hugepage_settings.h | 12 + + tools/mm/.gitignore | 1 + + tools/mm/Makefile | 11 +- + .../selftests/mm/gup_test.c => mm/gup_bench.c} | 192 ++--- + tools/mm/page-types.c | 2 - + tools/mm/page_owner_sort.c | 134 +++- + tools/sched_ext/include/scx/common.bpf.h | 2 - + tools/testing/memblock/Makefile | 3 +- + tools/testing/memblock/README | 12 +- + tools/testing/memblock/TODO | 5 - + tools/testing/memblock/asm/dma.h | 6 + + tools/testing/memblock/main.c | 2 + + tools/testing/memblock/tests/alloc_low_api.c | 148 ++++ + tools/testing/memblock/tests/alloc_low_api.h | 9 + + tools/testing/memblock/tests/basic_api.c | 20 +- + tools/testing/memblock/tests/common.c | 12 +- + tools/testing/radix-tree/idr-test.c | 20 +- + tools/testing/radix-tree/maple.c | 12 +- + tools/testing/selftests/cgroup/test_zswap.c | 24 +- + tools/testing/selftests/damon/.gitignore | 1 + + tools/testing/selftests/damon/_damon_sysfs.py | 161 +++- + tools/testing/selftests/damon/damon_nr_regions.py | 3 + + .../selftests/damon/damos_apply_interval.py | 3 + + tools/testing/selftests/damon/damos_quota.py | 3 + + tools/testing/selftests/damon/damos_quota_goal.py | 3 + + .../testing/selftests/damon/damos_tried_regions.py | 3 + + .../selftests/damon/drgn_dump_damon_status.py | 32 + + tools/testing/selftests/damon/sysfs.py | 39 +- + tools/testing/selftests/damon/sysfs.sh | 31 + + .../selftests/damon/sysfs_memcg_path_leak.sh | 7 + + .../selftests/damon/sysfs_no_op_commit_break.py | 35 +- + .../sysfs_update_schemes_tried_regions_hang.py | 3 + + ..._update_schemes_tried_regions_wss_estimation.py | 3 + + tools/testing/selftests/mm/.gitignore | 2 + + tools/testing/selftests/mm/Makefile | 21 +- + tools/testing/selftests/mm/compaction_test.c | 2 +- + tools/testing/selftests/mm/cow.c | 1 - + tools/testing/selftests/mm/folio_order_check.c | 122 +++ + tools/testing/selftests/mm/folio_split_race_test.c | 5 +- + tools/testing/selftests/mm/guard-regions.c | 15 +- + tools/testing/selftests/mm/gup.c | 262 ++++++ + tools/testing/selftests/mm/gup_longterm.c | 1 - + tools/testing/selftests/mm/hmm-tests.c | 7 +- + tools/testing/selftests/mm/hugetlb-madvise.c | 1 - + tools/testing/selftests/mm/hugetlb-mmap.c | 1 - + tools/testing/selftests/mm/hugetlb-mremap.c | 1 - + tools/testing/selftests/mm/hugetlb-shm.c | 1 - + tools/testing/selftests/mm/hugetlb-soft-offline.c | 59 +- + tools/testing/selftests/mm/hugetlb_dio.c | 1 - + .../selftests/mm/hugetlb_fault_after_madv.c | 1 - + tools/testing/selftests/mm/hugetlb_madv_vs_map.c | 142 +++- + tools/testing/selftests/mm/khugepaged.c | 580 ++++++++++++-- + tools/testing/selftests/mm/khugepaged_race.c | 538 +++++++++++++ + tools/testing/selftests/mm/khugepaged_sync_check.c | 179 +++++ + tools/testing/selftests/mm/ksm_tests.c | 1 - + tools/testing/selftests/mm/memfd_secret.c | 3 +- + tools/testing/selftests/mm/memory-failure.c | 114 ++- + tools/testing/selftests/mm/merge.c | 95 +++ + tools/testing/selftests/mm/migration.c | 3 +- + tools/testing/selftests/mm/mlock-random-test.c | 1 - + tools/testing/selftests/mm/mlock2-tests.c | 2 +- + tools/testing/selftests/mm/mlock2.h | 8 +- + tools/testing/selftests/mm/mremap_test.c | 44 +- + tools/testing/selftests/mm/pagemap_ioctl.c | 95 +-- + tools/testing/selftests/mm/pkey-helpers.h | 7 - + tools/testing/selftests/mm/pkey_sighandler_tests.c | 4 +- + tools/testing/selftests/mm/prctl_thp_disable.c | 1 - + tools/testing/selftests/mm/protection_keys.c | 2 +- + tools/testing/selftests/mm/run_vmtests.sh | 90 +-- + tools/testing/selftests/mm/soft-dirty.c | 1 - + tools/testing/selftests/mm/split_huge_page_test.c | 87 +- + tools/testing/selftests/mm/thuge-gen.c | 1 - + tools/testing/selftests/mm/transhuge-stress.c | 1 - + tools/testing/selftests/mm/uffd-common.c | 4 +- + tools/testing/selftests/mm/uffd-common.h | 1 - + tools/testing/selftests/mm/uffd-wp-mremap.c | 4 +- + tools/testing/selftests/mm/va_high_addr_switch.c | 1 - + tools/testing/selftests/mm/vm_util.c | 392 +++++---- + tools/testing/selftests/mm/vm_util.h | 19 +- + tools/testing/selftests/proc/proc-maps-race.c | 186 ++++- + .../selftests/proc/proc-self-map-files-001.c | 2 +- + .../selftests/proc/proc-self-map-files-002.c | 2 +- + tools/testing/vma/include/dup.h | 97 ++- + tools/testing/vma/include/stubs.h | 28 +- + tools/testing/vma/shared.c | 9 + + tools/testing/vma/tests/merge.c | 10 +- + tools/testing/vma/tests/mmap.c | 105 +++ + tools/testing/vma/vma_internal.h | 1 - + 459 files changed, 14885 insertions(+), 8276 deletions(-) + create mode 100644 Documentation/mm/kernel-page-tables.rst + delete mode 100644 include/linux/hugetlb_inline.h + create mode 100644 include/linux/vmemmap-optimization.h + rename drivers/char/mem.c => mm/char-mem.c (97%) + create mode 100644 mm/collapse.h + create mode 100644 tools/lib/mm/file_utils.c + create mode 100644 tools/lib/mm/file_utils.h + rename tools/{testing/selftests => lib}/mm/hugepage_settings.c (76%) + rename tools/{testing/selftests => lib}/mm/hugepage_settings.h (90%) + rename tools/{testing/selftests/mm/gup_test.c => mm/gup_bench.c} (50%) + delete mode 100644 tools/testing/memblock/TODO + create mode 100644 tools/testing/memblock/tests/alloc_low_api.c + create mode 100644 tools/testing/memblock/tests/alloc_low_api.h + create mode 100644 tools/testing/selftests/mm/folio_order_check.c + create mode 100644 tools/testing/selftests/mm/gup.c + create mode 100644 tools/testing/selftests/mm/khugepaged_race.c + create mode 100644 tools/testing/selftests/mm/khugepaged_sync_check.c +Merging mm-nonmm-stable/mm-nonmm-stable (a243ede718463 Merge tag 'mtd/fixes-for-7.3-rc6' of git://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux) +$ git merge -m Merge branch 'mm-nonmm-stable' of https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm mm-nonmm-stable/mm-nonmm-stable +Already up to date. +Merging mm-nonmm-unstable/mm-nonmm-unstable (20e59dfe656fb kbuild: move GCOV flags to scripts/Makefile.gcov) +$ git merge -m Merge branch 'mm-nonmm-unstable' of https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm mm-nonmm-unstable/mm-nonmm-unstable +Auto-merging .mailmap +Auto-merging MAINTAINERS +Auto-merging arch/Kconfig +Auto-merging arch/s390/Kconfig +Auto-merging fs/proc/base.c +Auto-merging init/main.c +Auto-merging kernel/fork.c +Auto-merging kernel/taskstats.c +Auto-merging lib/Kconfig.debug +Auto-merging mm/shmem.c +Merge made by the 'ort' strategy. + .mailmap | 4 + + CREDITS | 181 ++++++++-------- + Documentation/admin-guide/sysctl/kernel.rst | 5 +- + MAINTAINERS | 2 + + Makefile | 22 +- + arch/Kconfig | 8 - + arch/alpha/include/uapi/asm/setup.h | 4 + + arch/arc/include/asm/setup.h | 2 +- + arch/arm/include/uapi/asm/setup.h | 6 +- + arch/arm64/include/uapi/asm/setup.h | 4 + + arch/loongarch/include/uapi/asm/setup.h | 4 + + arch/m68k/include/uapi/asm/setup.h | 6 +- + arch/microblaze/include/uapi/asm/setup.h | 4 + + arch/mips/include/uapi/asm/setup.h | 4 + + arch/parisc/include/uapi/asm/setup.h | 4 + + arch/powerpc/include/uapi/asm/setup.h | 4 + + arch/riscv/include/uapi/asm/setup.h | 4 + + arch/s390/Kconfig | 8 - + arch/s390/kernel/traps.c | 7 + + arch/sparc/include/uapi/asm/setup.h | 10 +- + arch/sparc/kernel/setup.c | 9 + + arch/um/include/asm/setup.h | 2 +- + arch/x86/include/asm/setup.h | 2 +- + arch/xtensa/include/uapi/asm/setup.h | 4 + + drivers/media/v4l2-core/v4l2-vp9.c | 30 +-- + drivers/rapidio/devices/rio_mport_cdev.c | 9 +- + drivers/usb/gadget/legacy/inode.c | 4 +- + drivers/watchdog/hpwdt.c | 2 +- + fs/fat/inode.c | 23 +- + fs/ocfs2/alloc.c | 36 +++- + fs/ocfs2/cluster/heartbeat.c | 12 +- + fs/ocfs2/dir.c | 55 +++++ + fs/ocfs2/dlmglue.c | 33 ++- + fs/ocfs2/dlmglue.h | 5 +- + fs/ocfs2/export.c | 10 +- + fs/ocfs2/inode.c | 58 ++++- + fs/ocfs2/journal.c | 3 +- + fs/ocfs2/journal.h | 1 - + fs/ocfs2/ocfs2.h | 17 ++ + fs/ocfs2/quota_local.c | 33 ++- + fs/ocfs2/refcounttree.c | 27 +++ + fs/ocfs2/suballoc.c | 235 +++++++++++++++++--- + fs/ocfs2/suballoc.h | 2 +- + fs/ocfs2/super.c | 32 ++- + fs/ocfs2/xattr.c | 204 ++++++++++++++---- + fs/proc/base.c | 8 +- + fs/squashfs/fragment.c | 6 +- + fs/squashfs/squashfs_fs.h | 2 +- + include/linux/kdev_t.h | 2 +- + include/linux/minmax.h | 6 +- + include/linux/panic.h | 5 +- + include/uapi/asm-generic/setup.h | 4 + + init/Kconfig | 19 +- + init/Makefile | 6 +- + init/main.c | 12 +- + init/version.c | 11 +- + ipc/mqueue.c | 7 + + kernel/fork.c | 3 +- + kernel/gcov/fs.c | 2 +- + kernel/hung_task.c | 55 ++++- + kernel/kallsyms.c | 111 ++++++---- + kernel/kallsyms_internal.h | 11 + + kernel/kcov.c | 8 + + kernel/panic.c | 104 +++++---- + kernel/resource.c | 14 +- + kernel/resource_kunit.c | 3 + + kernel/taskstats.c | 54 ++--- + lib/Kconfig | 14 +- + lib/Kconfig.debug | 16 +- + lib/bootconfig.c | 46 ++-- + lib/decompress_bunzip2.c | 6 +- + lib/decompress_unlz4.c | 4 + + lib/decompress_unxz.c | 12 +- + lib/dynamic_debug.c | 15 +- + lib/group_cpus.c | 87 +++++++- + lib/klist.c | 16 +- + lib/percpu_counter.c | 10 +- + lib/plist.c | 7 + + lib/raid/Kconfig | 2 + + lib/raid/raid6/x86/avx2.c | 6 + + lib/raid/raid6/x86/avx512.c | 6 + + lib/raid/raid6/x86/recov_avx2.c | 2 + + lib/raid/raid6/x86/recov_avx512.c | 2 + + lib/raid/xor/x86/xor-avx.c | 1 + + lib/string_helpers.c | 5 +- + lib/tests/Makefile | 1 + + lib/tests/errseq_kunit.c | 237 +++++++++++++++++++++ + lib/tests/string_helpers_kunit.c | 46 ++-- + mm/shmem.c | 5 + + scripts/Makefile.gcov | 21 ++ + scripts/checkpatch.pl | 38 +++- + scripts/checkstack.pl | 6 +- + scripts/kallsyms.c | 14 +- + scripts/spelling.txt | 4 +- + tools/bootconfig/main.c | 104 ++++----- + tools/bootconfig/test-bootconfig.sh | 12 ++ + tools/testing/selftests/core/unshare_test.c | 19 +- + tools/testing/selftests/filelock/ofdlocks.c | 2 +- + .../filesystems/epoll/epoll_wakeup_test.c | 32 ++- + .../selftests/membarrier/membarrier_test_impl.h | 17 +- + tools/testing/selftests/uevent/uevent_filtering.c | 8 - + 101 files changed, 1808 insertions(+), 609 deletions(-) + create mode 100644 lib/tests/errseq_kunit.c + create mode 100644 scripts/Makefile.gcov +Merging kbuild/kbuild-for-next (2eaa399a85efa Merge branch 'kbuild-next-unstable' into kbuild-for-next) +$ git merge -m Merge branch 'kbuild-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/kbuild/linux.git kbuild/kbuild-for-next +Auto-merging MAINTAINERS +Auto-merging Makefile +Auto-merging arch/x86/Kconfig +Auto-merging arch/x86/Makefile +Auto-merging init/Kconfig +Auto-merging scripts/kallsyms.c +CONFLICT (content): Merge conflict in scripts/kallsyms.c +Resolved 'scripts/kallsyms.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 1051a15c2fb7a] Merge branch 'kbuild-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/kbuild/linux.git +$ git diff -M --stat --summary HEAD^.. + .gitignore | 1 + + Documentation/dev-tools/container.rst | 118 ++++++- + .../early-userspace/early_userspace_support.rst | 16 +- + .../filesystems/ramfs-rootfs-initramfs.rst | 10 +- + Documentation/kbuild/kbuild.rst | 17 + + Documentation/kbuild/kconfig-language.rst | 13 +- + Documentation/kbuild/kconfig-macro-language.rst | 8 +- + Documentation/process/changes.rst | 8 + + MAINTAINERS | 2 + + Makefile | 76 ++-- + arch/arm64/kernel/pi/Makefile | 2 +- + arch/riscv/kernel/pi/Makefile | 2 +- + arch/x86/Kconfig | 17 + + arch/x86/Makefile | 12 +- + arch/x86/boot/Makefile | 2 +- + arch/x86/boot/compressed/Makefile | 2 +- + drivers/firmware/efi/libstub/Makefile | 2 +- + drivers/firmware/qcom/Kconfig | 26 +- + drivers/firmware/qcom/qcom_tzmem.c | 4 +- + fs/erofs/Kconfig | 8 +- + include/asm-generic/vmlinux.lds.h | 2 +- + init/Kconfig | 197 +---------- + kernel/Makefile | 6 +- + scripts/.gitignore | 1 + + scripts/Kconfig.include | 29 +- + scripts/Kconfig.toolchain | 272 +++++++++++++++ + scripts/Makefile | 8 +- + scripts/Makefile.build | 21 ++ + scripts/Makefile.gcc-plugins | 4 + + scripts/Makefile.lib | 4 +- + scripts/Makefile.package | 2 +- + scripts/Makefile.vmlinux | 20 +- + scripts/Makefile.warn | 28 +- + scripts/check-function-names.sh | 3 +- + scripts/container | 86 ++++- + scripts/elf-parse.c | 103 +++++- + scripts/elf-parse.h | 54 +++ + {usr => scripts}/gen_init_cpio.c | 0 + {usr => scripts}/gen_initramfs.sh | 4 +- + scripts/jobserver-exec | 3 +- + scripts/kallsyms-sysmap.c | 248 +++++++++++++ + scripts/kallsyms.c | 382 +++++++++++++++------ + scripts/kallsyms.h | 44 +++ + scripts/kconfig/conf.c | 16 +- + scripts/kconfig/confdata.c | 6 + + scripts/kconfig/kconfig-sym-check.pl | 2 +- + scripts/kconfig/lexer.l | 3 + + scripts/kconfig/lkc.h | 2 +- + scripts/kconfig/lkc_proto.h | 1 + + scripts/kconfig/menu.c | 64 +++- + scripts/kconfig/parser.y | 8 +- + scripts/kconfig/symbol.c | 115 +++++-- + scripts/kconfig/tests/conftest.py | 27 +- + scripts/kconfig/tests/def_type/Kconfig | 28 ++ + scripts/kconfig/tests/def_type/__init__.py | 18 + + scripts/kconfig/tests/def_type/expected_guard_n | 9 + + scripts/kconfig/tests/def_type/expected_guard_y | 11 + + scripts/kconfig/tests/def_type/guard_n.config | 1 + + scripts/kconfig/tests/def_type/guard_y.config | 1 + + scripts/kconfig/tests/err_num_bounds/Kconfig | 79 +++++ + scripts/kconfig/tests/err_num_bounds/__init__.py | 12 + + .../kconfig/tests/err_num_bounds/expected_stderr | 10 + + scripts/kconfig/tests/err_num_mismatch/Kconfig | 38 ++ + scripts/kconfig/tests/err_num_mismatch/__init__.py | 9 + + .../kconfig/tests/err_num_mismatch/expected_stderr | 4 + + .../kconfig/tests/err_num_non_numeric_ref/Kconfig | 75 ++++ + .../tests/err_num_non_numeric_ref/__init__.py | 9 + + .../tests/err_num_non_numeric_ref/expected_stderr | 16 + + scripts/kconfig/tests/err_recursive_dep/Kconfig | 24 ++ + .../tests/err_recursive_dep/expected_stderr | 13 + + .../kconfig/tests/randconfig_probability/Kconfig | 12 + + .../tests/randconfig_probability/__init__.py | 130 +++++++ + scripts/kconfig/tests/savedefconfig_range/Kconfig | 60 ++++ + .../kconfig/tests/savedefconfig_range/__init__.py | 8 + + scripts/kconfig/tests/savedefconfig_range/config | 7 + + .../tests/savedefconfig_range/expected_defconfig | 0 + scripts/kconfig/tests/warn_changed_input/Kconfig | 5 + + .../kconfig/tests/warn_changed_input/__init__.py | 21 +- + .../tests/warn_changed_input/expected_config | 1 + + scripts/kconfig/tests/warn_num_bounds/Kconfig | 24 ++ + scripts/kconfig/tests/warn_num_bounds/__init__.py | 25 ++ + scripts/kconfig/tests/warn_num_bounds/config | 7 + + .../kconfig/tests/warn_num_bounds/expected_config | 11 + + .../tests/warn_num_bounds/expected_config_stderr | 3 + + .../tests/warn_num_bounds/expected_frontend_config | 11 + + .../tests/warn_num_bounds/expected_frontend_stderr | 0 + scripts/link-vmlinux.sh | 24 +- + scripts/mkcompile_h | 2 +- + scripts/mksysmap | 94 ----- + scripts/remove-stale-files | 2 + + .../sbom/tests/cmd_graph/test_savedcmd_parser.py | 2 +- + scripts/setlocalversion | 3 +- + scripts/sorttable.c | 15 +- + scripts/ver_linux | 1 + + tools/lib/python/jobserver.py | 13 + + tools/testing/selftests/kho/vmtest.sh | 2 +- + tools/testing/selftests/liveupdate/vmtest.sh | 4 +- + tools/testing/selftests/nolibc/Makefile.nolibc | 2 +- + usr/.gitignore | 4 +- + usr/Kconfig | 2 +- + usr/Makefile | 4 +- + 101 files changed, 2314 insertions(+), 653 deletions(-) + create mode 100644 scripts/Kconfig.toolchain + rename {usr => scripts}/gen_init_cpio.c (100%) + rename {usr => scripts}/gen_initramfs.sh (97%) + create mode 100644 scripts/kallsyms-sysmap.c + create mode 100644 scripts/kallsyms.h + create mode 100644 scripts/kconfig/tests/def_type/Kconfig + create mode 100644 scripts/kconfig/tests/def_type/__init__.py + create mode 100644 scripts/kconfig/tests/def_type/expected_guard_n + create mode 100644 scripts/kconfig/tests/def_type/expected_guard_y + create mode 100644 scripts/kconfig/tests/def_type/guard_n.config + create mode 100644 scripts/kconfig/tests/def_type/guard_y.config + create mode 100644 scripts/kconfig/tests/err_num_bounds/Kconfig + create mode 100644 scripts/kconfig/tests/err_num_bounds/__init__.py + create mode 100644 scripts/kconfig/tests/err_num_bounds/expected_stderr + create mode 100644 scripts/kconfig/tests/err_num_mismatch/Kconfig + create mode 100644 scripts/kconfig/tests/err_num_mismatch/__init__.py + create mode 100644 scripts/kconfig/tests/err_num_mismatch/expected_stderr + create mode 100644 scripts/kconfig/tests/err_num_non_numeric_ref/Kconfig + create mode 100644 scripts/kconfig/tests/err_num_non_numeric_ref/__init__.py + create mode 100644 scripts/kconfig/tests/err_num_non_numeric_ref/expected_stderr + create mode 100644 scripts/kconfig/tests/randconfig_probability/Kconfig + create mode 100644 scripts/kconfig/tests/randconfig_probability/__init__.py + create mode 100644 scripts/kconfig/tests/savedefconfig_range/Kconfig + create mode 100644 scripts/kconfig/tests/savedefconfig_range/__init__.py + create mode 100644 scripts/kconfig/tests/savedefconfig_range/config + create mode 100644 scripts/kconfig/tests/savedefconfig_range/expected_defconfig + create mode 100644 scripts/kconfig/tests/warn_num_bounds/Kconfig + create mode 100644 scripts/kconfig/tests/warn_num_bounds/__init__.py + create mode 100644 scripts/kconfig/tests/warn_num_bounds/config + create mode 100644 scripts/kconfig/tests/warn_num_bounds/expected_config + create mode 100644 scripts/kconfig/tests/warn_num_bounds/expected_config_stderr + create mode 100644 scripts/kconfig/tests/warn_num_bounds/expected_frontend_config + create mode 100644 scripts/kconfig/tests/warn_num_bounds/expected_frontend_stderr + delete mode 100755 scripts/mksysmap +Merging clang-fixes/clang-fixes-for-next (0bb666d5f5a23 once_lite: Simplify condition handling and fix context analysis) +$ git merge -m Merge branch 'clang-fixes-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/nathan/linux.git clang-fixes/clang-fixes-for-next +Merge made by the 'ort' strategy. + include/linux/once_lite.h | 8 +++++--- + 1 file changed, 5 insertions(+), 3 deletions(-) +Merging clang-format/clang-format (8f0b4cce4481f Linux 6.19-rc1) +$ git merge -m Merge branch 'clang-format' of https://github.com/ojeda/linux.git clang-format/clang-format +Already up to date. +Merging perf/perf-tools-next (2ed38aa8a52d3 perf python: Track linked libraries as extension dependencies) +$ git merge -m Merge branch 'perf-tools-next' of https://git.kernel.org/pub/scm/linux/kernel/git/perf/perf-tools-next.git perf/perf-tools-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 2 +- + tools/arch/x86/include/asm/msr-index.h | 7 + + tools/arch/x86/include/uapi/asm/perf_regs.h | 53 + + tools/build/Makefile.feature | 8 +- + tools/build/feature/Makefile | 42 +- + tools/build/feature/test-all.c | 6 +- + tools/build/feature/test-gettid.c | 1 + + tools/build/feature/test-gtk2-infobar.c | 12 - + tools/build/feature/{test-gtk2.c => test-gtk4.c} | 4 +- + tools/build/feature/test-libdw.c | 10 +- + tools/build/feature/test-libperl.c | 10 - + tools/build/feature/test-libpython.c | 10 - + tools/build/feature/test-python-module.c | 13 + + tools/build/feature/test-timerfd.c | 2 +- + tools/include/tools/dis-asm-compat.h | 1 + + tools/include/uapi/linux/perf_event.h | 49 +- + tools/lib/api/fd/array.c | 4 + + tools/lib/perf/evlist.c | 5 +- + tools/lib/perf/include/internal/evsel.h | 2 + + tools/perf/.clang-format | 1 + + tools/perf/Build | 17 +- + tools/perf/Documentation/build-docdep.perl | 2 +- + tools/perf/Documentation/db-export.txt | 20 +- + tools/perf/Documentation/perf-annotate.txt | 3 + + tools/perf/Documentation/perf-c2c.txt | 14 +- + tools/perf/Documentation/perf-check.txt | 3 +- + tools/perf/Documentation/perf-config.txt | 10 + + tools/perf/Documentation/perf-mem.txt | 4 + + tools/perf/Documentation/perf-record.txt | 27 +- + tools/perf/Documentation/perf-report.txt | 13 +- + tools/perf/Documentation/perf-sched.txt | 6 +- + tools/perf/Documentation/perf-script-perl.txt | 216 - + tools/perf/Documentation/perf-script-python.txt | 752 +-- + tools/perf/Documentation/perf-script.txt | 82 +- + tools/perf/Documentation/perf-stat.txt | 12 +- + tools/perf/Documentation/perf-top.txt | 11 + + tools/perf/Documentation/perf-trace.txt | 6 + + tools/perf/Documentation/perf.data-file-format.txt | 13 + + tools/perf/Documentation/tips.txt | 3 +- + tools/perf/Makefile | 51 +- + tools/perf/Makefile.config | 106 +- + tools/perf/Makefile.perf | 197 +- + tools/perf/arch/arm/tests/dwarf-unwind.c | 1 + + tools/perf/arch/arm64/tests/dwarf-unwind.c | 1 + + tools/perf/arch/loongarch/util/header.c | 2 +- + tools/perf/arch/powerpc/tests/dwarf-unwind.c | 1 + + tools/perf/arch/powerpc/util/skip-callchain-idx.c | 4 +- + tools/perf/arch/riscv/Build | 1 + + tools/perf/arch/riscv/include/arch-tests.h | 9 + + tools/perf/arch/riscv/include/perf_regs.h | 2 + + tools/perf/arch/riscv/tests/Build | 4 + + tools/perf/arch/riscv/tests/arch-tests.c | 10 + + tools/perf/arch/riscv/tests/dwarf-unwind.c | 64 + + tools/perf/arch/riscv/tests/regs_load.S | 58 + + tools/perf/arch/x86/tests/dwarf-unwind.c | 4 + + tools/perf/arch/x86/tests/topdown.c | 30 + + tools/perf/arch/x86/util/mem-events.c | 17 + + tools/perf/arch/x86/util/mem-events.h | 2 + + tools/perf/arch/x86/util/pmu.c | 226 +- + tools/perf/arch/x86/util/topdown.c | 3 +- + tools/perf/bench/futex.h | 11 + + tools/perf/bench/numa.c | 2 + + tools/perf/bench/sched-messaging.c | 4 +- + tools/perf/builtin-annotate.c | 96 +- + tools/perf/builtin-buildid-list.c | 17 +- + tools/perf/builtin-c2c.c | 229 +- + tools/perf/builtin-check.c | 3 +- + tools/perf/builtin-config.c | 18 +- + tools/perf/builtin-ftrace.c | 28 +- + tools/perf/builtin-help.c | 30 +- + tools/perf/builtin-inject.c | 4 + + tools/perf/builtin-kmem.c | 3 +- + tools/perf/builtin-kvm.c | 1 + + tools/perf/builtin-list.c | 2 +- + tools/perf/builtin-mem.c | 7 +- + tools/perf/builtin-record.c | 15 + + tools/perf/builtin-report.c | 43 +- + tools/perf/builtin-sched.c | 202 +- + tools/perf/builtin-script.c | 1124 +++-- + tools/perf/builtin-stat.c | 4 +- + tools/perf/builtin-timechart.c | 5 - + tools/perf/builtin-top.c | 38 +- + tools/perf/builtin-trace.c | 712 +-- + tools/perf/builtin.h | 3 + + tools/perf/perf.c | 120 +- + tools/perf/pmu-events/Build | 72 +- + tools/perf/pmu-events/amd_metrics.py | 16 +- + .../perf/pmu-events/arch/x86/alderlake/cache.json | 4 +- + .../perf/pmu-events/arch/x86/alderlake/memory.json | 13 + + .../perf/pmu-events/arch/x86/alderlake/other.json | 10 + + .../pmu-events/arch/x86/alderlake/pipeline.json | 6 +- + .../perf/pmu-events/arch/x86/alderlaken/cache.json | 4 +- + .../perf/pmu-events/arch/x86/arrowlake/cache.json | 1 + + .../perf/pmu-events/arch/x86/arrowlake/other.json | 19 + + .../pmu-events/arch/x86/arrowlake/pipeline.json | 10 +- + .../pmu-events/arch/x86/broadwell/bdw-metrics.json | 51 +- + .../arch/x86/broadwell/metricgroups.json | 3 +- + .../arch/x86/broadwellde/bdwde-metrics.json | 49 +- + .../arch/x86/broadwellde/metricgroups.json | 3 +- + .../arch/x86/broadwellx/bdx-metrics.json | 63 +- + .../arch/x86/broadwellx/metricgroups.json | 3 +- + .../arch/x86/cascadelakex/clx-metrics.json | 94 +- + .../arch/x86/cascadelakex/metricgroups.json | 3 +- + .../arch/x86/cascadelakex/uncore-interconnect.json | 6 + + .../arch/x86/clearwaterforest/pipeline.json | 3 +- + .../arch/x86/clearwaterforest/uncore-cache.json | 5 + + .../x86/clearwaterforest/uncore-interconnect.json | 2 + + .../pmu-events/arch/x86/emeraldrapids/cache.json | 2 +- + .../pmu-events/arch/x86/emeraldrapids/memory.json | 35 + + .../pmu-events/arch/x86/emeraldrapids/other.json | 9 + + .../arch/x86/emeraldrapids/pipeline.json | 10 +- + .../arch/x86/emeraldrapids/uncore-cache.json | 42 + + .../x86/emeraldrapids/uncore-interconnect.json | 12 + + .../arch/x86/emeraldrapids/uncore-io.json | 54 +- + .../arch/x86/emeraldrapids/uncore-memory.json | 10 + + .../pmu-events/arch/x86/graniterapids/other.json | 9 + + .../arch/x86/graniterapids/pipeline.json | 6 +- + .../arch/x86/graniterapids/uncore-cache.json | 15 + + .../x86/graniterapids/uncore-interconnect.json | 2 + + .../arch/x86/graniterapids/uncore-io.json | 12 + + .../pmu-events/arch/x86/haswell/hsw-metrics.json | 49 +- + .../pmu-events/arch/x86/haswell/metricgroups.json | 3 +- + .../pmu-events/arch/x86/haswellx/hsx-metrics.json | 61 +- + .../pmu-events/arch/x86/haswellx/metricgroups.json | 3 +- + .../pmu-events/arch/x86/icelake/icl-metrics.json | 84 +- + tools/perf/pmu-events/arch/x86/icelake/memory.json | 24 + + .../pmu-events/arch/x86/icelake/metricgroups.json | 2 +- + tools/perf/pmu-events/arch/x86/icelake/other.json | 9 + + .../perf/pmu-events/arch/x86/icelake/pipeline.json | 9 +- + .../pmu-events/arch/x86/icelakex/icx-metrics.json | 101 +- + .../perf/pmu-events/arch/x86/icelakex/memory.json | 24 + + .../pmu-events/arch/x86/icelakex/metricgroups.json | 2 +- + tools/perf/pmu-events/arch/x86/icelakex/other.json | 9 + + .../pmu-events/arch/x86/icelakex/pipeline.json | 7 +- + .../pmu-events/arch/x86/icelakex/uncore-cache.json | 79 + + .../arch/x86/icelakex/uncore-interconnect.json | 22 + + .../pmu-events/arch/x86/ivybridge/ivb-metrics.json | 51 +- + .../arch/x86/ivybridge/metricgroups.json | 3 +- + .../pmu-events/arch/x86/ivytown/ivt-metrics.json | 63 +- + .../pmu-events/arch/x86/ivytown/metricgroups.json | 3 +- + .../pmu-events/arch/x86/jaketown/jkt-metrics.json | 41 +- + .../pmu-events/arch/x86/jaketown/metricgroups.json | 3 +- + .../perf/pmu-events/arch/x86/lunarlake/cache.json | 3 + + .../pmu-events/arch/x86/lunarlake/pipeline.json | 2 + + tools/perf/pmu-events/arch/x86/mapfile.csv | 26 +- + .../perf/pmu-events/arch/x86/meteorlake/other.json | 10 + + .../pmu-events/arch/x86/meteorlake/pipeline.json | 6 +- + tools/perf/pmu-events/arch/x86/novalake/cache.json | 500 +- + .../arch/x86/novalake/floating-point.json | 423 ++ + .../pmu-events/arch/x86/novalake/frontend.json | 147 +- + .../perf/pmu-events/arch/x86/novalake/memory.json | 24 + + tools/perf/pmu-events/arch/x86/novalake/other.json | 213 + + .../pmu-events/arch/x86/novalake/pipeline.json | 508 +- + .../arch/x86/novalake/virtual-memory.json | 173 + + .../pmu-events/arch/x86/pantherlake/cache.json | 3 + + .../pmu-events/arch/x86/pantherlake/other.json | 10 + + .../pmu-events/arch/x86/pantherlake/pipeline.json | 10 +- + .../pmu-events/arch/x86/rocketlake/memory.json | 24 + + .../arch/x86/rocketlake/metricgroups.json | 2 +- + .../perf/pmu-events/arch/x86/rocketlake/other.json | 9 + + .../pmu-events/arch/x86/rocketlake/pipeline.json | 9 +- + .../arch/x86/rocketlake/rkl-metrics.json | 84 +- + .../arch/x86/sandybridge/metricgroups.json | 3 +- + .../arch/x86/sandybridge/snb-metrics.json | 35 +- + .../pmu-events/arch/x86/sapphirerapids/cache.json | 2 +- + .../pmu-events/arch/x86/sapphirerapids/memory.json | 35 + + .../arch/x86/sapphirerapids/metricgroups.json | 2 +- + .../pmu-events/arch/x86/sapphirerapids/other.json | 9 + + .../arch/x86/sapphirerapids/pipeline.json | 10 +- + .../arch/x86/sapphirerapids/spr-metrics.json | 93 +- + .../arch/x86/sapphirerapids/uncore-cache.json | 42 + + .../x86/sapphirerapids/uncore-interconnect.json | 12 + + .../arch/x86/sapphirerapids/uncore-io.json | 54 +- + .../arch/x86/sapphirerapids/uncore-memory.json | 10 + + .../arch/x86/sierraforest/uncore-cache.json | 5 + + .../arch/x86/sierraforest/uncore-interconnect.json | 2 + + .../pmu-events/arch/x86/skylake/metricgroups.json | 3 +- + .../pmu-events/arch/x86/skylake/skl-metrics.json | 59 +- + .../pmu-events/arch/x86/skylakex/metricgroups.json | 3 +- + .../pmu-events/arch/x86/skylakex/skx-metrics.json | 94 +- + .../arch/x86/skylakex/uncore-interconnect.json | 6 + + .../arch/x86/snowridgex/uncore-cache.json | 68 + + .../arch/x86/snowridgex/uncore-interconnect.json | 10 + + .../arch/x86/tigerlake/metricgroups.json | 2 +- + .../perf/pmu-events/arch/x86/tigerlake/other.json | 9 + + .../pmu-events/arch/x86/tigerlake/pipeline.json | 7 +- + .../pmu-events/arch/x86/tigerlake/tgl-metrics.json | 82 +- + tools/perf/pmu-events/intel_metrics.py | 268 +- + tools/perf/pmu-events/jevents.py | 112 +- + tools/perf/pmu-events/make_legacy_cache.py | 34 +- + tools/perf/pmu-events/metric.py | 60 +- + tools/perf/python/SchedGui.py | 246 + + tools/perf/python/arm-cs-trace-disasm.py | 440 ++ + tools/perf/python/check-perf-trace.py | 223 + + tools/perf/python/compaction-times.py | 362 ++ + tools/perf/python/counting.py | 1 + + tools/perf/python/event_analyzing_sample.py | 317 ++ + tools/perf/python/export-to-postgresql.py | 1328 +++++ + tools/perf/python/export-to-sqlite.py | 946 ++++ + tools/perf/python/exported-sql-viewer.py | 5103 ++++++++++++++++++++ + tools/perf/python/failed-syscalls-by-pid.py | 161 + + tools/perf/python/failed-syscalls.py | 94 + + tools/perf/python/flamegraph.py | 283 ++ + tools/perf/python/futex-contention.py | 107 + + tools/perf/python/gecko.py | 411 ++ + tools/perf/python/ilist.py | 86 +- + tools/perf/python/intel-pt-events.py | 632 +++ + tools/perf/python/libxed.py | 123 + + tools/perf/python/mem-phys-addr.py | 136 + + tools/perf/python/net_dropmonitor.py | 173 + + tools/perf/python/netdev-times.py | 494 ++ + tools/perf/python/parallel-perf.py | 1250 +++++ + tools/perf/python/perf.pyi | 122 +- + tools/perf/python/perf_live.py | 10 +- + tools/perf/{scripts => }/python/powerpc-hcalls.py | 213 +- + tools/perf/python/rw-by-file.py | 113 + + tools/perf/python/rw-by-pid.py | 205 + + tools/perf/python/rwtop.py | 256 + + tools/perf/python/sched-migration.py | 496 ++ + tools/perf/python/sctop.py | 253 + + tools/perf/python/stackcollapse.py | 145 + + tools/perf/python/stat-cpi.py | 231 + + tools/perf/python/syscall-counts-by-pid.py | 112 + + tools/perf/python/syscall-counts.py | 95 + + tools/perf/python/task-analyzer.py | 908 ++++ + tools/perf/python/tracepoint.py | 5 +- + tools/perf/python/treport.py | 564 +++ + tools/perf/python/twatch.py | 77 +- + tools/perf/python/wakeup-latency.py | 103 + + tools/perf/scripts/Build | 30 - + tools/perf/scripts/install-build-deps.sh | 8 +- + tools/perf/scripts/perl/Perf-Trace-Util/Build | 9 - + tools/perf/scripts/perl/Perf-Trace-Util/Context.c | 122 - + tools/perf/scripts/perl/Perf-Trace-Util/Context.xs | 42 - + .../perf/scripts/perl/Perf-Trace-Util/Makefile.PL | 18 - + tools/perf/scripts/perl/Perf-Trace-Util/README | 59 - + .../perl/Perf-Trace-Util/lib/Perf/Trace/Context.pm | 55 - + .../perl/Perf-Trace-Util/lib/Perf/Trace/Core.pm | 192 - + .../perl/Perf-Trace-Util/lib/Perf/Trace/Util.pm | 94 - + tools/perf/scripts/perl/Perf-Trace-Util/typemap | 1 - + .../perf/scripts/perl/bin/check-perf-trace-record | 2 - + tools/perf/scripts/perl/bin/failed-syscalls-record | 3 - + tools/perf/scripts/perl/bin/failed-syscalls-report | 10 - + tools/perf/scripts/perl/bin/rw-by-file-record | 3 - + tools/perf/scripts/perl/bin/rw-by-file-report | 10 - + tools/perf/scripts/perl/bin/rw-by-pid-record | 2 - + tools/perf/scripts/perl/bin/rw-by-pid-report | 3 - + tools/perf/scripts/perl/bin/rwtop-record | 2 - + tools/perf/scripts/perl/bin/rwtop-report | 20 - + tools/perf/scripts/perl/bin/wakeup-latency-record | 6 - + tools/perf/scripts/perl/bin/wakeup-latency-report | 3 - + tools/perf/scripts/perl/check-perf-trace.pl | 106 - + tools/perf/scripts/perl/failed-syscalls.pl | 47 - + tools/perf/scripts/perl/rw-by-file.pl | 106 - + tools/perf/scripts/perl/rw-by-pid.pl | 184 - + tools/perf/scripts/perl/rwtop.pl | 203 - + tools/perf/scripts/perl/wakeup-latency.pl | 107 - + tools/perf/scripts/python/Perf-Trace-Util/Build | 4 - + .../perf/scripts/python/Perf-Trace-Util/Context.c | 225 - + .../python/Perf-Trace-Util/lib/Perf/Trace/Core.py | 116 - + .../Perf-Trace-Util/lib/Perf/Trace/EventClass.py | 97 - + .../Perf-Trace-Util/lib/Perf/Trace/SchedGui.py | 184 - + .../python/Perf-Trace-Util/lib/Perf/Trace/Util.py | 92 - + tools/perf/scripts/python/arm-cs-trace-disasm.py | 356 -- + .../scripts/python/bin/compaction-times-record | 2 - + .../scripts/python/bin/compaction-times-report | 4 - + .../python/bin/event_analyzing_sample-record | 8 - + .../python/bin/event_analyzing_sample-report | 3 - + .../scripts/python/bin/export-to-postgresql-record | 8 - + .../scripts/python/bin/export-to-postgresql-report | 29 - + .../scripts/python/bin/export-to-sqlite-record | 8 - + .../scripts/python/bin/export-to-sqlite-report | 29 - + .../python/bin/failed-syscalls-by-pid-record | 3 - + .../python/bin/failed-syscalls-by-pid-report | 10 - + tools/perf/scripts/python/bin/flamegraph-record | 2 - + tools/perf/scripts/python/bin/flamegraph-report | 3 - + .../scripts/python/bin/futex-contention-record | 2 - + .../scripts/python/bin/futex-contention-report | 4 - + tools/perf/scripts/python/bin/gecko-record | 2 - + tools/perf/scripts/python/bin/gecko-report | 7 - + .../perf/scripts/python/bin/intel-pt-events-record | 13 - + .../perf/scripts/python/bin/intel-pt-events-report | 3 - + tools/perf/scripts/python/bin/mem-phys-addr-record | 19 - + tools/perf/scripts/python/bin/mem-phys-addr-report | 3 - + .../perf/scripts/python/bin/net_dropmonitor-record | 2 - + .../perf/scripts/python/bin/net_dropmonitor-report | 4 - + tools/perf/scripts/python/bin/netdev-times-record | 8 - + tools/perf/scripts/python/bin/netdev-times-report | 5 - + .../perf/scripts/python/bin/powerpc-hcalls-record | 2 - + .../perf/scripts/python/bin/powerpc-hcalls-report | 2 - + .../perf/scripts/python/bin/sched-migration-record | 2 - + .../perf/scripts/python/bin/sched-migration-report | 3 - + tools/perf/scripts/python/bin/sctop-record | 3 - + tools/perf/scripts/python/bin/sctop-report | 24 - + tools/perf/scripts/python/bin/stackcollapse-record | 8 - + tools/perf/scripts/python/bin/stackcollapse-report | 3 - + .../python/bin/syscall-counts-by-pid-record | 3 - + .../python/bin/syscall-counts-by-pid-report | 10 - + .../perf/scripts/python/bin/syscall-counts-record | 3 - + .../perf/scripts/python/bin/syscall-counts-report | 10 - + tools/perf/scripts/python/bin/task-analyzer-record | 2 - + tools/perf/scripts/python/bin/task-analyzer-report | 3 - + tools/perf/scripts/python/check-perf-trace.py | 84 - + tools/perf/scripts/python/compaction-times.py | 311 -- + .../perf/scripts/python/event_analyzing_sample.py | 192 - + tools/perf/scripts/python/export-to-postgresql.py | 1114 ----- + tools/perf/scripts/python/export-to-sqlite.py | 799 --- + tools/perf/scripts/python/exported-sql-viewer.py | 5030 ------------------- + .../perf/scripts/python/failed-syscalls-by-pid.py | 79 - + tools/perf/scripts/python/flamegraph.py | 267 - + tools/perf/scripts/python/futex-contention.py | 57 - + tools/perf/scripts/python/gecko.py | 395 -- + tools/perf/scripts/python/intel-pt-events.py | 494 -- + tools/perf/scripts/python/libxed.py | 107 - + tools/perf/scripts/python/mem-phys-addr.py | 127 - + tools/perf/scripts/python/net_dropmonitor.py | 78 - + tools/perf/scripts/python/netdev-times.py | 473 -- + tools/perf/scripts/python/parallel-perf.py | 989 ---- + tools/perf/scripts/python/sched-migration.py | 462 -- + tools/perf/scripts/python/sctop.py | 89 - + tools/perf/scripts/python/stackcollapse.py | 127 - + tools/perf/scripts/python/stat-cpi.py | 79 - + tools/perf/scripts/python/syscall-counts-by-pid.py | 75 - + tools/perf/scripts/python/syscall-counts.py | 65 - + tools/perf/scripts/python/task-analyzer.py | 934 ---- + tools/perf/tests/Build | 19 +- + tools/perf/tests/builtin-test.c | 53 +- + tools/perf/tests/code-reading.c | 12 +- + tools/perf/tests/demangle-java-test.c | 1 + + tools/perf/tests/demangle-rust-v0-test.c | 1 + + tools/perf/tests/fdarray.c | 32 + + tools/perf/tests/hists_cumulate.c | 2 +- + tools/perf/tests/hists_filter.c | 2 +- + tools/perf/tests/hists_link.c | 2 +- + tools/perf/tests/hists_output.c | 2 +- + tools/perf/tests/hybrid-merge.c | 206 + + tools/perf/tests/make | 114 +- + tools/perf/tests/sample-parsing.c | 109 + + tools/perf/tests/shell/annotate.sh | 1 - + tools/perf/tests/shell/annotate_weight.sh | 63 + + .../tests/shell/base_probe/test_adding_kernel.sh | 1 - + tools/perf/tests/shell/c2c.sh | 114 + + tools/perf/tests/shell/common/init.sh | 21 +- + tools/perf/tests/shell/coresight/callchain.sh | 7 +- + .../shell/coresight/test_arm_coresight_disasm.sh | 15 +- + tools/perf/tests/shell/data_type_profiling.sh | 46 +- + tools/perf/tests/shell/diff.sh | 1 - + tools/perf/tests/shell/inject_aslr.sh | 1 - + tools/perf/tests/shell/jitdump-python.sh | 1 - + tools/perf/tests/shell/lib/attr.py | 119 +- + tools/perf/tests/shell/lib/perf_brstack_max.py | 40 + + .../perf/tests/shell/lib/perf_json_output_lint.py | 36 +- + .../perf/tests/shell/lib/perf_metric_validation.py | 53 +- + tools/perf/tests/shell/lib/perf_record.sh | 2 +- + tools/perf/tests/shell/lib/probe_vfs_getname.sh | 33 +- + tools/perf/tests/shell/lib/setup_python.sh | 29 +- + tools/perf/tests/shell/lib/waiting.sh | 34 +- + tools/perf/tests/shell/list.sh | 1 - + tools/perf/tests/shell/pipe_test.sh | 1 - + tools/perf/tests/shell/probe_vfs_getname.sh | 5 +- + tools/perf/tests/shell/python-use.sh | 1 - + .../tests/shell/record+probe_libc_inet_pton.sh | 87 +- + .../tests/shell/record+script_probe_vfs_getname.sh | 18 +- + tools/perf/tests/shell/record.sh | 278 +- + tools/perf/tests/shell/record_weak_term.sh | 1 - + tools/perf/tests/shell/report_hybrid_merge.sh | 121 + + tools/perf/tests/shell/schedstat_snapshots.sh | 155 + + tools/perf/tests/shell/script.sh | 46 +- + tools/perf/tests/shell/script_dlfilter.sh | 1 - + tools/perf/tests/shell/script_perl.sh | 102 - + tools/perf/tests/shell/script_python.sh | 113 - + tools/perf/tests/shell/stat+csv_output.sh | 1 - + tools/perf/tests/shell/stat+json_output.sh | 1 - + tools/perf/tests/shell/stat+std_output.sh | 1 - + tools/perf/tests/shell/stat.sh | 5 + + tools/perf/tests/shell/stat_metrics_cgrp.sh | 7 + + tools/perf/tests/shell/stat_metrics_values.sh | 1 - + tools/perf/tests/shell/test_arm_callgraph_fp.sh | 1 - + tools/perf/tests/shell/test_brstack.sh | 1 - + .../tests/shell/test_check_perf_trace_python.sh | 87 + + .../tests/shell/test_compaction_times_python.sh | 129 + + tools/perf/tests/shell/test_data_symbol.sh | 7 +- + .../shell/test_event_analyzing_sample_python.sh | 70 + + tools/perf/tests/shell/test_event_open_fallback.sh | 2 - + .../shell/test_export_to_postgresql_python.sh | 129 + + .../tests/shell/test_export_to_sqlite_python.sh | 115 + + .../shell/test_failed_syscalls_by_pid_python.sh | 106 + + .../tests/shell/test_failed_syscalls_python.sh | 90 + + tools/perf/tests/shell/test_flamegraph_python.sh | 123 + + .../tests/shell/test_futex_contention_python.sh | 116 + + tools/perf/tests/shell/test_gecko_python.sh | 96 + + tools/perf/tests/shell/test_intel_pt.sh | 46 +- + .../tests/shell/test_intel_pt_events_python.sh | 74 + + .../perf/tests/shell/test_mem_phys_addr_python.sh | 105 + + .../tests/shell/test_net_dropmonitor_python.sh | 103 + + tools/perf/tests/shell/test_netdev_times_python.sh | 104 + + .../tests/shell/test_perf_data_converter_json.sh | 1 - + .../perf/tests/shell/test_powerpc_hcalls_python.sh | 95 + + tools/perf/tests/shell/test_rw_by_file_python.sh | 69 + + tools/perf/tests/shell/test_rw_by_pid_python.sh | 75 + + tools/perf/tests/shell/test_rwtop_python.sh | 74 + + .../tests/shell/test_sched_migration_python.sh | 72 + + tools/perf/tests/shell/test_sctop_python.sh | 78 + + .../perf/tests/shell/test_stackcollapse_python.sh | 78 + + tools/perf/tests/shell/test_stat_cpi_python.sh | 121 + + .../shell/test_syscall_counts_by_pid_python.sh | 93 + + .../perf/tests/shell/test_syscall_counts_python.sh | 82 + + tools/perf/tests/shell/test_task_analyzer.sh | 93 +- + tools/perf/tests/shell/test_test_junit_output.sh | 1 - + .../tests/shell/test_uprobe_from_different_cu.sh | 8 +- + .../perf/tests/shell/test_wakeup_latency_python.sh | 83 + + tools/perf/tests/shell/top.sh | 73 +- + tools/perf/tests/shell/trace+probe_vfs_getname.sh | 6 +- + tools/perf/tests/shell/trace_btf_enum.sh | 1 - + tools/perf/tests/shell/trace_btf_general.sh | 9 +- + tools/perf/tests/shell/trace_exit_race.sh | 1 - + tools/perf/tests/shell/trace_ksym_beautifier.sh | 38 + + tools/perf/tests/shell/trace_record_replay.sh | 2 - + tools/perf/tests/shell/trace_summary.sh | 16 +- + tools/perf/tests/symbols.c | 96 +- + tools/perf/tests/tests.h | 1 + + tools/perf/tests/thread-maps-share.c | 3 +- + tools/perf/tests/tool_pmu.c | 79 + + tools/perf/tests/workloads/code_with_type.c | 2 +- + tools/perf/trace/beauty/arch_errno_names.sh | 4 +- + tools/perf/trace/beauty/beauty.h | 20 + + tools/perf/trace/beauty/include/uapi/linux/fs.h | 2 +- + tools/perf/trace/beauty/perf_event_open.c | 31 +- + tools/perf/trace/beauty/sockaddr.c | 49 +- + tools/perf/trace/beauty/syscalltbl.sh | 2 +- + tools/perf/trace/beauty/timespec.c | 2 +- + tools/perf/ui/browsers/annotate-data.c | 12 +- + tools/perf/ui/browsers/annotate.c | 21 +- + tools/perf/ui/browsers/hists.c | 8 +- + tools/perf/ui/browsers/scripts.c | 218 +- + tools/perf/ui/gtk/annotate.c | 36 +- + tools/perf/ui/gtk/browser.c | 96 +- + tools/perf/ui/gtk/gtk.h | 16 +- + tools/perf/ui/gtk/hists.c | 72 +- + tools/perf/ui/gtk/progress.c | 40 +- + tools/perf/ui/gtk/setup.c | 5 +- + tools/perf/ui/gtk/util.c | 97 +- + tools/perf/ui/hist.c | 248 +- + tools/perf/ui/libslang.h | 2 + + tools/perf/ui/setup.c | 2 +- + tools/perf/util/Build | 18 +- + tools/perf/util/addr2line.c | 42 +- + tools/perf/util/annotate-arch/Build | 1 + + tools/perf/util/annotate-arch/annotate-alpha.c | 185 + + tools/perf/util/annotate-arch/annotate-arm64.c | 435 +- + tools/perf/util/annotate-arch/annotate-loongarch.c | 4 + + tools/perf/util/annotate-arch/annotate-powerpc.c | 9 + + tools/perf/util/annotate-arch/annotate-x86.c | 79 + + tools/perf/util/annotate-data.c | 303 +- + tools/perf/util/annotate-data.h | 17 +- + tools/perf/util/annotate.c | 240 +- + tools/perf/util/annotate.h | 121 +- + tools/perf/util/arm-spe.c | 17 +- + tools/perf/util/aslr.c | 41 +- + tools/perf/util/aslr.h | 2 + + tools/perf/util/auxtrace.c | 12 +- + tools/perf/util/auxtrace.h | 7 +- + tools/perf/util/bpf-filter.c | 29 +- + tools/perf/util/bpf-filter.l | 1 + + .../util/bpf_skel/augmented_raw_syscalls.bpf.c | 220 +- + tools/perf/util/bpf_skel/perf_trace_u.h | 14 + + tools/perf/util/bpf_skel/vmlinux/vmlinux.h | 9 + + tools/perf/util/bpf_trace_augment.c | 91 +- + tools/perf/util/build-id.c | 12 +- + tools/perf/util/c2c-function.c | 9 +- + tools/perf/util/c2c.h | 2 + + tools/perf/util/cache.h | 31 - + tools/perf/util/capstone.c | 186 +- + tools/perf/util/config.c | 52 +- + tools/perf/util/cs-etm.c | 2 +- + tools/perf/util/debug.c | 2 +- + tools/perf/util/debuginfo.c | 51 +- + tools/perf/util/debuginfo.h | 13 +- + tools/perf/util/disasm.c | 28 +- + tools/perf/util/disasm.h | 6 + + tools/perf/util/dso.c | 194 +- + tools/perf/util/dso.h | 59 +- + tools/perf/util/dwarf-aux.c | 247 +- + tools/perf/util/dwarf-aux.h | 16 + + tools/perf/util/dwarf-regs-arch/dwarf-regs-arm64.c | 25 + + tools/perf/util/dwarf-regs-arch/dwarf-regs-csky.c | 2 +- + .../perf/util/dwarf-regs-arch/dwarf-regs-powerpc.c | 2 +- + tools/perf/util/dwarf-regs-arch/dwarf-regs-s390.c | 2 +- + tools/perf/util/dwarf-regs-arch/dwarf-regs-x86.c | 140 +- + tools/perf/util/dwarf-regs.c | 11 +- + tools/perf/util/env.c | 1 + + tools/perf/util/env.h | 10 + + tools/perf/util/evlist.c | 236 +- + tools/perf/util/evlist.h | 3 + + tools/perf/util/evsel.c | 277 +- + tools/perf/util/evsel.h | 9 + + tools/perf/util/expr.c | 5 +- + tools/perf/util/genelf_debug.c | 28 +- + tools/perf/util/header.c | 336 +- + tools/perf/util/header.h | 1 + + tools/perf/util/help-unknown-cmd.c | 19 +- + tools/perf/util/help-unknown-cmd.h | 0 + tools/perf/util/hist.h | 21 +- + tools/perf/util/hwmon_pmu.c | 2 +- + tools/perf/util/include/dwarf-regs.h | 8 +- + tools/perf/util/intel-bts.c | 2 +- + tools/perf/util/intel-pt.c | 3 +- + tools/perf/util/jitdump.c | 204 +- + tools/perf/util/libbfd.c | 94 +- + tools/perf/util/libbfd.h | 9 + + tools/perf/util/libdw.c | 39 +- + tools/perf/util/llvm.c | 56 +- + tools/perf/util/machine.c | 4 +- + tools/perf/util/machine.h | 10 +- + tools/perf/util/mem-events.c | 94 +- + tools/perf/util/mem-events.h | 5 +- + tools/perf/util/metricgroup.c | 37 +- + tools/perf/util/parse-events.c | 5 +- + tools/perf/util/parse-regs-options.c | 211 +- + tools/perf/util/path.c | 10 +- + tools/perf/util/path.h | 6 +- + tools/perf/util/perf-regs-arch/perf_regs_x86.c | 438 +- + tools/perf/util/perf_event_attr_fprintf.c | 13 + + tools/perf/util/perf_regs.c | 84 +- + tools/perf/util/perf_regs.h | 21 +- + tools/perf/util/pmu.c | 48 +- + tools/perf/util/pmu.h | 3 + + tools/perf/util/pmus.c | 21 +- + tools/perf/util/powerpc-vpadtl.c | 2 +- + tools/perf/util/probe-event.c | 4 +- + tools/perf/util/python.c | 1042 +++- + tools/perf/util/record.h | 7 + + tools/perf/util/sample.h | 5 + + tools/perf/util/scripting-engines/Build | 9 - + .../perf/util/scripting-engines/trace-event-perl.c | 770 --- + .../util/scripting-engines/trace-event-python.c | 2224 --------- + tools/perf/util/session.c | 119 +- + tools/perf/util/setup.py | 1 + + tools/perf/util/srcline.c | 66 +- + tools/perf/util/strbuf.c | 14 +- + tools/perf/util/symbol.c | 34 +- + tools/perf/util/symbol_conf.h | 16 +- + tools/perf/util/synthetic-events.c | 56 +- + tools/perf/util/thread.c | 18 +- + tools/perf/util/thread_map.c | 26 + + tools/perf/util/thread_map.h | 1 + + tools/perf/util/tool_pmu.c | 2 +- + tools/perf/util/tp_pmu.c | 8 +- + tools/perf/util/trace-event-parse.c | 65 - + tools/perf/util/trace-event-scripting.c | 407 -- + tools/perf/util/trace-event.c | 90 +- + tools/perf/util/trace-event.h | 73 +- + tools/perf/util/trace_augment.h | 35 +- + tools/perf/util/unwind-libdw.c | 59 +- + tools/perf/util/unwind-libunwind.c | 62 +- + tools/perf/util/usage.c | 34 - + tools/perf/util/util.h | 4 - + 557 files changed, 33893 insertions(+), 24305 deletions(-) + delete mode 100644 tools/build/feature/test-gtk2-infobar.c + rename tools/build/feature/{test-gtk2.c => test-gtk4.c} (76%) + delete mode 100644 tools/build/feature/test-libperl.c + delete mode 100644 tools/build/feature/test-libpython.c + create mode 100644 tools/build/feature/test-python-module.c + delete mode 100644 tools/perf/Documentation/perf-script-perl.txt + create mode 100644 tools/perf/arch/riscv/include/arch-tests.h + create mode 100644 tools/perf/arch/riscv/tests/Build + create mode 100644 tools/perf/arch/riscv/tests/arch-tests.c + create mode 100644 tools/perf/arch/riscv/tests/dwarf-unwind.c + create mode 100644 tools/perf/arch/riscv/tests/regs_load.S + create mode 100755 tools/perf/python/SchedGui.py + create mode 100755 tools/perf/python/arm-cs-trace-disasm.py + create mode 100755 tools/perf/python/check-perf-trace.py + create mode 100755 tools/perf/python/compaction-times.py + create mode 100755 tools/perf/python/event_analyzing_sample.py + create mode 100755 tools/perf/python/export-to-postgresql.py + create mode 100755 tools/perf/python/export-to-sqlite.py + create mode 100755 tools/perf/python/exported-sql-viewer.py + create mode 100755 tools/perf/python/failed-syscalls-by-pid.py + create mode 100755 tools/perf/python/failed-syscalls.py + create mode 100755 tools/perf/python/flamegraph.py + create mode 100755 tools/perf/python/futex-contention.py + create mode 100755 tools/perf/python/gecko.py + create mode 100755 tools/perf/python/intel-pt-events.py + create mode 100755 tools/perf/python/libxed.py + create mode 100755 tools/perf/python/mem-phys-addr.py + create mode 100755 tools/perf/python/net_dropmonitor.py + create mode 100755 tools/perf/python/netdev-times.py + create mode 100755 tools/perf/python/parallel-perf.py + rename tools/perf/{scripts => }/python/powerpc-hcalls.py (54%) + mode change 100644 => 100755 + create mode 100755 tools/perf/python/rw-by-file.py + create mode 100755 tools/perf/python/rw-by-pid.py + create mode 100755 tools/perf/python/rwtop.py + create mode 100755 tools/perf/python/sched-migration.py + create mode 100755 tools/perf/python/sctop.py + create mode 100755 tools/perf/python/stackcollapse.py + create mode 100755 tools/perf/python/stat-cpi.py + create mode 100755 tools/perf/python/syscall-counts-by-pid.py + create mode 100755 tools/perf/python/syscall-counts.py + create mode 100755 tools/perf/python/task-analyzer.py + create mode 100755 tools/perf/python/treport.py + create mode 100755 tools/perf/python/wakeup-latency.py + delete mode 100644 tools/perf/scripts/Build + delete mode 100644 tools/perf/scripts/perl/Perf-Trace-Util/Build + delete mode 100644 tools/perf/scripts/perl/Perf-Trace-Util/Context.c + delete mode 100644 tools/perf/scripts/perl/Perf-Trace-Util/Context.xs + delete mode 100644 tools/perf/scripts/perl/Perf-Trace-Util/Makefile.PL + delete mode 100644 tools/perf/scripts/perl/Perf-Trace-Util/README + delete mode 100644 tools/perf/scripts/perl/Perf-Trace-Util/lib/Perf/Trace/Context.pm + delete mode 100644 tools/perf/scripts/perl/Perf-Trace-Util/lib/Perf/Trace/Core.pm + delete mode 100644 tools/perf/scripts/perl/Perf-Trace-Util/lib/Perf/Trace/Util.pm + delete mode 100644 tools/perf/scripts/perl/Perf-Trace-Util/typemap + delete mode 100644 tools/perf/scripts/perl/bin/check-perf-trace-record + delete mode 100644 tools/perf/scripts/perl/bin/failed-syscalls-record + delete mode 100644 tools/perf/scripts/perl/bin/failed-syscalls-report + delete mode 100644 tools/perf/scripts/perl/bin/rw-by-file-record + delete mode 100644 tools/perf/scripts/perl/bin/rw-by-file-report + delete mode 100644 tools/perf/scripts/perl/bin/rw-by-pid-record + delete mode 100644 tools/perf/scripts/perl/bin/rw-by-pid-report + delete mode 100644 tools/perf/scripts/perl/bin/rwtop-record + delete mode 100644 tools/perf/scripts/perl/bin/rwtop-report + delete mode 100644 tools/perf/scripts/perl/bin/wakeup-latency-record + delete mode 100644 tools/perf/scripts/perl/bin/wakeup-latency-report + delete mode 100644 tools/perf/scripts/perl/check-perf-trace.pl + delete mode 100644 tools/perf/scripts/perl/failed-syscalls.pl + delete mode 100644 tools/perf/scripts/perl/rw-by-file.pl + delete mode 100644 tools/perf/scripts/perl/rw-by-pid.pl + delete mode 100644 tools/perf/scripts/perl/rwtop.pl + delete mode 100644 tools/perf/scripts/perl/wakeup-latency.pl + delete mode 100644 tools/perf/scripts/python/Perf-Trace-Util/Build + delete mode 100644 tools/perf/scripts/python/Perf-Trace-Util/Context.c + delete mode 100644 tools/perf/scripts/python/Perf-Trace-Util/lib/Perf/Trace/Core.py + delete mode 100755 tools/perf/scripts/python/Perf-Trace-Util/lib/Perf/Trace/EventClass.py + delete mode 100644 tools/perf/scripts/python/Perf-Trace-Util/lib/Perf/Trace/SchedGui.py + delete mode 100644 tools/perf/scripts/python/Perf-Trace-Util/lib/Perf/Trace/Util.py + delete mode 100755 tools/perf/scripts/python/arm-cs-trace-disasm.py + delete mode 100644 tools/perf/scripts/python/bin/compaction-times-record + delete mode 100644 tools/perf/scripts/python/bin/compaction-times-report + delete mode 100644 tools/perf/scripts/python/bin/event_analyzing_sample-record + delete mode 100644 tools/perf/scripts/python/bin/event_analyzing_sample-report + delete mode 100644 tools/perf/scripts/python/bin/export-to-postgresql-record + delete mode 100644 tools/perf/scripts/python/bin/export-to-postgresql-report + delete mode 100644 tools/perf/scripts/python/bin/export-to-sqlite-record + delete mode 100644 tools/perf/scripts/python/bin/export-to-sqlite-report + delete mode 100644 tools/perf/scripts/python/bin/failed-syscalls-by-pid-record + delete mode 100644 tools/perf/scripts/python/bin/failed-syscalls-by-pid-report + delete mode 100755 tools/perf/scripts/python/bin/flamegraph-record + delete mode 100755 tools/perf/scripts/python/bin/flamegraph-report + delete mode 100644 tools/perf/scripts/python/bin/futex-contention-record + delete mode 100644 tools/perf/scripts/python/bin/futex-contention-report + delete mode 100644 tools/perf/scripts/python/bin/gecko-record + delete mode 100755 tools/perf/scripts/python/bin/gecko-report + delete mode 100644 tools/perf/scripts/python/bin/intel-pt-events-record + delete mode 100644 tools/perf/scripts/python/bin/intel-pt-events-report + delete mode 100644 tools/perf/scripts/python/bin/mem-phys-addr-record + delete mode 100644 tools/perf/scripts/python/bin/mem-phys-addr-report + delete mode 100755 tools/perf/scripts/python/bin/net_dropmonitor-record + delete mode 100755 tools/perf/scripts/python/bin/net_dropmonitor-report + delete mode 100644 tools/perf/scripts/python/bin/netdev-times-record + delete mode 100644 tools/perf/scripts/python/bin/netdev-times-report + delete mode 100644 tools/perf/scripts/python/bin/powerpc-hcalls-record + delete mode 100644 tools/perf/scripts/python/bin/powerpc-hcalls-report + delete mode 100644 tools/perf/scripts/python/bin/sched-migration-record + delete mode 100644 tools/perf/scripts/python/bin/sched-migration-report + delete mode 100644 tools/perf/scripts/python/bin/sctop-record + delete mode 100644 tools/perf/scripts/python/bin/sctop-report + delete mode 100755 tools/perf/scripts/python/bin/stackcollapse-record + delete mode 100755 tools/perf/scripts/python/bin/stackcollapse-report + delete mode 100644 tools/perf/scripts/python/bin/syscall-counts-by-pid-record + delete mode 100644 tools/perf/scripts/python/bin/syscall-counts-by-pid-report + delete mode 100644 tools/perf/scripts/python/bin/syscall-counts-record + delete mode 100644 tools/perf/scripts/python/bin/syscall-counts-report + delete mode 100755 tools/perf/scripts/python/bin/task-analyzer-record + delete mode 100755 tools/perf/scripts/python/bin/task-analyzer-report + delete mode 100644 tools/perf/scripts/python/check-perf-trace.py + delete mode 100644 tools/perf/scripts/python/compaction-times.py + delete mode 100644 tools/perf/scripts/python/event_analyzing_sample.py + delete mode 100644 tools/perf/scripts/python/export-to-postgresql.py + delete mode 100644 tools/perf/scripts/python/export-to-sqlite.py + delete mode 100755 tools/perf/scripts/python/exported-sql-viewer.py + delete mode 100644 tools/perf/scripts/python/failed-syscalls-by-pid.py + delete mode 100755 tools/perf/scripts/python/flamegraph.py + delete mode 100644 tools/perf/scripts/python/futex-contention.py + delete mode 100644 tools/perf/scripts/python/gecko.py + delete mode 100644 tools/perf/scripts/python/intel-pt-events.py + delete mode 100644 tools/perf/scripts/python/libxed.py + delete mode 100644 tools/perf/scripts/python/mem-phys-addr.py + delete mode 100755 tools/perf/scripts/python/net_dropmonitor.py + delete mode 100644 tools/perf/scripts/python/netdev-times.py + delete mode 100755 tools/perf/scripts/python/parallel-perf.py + delete mode 100644 tools/perf/scripts/python/sched-migration.py + delete mode 100644 tools/perf/scripts/python/sctop.py + delete mode 100755 tools/perf/scripts/python/stackcollapse.py + delete mode 100644 tools/perf/scripts/python/stat-cpi.py + delete mode 100644 tools/perf/scripts/python/syscall-counts-by-pid.py + delete mode 100644 tools/perf/scripts/python/syscall-counts.py + delete mode 100755 tools/perf/scripts/python/task-analyzer.py + create mode 100644 tools/perf/tests/hybrid-merge.c + create mode 100755 tools/perf/tests/shell/annotate_weight.sh + create mode 100644 tools/perf/tests/shell/lib/perf_brstack_max.py + create mode 100755 tools/perf/tests/shell/report_hybrid_merge.sh + create mode 100755 tools/perf/tests/shell/schedstat_snapshots.sh + delete mode 100755 tools/perf/tests/shell/script_perl.sh + delete mode 100755 tools/perf/tests/shell/script_python.sh + create mode 100755 tools/perf/tests/shell/test_check_perf_trace_python.sh + create mode 100755 tools/perf/tests/shell/test_compaction_times_python.sh + create mode 100755 tools/perf/tests/shell/test_event_analyzing_sample_python.sh + create mode 100755 tools/perf/tests/shell/test_export_to_postgresql_python.sh + create mode 100755 tools/perf/tests/shell/test_export_to_sqlite_python.sh + create mode 100755 tools/perf/tests/shell/test_failed_syscalls_by_pid_python.sh + create mode 100755 tools/perf/tests/shell/test_failed_syscalls_python.sh + create mode 100755 tools/perf/tests/shell/test_flamegraph_python.sh + create mode 100755 tools/perf/tests/shell/test_futex_contention_python.sh + create mode 100755 tools/perf/tests/shell/test_gecko_python.sh + create mode 100755 tools/perf/tests/shell/test_intel_pt_events_python.sh + create mode 100755 tools/perf/tests/shell/test_mem_phys_addr_python.sh + create mode 100755 tools/perf/tests/shell/test_net_dropmonitor_python.sh + create mode 100755 tools/perf/tests/shell/test_netdev_times_python.sh + create mode 100755 tools/perf/tests/shell/test_powerpc_hcalls_python.sh + create mode 100755 tools/perf/tests/shell/test_rw_by_file_python.sh + create mode 100755 tools/perf/tests/shell/test_rw_by_pid_python.sh + create mode 100755 tools/perf/tests/shell/test_rwtop_python.sh + create mode 100755 tools/perf/tests/shell/test_sched_migration_python.sh + create mode 100755 tools/perf/tests/shell/test_sctop_python.sh + create mode 100755 tools/perf/tests/shell/test_stackcollapse_python.sh + create mode 100755 tools/perf/tests/shell/test_stat_cpi_python.sh + create mode 100755 tools/perf/tests/shell/test_syscall_counts_by_pid_python.sh + create mode 100755 tools/perf/tests/shell/test_syscall_counts_python.sh + create mode 100755 tools/perf/tests/shell/test_wakeup_latency_python.sh + create mode 100755 tools/perf/tests/shell/trace_ksym_beautifier.sh + create mode 100644 tools/perf/util/annotate-arch/annotate-alpha.c + create mode 100644 tools/perf/util/bpf_skel/perf_trace_u.h + delete mode 100644 tools/perf/util/cache.h + delete mode 100644 tools/perf/util/help-unknown-cmd.h + delete mode 100644 tools/perf/util/scripting-engines/Build + delete mode 100644 tools/perf/util/scripting-engines/trace-event-perl.c + delete mode 100644 tools/perf/util/scripting-engines/trace-event-python.c + delete mode 100644 tools/perf/util/trace-event-scripting.c + delete mode 100644 tools/perf/util/usage.c +Merging compiler-attributes/compiler-attributes (8f0b4cce4481f Linux 6.19-rc1) +$ git merge -m Merge branch 'compiler-attributes' of https://github.com/ojeda/linux.git compiler-attributes/compiler-attributes +Already up to date. +Merging dma-mapping/dma-mapping-for-next (57a57ae077f0f MAINTAINERS: Add files to the DMA MAPPING HELPERS entry) +$ git merge -m Merge branch 'dma-mapping-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mszyprowski/linux.git dma-mapping/dma-mapping-for-next +Auto-merging MAINTAINERS +Auto-merging drivers/iommu/dma-iommu.c +Auto-merging drivers/nvme/host/pci.c +Auto-merging mm/page_alloc.c +Merge made by the 'ort' strategy. + Documentation/core-api/dma-api.rst | 2 +- + MAINTAINERS | 1 + + drivers/iommu/dma-iommu.c | 12 +++++++----- + drivers/nvme/host/pci.c | 2 +- + drivers/scsi/scsi_transport_sas.c | 12 ++++++------ + include/linux/dma-map-ops.h | 4 ++-- + include/linux/dma-mapping.h | 4 ++-- + include/linux/gfp.h | 2 +- + include/linux/iommu-dma.h | 2 +- + kernel/dma/contiguous.c | 6 +++--- + kernel/dma/direct.c | 8 +++++--- + kernel/dma/mapping.c | 10 +++++----- + kernel/dma/ops_helpers.c | 11 +++++++---- + mm/page_alloc.c | 2 +- + 14 files changed, 43 insertions(+), 35 deletions(-) +Merging asm-generic/master (adbbd9714f805 scripts: headers_install.sh: Remove config leak ignore machinery) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/arnd/asm-generic asm-generic/master +Already up to date. +Merging alpha/alpha-next (d58041d2c63e0 MAINTAINERS: Add Magnus Lindholm as maintainer for alpha port) +$ git merge -m Merge branch 'alpha-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mattst88/alpha.git alpha/alpha-next +Already up to date. +Merging arm/for-next (1a89abc009cb5 Merge branches 'fixes' and 'misc' into for-linus) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/rmk/linux.git arm/for-next +Already up to date. +Merging arm64/for-next/core (e4e0f90fbf1bd Merge branches 'for-next/misc', 'for-next/be-gone', 'for-next/smccc-bus', 'for-next/preemptible-this-cpu', 'for-next/mpam' and 'for-next/fw-rmi' into for-next/core) +$ git merge -m Merge branch 'for-next/core' of https://git.kernel.org/pub/scm/linux/kernel/git/arm64/linux arm64/for-next/core +Auto-merging Documentation/admin-guide/kernel-parameters.txt +Auto-merging Documentation/arch/arm64/booting.rst +Auto-merging MAINTAINERS +Auto-merging arch/arm64/Kconfig +Auto-merging arch/arm64/include/asm/io.h +Auto-merging arch/arm64/kernel/cpufeature.c +Auto-merging arch/arm64/kernel/kexec_image.c +Auto-merging arch/arm64/kvm/sys_regs.c +Auto-merging arch/arm64/mm/fault.c +Auto-merging arch/arm64/mm/pageattr.c +CONFLICT (content): Merge conflict in arch/arm64/mm/pageattr.c +Auto-merging arch/arm64/net/bpf_jit_comp.c +Auto-merging kernel/power/hibernate.c +Resolved 'arch/arm64/mm/pageattr.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 71ebfe91dd28f] Merge branch 'for-next/core' of https://git.kernel.org/pub/scm/linux/kernel/git/arm64/linux +$ git diff -M --stat --summary HEAD^.. + Documentation/ABI/testing/sysfs-firmware-cca | 7 + + Documentation/admin-guide/kernel-parameters.txt | 3 +- + Documentation/arch/arm64/booting.rst | 2 +- + Documentation/arch/arm64/mpam.rst | 26 + + MAINTAINERS | 10 + + arch/arm/include/asm/archrandom.h | 2 +- + arch/arm64/Kconfig | 36 +- + arch/arm64/Makefile | 12 +- + arch/arm64/crypto/aes-ce-ccm-core.S | 4 +- + arch/arm64/crypto/aes-neonbs-core.S | 4 +- + arch/arm64/crypto/ghash-ce-core.S | 6 +- + arch/arm64/crypto/sm4-ce-cipher-core.S | 4 +- + arch/arm64/include/asm/alternative-macros.h | 30 +- + arch/arm64/include/asm/archrandom.h | 2 +- + arch/arm64/include/asm/assembler.h | 31 - + arch/arm64/include/asm/atomic_ll_sc.h | 27 +- + arch/arm64/include/asm/atomic_lse.h | 4 +- + arch/arm64/include/asm/compat.h | 14 - + arch/arm64/include/asm/elf.h | 12 - + arch/arm64/include/asm/gpr-num.h | 9 + + arch/arm64/include/asm/io.h | 2 +- + arch/arm64/include/asm/mem_encrypt.h | 5 +- + arch/arm64/include/asm/mte-kasan.h | 10 +- + arch/arm64/include/asm/percpu.h | 462 ++++++-- + arch/arm64/include/asm/pgtable-prot.h | 4 +- + arch/arm64/include/asm/preempt.h | 31 +- + arch/arm64/include/asm/processor.h | 2 - + arch/arm64/include/asm/ptrace.h | 18 +- + arch/arm64/include/asm/rsi.h | 70 -- + arch/arm64/include/asm/stage2_pgtable.h | 2 +- + arch/arm64/include/asm/sysreg.h | 52 +- + arch/arm64/include/asm/thread_info.h | 6 +- + arch/arm64/include/asm/word-at-a-time.h | 6 - + arch/arm64/include/asm/xwreg.h | 16 + + arch/arm64/include/uapi/asm/byteorder.h | 4 - + arch/arm64/include/uapi/asm/ptrace.h | 4 +- + arch/arm64/kernel/Makefile | 2 +- + arch/arm64/kernel/asm-offsets.c | 2 + + arch/arm64/kernel/cpufeature.c | 9 +- + arch/arm64/kernel/entry-common.c | 38 + + arch/arm64/kernel/entry.S | 41 +- + arch/arm64/kernel/fpsimd.c | 24 +- + arch/arm64/kernel/head.S | 3 +- + arch/arm64/kernel/hibernate.c | 14 + + arch/arm64/kernel/image.h | 18 +- + arch/arm64/kernel/kexec_image.c | 5 +- + arch/arm64/kernel/kgdb.c | 6 +- + arch/arm64/kernel/machine_kexec.c | 11 + + arch/arm64/kernel/ptrace.c | 8 +- + arch/arm64/kernel/setup.c | 2 +- + arch/arm64/kernel/signal32.c | 13 - + arch/arm64/kernel/sys32.c | 5 - + arch/arm64/kernel/vdso32/Makefile | 4 - + arch/arm64/kvm/hyp/include/nvhe/spinlock.h | 4 - + arch/arm64/kvm/hyp/nvhe/gen-hyprel.c | 16 - + arch/arm64/kvm/sys_regs.c | 2 + + arch/arm64/lib/csum.c | 17 - + arch/arm64/lib/memchr.S | 2 +- + arch/arm64/lib/memcmp.S | 2 - + arch/arm64/lib/strcmp.S | 34 +- + arch/arm64/lib/strlen.S | 26 - + arch/arm64/lib/strncmp.S | 93 +- + arch/arm64/lib/strnlen.S | 16 +- + arch/arm64/mm/extable.c | 5 - + arch/arm64/mm/fault.c | 35 +- + arch/arm64/mm/init.c | 3 +- + arch/arm64/mm/pageattr.c | 4 +- + arch/arm64/net/bpf_jit_comp.c | 11 - + drivers/acpi/arm64/mpam.c | 6 +- + drivers/char/hw_random/arm_smccc_trng.c | 32 +- + drivers/crypto/hisilicon/sec2/sec_main.c | 2 +- + drivers/firmware/Kconfig | 1 + + drivers/firmware/Makefile | 1 + + drivers/firmware/arm_rmm/Kconfig | 39 + + drivers/firmware/arm_rmm/Makefile | 3 + + drivers/firmware/arm_rmm/rmi.c | 1135 ++++++++++++++++++++ + .../kernel => drivers/firmware/arm_rmm}/rsi.c | 79 +- + drivers/firmware/smccc/Makefile | 2 +- + drivers/firmware/smccc/bus.c | 158 +++ + drivers/firmware/smccc/smccc.c | 66 +- + drivers/hv/Kconfig | 3 +- + drivers/misc/vmw_vmci/Kconfig | 2 +- + drivers/net/ethernet/google/Kconfig | 2 +- + drivers/net/ethernet/microsoft/Kconfig | 2 +- + drivers/resctrl/Kconfig | 2 + + drivers/resctrl/Makefile | 2 +- + drivers/resctrl/mpam_devices.c | 1051 +++++++++++++----- + drivers/resctrl/mpam_fb.c | 258 +++++ + drivers/resctrl/mpam_internal.h | 70 +- + drivers/resctrl/mpam_resctrl.c | 13 +- + drivers/soc/tegra/Kconfig | 5 - + drivers/virt/coco/arm-cca-guest/Kconfig | 3 +- + drivers/virt/coco/arm-cca-guest/Makefile | 2 + + .../coco/arm-cca-guest/{arm-cca-guest.c => main.c} | 52 +- + include/linux/arm-rmi-cmds.h | 581 ++++++++++ + .../asm/rsi_cmds.h => include/linux/arm-rsi-cmds.h | 76 +- + include/linux/arm-smccc-bus.h | 48 + + include/linux/arm-smccc-rmi.h | 516 +++++++++ + .../asm/rsi_smc.h => include/linux/arm-smccc-rsi.h | 6 +- + include/linux/arm_mpam.h | 2 +- + include/linux/device-id/arm_smccc.h | 19 + + include/linux/mod_devicetable.h | 1 + + include/linux/suspend.h | 1 + + kernel/power/hibernate.c | 8 +- + scripts/mod/devicetable-offsets.c | 3 + + scripts/mod/file2alias.c | 8 + + tools/testing/selftests/arm64/fp/fp-ptrace.c | 19 +- + .../testcases/fake_sigreturn_sve_change_vl.c | 2 +- + tools/testing/selftests/rseq/rseq-arm64.h | 5 - + 109 files changed, 4565 insertions(+), 1135 deletions(-) + create mode 100644 Documentation/ABI/testing/sysfs-firmware-cca + delete mode 100644 arch/arm64/include/asm/rsi.h + create mode 100644 arch/arm64/include/asm/xwreg.h + create mode 100644 drivers/firmware/arm_rmm/Kconfig + create mode 100644 drivers/firmware/arm_rmm/Makefile + create mode 100644 drivers/firmware/arm_rmm/rmi.c + rename {arch/arm64/kernel => drivers/firmware/arm_rmm}/rsi.c (71%) + create mode 100644 drivers/firmware/smccc/bus.c + create mode 100644 drivers/resctrl/mpam_fb.c + rename drivers/virt/coco/arm-cca-guest/{arm-cca-guest.c => main.c} (82%) + create mode 100644 include/linux/arm-rmi-cmds.h + rename arch/arm64/include/asm/rsi_cmds.h => include/linux/arm-rsi-cmds.h (69%) + create mode 100644 include/linux/arm-smccc-bus.h + create mode 100644 include/linux/arm-smccc-rmi.h + rename arch/arm64/include/asm/rsi_smc.h => include/linux/arm-smccc-rsi.h (98%) + create mode 100644 include/linux/device-id/arm_smccc.h +$ git am -3 ../patches/0001-arm64-fixup-merge-with-dropped-copy-of-code-getting-.patch +Applying: arm64: fixup merge with dropped copy of code getting kept +$ git reset HEAD^ +Unstaged changes after reset: +M arch/arm64/mm/pageattr.c +$ git add -A . +$ git commit -v -a --amend +warning: notes ref refs/notes/commits is invalid +[master cc1bfc5640278] Merge branch 'for-next/core' of https://git.kernel.org/pub/scm/linux/kernel/git/arm64/linux + Date: Sat Oct 3 00:33:53 2026 +0200 +Merging arm-perf/for-next/perf (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-next/perf' of https://git.kernel.org/pub/scm/linux/kernel/git/will/linux.git arm-perf/for-next/perf +Already up to date. +Merging arm-soc/for-next (bdb122e397cb0 soc: document merges) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/soc/soc.git arm-soc/for-next +Auto-merging .mailmap +Auto-merging MAINTAINERS +Auto-merging arch/arm/Kconfig +Auto-merging arch/arm/Kconfig.debug +Auto-merging arch/arm/configs/shmobile_defconfig +Auto-merging arch/arm/mm/init.c +Auto-merging scripts/mod/devicetable-offsets.c +Auto-merging scripts/mod/file2alias.c +Merge made by the 'ort' strategy. + .mailmap | 1 + + Documentation/arch/arm/index.rst | 9 - + Documentation/arch/arm/marvell.rst | 15 +- + Documentation/arch/arm/microchip.rst | 39 - + Documentation/arch/arm/netwinder.rst | 85 - + Documentation/arch/arm/sa1100/assabet.rst | 301 -- + Documentation/arch/arm/sa1100/cerf.rst | 35 - + Documentation/arch/arm/sa1100/index.rst | 13 - + Documentation/arch/arm/sa1100/lart.rst | 15 - + Documentation/arch/arm/sa1100/serial_uart.rst | 51 - + Documentation/arch/arm/stm32/overview.rst | 8 +- + .../arch/arm/stm32/stm32f429-overview.rst | 25 - + .../arch/arm/stm32/stm32f746-overview.rst | 32 - + .../arch/arm/stm32/stm32f769-overview.rst | 34 - + .../arch/arm/stm32/stm32h743-overview.rst | 33 - + .../arch/arm/stm32/stm32h750-overview.rst | 34 - + .../devicetree/bindings/arm/atmel-at91.yaml | 39 - + Documentation/devicetree/bindings/arm/axxia.yaml | 23 - + Documentation/devicetree/bindings/arm/cpus.yaml | 1 + + .../devicetree/bindings/arm/mediatek.yaml | 1 + + .../devicetree/bindings/arm/qcom-soc.yaml | 4 +- + Documentation/devicetree/bindings/arm/qcom.yaml | 15 + + .../devicetree/bindings/arm/rockchip.yaml | 7 + + .../bindings/clock/qcom,milos-camcc.yaml | 39 +- + .../bindings/clock/qcom,milos-videocc.yaml | 29 +- + .../bindings/clock/qcom,sm8450-gpucc.yaml | 3 + + .../devicetree/bindings/mmc/qcom,sdhci-msm.yaml | 1 + + .../bindings/net/bluetooth/qcom,wcn6750-bt.yaml | 10 +- + .../bindings/net/wireless/qcom,ath11k.yaml | 16 +- + MAINTAINERS | 51 +- + arch/arm/Kconfig | 63 +- + arch/arm/Kconfig.debug | 104 +- + arch/arm/Makefile | 15 - + arch/arm/arm-soc-for-next-contents.txt | 64 + + arch/arm/boot/bootp/Makefile | 2 - + arch/arm/boot/compressed/Makefile | 8 - + arch/arm/boot/compressed/head-sa1100.S | 45 - + arch/arm/boot/compressed/head-sharpsl.S | 151 - + arch/arm/boot/compressed/head.S | 7 - + arch/arm/boot/dts/arm/Makefile | 3 - + arch/arm/boot/dts/arm/mps2-an385.dts | 92 - + arch/arm/boot/dts/arm/mps2-an399.dts | 92 - + arch/arm/boot/dts/arm/mps2.dtsi | 220 - + arch/arm/boot/dts/intel/Makefile | 1 - + arch/arm/boot/dts/intel/axm/Makefile | 5 - + arch/arm/boot/dts/intel/axm/axm5516-amarillo.dts | 47 - + arch/arm/boot/dts/intel/axm/axm5516-cpus.dtsi | 200 - + arch/arm/boot/dts/intel/axm/axm55xx.dtsi | 200 - + arch/arm/boot/dts/mediatek/Makefile | 1 + + arch/arm/boot/dts/mediatek/mt8127-amazon-ford.dts | 54 + + arch/arm/boot/dts/mediatek/mt8127.dtsi | 7 + + arch/arm/boot/dts/nxp/imx/Makefile | 5 - + arch/arm/boot/dts/nxp/imx/imx31-bug.dts | 22 - + arch/arm/boot/dts/nxp/imx/imx31-lite.dts | 178 - + arch/arm/boot/dts/nxp/imx/imx31.dtsi | 371 -- + arch/arm/boot/dts/nxp/imx/imxrt1050-evk.dts | 72 - + arch/arm/boot/dts/nxp/imx/imxrt1050-pinfunc.h | 993 ----- + arch/arm/boot/dts/nxp/imx/imxrt1050.dtsi | 160 - + arch/arm/boot/dts/nxp/imx/imxrt1170-pinfunc.h | 1561 ------- + arch/arm/boot/dts/nxp/lpc/Makefile | 5 - + arch/arm/boot/dts/nxp/lpc/lpc18xx.dtsi | 543 --- + arch/arm/boot/dts/nxp/lpc/lpc4337-ciaa.dts | 221 - + arch/arm/boot/dts/nxp/lpc/lpc4350-hitex-eval.dts | 485 --- + arch/arm/boot/dts/nxp/lpc/lpc4350.dtsi | 48 - + .../arm/boot/dts/nxp/lpc/lpc4357-ea4357-devkit.dts | 624 --- + arch/arm/boot/dts/nxp/lpc/lpc4357-myd-lpc4357.dts | 621 --- + arch/arm/boot/dts/nxp/lpc/lpc4357.dtsi | 52 - + arch/arm/boot/dts/nxp/vf/Makefile | 2 - + arch/arm/boot/dts/nxp/vf/vf610m4-colibri.dts | 61 - + arch/arm/boot/dts/nxp/vf/vf610m4-cosmic.dts | 88 - + arch/arm/boot/dts/nxp/vf/vf610m4.dtsi | 61 - + arch/arm/boot/dts/renesas/r7s72100-gr-peach.dts | 2 + + .../boot/dts/renesas/r8a7740-armadillo800eva.dts | 2 + + arch/arm/boot/dts/renesas/r8a7743-sk-rzg1m.dts | 2 + + arch/arm/boot/dts/renesas/r8a7743.dtsi | 6 +- + arch/arm/boot/dts/renesas/r8a7744.dtsi | 6 +- + arch/arm/boot/dts/renesas/r8a7745-sk-rzg1e.dts | 2 + + arch/arm/boot/dts/renesas/r8a7745.dtsi | 6 +- + arch/arm/boot/dts/renesas/r8a77470.dtsi | 6 +- + arch/arm/boot/dts/renesas/r8a7790-lager.dts | 28 +- + arch/arm/boot/dts/renesas/r8a7790-stout.dts | 4 +- + arch/arm/boot/dts/renesas/r8a7790.dtsi | 4 +- + arch/arm/boot/dts/renesas/r8a7791-koelsch.dts | 2 + + arch/arm/boot/dts/renesas/r8a7791-porter.dts | 2 + + arch/arm/boot/dts/renesas/r8a7791.dtsi | 4 +- + arch/arm/boot/dts/renesas/r8a7792.dtsi | 4 +- + arch/arm/boot/dts/renesas/r8a7793-gose.dts | 2 + + arch/arm/boot/dts/renesas/r8a7793.dtsi | 4 +- + arch/arm/boot/dts/renesas/r8a7794-alt.dts | 2 + + arch/arm/boot/dts/renesas/r8a7794-silk.dts | 2 + + arch/arm/boot/dts/renesas/r8a7794.dtsi | 4 +- + .../arm/boot/dts/renesas/r9a06g032-rzn1d400-db.dts | 1 - + .../arm/boot/dts/renesas/r9a06g032-rzn1d400-eb.dts | 14 +- + arch/arm/boot/dts/renesas/r9a06g032.dtsi | 2 - + arch/arm/boot/dts/st/Makefile | 11 - + arch/arm/boot/dts/st/stm32429i-eval.dts | 349 -- + arch/arm/boot/dts/st/stm32746g-eval.dts | 236 - + arch/arm/boot/dts/st/stm32f4-pinctrl.dtsi | 482 -- + arch/arm/boot/dts/st/stm32f429-disco.dts | 243 -- + arch/arm/boot/dts/st/stm32f429-pinctrl.dtsi | 91 - + arch/arm/boot/dts/st/stm32f429.dtsi | 800 ---- + arch/arm/boot/dts/st/stm32f469-disco.dts | 260 -- + arch/arm/boot/dts/st/stm32f469-pinctrl.dtsi | 92 - + arch/arm/boot/dts/st/stm32f469.dtsi | 18 - + arch/arm/boot/dts/st/stm32f7-pinctrl.dtsi | 436 -- + arch/arm/boot/dts/st/stm32f746-disco.dts | 234 - + arch/arm/boot/dts/st/stm32f746-pinctrl.dtsi | 55 - + arch/arm/boot/dts/st/stm32f746.dtsi | 727 --- + .../boot/dts/st/stm32f769-disco-mb1166-reva09.dts | 13 - + arch/arm/boot/dts/st/stm32f769-disco.dts | 243 -- + arch/arm/boot/dts/st/stm32f769-pinctrl.dtsi | 55 - + arch/arm/boot/dts/st/stm32f769.dtsi | 37 - + arch/arm/boot/dts/st/stm32h7-pinctrl.dtsi | 301 -- + arch/arm/boot/dts/st/stm32h743.dtsi | 738 ---- + arch/arm/boot/dts/st/stm32h743i-disco.dts | 145 - + arch/arm/boot/dts/st/stm32h743i-eval.dts | 185 - + arch/arm/boot/dts/st/stm32h747i-disco.dts | 149 - + arch/arm/boot/dts/st/stm32h750.dtsi | 6 - + arch/arm/boot/dts/st/stm32h750i-art-pi.dts | 229 - + arch/arm/boot/dts/ti/omap/Makefile | 6 - + arch/arm/boot/dts/ti/omap/omap2.dtsi | 338 -- + arch/arm/boot/dts/ti/omap/omap2420-clocks.dtsi | 267 -- + arch/arm/boot/dts/ti/omap/omap2420-h4.dts | 63 - + arch/arm/boot/dts/ti/omap/omap2420-n800.dts | 9 - + arch/arm/boot/dts/ti/omap/omap2420-n810-wimax.dts | 9 - + arch/arm/boot/dts/ti/omap/omap2420-n810.dts | 75 - + .../arm/boot/dts/ti/omap/omap2420-n8x0-common.dtsi | 123 - + arch/arm/boot/dts/ti/omap/omap2420.dtsi | 262 -- + arch/arm/boot/dts/ti/omap/omap2430-clocks.dtsi | 341 -- + arch/arm/boot/dts/ti/omap/omap2430-sdp.dts | 70 - + arch/arm/boot/dts/ti/omap/omap2430.dtsi | 366 -- + arch/arm/boot/dts/ti/omap/omap24xx-clocks.dtsi | 1241 ------ + arch/arm/common/Kconfig | 13 - + arch/arm/common/Makefile | 4 - + arch/arm/common/locomo.c | 886 ---- + arch/arm/common/sa1111.c | 1413 ------ + arch/arm/common/scoop.c | 269 -- + arch/arm/common/sharpsl_param.c | 63 - + arch/arm/configs/am200epdkit_defconfig | 91 - + arch/arm/configs/assabet_defconfig | 57 - + arch/arm/configs/axm55xx_defconfig | 231 - + arch/arm/configs/collie_defconfig | 87 - + arch/arm/configs/dove_defconfig | 134 - + arch/arm/configs/footbridge_defconfig | 114 - + arch/arm/configs/h3600_defconfig | 63 - + arch/arm/configs/imx_v6_v7_defconfig | 1 - + arch/arm/configs/imxrt_defconfig | 35 - + arch/arm/configs/jornada720_defconfig | 95 - + arch/arm/configs/lpc18xx_defconfig | 158 - + arch/arm/configs/mps2_defconfig | 102 - + arch/arm/configs/mv78xx0_defconfig | 120 - + arch/arm/configs/neponset_defconfig | 88 - + arch/arm/configs/netwinder_defconfig | 88 - + arch/arm/configs/omap2plus_defconfig | 4 - + arch/arm/configs/shmobile_defconfig | 3 +- + arch/arm/configs/spitz_defconfig | 237 - + arch/arm/configs/stm32_defconfig | 80 - + arch/arm/configs/vf610m4_defconfig | 39 - + arch/arm/include/asm/cputype.h | 4 - + arch/arm/include/asm/hardware/dec21285.h | 138 - + arch/arm/include/asm/hardware/locomo.h | 217 - + arch/arm/include/asm/hardware/sa1111.h | 442 -- + arch/arm/include/asm/hardware/scoop.h | 67 - + arch/arm/include/asm/hardware/ssp.h | 25 - + arch/arm/include/asm/mach/pci.h | 86 - + arch/arm/include/asm/mach/sharpsl_param.h | 33 - + arch/arm/include/asm/pci.h | 3 - + arch/arm/include/debug/dc21285.S | 41 - + arch/arm/include/debug/imx-uart.h | 10 - + arch/arm/include/debug/sa1100.S | 67 - + arch/arm/kernel/atags_parse.c | 18 - + arch/arm/kernel/bios32.c | 254 -- + arch/arm/kernel/head.S | 10 - + arch/arm/mach-at91/Kconfig | 11 +- + arch/arm/mach-at91/Makefile | 1 - + arch/arm/mach-at91/samv7.c | 17 - + arch/arm/mach-axxia/Kconfig | 17 - + arch/arm/mach-axxia/Makefile | 3 - + arch/arm/mach-axxia/axxia.c | 19 - + arch/arm/mach-axxia/platsmp.c | 87 - + arch/arm/mach-dove/Kconfig | 32 - + arch/arm/mach-dove/Makefile | 7 - + arch/arm/mach-dove/bridge-regs.h | 50 - + arch/arm/mach-dove/cm-a510.c | 94 - + arch/arm/mach-dove/common.c | 449 -- + arch/arm/mach-dove/common.h | 46 - + arch/arm/mach-dove/dove.h | 185 - + arch/arm/mach-dove/irq.c | 81 - + arch/arm/mach-dove/irqs.h | 89 - + arch/arm/mach-dove/mpp.c | 159 - + arch/arm/mach-dove/mpp.h | 197 - + arch/arm/mach-dove/pcie.c | 226 - + arch/arm/mach-dove/pm.h | 58 - + arch/arm/mach-footbridge/Kconfig | 57 - + arch/arm/mach-footbridge/Makefile | 18 - + arch/arm/mach-footbridge/common.c | 280 -- + arch/arm/mach-footbridge/common.h | 15 - + arch/arm/mach-footbridge/dc21285-timer.c | 136 - + arch/arm/mach-footbridge/dc21285.c | 360 -- + arch/arm/mach-footbridge/dma-isa.c | 230 - + arch/arm/mach-footbridge/ebsa285-pci.c | 48 - + arch/arm/mach-footbridge/ebsa285.c | 124 - + arch/arm/mach-footbridge/include/mach/hardware.h | 90 - + arch/arm/mach-footbridge/include/mach/irqs.h | 97 - + arch/arm/mach-footbridge/include/mach/isa-dma.h | 18 - + arch/arm/mach-footbridge/include/mach/memory.h | 26 - + arch/arm/mach-footbridge/include/mach/uncompress.h | 34 - + arch/arm/mach-footbridge/isa-irq.c | 177 - + arch/arm/mach-footbridge/isa-rtc.c | 57 - + arch/arm/mach-footbridge/isa-timer.c | 36 - + arch/arm/mach-footbridge/isa.c | 94 - + arch/arm/mach-footbridge/netwinder-hw.c | 772 ---- + arch/arm/mach-footbridge/netwinder-pci.c | 62 - + arch/arm/mach-imx/Kconfig | 38 +- + arch/arm/mach-imx/Makefile | 4 - + arch/arm/mach-imx/common.h | 2 - + arch/arm/mach-imx/cpu-imx31.c | 72 - + arch/arm/mach-imx/mach-imx31.c | 18 - + arch/arm/mach-imx/mach-imx7d-cm4.c | 18 - + arch/arm/mach-imx/mach-imxrt.c | 19 - + arch/arm/mach-imx/mach-vf610.c | 1 - + arch/arm/mach-imx/mm-imx3.c | 44 - + arch/arm/mach-lpc18xx/Makefile | 2 - + arch/arm/mach-lpc18xx/board-dt.c | 19 - + arch/arm/mach-mv78xx0/Kconfig | 27 - + arch/arm/mach-mv78xx0/Makefile | 5 - + arch/arm/mach-mv78xx0/bridge-regs.h | 31 - + arch/arm/mach-mv78xx0/buffalo-wxl-setup.c | 183 - + arch/arm/mach-mv78xx0/common.c | 444 -- + arch/arm/mach-mv78xx0/common.h | 54 - + arch/arm/mach-mv78xx0/irq.c | 69 - + arch/arm/mach-mv78xx0/irqs.h | 87 - + arch/arm/mach-mv78xx0/mpp.c | 34 - + arch/arm/mach-mv78xx0/mpp.h | 337 -- + arch/arm/mach-mv78xx0/mv78xx0.h | 134 - + arch/arm/mach-mv78xx0/pcie.c | 280 -- + arch/arm/mach-mvebu/Makefile | 2 - + arch/arm/mach-omap2/Kconfig | 62 +- + arch/arm/mach-omap2/Makefile | 63 +- + arch/arm/mach-omap2/board-generic.c | 34 - + arch/arm/mach-omap2/board-n8x0.c | 512 --- + arch/arm/mach-omap2/clkt2xxx_dpll.c | 56 - + arch/arm/mach-omap2/clkt2xxx_dpllcore.c | 194 - + arch/arm/mach-omap2/clkt2xxx_virt_prcm_set.c | 259 -- + arch/arm/mach-omap2/clock.c | 12 +- + arch/arm/mach-omap2/clock2xxx.h | 17 - + arch/arm/mach-omap2/clockdomain.c | 4 +- + arch/arm/mach-omap2/clockdomain.h | 5 - + arch/arm/mach-omap2/clockdomains2420_data.c | 154 - + arch/arm/mach-omap2/clockdomains2430_data.c | 181 - + arch/arm/mach-omap2/clockdomains2xxx_3xxx_data.c | 29 +- + arch/arm/mach-omap2/cm-regbits-24xx.h | 51 - + arch/arm/mach-omap2/cm2xxx.c | 302 -- + arch/arm/mach-omap2/cm2xxx.h | 60 - + arch/arm/mach-omap2/cm_common.c | 12 - + arch/arm/mach-omap2/common-board-devices.h | 11 - + arch/arm/mach-omap2/common.h | 12 - + arch/arm/mach-omap2/control.h | 56 - + arch/arm/mach-omap2/display.c | 4 +- + arch/arm/mach-omap2/dma.c | 53 +- + arch/arm/mach-omap2/fb.c | 14 +- + arch/arm/mach-omap2/gpmc.h | 10 - + arch/arm/mach-omap2/i2c.c | 2 +- + arch/arm/mach-omap2/id.c | 80 +- + arch/arm/mach-omap2/io.c | 140 +- + arch/arm/mach-omap2/iomap.h | 46 - + arch/arm/mach-omap2/l3_2xxx.h | 15 - + arch/arm/mach-omap2/l4_2xxx.h | 19 - + arch/arm/mach-omap2/mmc.h | 18 - + arch/arm/mach-omap2/msdi.c | 75 - + arch/arm/mach-omap2/omap2-restart.c | 62 - + arch/arm/mach-omap2/omap24xx.h | 73 - + arch/arm/mach-omap2/omap_hwmod.c | 20 +- + arch/arm/mach-omap2/omap_hwmod.h | 2 - + arch/arm/mach-omap2/omap_hwmod_2420_data.c | 402 -- + arch/arm/mach-omap2/omap_hwmod_2430_data.c | 606 --- + .../mach-omap2/omap_hwmod_2xxx_interconnect_data.c | 256 -- + arch/arm/mach-omap2/omap_hwmod_2xxx_ipblock_data.c | 668 --- + arch/arm/mach-omap2/omap_hwmod_common_data.h | 66 - + arch/arm/mach-omap2/opp2420_data.c | 131 - + arch/arm/mach-omap2/opp2430_data.c | 136 - + arch/arm/mach-omap2/opp2xxx.h | 430 -- + arch/arm/mach-omap2/pdata-quirks.c | 37 +- + arch/arm/mach-omap2/pm.h | 5 - + arch/arm/mach-omap2/powerdomain.h | 3 - + arch/arm/mach-omap2/powerdomains2xxx_data.c | 134 - + arch/arm/mach-omap2/prcm-common.h | 147 - + arch/arm/mach-omap2/prm-regbits-24xx.h | 39 - + arch/arm/mach-omap2/prm2xxx.c | 229 - + arch/arm/mach-omap2/prm2xxx.h | 128 - + arch/arm/mach-omap2/prm2xxx_3xxx.c | 1 - + arch/arm/mach-omap2/prm_common.c | 11 - + arch/arm/mach-omap2/sdrc.h | 21 - + arch/arm/mach-omap2/sdrc2xxx.c | 164 - + arch/arm/mach-omap2/sleep24xx.S | 91 - + arch/arm/mach-omap2/soc.h | 59 - + arch/arm/mach-omap2/sram.c | 159 +- + arch/arm/mach-omap2/sram.h | 32 - + arch/arm/mach-omap2/sram242x.S | 317 -- + arch/arm/mach-omap2/sram243x.S | 317 -- + arch/arm/mach-omap2/usb-tusb6010.c | 217 - + arch/arm/mach-omap2/usb-tusb6010.h | 12 - + arch/arm/mach-omap2/voltage.h | 1 - + arch/arm/mach-omap2/voltagedomains2xxx_data.c | 29 - + arch/arm/mach-orion5x/Kconfig | 95 - + arch/arm/mach-orion5x/Makefile | 13 +- + arch/arm/mach-orion5x/board-d2net.c | 2 - + arch/arm/mach-orion5x/board-dt.c | 4 +- + arch/arm/mach-orion5x/board-mss2.c | 1 - + arch/arm/mach-orion5x/board-rd88f5182.c | 1 - + arch/arm/mach-orion5x/common.c | 238 +- + arch/arm/mach-orion5x/common.h | 34 +- + arch/arm/mach-orion5x/dns323-setup.c | 757 ---- + arch/arm/mach-orion5x/irq.c | 51 - + arch/arm/mach-orion5x/irqs.h | 9 - + arch/arm/mach-orion5x/kurobox_pro-setup.c | 407 -- + arch/arm/mach-orion5x/mpp.c | 41 - + arch/arm/mach-orion5x/mpp.h | 130 - + arch/arm/mach-orion5x/mv2120-setup.c | 256 -- + arch/arm/mach-orion5x/net2big-setup.c | 444 -- + arch/arm/mach-orion5x/pci.c | 429 +- + arch/arm/mach-orion5x/terastation_pro2-setup.c | 366 -- + arch/arm/mach-orion5x/ts209-setup.c | 331 -- + arch/arm/mach-orion5x/ts409-setup.c | 329 -- + arch/arm/mach-orion5x/ts78xx-fpga.h | 42 - + arch/arm/mach-orion5x/ts78xx-setup.c | 574 --- + arch/arm/mach-orion5x/tsx09-common.c | 129 - + arch/arm/mach-orion5x/tsx09-common.h | 21 - + arch/arm/mach-pxa/Kconfig | 89 - + arch/arm/mach-pxa/Makefile | 8 - + arch/arm/mach-pxa/am200epd.c | 395 -- + arch/arm/mach-pxa/am300epd.c | 296 -- + arch/arm/mach-pxa/devices.c | 705 --- + arch/arm/mach-pxa/devices.h | 68 - + arch/arm/mach-pxa/generic.h | 5 - + arch/arm/mach-pxa/gumstix.c | 237 - + arch/arm/mach-pxa/gumstix.h | 89 - + arch/arm/mach-pxa/pxa25x.c | 59 - + arch/arm/mach-pxa/pxa27x.c | 77 - + arch/arm/mach-pxa/pxa300.c | 1 - + arch/arm/mach-pxa/pxa320.c | 1 - + arch/arm/mach-pxa/pxa3xx.c | 6 - + arch/arm/mach-pxa/sharpsl_pm.c | 941 ---- + arch/arm/mach-pxa/sharpsl_pm.h | 108 - + arch/arm/mach-pxa/spitz.c | 1174 ----- + arch/arm/mach-pxa/spitz.h | 185 - + arch/arm/mach-pxa/spitz_pm.c | 257 -- + arch/arm/mach-sa1100/Kconfig | 92 - + arch/arm/mach-sa1100/Makefile | 20 - + arch/arm/mach-sa1100/assabet.c | 770 ---- + arch/arm/mach-sa1100/clock.c | 145 - + arch/arm/mach-sa1100/collie.c | 447 -- + arch/arm/mach-sa1100/generic.c | 470 -- + arch/arm/mach-sa1100/generic.h | 58 - + arch/arm/mach-sa1100/h3600.c | 134 - + arch/arm/mach-sa1100/h3xxx.c | 305 -- + arch/arm/mach-sa1100/include/mach/SA-1100.h | 1798 -------- + arch/arm/mach-sa1100/include/mach/assabet.h | 99 - + arch/arm/mach-sa1100/include/mach/bitfield.h | 113 - + arch/arm/mach-sa1100/include/mach/collie.h | 94 - + arch/arm/mach-sa1100/include/mach/generic.h | 1 - + arch/arm/mach-sa1100/include/mach/h3xxx.h | 81 - + arch/arm/mach-sa1100/include/mach/hardware.h | 56 - + arch/arm/mach-sa1100/include/mach/irqs.h | 101 - + arch/arm/mach-sa1100/include/mach/jornada720.h | 28 - + arch/arm/mach-sa1100/include/mach/memory.h | 37 - + arch/arm/mach-sa1100/include/mach/mtd-xip.h | 23 - + arch/arm/mach-sa1100/include/mach/neponset.h | 31 - + arch/arm/mach-sa1100/include/mach/reset.h | 18 - + arch/arm/mach-sa1100/include/mach/uncompress.h | 52 - + arch/arm/mach-sa1100/jornada720.c | 380 -- + arch/arm/mach-sa1100/jornada720_ssp.c | 202 - + arch/arm/mach-sa1100/neponset.c | 438 -- + arch/arm/mach-sa1100/pm.c | 128 - + arch/arm/mach-sa1100/sleep.S | 143 - + arch/arm/mach-sa1100/ssp.c | 240 - + arch/arm/mach-stm32/Kconfig | 38 +- + arch/arm/mach-stm32/board-dt.c | 7 - + arch/arm/mach-versatile/Kconfig | 10 - + arch/arm/mach-versatile/Makefile | 3 - + arch/arm/mach-versatile/v2m-mps2.c | 17 - + arch/arm/mm/Kconfig | 2 +- + arch/arm/mm/init.c | 5 - + arch/arm/mm/ioremap.c | 1 - + arch/arm/mm/mmu.c | 16 - + arch/arm/plat-orion/Makefile | 9 - + arch/arm/plat-orion/common.c | 826 ---- + arch/arm/plat-orion/gpio.c | 616 --- + arch/arm/plat-orion/include/plat/addr-map.h | 54 - + arch/arm/plat-orion/include/plat/common.h | 104 - + arch/arm/plat-orion/include/plat/irq.h | 15 - + arch/arm/plat-orion/include/plat/mpp.h | 34 - + arch/arm/plat-orion/include/plat/orion-gpio.h | 38 - + arch/arm/plat-orion/include/plat/pcie.h | 34 - + arch/arm/plat-orion/include/plat/time.h | 20 - + arch/arm/plat-orion/irq.c | 39 - + arch/arm/plat-orion/mpp.c | 82 - + arch/arm/plat-orion/pcie.c | 288 -- + arch/arm/plat-orion/time.c | 238 - + arch/arm64/boot/dts/arm/fvp-base-revc.dts | 42 +- + arch/arm64/boot/dts/freescale/fsl-lx216x.dtsi | 19 +- + .../boot/dts/freescale/imx8mp-var-dart-sonata.dts | 1 + + arch/arm64/boot/dts/mediatek/mt6878-pinfunc.h | 2 +- + arch/arm64/boot/dts/mediatek/mt6893-pinfunc.h | 2 +- + .../boot/dts/mediatek/mt7986a-bananapi-bpi-r3.dts | 13 + + arch/arm64/boot/dts/mediatek/mt7986a.dtsi | 1 + + .../mediatek/mt7988a-bananapi-bpi-r4-pro-cn13.dtso | 1 + + .../mediatek/mt7988a-bananapi-bpi-r4-pro-cn14.dtso | 1 + + .../mediatek/mt7988a-bananapi-bpi-r4-pro-cn15.dtso | 1 + + .../mediatek/mt7988a-bananapi-bpi-r4-pro-cn18.dtso | 1 + + .../dts/mediatek/mt7988a-bananapi-bpi-r4-pro.dtsi | 18 + + .../boot/dts/mediatek/mt7988a-bananapi-bpi-r4.dtsi | 13 + + arch/arm64/boot/dts/mediatek/mt8173-elm.dtsi | 5 + + arch/arm64/boot/dts/mediatek/mt8173.dtsi | 35 +- + arch/arm64/boot/dts/mediatek/mt8183-kukui.dtsi | 5 + + arch/arm64/boot/dts/mediatek/mt8186-corsola.dtsi | 5 + + arch/arm64/boot/dts/mediatek/mt8188-geralt.dtsi | 7 +- + arch/arm64/boot/dts/mediatek/mt8192-asurada.dtsi | 5 + + arch/arm64/boot/dts/mediatek/mt8195-cherry.dtsi | 5 + + arch/arm64/boot/dts/qcom/Makefile | 27 + + arch/arm64/boot/dts/qcom/agatti.dtsi | 76 +- + arch/arm64/boot/dts/qcom/eliza-cqs-som.dtsi | 15 + + arch/arm64/boot/dts/qcom/eliza-evk.dtsi | 243 ++ + arch/arm64/boot/dts/qcom/eliza-mtp.dts | 34 + + arch/arm64/boot/dts/qcom/eliza.dtsi | 774 +++- + .../dts/qcom/glymur-asus-zenbook-a14-ux3407na.dts | 1140 +++++ + .../dts/qcom/glymur-asus-zenbook-a16-ux3607oa.dts | 2 +- + arch/arm64/boot/dts/qcom/glymur-crd.dts | 8 + + arch/arm64/boot/dts/qcom/glymur-crd.dtsi | 13 +- + .../boot/dts/qcom/glymur-hp-elitebook-x-g2q.dts | 940 ++++ + .../boot/dts/qcom/glymur-lenovo-yoga-slim7x.dts | 1221 ++++++ + arch/arm64/boot/dts/qcom/glymur.dtsi | 396 +- + arch/arm64/boot/dts/qcom/hamoa-iot-evk.dts | 134 +- + arch/arm64/boot/dts/qcom/hamoa-iot-som.dtsi | 21 + + .../qcom/hamoa-lenovo-ideacentre-mini-01q8x10.dts | 21 + + arch/arm64/boot/dts/qcom/hamoa-pmics.dtsi | 1 + + arch/arm64/boot/dts/qcom/hamoa.dtsi | 40 +- + arch/arm64/boot/dts/qcom/ipq5018-rdp432-c2.dts | 10 + + arch/arm64/boot/dts/qcom/ipq5018.dtsi | 21 +- + arch/arm64/boot/dts/qcom/kodiak.dtsi | 1413 +++--- + .../boot/dts/qcom/lemans-evk-ifp-mezzanine.dtso | 12 +- + arch/arm64/boot/dts/qcom/lemans-pmics.dtsi | 116 + + arch/arm64/boot/dts/qcom/mahua-crd.dts | 5 + + arch/arm64/boot/dts/qcom/mahua.dtsi | 89 + + arch/arm64/boot/dts/qcom/milos-fairphone-fp6.dts | 191 + + arch/arm64/boot/dts/qcom/milos.dtsi | 86 +- + arch/arm64/boot/dts/qcom/monaco-ac-evk.dts | 938 ++++ + .../boot/dts/qcom/monaco-evk-ifp-mezzanine.dtso | 12 +- + arch/arm64/boot/dts/qcom/monaco-pmics.dtsi | 62 + + arch/arm64/boot/dts/qcom/monaco.dtsi | 1 - + arch/arm64/boot/dts/qcom/msm8917-xiaomi-riva.dts | 35 + + arch/arm64/boot/dts/qcom/nord-embedded.dtsi | 1820 ++++++++ + arch/arm64/boot/dts/qcom/nord-gearvm.dtsi | 2848 ++++++++++++ + arch/arm64/boot/dts/qcom/nord-ride.dts | 254 ++ + arch/arm64/boot/dts/qcom/nord-rrd.dts | 424 ++ + arch/arm64/boot/dts/qcom/nord.dtsi | 4605 ++++++++++++++++++++ + arch/arm64/boot/dts/qcom/purwa-iot-evk.dts | 133 +- + arch/arm64/boot/dts/qcom/purwa-iot-som.dtsi | 21 + + .../qcom/qcs6490-rb3gen2-industrial-mezzanine.dtso | 24 +- + .../dts/qcom/qcs6490-rb3gen2-vision-mezzanine.dtso | 13 +- + arch/arm64/boot/dts/qcom/qcs6490-rb3gen2.dts | 12 +- + .../dts/qcom/qcs6490-thundercomm-minipc-g1iot.dts | 12 +- + .../qcs6490-thundercomm-rubikpi3-cam1-imx219.dtso | 115 + + .../qcs6490-thundercomm-rubikpi3-cam2-imx219.dtso | 115 + + .../boot/dts/qcom/qcs6490-thundercomm-rubikpi3.dts | 23 + + arch/arm64/boot/dts/qcom/qcs8550-rb5gen2.dts | 12 +- + arch/arm64/boot/dts/qcom/sar2130p.dtsi | 1 - + arch/arm64/boot/dts/qcom/sc8280xp-crd.dts | 3 + + .../boot/dts/qcom/sc8280xp-huawei-gaokun3.dts | 3 + + .../dts/qcom/sc8280xp-lenovo-thinkpad-x13s.dts | 37 + + .../boot/dts/qcom/sc8280xp-microsoft-arcata.dts | 3 + + .../boot/dts/qcom/sc8280xp-microsoft-blackrock.dts | 3 + + arch/arm64/boot/dts/qcom/sc8280xp.dtsi | 574 ++- + arch/arm64/boot/dts/qcom/sdm670-google-common.dtsi | 42 + + arch/arm64/boot/dts/qcom/sdm845-google-common.dtsi | 36 +- + .../arm64/boot/dts/qcom/sdm845-oneplus-common.dtsi | 1 - + .../boot/dts/qcom/sdm845-sony-xperia-tama.dtsi | 2 + + arch/arm64/boot/dts/qcom/sdx75.dtsi | 2 - + arch/arm64/boot/dts/qcom/shikra-cqm-som.dtsi | 1 - + arch/arm64/boot/dts/qcom/shikra-evk.dtsi | 18 + + arch/arm64/boot/dts/qcom/shikra-iqs-som.dtsi | 1 - + arch/arm64/boot/dts/qcom/shikra.dtsi | 1490 ++++++- + arch/arm64/boot/dts/qcom/sm4450.dtsi | 1 - + arch/arm64/boot/dts/qcom/sm7225-fairphone-fp4.dts | 4 + + .../boot/dts/qcom/sm7325-nothing-spacewar.dts | 2 + + arch/arm64/boot/dts/qcom/sm8250.dtsi | 1 - + .../boot/dts/qcom/sm8350-sony-xperia-sagami.dtsi | 17 +- + arch/arm64/boot/dts/qcom/sm8350.dtsi | 1 - + arch/arm64/boot/dts/qcom/sm8450.dtsi | 1 - + arch/arm64/boot/dts/qcom/sm8550.dtsi | 2 - + .../boot/dts/qcom/sm8650-ayaneo-pocket-s2.dts | 232 +- + arch/arm64/boot/dts/qcom/sm8650.dtsi | 2 - + arch/arm64/boot/dts/qcom/sm8750.dtsi | 70 +- + .../boot/dts/qcom/talos-evk-rpi-display-2-5in.dtso | 88 + + arch/arm64/boot/dts/qcom/talos-evk-som.dtsi | 4 + + arch/arm64/boot/dts/qcom/talos.dtsi | 1 - + arch/arm64/boot/dts/qcom/x1-asus-vivobook-s15.dtsi | 21 + + arch/arm64/boot/dts/qcom/x1-asus-zenbook-a14.dtsi | 21 + + arch/arm64/boot/dts/qcom/x1-crd.dtsi | 21 + + arch/arm64/boot/dts/qcom/x1-dell-thena.dtsi | 21 + + arch/arm64/boot/dts/qcom/x1-hp-omnibook-x14.dtsi | 21 + + arch/arm64/boot/dts/qcom/x1-microsoft-denali.dtsi | 93 +- + arch/arm64/boot/dts/qcom/x1e001de-devkit.dts | 21 + + .../dts/qcom/x1e78100-lenovo-thinkpad-t14s.dtsi | 21 + + arch/arm64/boot/dts/qcom/x1e80100-crd.dts | 119 + + .../boot/dts/qcom/x1e80100-dell-xps13-9345.dts | 21 + + .../dts/qcom/x1e80100-honor-magicbook-art-14.dts | 21 + + .../boot/dts/qcom/x1e80100-lenovo-yoga-slim7x.dts | 21 + + .../dts/qcom/x1e80100-medion-sprchrgd-14-s1.dts | 25 +- + .../boot/dts/qcom/x1e80100-microsoft-romulus.dtsi | 21 + + arch/arm64/boot/dts/qcom/x1e80100-qcp.dts | 21 + + .../boot/dts/qcom/x1p42100-lenovo-thinkbook-16.dts | 21 + + .../boot/dts/qcom/x1p42100-microsoft-sp12in.dts | 21 + + arch/arm64/boot/dts/renesas/Makefile | 9 + + .../dts/renesas/aistarvision-mipi-adapter-2.1.dtsi | 2 +- + .../boot/dts/renesas/beacon-renesom-baseboard.dtsi | 6 +- + .../arm64/boot/dts/renesas/beacon-renesom-som.dtsi | 12 +- + arch/arm64/boot/dts/renesas/cat875.dtsi | 2 + + arch/arm64/boot/dts/renesas/condor-common.dtsi | 2 +- + arch/arm64/boot/dts/renesas/draak.dtsi | 2 +- + arch/arm64/boot/dts/renesas/ebisu.dtsi | 6 +- + arch/arm64/boot/dts/renesas/gray-hawk-single.dtsi | 8 +- + arch/arm64/boot/dts/renesas/hihope-common.dtsi | 4 +- + arch/arm64/boot/dts/renesas/hihope-rev2.dtsi | 8 +- + arch/arm64/boot/dts/renesas/hihope-rev4.dtsi | 4 +- + arch/arm64/boot/dts/renesas/hihope-rzg2-ex.dtsi | 6 +- + arch/arm64/boot/dts/renesas/r8a774a1.dtsi | 14 +- + arch/arm64/boot/dts/renesas/r8a774b1.dtsi | 12 +- + arch/arm64/boot/dts/renesas/r8a774c0-cat874.dts | 4 +- + arch/arm64/boot/dts/renesas/r8a774c0.dtsi | 12 +- + arch/arm64/boot/dts/renesas/r8a774e1.dtsi | 14 +- + arch/arm64/boot/dts/renesas/r8a77951.dtsi | 14 +- + arch/arm64/boot/dts/renesas/r8a77960.dtsi | 14 +- + arch/arm64/boot/dts/renesas/r8a77961.dtsi | 14 +- + arch/arm64/boot/dts/renesas/r8a77965.dtsi | 12 +- + .../renesas/r8a77970-eagle-function-expansion.dtso | 2 +- + arch/arm64/boot/dts/renesas/r8a77970-eagle.dts | 2 +- + arch/arm64/boot/dts/renesas/r8a77970-v3msk.dts | 2 +- + arch/arm64/boot/dts/renesas/r8a77970.dtsi | 2 +- + arch/arm64/boot/dts/renesas/r8a77980-v3hsk.dts | 2 +- + arch/arm64/boot/dts/renesas/r8a77980.dtsi | 4 +- + arch/arm64/boot/dts/renesas/r8a77990.dtsi | 10 +- + arch/arm64/boot/dts/renesas/r8a77995.dtsi | 6 +- + .../boot/dts/renesas/r8a779a0-falcon-cpu.dtsi | 2 +- + arch/arm64/boot/dts/renesas/r8a779a0-falcon.dts | 4 +- + arch/arm64/boot/dts/renesas/r8a779a0.dtsi | 2 +- + .../boot/dts/renesas/r8a779f0-spider-cpu.dtsi | 2 +- + arch/arm64/boot/dts/renesas/r8a779f0.dtsi | 36 +- + arch/arm64/boot/dts/renesas/r8a779f4-s4sk.dts | 2 +- + .../arm64/boot/dts/renesas/r8a779g0-white-hawk.dts | 94 + + arch/arm64/boot/dts/renesas/r8a779g0.dtsi | 69 +- + .../renesas/r8a779g3-sparrow-hawk-fan-argon40.dtso | 1 - + .../r8a779g3-sparrow-hawk-ws-2ch-canfd.dtso | 135 + + .../boot/dts/renesas/r8a779g3-sparrow-hawk.dts | 19 +- + arch/arm64/boot/dts/renesas/r8a779h0.dtsi | 26 +- + arch/arm64/boot/dts/renesas/r8a779md-geist.dts | 12 +- + arch/arm64/boot/dts/renesas/r8a78000.dtsi | 300 +- + .../arm64/boot/dts/renesas/r9a07g044l2-remi-pi.dts | 2 +- + arch/arm64/boot/dts/renesas/r9a08g045.dtsi | 5 +- + arch/arm64/boot/dts/renesas/r9a08g046.dtsi | 265 ++ + arch/arm64/boot/dts/renesas/r9a09g011.dtsi | 4 +- + arch/arm64/boot/dts/renesas/r9a09g047.dtsi | 2 +- + arch/arm64/boot/dts/renesas/r9a09g047e57-smarc.dts | 27 +- + arch/arm64/boot/dts/renesas/r9a09g056.dtsi | 2 +- + arch/arm64/boot/dts/renesas/r9a09g057.dtsi | 2 +- + .../boot/dts/renesas/r9a09g057h44-rzv2h-evk.dts | 6 +- + arch/arm64/boot/dts/renesas/r9a09g077.dtsi | 65 +- + .../dts/renesas/r9a09g077m44-evk-cn15-lcdc.dtso | 53 + + .../boot/dts/renesas/r9a09g077m44-rzt2h-evk.dts | 18 +- + arch/arm64/boot/dts/renesas/r9a09g087.dtsi | 65 +- + .../dts/renesas/r9a09g087m44-evk-cn20-lcdc.dtso | 63 + + .../boot/dts/renesas/r9a09g087m44-rzn2h-evk.dts | 32 +- + arch/arm64/boot/dts/renesas/renesas-smarc2.dtsi | 13 +- + .../boot/dts/renesas/rzg2l-smarc-pinfunction.dtsi | 16 +- + arch/arm64/boot/dts/renesas/rzg2l-smarc-som.dtsi | 20 +- + arch/arm64/boot/dts/renesas/rzg2l-smarc.dtsi | 2 +- + .../boot/dts/renesas/rzg2lc-smarc-pinfunction.dtsi | 16 +- + arch/arm64/boot/dts/renesas/rzg2lc-smarc-som.dtsi | 20 +- + arch/arm64/boot/dts/renesas/rzg2lc-smarc.dtsi | 2 +- + .../boot/dts/renesas/rzg2ul-smarc-pinfunction.dtsi | 16 +- + arch/arm64/boot/dts/renesas/rzg2ul-smarc-som.dtsi | 20 +- + arch/arm64/boot/dts/renesas/rzg3l-smarc-som.dtsi | 14 + + arch/arm64/boot/dts/renesas/rzg3s-smarc-som.dtsi | 25 +- + arch/arm64/boot/dts/renesas/rzg3s-smarc-switches.h | 4 + + arch/arm64/boot/dts/renesas/rzg3s-smarc.dtsi | 1 - + .../boot/dts/renesas/rzt2h-n2h-evk-common.dtsi | 7 + + .../boot/dts/renesas/rzt2h-n2h-evk-du-adv7513.dtsi | 72 + + arch/arm64/boot/dts/renesas/salvator-common.dtsi | 14 +- + arch/arm64/boot/dts/renesas/salvator-xs.dtsi | 2 +- + .../ulcb-kf-audio-graph-card2-mix+split.dtsi | 18 +- + arch/arm64/boot/dts/renesas/ulcb-kf.dtsi | 2 +- + arch/arm64/boot/dts/renesas/ulcb.dtsi | 8 +- + .../boot/dts/renesas/white-hawk-cpu-common.dtsi | 6 +- + arch/arm64/boot/dts/rockchip/Makefile | 6 + + .../boot/dts/rockchip/rk3399pro-vmarc-som.dtsi | 2 + + .../boot/dts/rockchip/rk3576-armsom-cm5-io.dts | 4 +- + arch/arm64/boot/dts/rockchip/rk3576.dtsi | 2 +- + arch/arm64/boot/dts/rockchip/rk3588-base.dtsi | 39 + + .../rockchip/rk3588-jaguar-can1-can2-uart4.dtso | 26 + + arch/arm64/boot/dts/rockchip/rk3588-jaguar.dts | 16 + + .../boot/dts/rockchip/rk3588-lubancat-5-btb.dtsi | 534 +++ + .../boot/dts/rockchip/rk3588-lubancat-5io.dts | 1032 +++++ + .../boot/dts/rockchip/rk3588-nanopc-t6-lts.dts | 17 - + arch/arm64/boot/dts/rockchip/rk3588-nanopc-t6.dtsi | 175 +- + .../boot/dts/rockchip/rk3588-tiger-haikou.dts | 4 + + arch/arm64/boot/dts/rockchip/rk3588-tiger.dtsi | 5 + + .../boot/dts/rockchip/rk3588s-gameforce-ace.dts | 7 +- + arch/arm64/configs/defconfig | 6 + + drivers/ata/Kconfig | 3 +- + drivers/clk/qcom/gcc-eliza.c | 6 +- + drivers/clk/qcom/gcc-hawi.c | 6 +- + drivers/clk/qcom/gcc-kaanapali.c | 6 +- + drivers/clk/qcom/gpucc-glymur.c | 2 +- + drivers/clk/qcom/gpucc-kaanapali.c | 2 +- + drivers/crypto/Kconfig | 2 +- + drivers/firmware/arm_scmi/bus.c | 109 +- + drivers/firmware/arm_scmi/common.h | 61 +- + drivers/firmware/arm_scmi/driver.c | 262 +- + drivers/firmware/arm_scmi/protocols.h | 9 +- + drivers/firmware/arm_scmi/raw_mode.c | 30 +- + drivers/firmware/arm_scmi/reset.c | 36 +- + drivers/firmware/arm_scmi/transports/Kconfig | 12 + + drivers/firmware/arm_scmi/transports/Makefile | 2 + + drivers/firmware/arm_scmi/transports/mailbox.c | 7 +- + drivers/firmware/arm_scmi/transports/optee.c | 7 +- + drivers/firmware/arm_scmi/transports/pcc.c | 875 ++++ + drivers/firmware/arm_scmi/transports/smc.c | 10 +- + drivers/firmware/arm_scmi/transports/virtio.c | 3 +- + drivers/i2c/busses/Kconfig | 2 +- + drivers/media/platform/ti/omap/Kconfig | 4 +- + drivers/mmc/host/Kconfig | 2 +- + drivers/mtd/nand/onenand/Kconfig | 4 +- + drivers/pci/controller/Kconfig | 2 +- + drivers/pci/controller/pci-mvebu.c | 5 - + drivers/phy/marvell/Kconfig | 2 +- + drivers/platform/cznic/turris-omnia-mcu-base.c | 1 + + drivers/rtc/Kconfig | 8 +- + drivers/rtc/rtc-sa1100.c | 17 +- + drivers/soc/Makefile | 1 - + drivers/soc/atmel/soc.c | 36 - + drivers/soc/atmel/soc.h | 26 - + drivers/soc/mediatek/mt8167-mmsys.h | 183 +- + drivers/soc/mediatek/mtk-regulator-coupler.c | 1 + + drivers/soc/renesas/Kconfig | 19 +- + drivers/soc/renesas/r9a08g046-sysc.c | 1 + + drivers/soc/renesas/rcar-mfis.c | 62 +- + drivers/soc/renesas/rcar-rst.c | 1 + + drivers/soc/renesas/renesas-soc.c | 3 + + drivers/soc/renesas/rz-sysc.c | 5 + + drivers/soc/renesas/rz-sysc.h | 2 + + drivers/tee/optee/ffa_abi.c | 38 +- + drivers/tee/optee/optee_msg.h | 4 +- + drivers/tee/optee/smc_abi.c | 17 +- + drivers/thermal/Kconfig | 2 +- + drivers/video/console/Kconfig | 6 +- + drivers/video/console/dummycon.c | 6 - + drivers/video/fbdev/cyber2000fb.c | 10 - + drivers/video/fbdev/omap2/omapfb/Kconfig | 2 +- + drivers/watchdog/Kconfig | 14 +- + include/dt-bindings/arm/qcom,ids.h | 1 + + .../dt-bindings/clock/qcom,eliza-cambistmclkcc.h | 32 + + include/dt-bindings/clock/qcom,eliza-camcc.h | 151 + + include/dt-bindings/clock/qcom,eliza-gpucc.h | 51 + + include/dt-bindings/clock/qcom,eliza-videocc.h | 37 + + include/linux/clk/ti.h | 8 - + include/linux/device-id/scmi.h | 17 + + include/linux/scmi_protocol.h | 20 +- + include/trace/events/scmi.h | 8 +- + scripts/mod/devicetable-offsets.c | 5 + + scripts/mod/file2alias.c | 12 + + sound/soc/kirkwood/Kconfig | 2 +- + 671 files changed, 26791 insertions(+), 57356 deletions(-) + delete mode 100644 Documentation/arch/arm/netwinder.rst + delete mode 100644 Documentation/arch/arm/sa1100/assabet.rst + delete mode 100644 Documentation/arch/arm/sa1100/cerf.rst + delete mode 100644 Documentation/arch/arm/sa1100/index.rst + delete mode 100644 Documentation/arch/arm/sa1100/lart.rst + delete mode 100644 Documentation/arch/arm/sa1100/serial_uart.rst + delete mode 100644 Documentation/arch/arm/stm32/stm32f429-overview.rst + delete mode 100644 Documentation/arch/arm/stm32/stm32f746-overview.rst + delete mode 100644 Documentation/arch/arm/stm32/stm32f769-overview.rst + delete mode 100644 Documentation/arch/arm/stm32/stm32h743-overview.rst + delete mode 100644 Documentation/arch/arm/stm32/stm32h750-overview.rst + delete mode 100644 Documentation/devicetree/bindings/arm/axxia.yaml + create mode 100644 arch/arm/arm-soc-for-next-contents.txt + delete mode 100644 arch/arm/boot/compressed/head-sa1100.S + delete mode 100644 arch/arm/boot/compressed/head-sharpsl.S + delete mode 100644 arch/arm/boot/dts/arm/mps2-an385.dts + delete mode 100644 arch/arm/boot/dts/arm/mps2-an399.dts + delete mode 100644 arch/arm/boot/dts/arm/mps2.dtsi + delete mode 100644 arch/arm/boot/dts/intel/axm/Makefile + delete mode 100644 arch/arm/boot/dts/intel/axm/axm5516-amarillo.dts + delete mode 100644 arch/arm/boot/dts/intel/axm/axm5516-cpus.dtsi + delete mode 100644 arch/arm/boot/dts/intel/axm/axm55xx.dtsi + create mode 100644 arch/arm/boot/dts/mediatek/mt8127-amazon-ford.dts + delete mode 100644 arch/arm/boot/dts/nxp/imx/imx31-bug.dts + delete mode 100644 arch/arm/boot/dts/nxp/imx/imx31-lite.dts + delete mode 100644 arch/arm/boot/dts/nxp/imx/imx31.dtsi + delete mode 100644 arch/arm/boot/dts/nxp/imx/imxrt1050-evk.dts + delete mode 100644 arch/arm/boot/dts/nxp/imx/imxrt1050-pinfunc.h + delete mode 100644 arch/arm/boot/dts/nxp/imx/imxrt1050.dtsi + delete mode 100644 arch/arm/boot/dts/nxp/imx/imxrt1170-pinfunc.h + delete mode 100644 arch/arm/boot/dts/nxp/lpc/lpc18xx.dtsi + delete mode 100644 arch/arm/boot/dts/nxp/lpc/lpc4337-ciaa.dts + delete mode 100644 arch/arm/boot/dts/nxp/lpc/lpc4350-hitex-eval.dts + delete mode 100644 arch/arm/boot/dts/nxp/lpc/lpc4350.dtsi + delete mode 100644 arch/arm/boot/dts/nxp/lpc/lpc4357-ea4357-devkit.dts + delete mode 100644 arch/arm/boot/dts/nxp/lpc/lpc4357-myd-lpc4357.dts + delete mode 100644 arch/arm/boot/dts/nxp/lpc/lpc4357.dtsi + delete mode 100644 arch/arm/boot/dts/nxp/vf/vf610m4-colibri.dts + delete mode 100644 arch/arm/boot/dts/nxp/vf/vf610m4-cosmic.dts + delete mode 100644 arch/arm/boot/dts/nxp/vf/vf610m4.dtsi + delete mode 100644 arch/arm/boot/dts/st/stm32429i-eval.dts + delete mode 100644 arch/arm/boot/dts/st/stm32746g-eval.dts + delete mode 100644 arch/arm/boot/dts/st/stm32f4-pinctrl.dtsi + delete mode 100644 arch/arm/boot/dts/st/stm32f429-disco.dts + delete mode 100644 arch/arm/boot/dts/st/stm32f429-pinctrl.dtsi + delete mode 100644 arch/arm/boot/dts/st/stm32f429.dtsi + delete mode 100644 arch/arm/boot/dts/st/stm32f469-disco.dts + delete mode 100644 arch/arm/boot/dts/st/stm32f469-pinctrl.dtsi + delete mode 100644 arch/arm/boot/dts/st/stm32f469.dtsi + delete mode 100644 arch/arm/boot/dts/st/stm32f7-pinctrl.dtsi + delete mode 100644 arch/arm/boot/dts/st/stm32f746-disco.dts + delete mode 100644 arch/arm/boot/dts/st/stm32f746-pinctrl.dtsi + delete mode 100644 arch/arm/boot/dts/st/stm32f746.dtsi + delete mode 100644 arch/arm/boot/dts/st/stm32f769-disco-mb1166-reva09.dts + delete mode 100644 arch/arm/boot/dts/st/stm32f769-disco.dts + delete mode 100644 arch/arm/boot/dts/st/stm32f769-pinctrl.dtsi + delete mode 100644 arch/arm/boot/dts/st/stm32f769.dtsi + delete mode 100644 arch/arm/boot/dts/st/stm32h7-pinctrl.dtsi + delete mode 100644 arch/arm/boot/dts/st/stm32h743.dtsi + delete mode 100644 arch/arm/boot/dts/st/stm32h743i-disco.dts + delete mode 100644 arch/arm/boot/dts/st/stm32h743i-eval.dts + delete mode 100644 arch/arm/boot/dts/st/stm32h747i-disco.dts + delete mode 100644 arch/arm/boot/dts/st/stm32h750.dtsi + delete mode 100644 arch/arm/boot/dts/st/stm32h750i-art-pi.dts + delete mode 100644 arch/arm/boot/dts/ti/omap/omap2.dtsi + delete mode 100644 arch/arm/boot/dts/ti/omap/omap2420-clocks.dtsi + delete mode 100644 arch/arm/boot/dts/ti/omap/omap2420-h4.dts + delete mode 100644 arch/arm/boot/dts/ti/omap/omap2420-n800.dts + delete mode 100644 arch/arm/boot/dts/ti/omap/omap2420-n810-wimax.dts + delete mode 100644 arch/arm/boot/dts/ti/omap/omap2420-n810.dts + delete mode 100644 arch/arm/boot/dts/ti/omap/omap2420-n8x0-common.dtsi + delete mode 100644 arch/arm/boot/dts/ti/omap/omap2420.dtsi + delete mode 100644 arch/arm/boot/dts/ti/omap/omap2430-clocks.dtsi + delete mode 100644 arch/arm/boot/dts/ti/omap/omap2430-sdp.dts + delete mode 100644 arch/arm/boot/dts/ti/omap/omap2430.dtsi + delete mode 100644 arch/arm/boot/dts/ti/omap/omap24xx-clocks.dtsi + delete mode 100644 arch/arm/common/locomo.c + delete mode 100644 arch/arm/common/sa1111.c + delete mode 100644 arch/arm/common/scoop.c + delete mode 100644 arch/arm/common/sharpsl_param.c + delete mode 100644 arch/arm/configs/am200epdkit_defconfig + delete mode 100644 arch/arm/configs/assabet_defconfig + delete mode 100644 arch/arm/configs/axm55xx_defconfig + delete mode 100644 arch/arm/configs/collie_defconfig + delete mode 100644 arch/arm/configs/dove_defconfig + delete mode 100644 arch/arm/configs/footbridge_defconfig + delete mode 100644 arch/arm/configs/h3600_defconfig + delete mode 100644 arch/arm/configs/imxrt_defconfig + delete mode 100644 arch/arm/configs/jornada720_defconfig + delete mode 100644 arch/arm/configs/lpc18xx_defconfig + delete mode 100644 arch/arm/configs/mps2_defconfig + delete mode 100644 arch/arm/configs/mv78xx0_defconfig + delete mode 100644 arch/arm/configs/neponset_defconfig + delete mode 100644 arch/arm/configs/netwinder_defconfig + delete mode 100644 arch/arm/configs/spitz_defconfig + delete mode 100644 arch/arm/configs/stm32_defconfig + delete mode 100644 arch/arm/configs/vf610m4_defconfig + delete mode 100644 arch/arm/include/asm/hardware/dec21285.h + delete mode 100644 arch/arm/include/asm/hardware/locomo.h + delete mode 100644 arch/arm/include/asm/hardware/sa1111.h + delete mode 100644 arch/arm/include/asm/hardware/scoop.h + delete mode 100644 arch/arm/include/asm/hardware/ssp.h + delete mode 100644 arch/arm/include/asm/mach/pci.h + delete mode 100644 arch/arm/include/asm/mach/sharpsl_param.h + delete mode 100644 arch/arm/include/debug/dc21285.S + delete mode 100644 arch/arm/include/debug/sa1100.S + delete mode 100644 arch/arm/mach-at91/samv7.c + delete mode 100644 arch/arm/mach-axxia/Kconfig + delete mode 100644 arch/arm/mach-axxia/Makefile + delete mode 100644 arch/arm/mach-axxia/axxia.c + delete mode 100644 arch/arm/mach-axxia/platsmp.c + delete mode 100644 arch/arm/mach-dove/Kconfig + delete mode 100644 arch/arm/mach-dove/Makefile + delete mode 100644 arch/arm/mach-dove/bridge-regs.h + delete mode 100644 arch/arm/mach-dove/cm-a510.c + delete mode 100644 arch/arm/mach-dove/common.c + delete mode 100644 arch/arm/mach-dove/common.h + delete mode 100644 arch/arm/mach-dove/dove.h + delete mode 100644 arch/arm/mach-dove/irq.c + delete mode 100644 arch/arm/mach-dove/irqs.h + delete mode 100644 arch/arm/mach-dove/mpp.c + delete mode 100644 arch/arm/mach-dove/mpp.h + delete mode 100644 arch/arm/mach-dove/pcie.c + delete mode 100644 arch/arm/mach-dove/pm.h + delete mode 100644 arch/arm/mach-footbridge/Kconfig + delete mode 100644 arch/arm/mach-footbridge/Makefile + delete mode 100644 arch/arm/mach-footbridge/common.c + delete mode 100644 arch/arm/mach-footbridge/common.h + delete mode 100644 arch/arm/mach-footbridge/dc21285-timer.c + delete mode 100644 arch/arm/mach-footbridge/dc21285.c + delete mode 100644 arch/arm/mach-footbridge/dma-isa.c + delete mode 100644 arch/arm/mach-footbridge/ebsa285-pci.c + delete mode 100644 arch/arm/mach-footbridge/ebsa285.c + delete mode 100644 arch/arm/mach-footbridge/include/mach/hardware.h + delete mode 100644 arch/arm/mach-footbridge/include/mach/irqs.h + delete mode 100644 arch/arm/mach-footbridge/include/mach/isa-dma.h + delete mode 100644 arch/arm/mach-footbridge/include/mach/memory.h + delete mode 100644 arch/arm/mach-footbridge/include/mach/uncompress.h + delete mode 100644 arch/arm/mach-footbridge/isa-irq.c + delete mode 100644 arch/arm/mach-footbridge/isa-rtc.c + delete mode 100644 arch/arm/mach-footbridge/isa-timer.c + delete mode 100644 arch/arm/mach-footbridge/isa.c + delete mode 100644 arch/arm/mach-footbridge/netwinder-hw.c + delete mode 100644 arch/arm/mach-footbridge/netwinder-pci.c + delete mode 100644 arch/arm/mach-imx/cpu-imx31.c + delete mode 100644 arch/arm/mach-imx/mach-imx31.c + delete mode 100644 arch/arm/mach-imx/mach-imx7d-cm4.c + delete mode 100644 arch/arm/mach-imx/mach-imxrt.c + delete mode 100644 arch/arm/mach-lpc18xx/Makefile + delete mode 100644 arch/arm/mach-lpc18xx/board-dt.c + delete mode 100644 arch/arm/mach-mv78xx0/Kconfig + delete mode 100644 arch/arm/mach-mv78xx0/Makefile + delete mode 100644 arch/arm/mach-mv78xx0/bridge-regs.h + delete mode 100644 arch/arm/mach-mv78xx0/buffalo-wxl-setup.c + delete mode 100644 arch/arm/mach-mv78xx0/common.c + delete mode 100644 arch/arm/mach-mv78xx0/common.h + delete mode 100644 arch/arm/mach-mv78xx0/irq.c + delete mode 100644 arch/arm/mach-mv78xx0/irqs.h + delete mode 100644 arch/arm/mach-mv78xx0/mpp.c + delete mode 100644 arch/arm/mach-mv78xx0/mpp.h + delete mode 100644 arch/arm/mach-mv78xx0/mv78xx0.h + delete mode 100644 arch/arm/mach-mv78xx0/pcie.c + delete mode 100644 arch/arm/mach-omap2/board-n8x0.c + delete mode 100644 arch/arm/mach-omap2/clkt2xxx_dpll.c + delete mode 100644 arch/arm/mach-omap2/clkt2xxx_dpllcore.c + delete mode 100644 arch/arm/mach-omap2/clkt2xxx_virt_prcm_set.c + delete mode 100644 arch/arm/mach-omap2/clock2xxx.h + delete mode 100644 arch/arm/mach-omap2/clockdomains2420_data.c + delete mode 100644 arch/arm/mach-omap2/clockdomains2430_data.c + delete mode 100644 arch/arm/mach-omap2/cm-regbits-24xx.h + delete mode 100644 arch/arm/mach-omap2/cm2xxx.c + delete mode 100644 arch/arm/mach-omap2/cm2xxx.h + delete mode 100644 arch/arm/mach-omap2/common-board-devices.h + delete mode 100644 arch/arm/mach-omap2/gpmc.h + delete mode 100644 arch/arm/mach-omap2/l3_2xxx.h + delete mode 100644 arch/arm/mach-omap2/l4_2xxx.h + delete mode 100644 arch/arm/mach-omap2/mmc.h + delete mode 100644 arch/arm/mach-omap2/msdi.c + delete mode 100644 arch/arm/mach-omap2/omap2-restart.c + delete mode 100644 arch/arm/mach-omap2/omap24xx.h + delete mode 100644 arch/arm/mach-omap2/omap_hwmod_2420_data.c + delete mode 100644 arch/arm/mach-omap2/omap_hwmod_2430_data.c + delete mode 100644 arch/arm/mach-omap2/omap_hwmod_2xxx_interconnect_data.c + delete mode 100644 arch/arm/mach-omap2/omap_hwmod_2xxx_ipblock_data.c + delete mode 100644 arch/arm/mach-omap2/opp2420_data.c + delete mode 100644 arch/arm/mach-omap2/opp2430_data.c + delete mode 100644 arch/arm/mach-omap2/opp2xxx.h + delete mode 100644 arch/arm/mach-omap2/powerdomains2xxx_data.c + delete mode 100644 arch/arm/mach-omap2/prm-regbits-24xx.h + delete mode 100644 arch/arm/mach-omap2/prm2xxx.c + delete mode 100644 arch/arm/mach-omap2/prm2xxx.h + delete mode 100644 arch/arm/mach-omap2/sdrc2xxx.c + delete mode 100644 arch/arm/mach-omap2/sleep24xx.S + delete mode 100644 arch/arm/mach-omap2/sram242x.S + delete mode 100644 arch/arm/mach-omap2/sram243x.S + delete mode 100644 arch/arm/mach-omap2/usb-tusb6010.c + delete mode 100644 arch/arm/mach-omap2/usb-tusb6010.h + delete mode 100644 arch/arm/mach-omap2/voltagedomains2xxx_data.c + delete mode 100644 arch/arm/mach-orion5x/dns323-setup.c + delete mode 100644 arch/arm/mach-orion5x/irq.c + delete mode 100644 arch/arm/mach-orion5x/kurobox_pro-setup.c + delete mode 100644 arch/arm/mach-orion5x/mpp.c + delete mode 100644 arch/arm/mach-orion5x/mpp.h + delete mode 100644 arch/arm/mach-orion5x/mv2120-setup.c + delete mode 100644 arch/arm/mach-orion5x/net2big-setup.c + delete mode 100644 arch/arm/mach-orion5x/terastation_pro2-setup.c + delete mode 100644 arch/arm/mach-orion5x/ts209-setup.c + delete mode 100644 arch/arm/mach-orion5x/ts409-setup.c + delete mode 100644 arch/arm/mach-orion5x/ts78xx-fpga.h + delete mode 100644 arch/arm/mach-orion5x/ts78xx-setup.c + delete mode 100644 arch/arm/mach-orion5x/tsx09-common.c + delete mode 100644 arch/arm/mach-orion5x/tsx09-common.h + delete mode 100644 arch/arm/mach-pxa/am200epd.c + delete mode 100644 arch/arm/mach-pxa/am300epd.c + delete mode 100644 arch/arm/mach-pxa/devices.h + delete mode 100644 arch/arm/mach-pxa/gumstix.c + delete mode 100644 arch/arm/mach-pxa/gumstix.h + delete mode 100644 arch/arm/mach-pxa/sharpsl_pm.c + delete mode 100644 arch/arm/mach-pxa/sharpsl_pm.h + delete mode 100644 arch/arm/mach-pxa/spitz.c + delete mode 100644 arch/arm/mach-pxa/spitz.h + delete mode 100644 arch/arm/mach-pxa/spitz_pm.c + delete mode 100644 arch/arm/mach-sa1100/Kconfig + delete mode 100644 arch/arm/mach-sa1100/Makefile + delete mode 100644 arch/arm/mach-sa1100/assabet.c + delete mode 100644 arch/arm/mach-sa1100/clock.c + delete mode 100644 arch/arm/mach-sa1100/collie.c + delete mode 100644 arch/arm/mach-sa1100/generic.c + delete mode 100644 arch/arm/mach-sa1100/generic.h + delete mode 100644 arch/arm/mach-sa1100/h3600.c + delete mode 100644 arch/arm/mach-sa1100/h3xxx.c + delete mode 100644 arch/arm/mach-sa1100/include/mach/SA-1100.h + delete mode 100644 arch/arm/mach-sa1100/include/mach/assabet.h + delete mode 100644 arch/arm/mach-sa1100/include/mach/bitfield.h + delete mode 100644 arch/arm/mach-sa1100/include/mach/collie.h + delete mode 100644 arch/arm/mach-sa1100/include/mach/generic.h + delete mode 100644 arch/arm/mach-sa1100/include/mach/h3xxx.h + delete mode 100644 arch/arm/mach-sa1100/include/mach/hardware.h + delete mode 100644 arch/arm/mach-sa1100/include/mach/irqs.h + delete mode 100644 arch/arm/mach-sa1100/include/mach/jornada720.h + delete mode 100644 arch/arm/mach-sa1100/include/mach/memory.h + delete mode 100644 arch/arm/mach-sa1100/include/mach/mtd-xip.h + delete mode 100644 arch/arm/mach-sa1100/include/mach/neponset.h + delete mode 100644 arch/arm/mach-sa1100/include/mach/reset.h + delete mode 100644 arch/arm/mach-sa1100/include/mach/uncompress.h + delete mode 100644 arch/arm/mach-sa1100/jornada720.c + delete mode 100644 arch/arm/mach-sa1100/jornada720_ssp.c + delete mode 100644 arch/arm/mach-sa1100/neponset.c + delete mode 100644 arch/arm/mach-sa1100/pm.c + delete mode 100644 arch/arm/mach-sa1100/sleep.S + delete mode 100644 arch/arm/mach-sa1100/ssp.c + delete mode 100644 arch/arm/mach-versatile/v2m-mps2.c + delete mode 100644 arch/arm/plat-orion/Makefile + delete mode 100644 arch/arm/plat-orion/common.c + delete mode 100644 arch/arm/plat-orion/gpio.c + delete mode 100644 arch/arm/plat-orion/include/plat/addr-map.h + delete mode 100644 arch/arm/plat-orion/include/plat/common.h + delete mode 100644 arch/arm/plat-orion/include/plat/irq.h + delete mode 100644 arch/arm/plat-orion/include/plat/mpp.h + delete mode 100644 arch/arm/plat-orion/include/plat/orion-gpio.h + delete mode 100644 arch/arm/plat-orion/include/plat/pcie.h + delete mode 100644 arch/arm/plat-orion/include/plat/time.h + delete mode 100644 arch/arm/plat-orion/irq.c + delete mode 100644 arch/arm/plat-orion/mpp.c + delete mode 100644 arch/arm/plat-orion/pcie.c + delete mode 100644 arch/arm/plat-orion/time.c + create mode 100644 arch/arm64/boot/dts/qcom/glymur-asus-zenbook-a14-ux3407na.dts + create mode 100644 arch/arm64/boot/dts/qcom/glymur-hp-elitebook-x-g2q.dts + create mode 100644 arch/arm64/boot/dts/qcom/glymur-lenovo-yoga-slim7x.dts + create mode 100644 arch/arm64/boot/dts/qcom/monaco-ac-evk.dts + create mode 100644 arch/arm64/boot/dts/qcom/nord-embedded.dtsi + create mode 100644 arch/arm64/boot/dts/qcom/nord-gearvm.dtsi + create mode 100644 arch/arm64/boot/dts/qcom/nord-ride.dts + create mode 100644 arch/arm64/boot/dts/qcom/nord-rrd.dts + create mode 100644 arch/arm64/boot/dts/qcom/nord.dtsi + create mode 100644 arch/arm64/boot/dts/qcom/qcs6490-thundercomm-rubikpi3-cam1-imx219.dtso + create mode 100644 arch/arm64/boot/dts/qcom/qcs6490-thundercomm-rubikpi3-cam2-imx219.dtso + create mode 100644 arch/arm64/boot/dts/qcom/talos-evk-rpi-display-2-5in.dtso + create mode 100644 arch/arm64/boot/dts/renesas/r8a779g3-sparrow-hawk-ws-2ch-canfd.dtso + create mode 100644 arch/arm64/boot/dts/renesas/r9a09g077m44-evk-cn15-lcdc.dtso + create mode 100644 arch/arm64/boot/dts/renesas/r9a09g087m44-evk-cn20-lcdc.dtso + create mode 100644 arch/arm64/boot/dts/renesas/rzt2h-n2h-evk-du-adv7513.dtsi + create mode 100644 arch/arm64/boot/dts/rockchip/rk3588-jaguar-can1-can2-uart4.dtso + create mode 100644 arch/arm64/boot/dts/rockchip/rk3588-lubancat-5-btb.dtsi + create mode 100644 arch/arm64/boot/dts/rockchip/rk3588-lubancat-5io.dts + create mode 100644 drivers/firmware/arm_scmi/transports/pcc.c + create mode 100644 include/dt-bindings/clock/qcom,eliza-cambistmclkcc.h + create mode 100644 include/dt-bindings/clock/qcom,eliza-camcc.h + create mode 100644 include/dt-bindings/clock/qcom,eliza-gpucc.h + create mode 100644 include/dt-bindings/clock/qcom,eliza-videocc.h + create mode 100644 include/linux/device-id/scmi.h +Merging amlogic/for-next (8610d31587e13 Merge branch 'v7.4/arm64-dt' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/amlogic/linux.git amlogic/for-next +Merge made by the 'ort' strategy. + .../soc/amlogic/amlogic,meson-gx-clk-measure.yaml | 16 +- + arch/arm64/boot/dts/amlogic/amlogic-a4.dtsi | 5 + + arch/arm64/boot/dts/amlogic/amlogic-a5.dtsi | 5 + + arch/arm64/boot/dts/amlogic/amlogic-a9.dtsi | 5 + + arch/arm64/boot/dts/amlogic/amlogic-s6.dtsi | 5 + + arch/arm64/boot/dts/amlogic/amlogic-s7.dtsi | 5 + + arch/arm64/boot/dts/amlogic/amlogic-s7d.dtsi | 5 + + .../dts/amlogic/amlogic-t7-a311d2-khadas-vim4.dts | 13 + + arch/arm64/boot/dts/amlogic/amlogic-t7.dtsi | 25 +- + arch/arm64/boot/dts/amlogic/meson-g12-common.dtsi | 9 + + .../dts/amlogic/meson-g12b-odroid-go-ultra.dts | 2 +- + .../boot/dts/amlogic/meson-gxbb-nanopi-k2.dts | 2 +- + arch/arm64/boot/dts/amlogic/meson-gxbb.dtsi | 2 +- + arch/arm64/boot/dts/amlogic/meson-gxl.dtsi | 2 +- + .../boot/dts/amlogic/meson-sm1-odroid-hc4.dts | 4 +- + drivers/soc/amlogic/meson-clk-measure.c | 927 ++++++++++++++++++++- + 16 files changed, 1000 insertions(+), 32 deletions(-) +Merging asahi-soc/asahi-soc/for-next (9378cd5ddcebc Merge branch 'apple-soc/drivers-7.4' into asahi-soc/for-next) +$ git merge -m Merge branch 'asahi-soc/for-next' of https://github.com/AsahiLinux/linux.git asahi-soc/asahi-soc/for-next +Auto-merging Documentation/devicetree/bindings/arm/cpus.yaml +Merge made by the 'ort' strategy. + Documentation/devicetree/bindings/arm/apple.yaml | 32 + + .../devicetree/bindings/arm/apple/apple,pmgr.yaml | 2 + + Documentation/devicetree/bindings/arm/cpus.yaml | 4 + + .../devicetree/bindings/i2c/apple,i2c.yaml | 1 + + .../bindings/interrupt-controller/apple,aic2.yaml | 2 + + .../devicetree/bindings/pinctrl/apple,pinctrl.yaml | 1 + + .../bindings/power/apple,pmgr-pwrstate.yaml | 2 + + .../devicetree/bindings/pwm/apple,s5l-fpwm.yaml | 1 + + arch/arm64/Kconfig.platforms | 1 + + arch/arm64/boot/dts/apple/Makefile | 7 + + arch/arm64/boot/dts/apple/t6001.dtsi | 11 +- + arch/arm64/boot/dts/apple/t6002.dtsi | 18 +- + arch/arm64/boot/dts/apple/t600x-die0.dtsi | 3 +- + arch/arm64/boot/dts/apple/t600x-dieX.dtsi | 2 + + arch/arm64/boot/dts/apple/t600x-nvme.dtsi | 2 + + arch/arm64/boot/dts/apple/t6021.dtsi | 10 +- + arch/arm64/boot/dts/apple/t6022.dtsi | 18 +- + arch/arm64/boot/dts/apple/t602x-common.dtsi | 2 +- + arch/arm64/boot/dts/apple/t602x-die0.dtsi | 2 + + arch/arm64/boot/dts/apple/t602x-dieX.dtsi | 2 + + arch/arm64/boot/dts/apple/t602x-nvme.dtsi | 2 + + arch/arm64/boot/dts/apple/t6030-pmgr.dtsi | 1 - + arch/arm64/boot/dts/apple/t6030.dtsi | 277 ++--- + arch/arm64/boot/dts/apple/t6031-base.dtsi | 191 ++-- + arch/arm64/boot/dts/apple/t6031-die0.dtsi | 63 +- + arch/arm64/boot/dts/apple/t6031-dieX.dtsi | 94 +- + arch/arm64/boot/dts/apple/t6031-gpio-pins.dtsi | 18 +- + arch/arm64/boot/dts/apple/t6031-pmgr.dtsi | 6 + + arch/arm64/boot/dts/apple/t6031.dtsi | 16 +- + arch/arm64/boot/dts/apple/t6032-j575d.dts | 10 +- + arch/arm64/boot/dts/apple/t6032.dtsi | 193 ++-- + arch/arm64/boot/dts/apple/t603x-j514-j516.dtsi | 15 +- + arch/arm64/boot/dts/apple/t8122-j504.dts | 10 +- + arch/arm64/boot/dts/apple/t8122-j613.dts | 9 +- + arch/arm64/boot/dts/apple/t8122-j615.dts | 9 +- + arch/arm64/boot/dts/apple/t8122-jxxx.dtsi | 5 +- + arch/arm64/boot/dts/apple/t8122-pmgr.dtsi | 1 - + arch/arm64/boot/dts/apple/t8122.dtsi | 256 +++-- + arch/arm64/boot/dts/apple/t8132-j604.dts | 35 + + arch/arm64/boot/dts/apple/t8132-j623.dts | 18 + + arch/arm64/boot/dts/apple/t8132-j624.dts | 18 + + arch/arm64/boot/dts/apple/t8132-j713.dts | 35 + + arch/arm64/boot/dts/apple/t8132-j715.dts | 35 + + arch/arm64/boot/dts/apple/t8132-j773g.dts | 25 + + arch/arm64/boot/dts/apple/t8132-jxxx.dtsi | 48 + + arch/arm64/boot/dts/apple/t8132-pmgr.dtsi | 1130 ++++++++++++++++++++ + arch/arm64/boot/dts/apple/t8132.dtsi | 467 ++++++++ + arch/arm64/boot/dts/apple/t8140-j700.dts | 54 + + arch/arm64/boot/dts/apple/t8140-pmgr.dtsi | 777 ++++++++++++++ + arch/arm64/boot/dts/apple/t8140.dtsi | 207 ++++ + drivers/soc/apple/rtkit-crashlog.c | 141 ++- + 51 files changed, 3715 insertions(+), 574 deletions(-) + create mode 100644 arch/arm64/boot/dts/apple/t8132-j604.dts + create mode 100644 arch/arm64/boot/dts/apple/t8132-j623.dts + create mode 100644 arch/arm64/boot/dts/apple/t8132-j624.dts + create mode 100644 arch/arm64/boot/dts/apple/t8132-j713.dts + create mode 100644 arch/arm64/boot/dts/apple/t8132-j715.dts + create mode 100644 arch/arm64/boot/dts/apple/t8132-j773g.dts + create mode 100644 arch/arm64/boot/dts/apple/t8132-jxxx.dtsi + create mode 100644 arch/arm64/boot/dts/apple/t8132-pmgr.dtsi + create mode 100644 arch/arm64/boot/dts/apple/t8132.dtsi + create mode 100644 arch/arm64/boot/dts/apple/t8140-j700.dts + create mode 100644 arch/arm64/boot/dts/apple/t8140-pmgr.dtsi + create mode 100644 arch/arm64/boot/dts/apple/t8140.dtsi +Merging at91/at91-next (bd49a1c72abbd Merge branch 'microchip-dt64' into at91-next) +$ git merge -m Merge branch 'at91-next' of https://git.kernel.org/pub/scm/linux/kernel/git/at91/linux.git at91/at91-next +Auto-merging drivers/soc/atmel/soc.c +Merge made by the 'ort' strategy. + arch/arm/mach-at91/pm.h | 2 +- + arch/arm64/boot/dts/microchip/lan9696-ev23x71a.dts | 13 +++++++++++++ + drivers/clk/at91/clk-main.c | 2 +- + drivers/soc/atmel/soc.c | 4 +--- + 4 files changed, 16 insertions(+), 5 deletions(-) +Merging bmc/for-next (cd7d1ef7d74ed Merge branches 'aspeed/drivers', 'aspeed/arm/dt', 'aspeed/fixes/drivers', 'aspeed/maintainers', 'nuvoton/arm/dt', 'nuvoton/arm/fixes' and 'nuvoton/arm64/dt' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/bmc/linux.git bmc/for-next +Merge made by the 'ort' strategy. +Merging broadcom/next (c0f2033a9ce4b Merge branch 'devicetree-arm64/next' into next) +$ git merge -m Merge branch 'next' of https://github.com/Broadcom/stblinux.git broadcom/next +Merge made by the 'ort' strategy. + .../devicetree/bindings/arm/bcm/bcm2835.yaml | 6 ++ + arch/arm/boot/dts/broadcom/bcm-ns.dtsi | 13 +++ + .../dts/broadcom/bcm4708-linksys-ea6300-v1.dts | 4 + + .../boot/dts/broadcom/bcm4708-smartrg-sr400ac.dts | 18 ++++ + .../boot/dts/broadcom/bcm4709-linksys-ea9200.dts | 62 +++++++++++ + .../boot/dts/broadcom/bcm4709-netgear-r8000.dts | 12 +++ + .../boot/dts/broadcom/bcm47094-dlink-dir-890l.dts | 6 +- + .../dts/broadcom/bcm47094-linksys-panamera.dts | 2 +- + arch/arm/boot/dts/broadcom/bcm47189-tenda-ac9.dts | 20 ++++ + arch/arm/boot/dts/broadcom/bcm53573.dtsi | 9 +- + arch/arm/boot/dts/broadcom/bcm7445.dtsi | 2 +- + arch/arm64/boot/dts/broadcom/bcm2712-d-rpi-5-b.dts | 5 + + .../boot/dts/broadcom/bcm2712-rpi-5-b-base.dtsi | 41 +++++++ + arch/arm64/boot/dts/broadcom/bcm2712.dtsi | 77 +++++++++++++ + arch/arm64/boot/dts/broadcom/rp1-common.dtsi | 119 +++++++++++++++++++++ + 15 files changed, 392 insertions(+), 4 deletions(-) +Merging cix/for-next (a0cffbd8878c5 Merge remote-tracking branch 'cix/dt' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/peter.chen/cix.git cix/for-next +Merge made by the 'ort' strategy. +Merging davinci/davinci/for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'davinci/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git davinci/davinci/for-next +Already up to date. +Merging drivers-memory/for-next (a22355280361d memory: stm32_omm: fix child clock leak on set_amcr() error path) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-mem-ctrl.git drivers-memory/for-next +Merge made by the 'ort' strategy. + .../memory-controllers/mediatek,smi-common.yaml | 21 +- + .../memory-controllers/mediatek,smi-larb.yaml | 3 + + drivers/memory/brcmstb_dpfe.c | 41 ++- + drivers/memory/emif.c | 6 +- + drivers/memory/emif.h | 4 +- + drivers/memory/mtk-smi.c | 45 +++ + drivers/memory/renesas-rpc-if.c | 87 ++--- + drivers/memory/samsung/exynos5422-dmc.c | 11 +- + drivers/memory/stm32_omm.c | 2 + + drivers/memory/tegra/tegra264.c | 392 +++++++++++++-------- + 10 files changed, 398 insertions(+), 214 deletions(-) +Merging fsl/soc_fsl (df0fd0f5af4dd bus: fsl-mc: Annotate fsl_mc_io.portal_virt_addr with __counted_by_ptr) +$ git merge -m Merge branch 'soc_fsl' of https://git.kernel.org/pub/scm/linux/kernel/git/chleroy/linux.git fsl/soc_fsl +Merge made by the 'ort' strategy. + drivers/bus/fsl-mc/fsl-mc-bus.c | 64 +++++++++++++++++++++++++++------------ + drivers/bus/fsl-mc/fsl-mc-msi.c | 2 +- + drivers/soc/fsl/qe/gpio.c | 67 ++++++++++++++++------------------------- + drivers/soc/fsl/qe/tsa.c | 8 ++--- + drivers/usb/host/fhci-hcd.c | 2 +- + include/linux/fsl/mc.h | 2 +- + include/soc/fsl/qe/qe.h | 6 ++-- + 7 files changed, 82 insertions(+), 69 deletions(-) +Merging imx-mxs/for-next (31f9ec9f38735 Merge branches 'imx/dt' and 'imx/dt64' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/frank.li/linux.git imx-mxs/for-next +Auto-merging arch/arm/boot/dts/nxp/imx/Makefile +Auto-merging arch/arm/boot/dts/nxp/vf/Makefile +CONFLICT (content): Merge conflict in arch/arm/boot/dts/nxp/vf/Makefile +Resolved 'arch/arm/boot/dts/nxp/vf/Makefile' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master db5be32e3c1f1] Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/frank.li/linux.git +$ git diff -M --stat --summary HEAD^.. + Documentation/ABI/testing/se-cdev | 45 + + Documentation/devicetree/bindings/arm/fsl.yaml | 70 +- + .../bindings/display/bridge/fsl,ldb.yaml | 23 +- + .../devicetree/bindings/display/fsl,lcdif.yaml | 1 + + .../devicetree/bindings/firmware/fsl,imx-se.yaml | 91 + + .../devicetree/bindings/mfd/st,stmpe.yaml | 25 +- + .../bindings/soc/imx/fsl,imx-iomuxc-gpr.yaml | 62 + + .../devicetree/bindings/vendor-prefixes.yaml | 2 + + .../driver-api/firmware/other_interfaces.rst | 238 ++ + arch/arm/boot/dts/nxp/imx/Makefile | 51 + + arch/arm/boot/dts/nxp/imx/e60k02.dtsi | 8 +- + arch/arm/boot/dts/nxp/imx/e70k02.dtsi | 8 +- + arch/arm/boot/dts/nxp/imx/imx50-amazon-yoshi.dts | 59 + + arch/arm/boot/dts/nxp/imx/imx50-kobo-aura.dts | 7 +- + arch/arm/boot/dts/nxp/imx/imx51-zii-rdu1.dts | 2 +- + arch/arm/boot/dts/nxp/imx/imx53-m53.dtsi | 3 - + arch/arm/boot/dts/nxp/imx/imx53-qsb.dts | 2 +- + .../arm/boot/dts/nxp/imx/imx53-voipac-dmm-668.dtsi | 2 +- + arch/arm/boot/dts/nxp/imx/imx6dl-b125pv2.dts | 2 + + arch/arm/boot/dts/nxp/imx/imx6dl-b125v2.dts | 2 + + arch/arm/boot/dts/nxp/imx/imx6dl-mamoj.dts | 2 +- + .../dts/nxp/imx/imx6dl-phytec-mira-rdk-emmc.dts | 75 + + .../dts/nxp/imx/imx6dl-phytec-mira-rdk-nand.dts | 2 +- + .../dts/nxp/imx/imx6dl-phytec-phycore-som.dtsi | 23 + + arch/arm/boot/dts/nxp/imx/imx6dl-plym2m.dts | 2 +- + arch/arm/boot/dts/nxp/imx/imx6dl-prtvt7.dts | 2 +- + arch/arm/boot/dts/nxp/imx/imx6dl-victgo.dts | 2 +- + arch/arm/boot/dts/nxp/imx/imx6dl.dtsi | 2 +- + arch/arm/boot/dts/nxp/imx/imx6q-bosch-acc.dts | 4 +- + arch/arm/boot/dts/nxp/imx/imx6q-novena.dts | 3 - + .../dts/nxp/imx/imx6q-phytec-mira-rdk-emmc.dts | 2 +- + .../dts/nxp/imx/imx6q-phytec-mira-rdk-nand.dts | 2 +- + .../boot/dts/nxp/imx/imx6q-phytec-phycore-som.dtsi | 75 + + arch/arm/boot/dts/nxp/imx/imx6qdl-apalis.dtsi | 3 - + arch/arm/boot/dts/nxp/imx/imx6qdl-colibri.dtsi | 3 - + arch/arm/boot/dts/nxp/imx/imx6qdl-emcon.dtsi | 1 - + arch/arm/boot/dts/nxp/imx/imx6qdl-gw560x.dtsi | 20 +- + arch/arm/boot/dts/nxp/imx/imx6qdl-pico.dtsi | 4 +- + arch/arm/boot/dts/nxp/imx/imx6qdl-skov-cpu.dtsi | 2 +- + arch/arm/boot/dts/nxp/imx/imx6qdl-wandboard.dtsi | 4 +- + arch/arm/boot/dts/nxp/imx/imx6qdl-zii-rdu2.dtsi | 10 +- + arch/arm/boot/dts/nxp/imx/imx6qdl.dtsi | 10 +- + arch/arm/boot/dts/nxp/imx/imx6sl-kobo-aura2.dts | 8 +- + .../boot/dts/nxp/imx/imx6sl-tolino-shine2hd.dts | 8 +- + arch/arm/boot/dts/nxp/imx/imx6sll-evk.dts | 2 +- + arch/arm/boot/dts/nxp/imx/imx6sll.dtsi | 11 - + arch/arm/boot/dts/nxp/imx/imx6ul-14x14-evk.dtsi | 1 - + .../boot/dts/nxp/imx/imx6ul-kontron-bl-common.dtsi | 38 +- + arch/arm/boot/dts/nxp/imx/imx6ul-tx6ul.dtsi | 2 +- + arch/arm/boot/dts/nxp/imx/imx6ull-colibri.dtsi | 1 + + .../arm/boot/dts/nxp/imx/imx7-colibri-iris-v2.dtsi | 1 - + arch/arm/boot/dts/nxp/imx/imx7-colibri.dtsi | 1 + + arch/arm/boot/dts/nxp/imx/imx7-tqma7.dtsi | 52 +- + ...nel-cap-touch-7inch-parallel-touch-adapter.dtso | 58 + + ...olibri-emmc-panel-cap-touch-7inch-parallel.dtso | 43 + + ...olibri-emmc-panel-res-touch-7inch-parallel.dtso | 32 + + .../boot/dts/nxp/imx/imx7d-colibri-emmc-vga.dtso | 59 + + arch/arm/boot/dts/nxp/imx/imx7d-meerkat96.dts | 2 +- + arch/arm/boot/dts/nxp/imx/imx7d-nitrogen7.dts | 1 - + .../nxp/imx/imx7d-var-som-emmc-mx7customboard.dts | 23 + + .../imx7d-var-som-emmc-wm8731-mx7customboard.dts | 23 + + arch/arm/boot/dts/nxp/imx/imx7d-var-som-emmc.dtsi | 56 + + .../dts/nxp/imx/imx7d-var-som-mx7customboard.dtsi | 358 +++ + .../nxp/imx/imx7d-var-som-nand-mx7customboard.dts | 23 + + .../imx7d-var-som-nand-wm8731-mx7customboard.dts | 23 + + arch/arm/boot/dts/nxp/imx/imx7d-var-som-nand.dtsi | 73 + + .../imx/imx7d-var-som-v2-emmc-mx7customboard.dts | 24 + + .../imx/imx7d-var-som-v2-nand-mx7customboard.dts | 24 + + arch/arm/boot/dts/nxp/imx/imx7d-var-som-v2.dtsi | 66 + + .../arm/boot/dts/nxp/imx/imx7d-var-som-wm8731.dtsi | 102 + + .../arm/boot/dts/nxp/imx/imx7d-var-som-wm8904.dtsi | 118 + + arch/arm/boot/dts/nxp/imx/imx7d-var-som.dtsi | 533 +++++ + arch/arm/boot/dts/nxp/imx/imx7s-warp.dts | 2 +- + arch/arm/boot/dts/nxp/imx/mba6ulx.dtsi | 36 +- + arch/arm/boot/dts/nxp/ls/ls1021a.dtsi | 10 +- + arch/arm/boot/dts/nxp/vf/Makefile | 2 + + arch/arm/boot/dts/nxp/vf/vf-colibri-iris.dtsi | 126 + + arch/arm/boot/dts/nxp/vf/vf-colibri.dtsi | 56 + + arch/arm/boot/dts/nxp/vf/vf500-colibri-iris.dts | 14 + + arch/arm/boot/dts/nxp/vf/vf500.dtsi | 6 + + arch/arm/boot/dts/nxp/vf/vf610-colibri-iris.dts | 14 + + arch/arm/boot/dts/nxp/vf/vf610.dtsi | 3 + + arch/arm/include/asm/linkage.h | 29 + + arch/arm/mach-imx/hardware.h | 2 +- + arch/arm/mach-imx/mach-imx6q.c | 32 - + arch/arm/mach-imx/mxc.h | 2 +- + arch/arm/mach-imx/pm-imx6.c | 26 +- + arch/arm/mach-imx/suspend-imx6.S | 6 +- + arch/arm64/boot/dts/freescale/Makefile | 267 ++- + arch/arm64/boot/dts/freescale/fsl-ls1028a.dtsi | 2 +- + .../arm64/boot/dts/freescale/fsl-lx2160a-nbxv3.dts | 34 + + .../boot/dts/freescale/fsl-lx2160a-nbxv3.dtsi | 330 +++ + arch/arm64/boot/dts/freescale/fsl-lx216x.dtsi | 1 + + .../arm64/boot/dts/freescale/imx8-apalis-eval.dtsi | 6 +- + .../boot/dts/freescale/imx8-apalis-ixora-v1.1.dtsi | 6 +- + .../boot/dts/freescale/imx8-apalis-ixora-v1.2.dtsi | 6 +- + .../arm64/boot/dts/freescale/imx8-apalis-v1.1.dtsi | 16 +- + arch/arm64/boot/dts/freescale/imx8dxl-evk.dts | 14 +- + arch/arm64/boot/dts/freescale/imx8dxl-sr-som.dtsi | 14 +- + .../dts/freescale/imx8mm-beacon-baseboard.dtsi | 2 +- + .../imx8mm-data-modul-edm-sbc-overlay-cm4.dtso | 56 + + ...odul-edm-sbc-overlay-edm-mod-imx8mm-common.dtsi | 59 + + ...-edm-sbc-overlay-edm-mod-imx8mm-fio1-audio.dtsi | 99 + + ...-edm-sbc-overlay-edm-mod-imx8mm-fio1-audio.dtso | 79 + + ...-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1.dtsi | 69 + + ...-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1.dtso | 59 + + ...-modul-edm-sbc-overlay-edm-mod-imx8mm-hdmi.dtsi | 94 + + ...-modul-edm-sbc-overlay-edm-mod-imx8mm-hdmi.dtso | 17 + + ...edm-sbc-overlay-edm-mod-imx8mm-lvds-common.dtsi | 118 + + ...l-edm-sbc-overlay-edm-mod-imx8mm-lvds-dual.dtsi | 32 + + ...sbc-overlay-edm-mod-imx8mm-lvds-g070y2-l01.dtsi | 12 + + ...sbc-overlay-edm-mod-imx8mm-lvds-g070y2-l01.dtso | 7 + + ...bc-overlay-edm-mod-imx8mm-lvds-g101ice-l01.dtsi | 12 + + ...bc-overlay-edm-mod-imx8mm-lvds-g101ice-l01.dtso | 7 + + ...bc-overlay-edm-mod-imx8mm-lvds-g121xce-l01.dtsi | 12 + + ...bc-overlay-edm-mod-imx8mm-lvds-g121xce-l01.dtso | 7 + + ...bc-overlay-edm-mod-imx8mm-lvds-g156hce-l01.dtsi | 12 + + ...bc-overlay-edm-mod-imx8mm-lvds-g156hce-l01.dtso | 7 + + ...sbc-overlay-edm-mod-imx8mm-lvds-g215hvn011.dtsi | 12 + + ...sbc-overlay-edm-mod-imx8mm-lvds-g215hvn011.dtso | 7 + + ...c-overlay-edm-mod-imx8mm-lvds-mi0700a2t-30.dtsi | 12 + + ...c-overlay-edm-mod-imx8mm-lvds-mi0700a2t-30.dtso | 7 + + ...verlay-edm-mod-imx8mm-lvds-mi1010z1t-1cp11.dtsi | 12 + + ...verlay-edm-mod-imx8mm-lvds-mi1010z1t-1cp11.dtso | 7 + + ...edm-sbc-overlay-edm-mod-imx8mm-lvds-single.dtsi | 20 + + ...-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds.dtsi | 22 + + ...odul-edm-sbc-overlay-edm-sbc-imx8mm-rev900.dtso | 18 + + ...imx8mm-data-modul-edm-sbc-overlay-lvds-3v3.dtsi | 19 + + ...imx8mm-data-modul-edm-sbc-overlay-lvds-5v0.dtsi | 19 + + ...mx8mm-data-modul-edm-sbc-overlay-lvds-dual.dtsi | 29 + + ...data-modul-edm-sbc-overlay-lvds-g070y2-l01.dtsi | 31 + + ...ata-modul-edm-sbc-overlay-lvds-g101ice-l01.dtsi | 31 + + ...ata-modul-edm-sbc-overlay-lvds-g121xce-l01.dtsi | 31 + + ...ata-modul-edm-sbc-overlay-lvds-g156hce-l01.dtsi | 31 + + ...data-modul-edm-sbc-overlay-lvds-g215hvn011.dtsi | 30 + + ...ta-modul-edm-sbc-overlay-lvds-mi0700a2t-30.dtsi | 31 + + ...modul-edm-sbc-overlay-lvds-mi1010z1t-1cp11.dtsi | 31 + + ...8mm-data-modul-edm-sbc-overlay-lvds-single.dtsi | 13 + + .../dts/freescale/imx8mm-data-modul-edm-sbc.dts | 64 +- + .../arm64/boot/dts/freescale/imx8mm-kontron-bl.dts | 107 +- + .../imx8mm-tx8m-1610-moduline-iv-306-d.dts | 1 - + .../imx8mm-tx8m-1610-moduline-mini-111.dts | 1 - + arch/arm64/boot/dts/freescale/imx8mm-var-dart.dtsi | 21 +- + .../boot/dts/freescale/imx8mm-verdin-ivy.dtsi | 4 +- + arch/arm64/boot/dts/freescale/imx8mm-verdin.dtsi | 2 +- + .../dts/freescale/imx8mn-solidsense-n8-compact.dts | 2 - + .../freescale/imx8mn-vhip4-evalboard-common.dtsi | 3 +- + .../dts/freescale/imx8mn-vhip4-evalboard-v1.dts | 8 + + .../dts/freescale/imx8mn-vhip4-evalboard-v2.dts | 28 +- + .../imx8mn-vhip4-overlay-eeprom-1000.dtso | 103 + + .../imx8mn-vhip4-overlay-eeprom-1100.dtso | 103 + + .../imx8mn-vhip4-overlay-eeprom-2000.dtso | 107 + + .../imx8mp-data-modul-edm-sbc-overlay-cm7.dtso | 57 + + ...-edm-sbc-overlay-edm-mod-imx8mm-fio1-audio.dtso | 67 + + ...-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1.dtso | 46 + + ...-modul-edm-sbc-overlay-edm-mod-imx8mm-hdmi.dtso | 17 + + ...sbc-overlay-edm-mod-imx8mm-lvds-g070y2-l01.dtso | 7 + + ...bc-overlay-edm-mod-imx8mm-lvds-g101ice-l01.dtso | 7 + + ...bc-overlay-edm-mod-imx8mm-lvds-g121xce-l01.dtso | 7 + + ...bc-overlay-edm-mod-imx8mm-lvds-g156hce-l01.dtso | 7 + + ...sbc-overlay-edm-mod-imx8mm-lvds-g215hvn011.dtso | 11 + + ...c-overlay-edm-mod-imx8mm-lvds-mi0700a2t-30.dtso | 7 + + ...verlay-edm-mod-imx8mm-lvds-mi1010z1t-1cp11.dtso | 7 + + ...-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds.dtsi | 35 + + ...sbc-overlay-edm-sbc-imx8mp-lvds-g070y2-l01.dtso | 24 + + ...bc-overlay-edm-sbc-imx8mp-lvds-g101ice-l01.dtso | 24 + + ...bc-overlay-edm-sbc-imx8mp-lvds-g121xce-l01.dtso | 24 + + ...bc-overlay-edm-sbc-imx8mp-lvds-g156hce-l01.dtso | 32 + + ...sbc-overlay-edm-sbc-imx8mp-lvds-g215hvn011.dtso | 36 + + ...c-overlay-edm-sbc-imx8mp-lvds-mi0700a2t-30.dtso | 24 + + ...verlay-edm-sbc-imx8mp-lvds-mi1010z1t-1cp11.dtso | 24 + + ...edm-sbc-overlay-edm-sbc-imx8mp-lvds-rev900.dtso | 41 + + ...edm-sbc-overlay-edm-sbc-imx8mp-lvds-rev902.dtso | 14 + + ...-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds.dtsi | 79 + + ...odul-edm-sbc-overlay-edm-sbc-imx8mp-rev900.dtso | 97 + + ...odul-edm-sbc-overlay-edm-sbc-imx8mp-rev902.dtso | 69 + + .../dts/freescale/imx8mp-data-modul-edm-sbc.dts | 176 +- + .../dts/freescale/imx8mp-icore-mx8mp-edimm2.2.dts | 2 +- + .../boot/dts/freescale/imx8mp-icore-mx8mp.dtsi | 2 +- + .../boot/dts/freescale/imx8mp-kontron-dl.dtso | 11 + + .../imx8mp-phyboard-pollux-peb-av-10.dtsi | 5 +- + .../imx8mp-phyboard-pollux-peb-wlbt-07.dtso | 76 + + .../boot/dts/freescale/imx8mp-toradex-smarc.dtsi | 2 +- + ...8mp-tqma8mpqs-mb-smarc-2-lvds0-tm070jvhg33.dtso | 2 +- + ...8mp-tqma8mpqs-mb-smarc-2-lvds1-tm070jvhg33.dtso | 2 +- + .../imx8mp-tx8p-ml81-moduline-display-106.dts | 1 - + .../boot/dts/freescale/imx8mp-verdin-ivy.dtsi | 4 +- + .../boot/dts/freescale/imx8mp-verdin-yavia.dtsi | 7 +- + arch/arm64/boot/dts/freescale/imx8mp.dtsi | 2 +- + arch/arm64/boot/dts/freescale/imx8mq-evk.dts | 6 + + .../boot/dts/freescale/imx8mq-librem5-devkit.dts | 4 +- + arch/arm64/boot/dts/freescale/imx8mq-librem5.dtsi | 4 +- + arch/arm64/boot/dts/freescale/imx8mq-nitrogen.dts | 2 +- + .../arm64/boot/dts/freescale/imx8mq-zii-ultra.dtsi | 16 +- + arch/arm64/boot/dts/freescale/imx8mq.dtsi | 168 +- + arch/arm64/boot/dts/freescale/imx8qm-mek.dts | 32 +- + arch/arm64/boot/dts/freescale/imx8qm-ss-conn.dtsi | 39 + + .../boot/dts/freescale/imx8qm-var-som-symphony.dts | 2 +- + arch/arm64/boot/dts/freescale/imx8qm-var-som.dtsi | 6 +- + arch/arm64/boot/dts/freescale/imx8qxp-mek.dts | 20 +- + arch/arm64/boot/dts/freescale/imx8ulp-evk.dts | 7 +- + .../arm64/boot/dts/freescale/imx8ulp-firmware.dtsi | 31 + + arch/arm64/boot/dts/freescale/imx8ulp.dtsi | 12 +- + arch/arm64/boot/dts/freescale/imx9-mqs.dtso | 57 + + arch/arm64/boot/dts/freescale/imx91-11x11-evk.dts | 55 +- + .../boot/dts/freescale/imx91-11x11-frdm-s.dts | 2 - + arch/arm64/boot/dts/freescale/imx91-11x11-frdm.dts | 2 - + .../freescale/imx91-9x9-qsb-tianma-tm050rdh03.dtso | 104 + + arch/arm64/boot/dts/freescale/imx91-9x9-qsb.dts | 52 + + .../dts/freescale/imx91-lino-verdin-dahlia.dts | 24 + + .../boot/dts/freescale/imx91-lino-verdin-dev.dts | 24 + + arch/arm64/boot/dts/freescale/imx91-lino.dtsi | 305 +++ + .../boot/dts/freescale/imx91-toradex-osm-dev.dts | 22 + + .../boot/dts/freescale/imx91-toradex-osm.dtsi | 303 +++ + arch/arm64/boot/dts/freescale/imx91.dtsi | 11 + + arch/arm64/boot/dts/freescale/imx91_93_common.dtsi | 6 + + .../boot/dts/freescale/imx93-11x11-evk-common.dtsi | 5 +- + arch/arm64/boot/dts/freescale/imx93-11x11-evk.dts | 54 +- + .../dts/freescale/imx93-11x11-frdm-common.dtsi | 681 ++++++ + arch/arm64/boot/dts/freescale/imx93-11x11-frdm.dts | 682 +----- + arch/arm64/boot/dts/freescale/imx93-14x14-evk.dts | 59 +- + arch/arm64/boot/dts/freescale/imx93-9x9-qsb.dts | 7 +- + .../dts/freescale/imx93-imx91-lino-common.dtsi | 536 +++++ + .../freescale/imx93-imx91-lino-verdin-common.dtsi | 175 ++ + .../imx93-imx91-lino-verdin-dahlia-common.dtsi | 198 ++ + .../imx93-imx91-lino-verdin-dev-common.dtsi | 216 ++ + .../freescale/imx93-imx91-toradex-osm-common.dtsi | 526 +++++ + .../imx93-imx91-toradex-osm-dev-common.dtsi | 324 +++ + .../boot/dts/freescale/imx93-kontron-bl-osm-s.dts | 6 +- + .../boot/dts/freescale/imx93-kontron-osm-s.dtsi | 32 +- + .../dts/freescale/imx93-lino-verdin-dahlia.dts | 24 + + .../boot/dts/freescale/imx93-lino-verdin-dev.dts | 24 + + arch/arm64/boot/dts/freescale/imx93-lino.dtsi | 359 +++ + .../boot/dts/freescale/imx93-toradex-osm-dev.dts | 22 + + .../boot/dts/freescale/imx93-toradex-osm.dtsi | 357 +++ + arch/arm64/boot/dts/freescale/imx93w-frdm.dts | 23 + + arch/arm64/boot/dts/freescale/imx94.dtsi | 5 +- + .../boot/dts/freescale/imx943-evk-sdwifi.dtso | 25 + + arch/arm64/boot/dts/freescale/imx943-evk.dts | 138 +- + arch/arm64/boot/dts/freescale/imx943.dtsi | 2 +- + arch/arm64/boot/dts/freescale/imx95-15x15-evk.dts | 48 +- + arch/arm64/boot/dts/freescale/imx95-15x15-frdm.dts | 38 +- + arch/arm64/boot/dts/freescale/imx95-19x19-evk.dts | 55 +- + .../boot/dts/freescale/imx95-19x19-frdm-pro.dts | 2 - + .../boot/dts/freescale/imx95-aquila-clover.dts | 23 + + arch/arm64/boot/dts/freescale/imx95-aquila-dev.dts | 18 + + arch/arm64/boot/dts/freescale/imx95-aquila.dtsi | 15 +- + arch/arm64/boot/dts/freescale/imx95-navq.dts | 228 ++ + .../boot/dts/freescale/imx95-toradex-osm-dev.dts | 404 ++++ + .../boot/dts/freescale/imx95-toradex-osm.dtsi | 1032 +++++++++ + .../boot/dts/freescale/imx95-toradex-smarc-dev.dts | 8 + + .../boot/dts/freescale/imx95-toradex-smarc.dtsi | 8 + + .../dts/freescale/imx95-tqma9596sa-mb-smarc-2.dts | 2 + + .../arm64/boot/dts/freescale/imx95-tqma9596sa.dtsi | 75 + + .../boot/dts/freescale/imx95-var-dart-sonata.dts | 2 +- + arch/arm64/boot/dts/freescale/imx95.dtsi | 18 +- + arch/arm64/boot/dts/freescale/imx952-evk.dts | 77 +- + arch/arm64/boot/dts/freescale/imx952-frdm.dts | 9 + + arch/arm64/boot/dts/freescale/imx952-frdm.dtsi | 726 ++++++ + arch/arm64/boot/dts/freescale/imx952.dtsi | 19 +- + arch/arm64/boot/dts/freescale/s32g2.dtsi | 2 +- + arch/arm64/boot/dts/freescale/s32g3.dtsi | 2 +- + drivers/firmware/imx/Kconfig | 12 + + drivers/firmware/imx/Makefile | 2 + + drivers/firmware/imx/ele_base_msg.c | 395 ++++ + drivers/firmware/imx/ele_base_msg.h | 119 + + drivers/firmware/imx/ele_common.c | 1012 ++++++++ + drivers/firmware/imx/ele_common.h | 147 ++ + drivers/firmware/imx/ele_fw_api.c | 406 ++++ + drivers/firmware/imx/ele_fw_api.h | 104 + + drivers/firmware/imx/ele_msg_addr_field.c | 619 +++++ + drivers/firmware/imx/imx-dsp.c | 6 +- + drivers/firmware/imx/se_ctrl.c | 2434 ++++++++++++++++++++ + drivers/firmware/imx/se_ctrl.h | 309 +++ + include/linux/firmware/imx/se_api.h | 14 + + include/linux/micrel_phy.h | 5 - + include/soc/imx/cpu.h | 2 +- + include/uapi/linux/se_ioctl.h | 97 + + 278 files changed, 19568 insertions(+), 1288 deletions(-) + create mode 100644 Documentation/ABI/testing/se-cdev + create mode 100644 Documentation/devicetree/bindings/firmware/fsl,imx-se.yaml + create mode 100644 arch/arm/boot/dts/nxp/imx/imx50-amazon-yoshi.dts + create mode 100644 arch/arm/boot/dts/nxp/imx/imx6dl-phytec-mira-rdk-emmc.dts + create mode 100644 arch/arm/boot/dts/nxp/imx/imx6dl-phytec-phycore-som.dtsi + create mode 100644 arch/arm/boot/dts/nxp/imx/imx6q-phytec-phycore-som.dtsi + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-colibri-emmc-panel-cap-touch-7inch-parallel-touch-adapter.dtso + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-colibri-emmc-panel-cap-touch-7inch-parallel.dtso + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-colibri-emmc-panel-res-touch-7inch-parallel.dtso + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-colibri-emmc-vga.dtso + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som-emmc-mx7customboard.dts + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som-emmc-wm8731-mx7customboard.dts + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som-emmc.dtsi + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som-mx7customboard.dtsi + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som-nand-mx7customboard.dts + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som-nand-wm8731-mx7customboard.dts + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som-nand.dtsi + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som-v2-emmc-mx7customboard.dts + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som-v2-nand-mx7customboard.dts + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som-v2.dtsi + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som-wm8731.dtsi + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som-wm8904.dtsi + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som.dtsi + create mode 100644 arch/arm/boot/dts/nxp/vf/vf-colibri-iris.dtsi + create mode 100644 arch/arm/boot/dts/nxp/vf/vf500-colibri-iris.dts + create mode 100644 arch/arm/boot/dts/nxp/vf/vf610-colibri-iris.dts + create mode 100644 arch/arm64/boot/dts/freescale/fsl-lx2160a-nbxv3.dts + create mode 100644 arch/arm64/boot/dts/freescale/fsl-lx2160a-nbxv3.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-cm4.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-common.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1-audio.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1-audio.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-hdmi.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-hdmi.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-common.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-dual.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g070y2-l01.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g070y2-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g101ice-l01.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g101ice-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g121xce-l01.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g121xce-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g156hce-l01.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g156hce-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g215hvn011.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g215hvn011.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-mi0700a2t-30.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-mi0700a2t-30.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-mi1010z1t-1cp11.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-mi1010z1t-1cp11.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-single.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-sbc-imx8mm-rev900.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-3v3.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-5v0.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-dual.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-g070y2-l01.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-g101ice-l01.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-g121xce-l01.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-g156hce-l01.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-g215hvn011.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-mi0700a2t-30.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-mi1010z1t-1cp11.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-single.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mn-vhip4-overlay-eeprom-1000.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mn-vhip4-overlay-eeprom-1100.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mn-vhip4-overlay-eeprom-2000.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-cm7.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1-audio.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-hdmi.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g070y2-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g101ice-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g121xce-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g156hce-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g215hvn011.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-mi0700a2t-30.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-mi1010z1t-1cp11.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-g070y2-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-g101ice-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-g121xce-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-g156hce-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-g215hvn011.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-mi0700a2t-30.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-mi1010z1t-1cp11.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-rev900.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-rev902.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-rev900.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-rev902.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-phyboard-pollux-peb-wlbt-07.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8ulp-firmware.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx9-mqs.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx91-9x9-qsb-tianma-tm050rdh03.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx91-lino-verdin-dahlia.dts + create mode 100644 arch/arm64/boot/dts/freescale/imx91-lino-verdin-dev.dts + create mode 100644 arch/arm64/boot/dts/freescale/imx91-lino.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx91-toradex-osm-dev.dts + create mode 100644 arch/arm64/boot/dts/freescale/imx91-toradex-osm.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx93-11x11-frdm-common.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx93-imx91-lino-common.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx93-imx91-lino-verdin-common.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx93-imx91-lino-verdin-dahlia-common.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx93-imx91-lino-verdin-dev-common.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx93-imx91-toradex-osm-common.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx93-imx91-toradex-osm-dev-common.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx93-lino-verdin-dahlia.dts + create mode 100644 arch/arm64/boot/dts/freescale/imx93-lino-verdin-dev.dts + create mode 100644 arch/arm64/boot/dts/freescale/imx93-lino.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx93-toradex-osm-dev.dts + create mode 100644 arch/arm64/boot/dts/freescale/imx93-toradex-osm.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx93w-frdm.dts + create mode 100644 arch/arm64/boot/dts/freescale/imx95-navq.dts + create mode 100644 arch/arm64/boot/dts/freescale/imx95-toradex-osm-dev.dts + create mode 100644 arch/arm64/boot/dts/freescale/imx95-toradex-osm.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx952-frdm.dts + create mode 100644 arch/arm64/boot/dts/freescale/imx952-frdm.dtsi + create mode 100644 drivers/firmware/imx/ele_base_msg.c + create mode 100644 drivers/firmware/imx/ele_base_msg.h + create mode 100644 drivers/firmware/imx/ele_common.c + create mode 100644 drivers/firmware/imx/ele_common.h + create mode 100644 drivers/firmware/imx/ele_fw_api.c + create mode 100644 drivers/firmware/imx/ele_fw_api.h + create mode 100644 drivers/firmware/imx/ele_msg_addr_field.c + create mode 100644 drivers/firmware/imx/se_ctrl.c + create mode 100644 drivers/firmware/imx/se_ctrl.h + create mode 100644 include/linux/firmware/imx/se_api.h + create mode 100644 include/uapi/linux/se_ioctl.h +Merging mediatek/for-next (4bfec57314396 Merge branches 'v7.3-next/dts32', 'v7.3-next/dts64' and 'v7.3-next/soc' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mediatek/linux.git mediatek/for-next +Merge made by the 'ort' strategy. +Merging mvebu/for-next (73b0de63d7f7d Merge branch 'mvebu/dt64' into mvebu/for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/gclement/mvebu.git mvebu/for-next +Auto-merging Documentation/devicetree/bindings/vendor-prefixes.yaml +Merge made by the 'ort' strategy. + .../bindings/arm/marvell/armada-7k-8k.yaml | 7 + + .../devicetree/bindings/vendor-prefixes.yaml | 2 + + .../boot/dts/marvell/armada-385-clearfog-gtr.dtsi | 2 +- + arch/arm/boot/dts/marvell/armada-388-clearfog.dtsi | 7 +- + arch/arm/boot/dts/marvell/armada-388-helios4.dts | 36 ++-- + arch/arm/boot/dts/marvell/armada-38x.dtsi | 1 + + arch/arm/boot/dts/marvell/armada-395-gp.dts | 1 - + arch/arm/boot/dts/marvell/armada-39x.dtsi | 1 + + arch/arm/boot/dts/marvell/dove.dtsi | 22 +-- + arch/arm/boot/dts/marvell/kirkwood-6282.dtsi | 15 +- + .../dts/marvell/kirkwood-guruplug-server-plus.dts | 4 +- + arch/arm/boot/dts/marvell/kirkwood-lsxl.dtsi | 4 +- + arch/arm/boot/dts/marvell/kirkwood-synology.dtsi | 12 +- + .../arm/boot/dts/marvell/orion5x-rd88f5182-nas.dts | 2 +- + arch/arm/mach-mvebu/Kconfig | 4 +- + arch/arm64/boot/dts/marvell/Makefile | 1 + + .../boot/dts/marvell/cn9130-sophos-xgs107w.dts | 188 +++++++++++++++++++++ + 17 files changed, 247 insertions(+), 62 deletions(-) + create mode 100644 arch/arm64/boot/dts/marvell/cn9130-sophos-xgs107w.dts +Merging omap/for-next (7cdd46c9d6c5a Merge branch 'omap-for-v7.4/soc' into tmp/omap-next-20260921.091322) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/khilman/linux-omap.git omap/for-next +Auto-merging arch/arm/configs/omap2plus_defconfig +Auto-merging arch/arm/mach-omap2/control.h +CONFLICT (content): Merge conflict in arch/arm/mach-omap2/control.h +Auto-merging arch/arm/mach-omap2/soc.h +Auto-merging arch/arm/mach-omap2/sram.h +Resolved 'arch/arm/mach-omap2/control.h' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 03bcfb09e04ad] Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/khilman/linux-omap.git +$ git diff -M --stat --summary HEAD^.. + .../devicetree/bindings/input/omap-keypad.txt | 28 ------- + .../devicetree/bindings/input/ti,omap4-keypad.yaml | 63 +++++++++++++++ + .../boot/dts/ti/omap/am335x-boneblack-hdmi.dtsi | 2 +- + arch/arm/boot/dts/ti/omap/am335x-bonegreen-eco.dts | 2 +- + arch/arm/boot/dts/ti/omap/am335x-shc.dts | 4 - + arch/arm/boot/dts/ti/omap/am437x-l4.dtsi | 2 +- + arch/arm/boot/dts/ti/omap/am57-pruss.dtsi | 11 +++ + arch/arm/boot/dts/ti/omap/am571x-idk.dts | 65 +++++++++++++++- + arch/arm/boot/dts/ti/omap/am5729-beagleboneai.dts | 2 +- + .../boot/dts/ti/omap/am57xx-beagle-x15-common.dtsi | 2 +- + arch/arm/boot/dts/ti/omap/am57xx-idk-common.dtsi | 3 +- + arch/arm/boot/dts/ti/omap/dra7-l4.dtsi | 2 +- + .../boot/dts/ti/omap/motorola-mapphone-common.dtsi | 8 +- + .../dts/ti/omap/motorola-mapphone-mz607-mz617.dtsi | 2 + + arch/arm/boot/dts/ti/omap/omap3-igep.dtsi | 2 +- + arch/arm/boot/dts/ti/omap/omap3-n9.dts | 5 +- + arch/arm/boot/dts/ti/omap/omap3-n900.dts | 3 +- + arch/arm/boot/dts/ti/omap/omap3-n950.dts | 5 +- + arch/arm/boot/dts/ti/omap/omap3-tao3530.dtsi | 8 +- + .../boot/dts/ti/omap/omap4-droid-bionic-xt875.dts | 2 + + arch/arm/boot/dts/ti/omap/omap4-droid4-xt894.dts | 2 + + arch/arm/boot/dts/ti/omap/omap4-epson-embt2ws.dts | 2 + + arch/arm/boot/dts/ti/omap/omap4-l4.dtsi | 1 + + .../dts/ti/omap/omap4-samsung-espresso-common.dtsi | 6 +- + arch/arm/boot/dts/ti/omap/omap4-sdp.dts | 2 + + arch/arm/boot/dts/ti/omap/omap4.dtsi | 6 +- + arch/arm/boot/dts/ti/omap/omap5-l4.dtsi | 1 + + arch/arm/boot/dts/ti/omap/omap5.dtsi | 2 +- + arch/arm/configs/multi_v7_defconfig | 1 + + arch/arm/configs/omap2plus_defconfig | 3 + + arch/arm/mach-omap2/control.h | 8 +- + arch/arm/mach-omap2/ctrl_module_wkup_44xx.h | 89 ---------------------- + arch/arm/mach-omap2/soc.h | 4 +- + arch/arm/mach-omap2/sram.h | 4 +- + 34 files changed, 195 insertions(+), 157 deletions(-) + delete mode 100644 Documentation/devicetree/bindings/input/omap-keypad.txt + create mode 100644 Documentation/devicetree/bindings/input/ti,omap4-keypad.yaml + delete mode 100644 arch/arm/mach-omap2/ctrl_module_wkup_44xx.h +Merging qcom/for-next (705529dab1fa8 Merge branches 'arm32-for-7.4', 'arm64-defconfig-for-7.4', 'arm64-fixes-for-7.3', 'arm64-for-7.4', 'clk-fixes-for-7.3', 'clk-for-7.4', 'drivers-fixes-for-7.3' and 'drivers-for-7.4' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/qcom/linux.git qcom/for-next +Auto-merging Documentation/devicetree/bindings/arm/cpus.yaml +Auto-merging Documentation/devicetree/bindings/vendor-prefixes.yaml +Auto-merging MAINTAINERS +Auto-merging arch/arm/configs/multi_v7_defconfig +Auto-merging drivers/firmware/qcom/Kconfig +Auto-merging drivers/firmware/qcom/qcom_tzmem.c +Auto-merging include/linux/mod_devicetable.h +CONFLICT (content): Merge conflict in include/linux/mod_devicetable.h +Auto-merging include/linux/soc/qcom/geni-se.h +Resolved 'include/linux/mod_devicetable.h' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master aad48d2947b6c] Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/qcom/linux.git +$ git diff -M --stat --summary HEAD^.. + Documentation/devicetree/bindings/arm/cpus.yaml | 3 + + .../bindings/arm/qcom,coresight-ctcu.yaml | 1 + + Documentation/devicetree/bindings/arm/qcom.yaml | 45 + + .../devicetree/bindings/cache/qcom,llcc.yaml | 55 +- + .../bindings/clock/qcom,gcc-msm8952.yaml | 65 + + .../devicetree/bindings/clock/qcom,gcc-sdm845.yaml | 6 +- + .../devicetree/bindings/clock/qcom,gcc-sm6350.yaml | 7 + + .../devicetree/bindings/clock/qcom,gcc-sm8150.yaml | 7 + + .../devicetree/bindings/clock/qcom,gcc-sm8250.yaml | 7 + + .../devicetree/bindings/clock/qcom,gcc-sm8350.yaml | 7 + + .../devicetree/bindings/clock/qcom,gcc-sm8450.yaml | 7 + + .../devicetree/bindings/clock/qcom,hawi-gpucc.yaml | 80 + + .../bindings/clock/qcom,ipq9574-cmn-pll.yaml | 1 + + .../bindings/clock/qcom,kaanapali-gxclkctl.yaml | 23 +- + .../devicetree/bindings/clock/qcom,kuno-gcc.yaml | 52 + + .../devicetree/bindings/clock/qcom,qcs615-gcc.yaml | 7 + + .../bindings/clock/qcom,qcs8300-gcc.yaml | 7 + + .../devicetree/bindings/clock/qcom,rpmcc.yaml | 5 + + .../devicetree/bindings/clock/qcom,rpmhcc.yaml | 2 + + .../devicetree/bindings/clock/qcom,sdx75-gcc.yaml | 7 + + .../devicetree/bindings/clock/qcom,sm4450-gcc.yaml | 7 + + .../devicetree/bindings/clock/qcom,sm6375-gcc.yaml | 9 +- + .../devicetree/bindings/clock/qcom,sm7150-gcc.yaml | 53 - + .../bindings/clock/qcom,sm8450-camcc.yaml | 47 +- + .../bindings/clock/qcom,sm8450-videocc.yaml | 2 + + .../devicetree/bindings/clock/qcom,sm8550-gcc.yaml | 7 + + .../bindings/clock/qcom,sm8550-tcsr.yaml | 1 - + .../devicetree/bindings/clock/qcom,sm8650-gcc.yaml | 7 + + .../devicetree/bindings/clock/qcom,sm8750-gcc.yaml | 7 + + .../bindings/clock/qcom,x1e80100-tcsr.yaml | 114 + + .../embedded-controller/qcom,hamoa-crd-ec.yaml | 3 + + .../devicetree/bindings/firmware/qcom,scm.yaml | 4 + + .../bindings/interconnect/qcom,msm8998-bwmon.yaml | 1 + + .../bindings/mailbox/qcom,cpucp-mbox.yaml | 1 + + Documentation/devicetree/bindings/mfd/syscon.yaml | 1 + + .../devicetree/bindings/soc/qcom/qcom,apr.yaml | 4 +- + .../devicetree/bindings/soc/qcom/qcom-stats.yaml | 25 +- + .../bindings/spmi/qcom,glymur-spmi-pmic-arb.yaml | 1 + + .../devicetree/bindings/sram/qcom,imem.yaml | 1 - + Documentation/devicetree/bindings/sram/sram.yaml | 1 + + .../devicetree/bindings/usb/qcom,snps-dwc3.yaml | 3 + + .../devicetree/bindings/vendor-prefixes.yaml | 8 + + MAINTAINERS | 2 + + .../boot/dts/qcom/qcom-msm8960-sony-huashan.dts | 15 + + arch/arm/boot/dts/qcom/qcom-msm8960.dtsi | 169 +- + arch/arm/boot/dts/qcom/qcom-msm8974.dtsi | 3 + + arch/arm/boot/dts/qcom/qcom-sdx55-t55.dts | 2 +- + arch/arm/configs/multi_v7_defconfig | 9 - + arch/arm/configs/qcom_defconfig | 14 - + arch/arm64/boot/dts/qcom/Makefile | 23 + + arch/arm64/boot/dts/qcom/agatti.dtsi | 13 +- + arch/arm64/boot/dts/qcom/apq8096-db820c.dtsi | 15 +- + .../dts/qcom/glymur-asus-zenbook-a16-ux3607oa.dts | 1 - + arch/arm64/boot/dts/qcom/glymur-crd.dtsi | 23 + + arch/arm64/boot/dts/qcom/glymur-qcb.dts | 699 ++ + arch/arm64/boot/dts/qcom/glymur.dtsi | 100 +- + .../boot/dts/qcom/hamoa-advantech-som-6820.dtsi | 54 + + .../boot/dts/qcom/hamoa-advantech-som-db5830.dts | 119 + + arch/arm64/boot/dts/qcom/hamoa-iot-evk.dts | 131 +- + arch/arm64/boot/dts/qcom/hamoa.dtsi | 4 +- + arch/arm64/boot/dts/qcom/hawi-ipcc.h | 61 + + arch/arm64/boot/dts/qcom/hawi-mtp.dts | 997 +++ + arch/arm64/boot/dts/qcom/hawi.dtsi | 6844 ++++++++++++++++++++ + arch/arm64/boot/dts/qcom/ipq5332-rdp-common.dtsi | 2 + + arch/arm64/boot/dts/qcom/ipq5424-rdp466.dts | 2 + + arch/arm64/boot/dts/qcom/ipq5424.dtsi | 16 + + arch/arm64/boot/dts/qcom/ipq9574-rdp-common.dtsi | 2 + + arch/arm64/boot/dts/qcom/ipq9574.dtsi | 3 +- + arch/arm64/boot/dts/qcom/ipq9650-rdp488.dts | 119 + + arch/arm64/boot/dts/qcom/ipq9650.dtsi | 666 +- + arch/arm64/boot/dts/qcom/kaanapali-mtp.dts | 10 +- + arch/arm64/boot/dts/qcom/kaanapali-qrd.dts | 8 + + arch/arm64/boot/dts/qcom/kaanapali.dtsi | 472 +- + arch/arm64/boot/dts/qcom/kodiak.dtsi | 31 +- + arch/arm64/boot/dts/qcom/lemans-el2.dtso | 4 +- + .../boot/dts/qcom/lemans-evk-ifp-mezzanine.dtso | 3 +- + arch/arm64/boot/dts/qcom/lemans-evk.dts | 16 +- + arch/arm64/boot/dts/qcom/lemans-ride-common.dtsi | 21 +- + arch/arm64/boot/dts/qcom/lemans.dtsi | 24 +- + .../qcom/mahua-lenovo-thinkpad-t14s-gen7-lcd.dts | 1272 ++++ + arch/arm64/boot/dts/qcom/milos.dtsi | 393 ++ + arch/arm64/boot/dts/qcom/monaco-arduino-monza.dts | 201 +- + arch/arm64/boot/dts/qcom/monaco-evk.dts | 69 +- + arch/arm64/boot/dts/qcom/monaco.dtsi | 15 + + arch/arm64/boot/dts/qcom/msm8939-asus-z00t.dts | 8 + + .../boot/dts/qcom/msm8939-longcheer-l9100.dts | 8 + + arch/arm64/boot/dts/qcom/msm8939.dtsi | 26 + + arch/arm64/boot/dts/qcom/msm8953-motorola-deen.dts | 440 ++ + arch/arm64/boot/dts/qcom/msm8976-leeco-s2.dts | 442 ++ + arch/arm64/boot/dts/qcom/msm8976.dtsi | 2 +- + .../boot/dts/qcom/msm8996-oneplus-common.dtsi | 5 +- + .../boot/dts/qcom/msm8996-sony-xperia-tone.dtsi | 7 +- + .../arm64/boot/dts/qcom/msm8996-xiaomi-common.dtsi | 6 +- + arch/arm64/boot/dts/qcom/msm8996.dtsi | 45 +- + arch/arm64/boot/dts/qcom/msm8998.dtsi | 6 +- + arch/arm64/boot/dts/qcom/nord-embedded.dtsi | 108 + + arch/arm64/boot/dts/qcom/nord.dtsi | 45 + + arch/arm64/boot/dts/qcom/pmh0104-hawi.dtsi | 103 + + arch/arm64/boot/dts/qcom/pmi8996.dtsi | 12 +- + arch/arm64/boot/dts/qcom/purwa-iot-evk.dts | 91 +- + arch/arm64/boot/dts/qcom/qcm6490-fairphone-fp5.dts | 54 + + arch/arm64/boot/dts/qcom/qcm6490-idp.dts | 6 + + .../boot/dts/qcom/qcm6490-particle-tachyon.dts | 3 +- + arch/arm64/boot/dts/qcom/qcs404-evb.dtsi | 6 +- + arch/arm64/boot/dts/qcom/qcs404.dtsi | 7 +- + arch/arm64/boot/dts/qcom/qcs615-ride.dts | 5 +- + .../boot/dts/qcom/qcs6490-radxa-dragon-q6a.dts | 4 +- + .../qcom/qcs6490-rb3gen2-industrial-mezzanine.dtso | 6 +- + arch/arm64/boot/dts/qcom/qcs6490-rb3gen2.dts | 12 +- + .../dts/qcom/qcs6490-thundercomm-minipc-g1iot.dts | 4 +- + .../boot/dts/qcom/qcs6490-thundercomm-rubikpi3.dts | 2 +- + .../boot/dts/qcom/qcs6490-vicharak-axon-mini.dts | 16 +- + arch/arm64/boot/dts/qcom/qcs8300-ride.dts | 8 +- + arch/arm64/boot/dts/qcom/qcs8550-aim300.dtsi | 16 +- + .../boot/dts/qcom/qcs8550-ayaneo-pocket-ds.dts | 1742 +++++ + .../arm64/boot/dts/qcom/qcs8550-ayntec-common.dtsi | 1819 ++++++ + .../boot/dts/qcom/qcs8550-ayntec-odin2mini.dts | 44 + + .../boot/dts/qcom/qcs8550-ayntec-odin2portal.dts | 99 + + arch/arm64/boot/dts/qcom/qcs8550-ayntec-thor.dts | 250 + + arch/arm64/boot/dts/qcom/qcs8550-imdt-sbc.dts | 401 ++ + arch/arm64/boot/dts/qcom/qcs8550-imdt-som.dtsi | 315 + + arch/arm64/boot/dts/qcom/qcs8550-rb5gen2.dts | 12 +- + .../boot/dts/qcom/qcs8550-retroidpocket-nova.dts | 116 + + .../boot/dts/qcom/qcs8550-retroidpocket-rp6.dts | 120 + + arch/arm64/boot/dts/qcom/qdu1000.dtsi | 6 +- + arch/arm64/boot/dts/qcom/sa8540p-ride.dts | 4 +- + arch/arm64/boot/dts/qcom/sar2130p-qar2130p.dts | 6 +- + arch/arm64/boot/dts/qcom/sar2130p.dtsi | 13 +- + .../boot/dts/qcom/sc7180-trogdor-homestar.dtsi | 2 +- + .../qcom/sc7180-trogdor-lazor-limozeen-nots-r4.dts | 2 +- + .../boot/dts/qcom/sc7180-trogdor-lazor-r1.dts | 4 +- + .../boot/dts/qcom/sc7180-trogdor-pompom-r1.dts | 4 +- + arch/arm64/boot/dts/qcom/sc7180-trogdor-r1.dts | 4 +- + .../arm64/boot/dts/qcom/sc8180x-lenovo-flex-5g.dts | 7 +- + arch/arm64/boot/dts/qcom/sc8180x-primus.dts | 7 +- + arch/arm64/boot/dts/qcom/sc8180x.dtsi | 24 +- + .../boot/dts/qcom/sc8280xp-huawei-gaokun3.dts | 10 +- + .../boot/dts/qcom/sc8280xp-microsoft-arcata.dts | 2 +- + arch/arm64/boot/dts/qcom/sc8280xp.dtsi | 116 +- + arch/arm64/boot/dts/qcom/sdm630.dtsi | 16 +- + arch/arm64/boot/dts/qcom/sdm670-google-common.dtsi | 8 +- + arch/arm64/boot/dts/qcom/sdm670.dtsi | 272 + + arch/arm64/boot/dts/qcom/sdm845-db845c.dts | 13 +- + arch/arm64/boot/dts/qcom/sdm845-lg-common.dtsi | 13 +- + arch/arm64/boot/dts/qcom/sdm845-mtp.dts | 12 +- + .../arm64/boot/dts/qcom/sdm845-oneplus-common.dtsi | 6 +- + .../boot/dts/qcom/sdm845-oneplus-enchilada.dts | 5 + + arch/arm64/boot/dts/qcom/sdm845-oneplus-fajita.dts | 1 + + .../dts/qcom/sdm845-xiaomi-beryllium-common.dtsi | 6 + + arch/arm64/boot/dts/qcom/sdm845-xiaomi-polaris.dts | 6 + + arch/arm64/boot/dts/qcom/sdm845.dtsi | 20 +- + arch/arm64/boot/dts/qcom/sdx75.dtsi | 1 + + .../dts/qcom/shikra-cqm-cqs-evk-imx577-camera.dtso | 72 + + arch/arm64/boot/dts/qcom/shikra-cqm-evk.dts | 56 + + arch/arm64/boot/dts/qcom/shikra-cqm-som.dtsi | 41 + + arch/arm64/boot/dts/qcom/shikra-cqs-evk.dts | 52 + + arch/arm64/boot/dts/qcom/shikra-evk.dtsi | 23 + + .../dts/qcom/shikra-iqs-evk-imx577-camera.dtso | 72 + + arch/arm64/boot/dts/qcom/shikra-iqs-evk.dts | 32 + + arch/arm64/boot/dts/qcom/shikra-iqs-som.dtsi | 2 +- + arch/arm64/boot/dts/qcom/shikra.dtsi | 546 +- + arch/arm64/boot/dts/qcom/sm4450.dtsi | 1 + + arch/arm64/boot/dts/qcom/sm6115-xiaomi-lemon.dts | 497 ++ + arch/arm64/boot/dts/qcom/sm6115.dtsi | 2 +- + arch/arm64/boot/dts/qcom/sm6350.dtsi | 234 + + arch/arm64/boot/dts/qcom/sm7325-xiaomi-lisa.dts | 1108 ++++ + arch/arm64/boot/dts/qcom/sm8150.dtsi | 26 +- + arch/arm64/boot/dts/qcom/sm8250.dtsi | 49 +- + arch/arm64/boot/dts/qcom/sm8350-hdk.dts | 18 +- + arch/arm64/boot/dts/qcom/sm8350.dtsi | 15 +- + arch/arm64/boot/dts/qcom/sm8450.dtsi | 25 +- + arch/arm64/boot/dts/qcom/sm8550-hdk.dts | 28 +- + arch/arm64/boot/dts/qcom/sm8550-mtp.dts | 16 +- + arch/arm64/boot/dts/qcom/sm8550-qrd.dts | 20 +- + arch/arm64/boot/dts/qcom/sm8550-samsung-q5q.dts | 7 +- + .../dts/qcom/sm8550-sony-xperia-yodo-pdx234.dts | 8 +- + arch/arm64/boot/dts/qcom/sm8550.dtsi | 63 +- + .../boot/dts/qcom/sm8650-ayaneo-pocket-s2.dts | 11 +- + arch/arm64/boot/dts/qcom/sm8650-hdk.dts | 22 +- + arch/arm64/boot/dts/qcom/sm8650-mtp.dts | 16 +- + arch/arm64/boot/dts/qcom/sm8650-qrd.dts | 14 +- + arch/arm64/boot/dts/qcom/sm8650-valve-deckard.dts | 1288 ++++ + arch/arm64/boot/dts/qcom/sm8650.dtsi | 18 +- + arch/arm64/boot/dts/qcom/sm8750-mtp.dts | 10 +- + arch/arm64/boot/dts/qcom/sm8750-qrd.dts | 8 + + arch/arm64/boot/dts/qcom/sm8750.dtsi | 339 +- + arch/arm64/boot/dts/qcom/talos-evk-som.dtsi | 5 +- + arch/arm64/boot/dts/qcom/talos-evk.dts | 2 +- + arch/arm64/boot/dts/qcom/talos.dtsi | 12 +- + arch/arm64/boot/dts/qcom/x1-asus-vivobook-s15.dtsi | 1 + + arch/arm64/boot/dts/qcom/x1-asus-zenbook-a14.dtsi | 1 + + arch/arm64/boot/dts/qcom/x1-crd.dtsi | 1 + + arch/arm64/boot/dts/qcom/x1-dell-thena.dtsi | 3 +- + arch/arm64/boot/dts/qcom/x1-hp-omnibook-x14.dtsi | 1 + + arch/arm64/boot/dts/qcom/x1e001de-devkit.dts | 3 +- + .../dts/qcom/x1e78100-lenovo-thinkpad-t14s.dtsi | 1 + + .../qcom/x1e80100-dell-inspiron-14-plus-7441.dts | 31 +- + .../boot/dts/qcom/x1e80100-dell-xps13-9345.dts | 1 + + .../dts/qcom/x1e80100-honor-magicbook-art-14.dts | 3 +- + .../boot/dts/qcom/x1e80100-lenovo-yoga-slim7x.dts | 23 + + .../dts/qcom/x1e80100-medion-sprchrgd-14-s1.dts | 1 + + arch/arm64/boot/dts/qcom/x1e80100-qcp.dts | 1 + + .../boot/dts/qcom/x1p42100-lenovo-thinkbook-16.dts | 1 + + arch/arm64/configs/defconfig | 121 +- + drivers/clk/qcom/Kconfig | 167 +- + drivers/clk/qcom/Makefile | 9 + + drivers/clk/qcom/cambistmclkcc-eliza.c | 464 ++ + drivers/clk/qcom/cambistmclkcc-kaanapali.c | 7 + + drivers/clk/qcom/cambistmclkcc-sm8750.c | 7 + + drivers/clk/qcom/camcc-eliza.c | 2803 ++++++++ + drivers/clk/qcom/camcc-glymur.c | 7 + + drivers/clk/qcom/camcc-hawi.c | 3152 +++++++++ + drivers/clk/qcom/camcc-nord.c | 2941 +++++++++ + drivers/clk/qcom/camcc-sa8775p.c | 36 +- + drivers/clk/qcom/camcc-sc8280xp.c | 13 +- + drivers/clk/qcom/camcc-sm7150.c | 13 +- + drivers/clk/qcom/camcc-sm8150.c | 13 +- + drivers/clk/qcom/clk-alpha-pll.c | 9 + + drivers/clk/qcom/clk-rcg2.c | 35 +- + drivers/clk/qcom/clk-regmap-divider.c | 16 +- + drivers/clk/qcom/clk-regmap-divider.h | 1 + + drivers/clk/qcom/clk-rpmh.c | 44 + + drivers/clk/qcom/dispcc-sc7280.c | 13 +- + drivers/clk/qcom/dispcc-sc8280xp.c | 18 +- + drivers/clk/qcom/dispcc-sm4450.c | 15 +- + drivers/clk/qcom/dispcc-sm6115.c | 13 +- + drivers/clk/qcom/dispcc-sm7150.c | 13 +- + drivers/clk/qcom/dispcc-sm8250.c | 13 +- + drivers/clk/qcom/dispcc-sm8450.c | 40 +- + drivers/clk/qcom/dispcc-sm8550.c | 13 +- + drivers/clk/qcom/dispcc-sm8750.c | 19 +- + drivers/clk/qcom/dispcc-x1e80100.c | 27 +- + drivers/clk/qcom/dispcc0-sa8775p.c | 15 +- + drivers/clk/qcom/dispcc1-sa8775p.c | 15 +- + drivers/clk/qcom/gcc-eliza.c | 13 +- + drivers/clk/qcom/gcc-glymur.c | 81 +- + drivers/clk/qcom/gcc-hawi.c | 12 +- + drivers/clk/qcom/gcc-ipq5018.c | 10 + + drivers/clk/qcom/gcc-ipq5210.c | 1 + + drivers/clk/qcom/gcc-ipq5332.c | 1 + + drivers/clk/qcom/gcc-ipq5424.c | 4 +- + drivers/clk/qcom/gcc-ipq9650.c | 22 + + drivers/clk/qcom/gcc-kaanapali.c | 13 +- + drivers/clk/qcom/gcc-kuno.c | 1473 +++++ + drivers/clk/qcom/gcc-milos.c | 12 +- + drivers/clk/qcom/gcc-msm8939.c | 4 + + drivers/clk/qcom/gcc-msm8952.c | 3548 ++++++++++ + drivers/clk/qcom/gcc-nord.c | 12 +- + drivers/clk/qcom/gcc-qcm2290.c | 12 +- + drivers/clk/qcom/gcc-qcs615.c | 69 +- + drivers/clk/qcom/gcc-qcs8300.c | 167 +- + drivers/clk/qcom/gcc-qdu1000.c | 12 +- + drivers/clk/qcom/gcc-sa8775p.c | 56 +- + drivers/clk/qcom/gcc-sar2130p.c | 58 +- + drivers/clk/qcom/gcc-sc7180.c | 60 +- + drivers/clk/qcom/gcc-sc7280.c | 67 +- + drivers/clk/qcom/gcc-sc8280xp.c | 47 +- + drivers/clk/qcom/gcc-sdx55.c | 43 +- + drivers/clk/qcom/gcc-sdx65.c | 43 +- + drivers/clk/qcom/gcc-sdx75.c | 44 +- + drivers/clk/qcom/gcc-shikra.c | 12 +- + drivers/clk/qcom/gcc-sm4450.c | 77 +- + drivers/clk/qcom/gcc-sm6115.c | 12 +- + drivers/clk/qcom/gcc-sm6125.c | 12 +- + drivers/clk/qcom/gcc-sm6375.c | 12 +- + drivers/clk/qcom/gcc-sm7150.c | 72 +- + drivers/clk/qcom/gcc-sm8150.c | 12 +- + drivers/clk/qcom/gcc-sm8250.c | 69 +- + drivers/clk/qcom/gcc-sm8350.c | 65 +- + drivers/clk/qcom/gcc-sm8450.c | 64 +- + drivers/clk/qcom/gcc-sm8550.c | 70 +- + drivers/clk/qcom/gcc-sm8650.c | 72 +- + drivers/clk/qcom/gcc-sm8750.c | 89 +- + drivers/clk/qcom/gcc-x1e80100.c | 80 +- + drivers/clk/qcom/gpucc-eliza.c | 606 ++ + drivers/clk/qcom/gpucc-hawi.c | 477 ++ + drivers/clk/qcom/gpucc-sar2130p.c | 13 +- + drivers/clk/qcom/gpucc-sc7280.c | 22 +- + drivers/clk/qcom/gpucc-sc8280xp.c | 15 +- + drivers/clk/qcom/gpucc-sm4450.c | 17 +- + drivers/clk/qcom/gpucc-sm8550.c | 15 +- + drivers/clk/qcom/gpucc-x1e80100.c | 13 +- + drivers/clk/qcom/gpucc-x1p42100.c | 17 +- + drivers/clk/qcom/ipq-cmn-pll.c | 538 +- + drivers/clk/qcom/lpasscc-sc7280.c | 12 +- + drivers/clk/qcom/lpasscc-sdm845.c | 12 +- + drivers/clk/qcom/lpasscc-sm6115.c | 4 +- + drivers/clk/qcom/lpasscorecc-sc7180.c | 31 +- + drivers/clk/qcom/mmcc-msm8960.c | 3 +- + drivers/clk/qcom/mmcc-msm8996.c | 4 +- + drivers/clk/qcom/mmcc-sdm660.c | 2 +- + drivers/clk/qcom/negcc-nord.c | 5 +- + drivers/clk/qcom/nwgcc-nord.c | 5 +- + drivers/clk/qcom/segcc-nord.c | 40 +- + drivers/clk/qcom/tcsrcc-eliza.c | 12 +- + drivers/clk/qcom/tcsrcc-glymur.c | 12 +- + drivers/clk/qcom/tcsrcc-kaanapali.c | 12 +- + drivers/clk/qcom/tcsrcc-sm8550.c | 12 +- + drivers/clk/qcom/tcsrcc-sm8650.c | 12 +- + drivers/clk/qcom/tcsrcc-sm8750.c | 12 +- + drivers/clk/qcom/tcsrcc-x1e80100.c | 349 +- + drivers/clk/qcom/videocc-eliza.c | 404 ++ + drivers/clk/qcom/videocc-glymur.c | 38 + + drivers/clk/qcom/videocc-sa8775p.c | 17 +- + drivers/clk/qcom/videocc-sm6350.c | 13 +- + drivers/clk/qcom/videocc-sm7150.c | 13 +- + drivers/clk/qcom/videocc-sm8250.c | 15 +- + drivers/clk/qcom/videocc-sm8350.c | 31 +- + drivers/clk/qcom/videocc-sm8750.c | 12 +- + drivers/clk/renesas/r9a06g032-clocks.c | 8 +- + drivers/clk/renesas/renesas-cpg-mssr.c | 7 +- + drivers/clk/renesas/rzg2l-cpg.c | 7 +- + drivers/clk/renesas/rzv2h-cpg.c | 7 +- + drivers/firmware/qcom/Kconfig | 5 +- + drivers/firmware/qcom/qcom_scm-legacy.c | 2 +- + drivers/firmware/qcom/qcom_scm.c | 187 +- + drivers/firmware/qcom/qcom_tzmem.c | 3 +- + drivers/soc/qcom/Kconfig | 1 - + drivers/soc/qcom/apr.c | 75 +- + drivers/soc/qcom/llcc-qcom.c | 199 +- + drivers/soc/qcom/pmic_glink.c | 3 +- + drivers/soc/qcom/pmic_glink_altmode.c | 4 +- + drivers/soc/qcom/pmic_pdcharger_ulog.c | 2 +- + drivers/soc/qcom/qcom-geni-se.c | 69 +- + drivers/soc/qcom/qcom_aoss.c | 4 +- + drivers/soc/qcom/qcom_stats.c | 9 + + drivers/soc/qcom/rpmh-internal.h | 16 +- + drivers/soc/qcom/rpmh-rsc.c | 4 +- + drivers/soc/qcom/rpmh.c | 40 +- + drivers/soc/qcom/smem_dramc.c | 81 +- + drivers/soc/qcom/smp2p.c | 22 +- + drivers/soc/qcom/smsm.c | 36 +- + drivers/soc/qcom/socinfo.c | 3 + + drivers/soc/qcom/ubwc_config.c | 12 + + include/dt-bindings/arm/qcom,ids.h | 2 + + include/dt-bindings/clock/qcom,gcc-msm8952.h | 215 + + include/dt-bindings/clock/qcom,hawi-camcc.h | 165 + + include/dt-bindings/clock/qcom,hawi-gpucc.h | 47 + + include/dt-bindings/clock/qcom,ipq5210-cmn-pll.h | 30 + + include/dt-bindings/clock/qcom,kuno-gcc.h | 100 + + include/dt-bindings/clock/qcom,nord-camcc.h | 167 + + include/dt-bindings/clock/qcom,nord-negcc.h | 1 + + include/dt-bindings/clock/qcom,nord-nwgcc.h | 3 + + include/dt-bindings/clock/qcom,nord-segcc.h | 2 + + include/dt-bindings/clock/qcom,qcs8300-gcc.h | 6 + + include/dt-bindings/interconnect/qcom,ipq9650.h | 28 + + include/linux/device-id/apr.h | 21 - + include/linux/device/driver.h | 31 + + include/linux/firmware/qcom/qcom_scm.h | 29 - + include/linux/mod_devicetable.h | 1 - + include/linux/platform_device.h | 30 + + include/linux/soc/qcom/apr.h | 4 +- + include/linux/soc/qcom/geni-se.h | 2 +- + include/linux/soc/qcom/llcc-qcom.h | 6 + + include/linux/soc/qcom/ubwc.h | 3 + + 355 files changed, 43776 insertions(+), 2601 deletions(-) + create mode 100644 Documentation/devicetree/bindings/clock/qcom,gcc-msm8952.yaml + create mode 100644 Documentation/devicetree/bindings/clock/qcom,hawi-gpucc.yaml + create mode 100644 Documentation/devicetree/bindings/clock/qcom,kuno-gcc.yaml + delete mode 100644 Documentation/devicetree/bindings/clock/qcom,sm7150-gcc.yaml + create mode 100644 Documentation/devicetree/bindings/clock/qcom,x1e80100-tcsr.yaml + create mode 100644 arch/arm64/boot/dts/qcom/glymur-qcb.dts + create mode 100644 arch/arm64/boot/dts/qcom/hamoa-advantech-som-6820.dtsi + create mode 100644 arch/arm64/boot/dts/qcom/hamoa-advantech-som-db5830.dts + create mode 100644 arch/arm64/boot/dts/qcom/hawi-ipcc.h + create mode 100644 arch/arm64/boot/dts/qcom/hawi-mtp.dts + create mode 100644 arch/arm64/boot/dts/qcom/hawi.dtsi + create mode 100644 arch/arm64/boot/dts/qcom/mahua-lenovo-thinkpad-t14s-gen7-lcd.dts + create mode 100644 arch/arm64/boot/dts/qcom/msm8953-motorola-deen.dts + create mode 100644 arch/arm64/boot/dts/qcom/msm8976-leeco-s2.dts + create mode 100644 arch/arm64/boot/dts/qcom/pmh0104-hawi.dtsi + create mode 100644 arch/arm64/boot/dts/qcom/qcs8550-ayaneo-pocket-ds.dts + create mode 100644 arch/arm64/boot/dts/qcom/qcs8550-ayntec-common.dtsi + create mode 100644 arch/arm64/boot/dts/qcom/qcs8550-ayntec-odin2mini.dts + create mode 100644 arch/arm64/boot/dts/qcom/qcs8550-ayntec-odin2portal.dts + create mode 100644 arch/arm64/boot/dts/qcom/qcs8550-ayntec-thor.dts + create mode 100644 arch/arm64/boot/dts/qcom/qcs8550-imdt-sbc.dts + create mode 100644 arch/arm64/boot/dts/qcom/qcs8550-imdt-som.dtsi + create mode 100644 arch/arm64/boot/dts/qcom/qcs8550-retroidpocket-nova.dts + create mode 100644 arch/arm64/boot/dts/qcom/qcs8550-retroidpocket-rp6.dts + create mode 100644 arch/arm64/boot/dts/qcom/shikra-cqm-cqs-evk-imx577-camera.dtso + create mode 100644 arch/arm64/boot/dts/qcom/shikra-iqs-evk-imx577-camera.dtso + create mode 100644 arch/arm64/boot/dts/qcom/sm6115-xiaomi-lemon.dts + create mode 100644 arch/arm64/boot/dts/qcom/sm7325-xiaomi-lisa.dts + create mode 100644 arch/arm64/boot/dts/qcom/sm8650-valve-deckard.dts + create mode 100644 drivers/clk/qcom/cambistmclkcc-eliza.c + create mode 100644 drivers/clk/qcom/camcc-eliza.c + create mode 100644 drivers/clk/qcom/camcc-hawi.c + create mode 100644 drivers/clk/qcom/camcc-nord.c + create mode 100644 drivers/clk/qcom/gcc-kuno.c + create mode 100644 drivers/clk/qcom/gcc-msm8952.c + create mode 100644 drivers/clk/qcom/gpucc-eliza.c + create mode 100644 drivers/clk/qcom/gpucc-hawi.c + create mode 100644 drivers/clk/qcom/videocc-eliza.c + create mode 100644 include/dt-bindings/clock/qcom,gcc-msm8952.h + create mode 100644 include/dt-bindings/clock/qcom,hawi-camcc.h + create mode 100644 include/dt-bindings/clock/qcom,hawi-gpucc.h + create mode 100644 include/dt-bindings/clock/qcom,ipq5210-cmn-pll.h + create mode 100644 include/dt-bindings/clock/qcom,kuno-gcc.h + create mode 100644 include/dt-bindings/clock/qcom,nord-camcc.h + create mode 100644 include/dt-bindings/interconnect/qcom,ipq9650.h + delete mode 100644 include/linux/device-id/apr.h +Merging realtek/for-next (3c778f0c9fa36 arm64: dts: realtek: Add GPIO support for RTD1625) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/yu_chun/linux.git realtek/for-next +Already up to date. +Merging renesas/next (fb050251fd4fe Merge branches 'renesas-arm-defconfig-for-v7.4', 'renesas-drivers-for-v7.4' and 'renesas-dts-for-v7.4' into renesas-next) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel.git renesas/next +Merge made by the 'ort' strategy. + .../devicetree/bindings/power/qcom,rpmpd.yaml | 2 + + .../bindings/power/renesas,rcar-sysc.yaml | 7 +- + .../bindings/power/riscv,rpmi-device-power.yaml | 65 +++ + .../power/riscv,rpmi-mpxy-device-power.yaml | 65 +++ + .../devicetree/bindings/soc/renesas/renesas.yaml | 37 ++ + arch/arm64/boot/dts/renesas/Makefile | 74 +++ + .../boot/dts/renesas/r8a774a3-hihope-rzg2m-ex.dts | 14 + + .../boot/dts/renesas/r8a774a3-hihope-rzg2m.dts | 49 ++ + arch/arm64/boot/dts/renesas/r8a774a3.dtsi | 302 +++++++++++ + arch/arm64/boot/dts/renesas/r8a779f0.dtsi | 16 +- + arch/arm64/boot/dts/renesas/r8a779g0.dtsi | 16 +- + arch/arm64/boot/dts/renesas/r8a779h0.dtsi | 8 +- + arch/arm64/boot/dts/renesas/r8a78000-ironhide.dts | 12 +- + arch/arm64/boot/dts/renesas/r8a78000.dtsi | 216 ++++++++ + .../renesas/r9a07g043u12-hummingboard-ripple.dts | 89 ++++ + .../dts/renesas/r9a07g044c2-hummingboard-iiot.dts | 20 + + .../renesas/r9a07g044c2-hummingboard-ripple.dts | 93 ++++ + .../dts/renesas/r9a07g044l2-hummingboard-iiot.dts | 16 + + .../renesas/r9a07g044l2-hummingboard-ripple.dts | 18 + + .../dts/renesas/r9a07g054l2-hummingboard-iiot.dts | 16 + + .../renesas/r9a07g054l2-hummingboard-ripple.dts | 18 + + arch/arm64/boot/dts/renesas/r9a08g046.dtsi | 70 ++- + arch/arm64/boot/dts/renesas/r9a08g046l48-smarc.dts | 36 ++ + arch/arm64/boot/dts/renesas/r9a09g047.dtsi | 124 +++++ + arch/arm64/boot/dts/renesas/r9a09g047e57-smarc.dts | 31 ++ + arch/arm64/boot/dts/renesas/r9a09g077.dtsi | 24 +- + arch/arm64/boot/dts/renesas/r9a09g087.dtsi | 24 +- + arch/arm64/boot/dts/renesas/renesas-smarc2.dtsi | 25 + + .../renesas/rzg2l-hummingboard-iiot-common.dtsi | 560 +++++++++++++++++++++ + .../renesas/rzg2l-hummingboard-iiot-microsd.dtso | 26 + + ...hummingboard-iiot-panel-dsi-WJ70N3TYJHMNG0.dtso | 79 +++ + .../renesas/rzg2l-hummingboard-iiot-rs485-a.dtso | 17 + + .../renesas/rzg2l-hummingboard-iiot-rs485-b.dtso | 17 + + .../boot/dts/renesas/rzg2l-hummingboard-iiot.dtsi | 49 ++ + .../renesas/rzg2l-hummingboard-pulse-common.dtsi | 116 +++++ + .../rzg2l-hummingboard-pulse-micro-hdmi.dtsi | 79 +++ + .../dts/renesas/rzg2l-hummingboard-ripple.dtsi | 122 +++++ + arch/arm64/boot/dts/renesas/rzg2l-sr-som-emmc.dtso | 44 ++ + arch/arm64/boot/dts/renesas/rzg2l-sr-som.dtsi | 473 +++++++++++++++++ + .../renesas/rzg2lc-hummingboard-pulse-leds.dtso | 48 ++ + arch/arm64/boot/dts/renesas/rzg2lc-sr-som.dtsi | 436 ++++++++++++++++ + .../renesas/rzg2ul-hummingboard-pulse-leds.dtso | 48 ++ + arch/arm64/boot/dts/renesas/rzg2ul-sr-som.dtsi | 418 +++++++++++++++ + arch/arm64/boot/dts/renesas/rzg3l-smarc-som.dtsi | 11 + + drivers/soc/renesas/Kconfig | 3 + + .../dt-bindings/clock/renesas,r8a774a3-cpg-mssr.h | 59 +++ + include/dt-bindings/clock/renesas,r8a78000-cpg.h | 2 + + include/dt-bindings/power/qcom,rpmhpd.h | 1 + + include/dt-bindings/power/renesas,r8a774a3-sysc.h | 30 ++ + 49 files changed, 4071 insertions(+), 54 deletions(-) + create mode 100644 Documentation/devicetree/bindings/power/riscv,rpmi-device-power.yaml + create mode 100644 Documentation/devicetree/bindings/power/riscv,rpmi-mpxy-device-power.yaml + create mode 100644 arch/arm64/boot/dts/renesas/r8a774a3-hihope-rzg2m-ex.dts + create mode 100644 arch/arm64/boot/dts/renesas/r8a774a3-hihope-rzg2m.dts + create mode 100644 arch/arm64/boot/dts/renesas/r8a774a3.dtsi + create mode 100644 arch/arm64/boot/dts/renesas/r9a07g043u12-hummingboard-ripple.dts + create mode 100644 arch/arm64/boot/dts/renesas/r9a07g044c2-hummingboard-iiot.dts + create mode 100644 arch/arm64/boot/dts/renesas/r9a07g044c2-hummingboard-ripple.dts + create mode 100644 arch/arm64/boot/dts/renesas/r9a07g044l2-hummingboard-iiot.dts + create mode 100644 arch/arm64/boot/dts/renesas/r9a07g044l2-hummingboard-ripple.dts + create mode 100644 arch/arm64/boot/dts/renesas/r9a07g054l2-hummingboard-iiot.dts + create mode 100644 arch/arm64/boot/dts/renesas/r9a07g054l2-hummingboard-ripple.dts + create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-common.dtsi + create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-microsd.dtso + create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-panel-dsi-WJ70N3TYJHMNG0.dtso + create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-rs485-a.dtso + create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-rs485-b.dtso + create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot.dtsi + create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-pulse-common.dtsi + create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-pulse-micro-hdmi.dtsi + create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-ripple.dtsi + create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-sr-som-emmc.dtso + create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-sr-som.dtsi + create mode 100644 arch/arm64/boot/dts/renesas/rzg2lc-hummingboard-pulse-leds.dtso + create mode 100644 arch/arm64/boot/dts/renesas/rzg2lc-sr-som.dtsi + create mode 100644 arch/arm64/boot/dts/renesas/rzg2ul-hummingboard-pulse-leds.dtso + create mode 100644 arch/arm64/boot/dts/renesas/rzg2ul-sr-som.dtsi + create mode 100644 include/dt-bindings/clock/renesas,r8a774a3-cpg-mssr.h + create mode 100644 include/dt-bindings/power/renesas,r8a774a3-sysc.h +Merging reset/reset/next (d373605cd5148 Merge tag 'reset-fixes-for-v7.0-2' into reset/next) +$ git merge -m Merge branch 'reset/next' of https://git.kernel.org/pub/scm/linux/kernel/git/pza/linux reset/reset/next +Already up to date. +Merging rockchip/for-next (f4d0e3d48d590 Merge branch 'v7.4-armsoc/dts64' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mmind/linux-rockchip.git rockchip/for-next +Merge made by the 'ort' strategy. + .../devicetree/bindings/arm/rockchip.yaml | 22 +- + arch/arm/boot/dts/rockchip/rk3229-xms6.dts | 4 +- + arch/arm/boot/dts/rockchip/rv1103b-omega4.dtsi | 1 - + arch/arm64/boot/dts/rockchip/Makefile | 3 + + .../boot/dts/rockchip/rk3308-unihiker-m10-v1.2.dts | 248 +++++++++++++ + arch/arm64/boot/dts/rockchip/rk3399-evb-ind.dts | 1 - + .../boot/dts/rockchip/rk3399-pinebook-pro.dts | 4 + + .../boot/dts/rockchip/rk3399-rockpro64-v2.dts | 4 + + arch/arm64/boot/dts/rockchip/rk3399-rockpro64.dts | 4 + + .../boot/dts/rockchip/rk3528-mangopi-m28k.dts | 412 +++++++++++++++++++++ + .../arm64/boot/dts/rockchip/rk3528-nanopi-r28s.dts | 160 ++++++++ + .../boot/dts/rockchip/rk3528-nanopi-zero2.dts | 290 +-------------- + arch/arm64/boot/dts/rockchip/rk3528-nanopi.dtsi | 291 +++++++++++++++ + arch/arm64/boot/dts/rockchip/rk3562-evb2-v10.dts | 1 - + arch/arm64/boot/dts/rockchip/rk3566-nanopi-r3s.dts | 1 - + .../boot/dts/rockchip/rk3568-fastrhino-r66s.dts | 1 - + arch/arm64/boot/dts/rockchip/rk3568-lubancat-2.dts | 1 - + .../arm64/boot/dts/rockchip/rk3568-nanopi-r5s.dtsi | 1 - + arch/arm64/boot/dts/rockchip/rk3568-rock-3b.dts | 51 +++ + .../dts/rockchip/rk3576-anbernic-rg-vita-pro.dts | 2 - + .../boot/dts/rockchip/rk3576-armsom-cm5-io.dts | 1 - + .../boot/dts/rockchip/rk3576-armsom-sige5.dts | 1 - + arch/arm64/boot/dts/rockchip/rk3576-evb1-v10.dts | 1 - + arch/arm64/boot/dts/rockchip/rk3576-nanopi-m5.dts | 1 - + .../arm64/boot/dts/rockchip/rk3576-nanopi-r76s.dts | 1 - + arch/arm64/boot/dts/rockchip/rk3576-roc-pc.dts | 1 - + arch/arm64/boot/dts/rockchip/rk3576-rock-4d.dts | 1 - + arch/arm64/boot/dts/rockchip/rk3576.dtsi | 5 + + .../boot/dts/rockchip/rk3588-armsom-sige7.dts | 1 - + arch/arm64/boot/dts/rockchip/rk3588-armsom-w3.dts | 1 - + .../arm64/boot/dts/rockchip/rk3588-coolpi-cm5.dtsi | 1 - + .../boot/dts/rockchip/rk3588-edgeble-neu6a-io.dtsi | 1 - + .../boot/dts/rockchip/rk3588-lubancat-5io.dts | 1 - + arch/arm64/boot/dts/rockchip/rk3588-nanopc-t6.dtsi | 1 - + arch/arm64/boot/dts/rockchip/rk3588-ok3588-c.dts | 1 - + arch/arm64/boot/dts/rockchip/rk3588-rock-5-itx.dts | 2 - + .../boot/dts/rockchip/rk3588-rock-5b-5bp-5t.dtsi | 1 - + .../boot/dts/rockchip/rk3588-vicharak-axon.dts | 1 - + .../boot/dts/rockchip/rk3588-vicharak-vaaman2.dts | 1 - + .../boot/dts/rockchip/rk3588-youyeetoo-yy3588.dts | 1 - + arch/arm64/boot/dts/rockchip/rk3588s-coolpi-4b.dts | 1 - + arch/arm64/boot/dts/rockchip/rk3588s-evb1-v10.dts | 1 - + .../boot/dts/rockchip/rk3588s-gameforce-ace.dts | 1 - + .../boot/dts/rockchip/rk3588s-indiedroid-nova.dts | 1 - + .../arm64/boot/dts/rockchip/rk3588s-lubancat-4.dts | 1 - + arch/arm64/boot/dts/rockchip/rk3588s-rock-5a.dts | 1 - + arch/arm64/boot/dts/rockchip/rk3588s-rock-5c.dts | 1 - + 47 files changed, 1206 insertions(+), 328 deletions(-) + create mode 100644 arch/arm64/boot/dts/rockchip/rk3308-unihiker-m10-v1.2.dts + create mode 100644 arch/arm64/boot/dts/rockchip/rk3528-mangopi-m28k.dts + create mode 100644 arch/arm64/boot/dts/rockchip/rk3528-nanopi-r28s.dts + create mode 100644 arch/arm64/boot/dts/rockchip/rk3528-nanopi.dtsi +Merging samsung-krzk/for-next (08df370772f32 Merge branch 'for-v7.4/google-lga' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux.git samsung-krzk/for-next +Auto-merging Documentation/devicetree/bindings/sram/sram.yaml +Auto-merging MAINTAINERS +Auto-merging arch/arm64/Kconfig.platforms +Auto-merging arch/arm64/configs/defconfig +Merge made by the 'ort' strategy. + Documentation/devicetree/bindings/arm/google.yaml | 48 +- + .../bindings/arm/samsung/samsung-boards.yaml | 1 + + .../bindings/clock/samsung,exynos5515-cmu.yaml | 145 ++ + .../devicetree/bindings/gpu/arm,mali-bifrost.yaml | 1 + + .../bindings/hwinfo/samsung,exynos-chipid.yaml | 1 + + .../bindings/input/samsung,s3c6410-keypad.yaml | 53 +- + Documentation/devicetree/bindings/sram/sram.yaml | 2 + + MAINTAINERS | 1 + + arch/arm/boot/dts/samsung/exynos4210-i9100.dts | 9 +- + arch/arm/boot/dts/samsung/exynos4210-trats.dts | 5 + + .../boot/dts/samsung/exynos4210-universal_c210.dts | 5 + + arch/arm/boot/dts/samsung/exynos4412-midas.dtsi | 5 +- + arch/arm/mach-exynos/smc.h | 4 +- + arch/arm/mach-s3c/Kconfig | 5 - + arch/arm/mach-s3c/Kconfig.s3c64xx | 7 - + arch/arm/mach-s3c/Makefile.s3c64xx | 1 - + arch/arm/mach-s3c/devs.c | 27 - + arch/arm/mach-s3c/devs.h | 1 - + arch/arm/mach-s3c/gpio-core.h | 3 + + arch/arm/mach-s3c/gpio-samsung-s3c64xx.h | 5 + + arch/arm/mach-s3c/gpio-samsung.c | 72 +- + arch/arm/mach-s3c/keypad.h | 27 - + arch/arm/mach-s3c/mach-crag6410.c | 135 +- + arch/arm/mach-s3c/setup-keypad-s3c64xx.c | 20 - + arch/arm64/Kconfig.platforms | 8 + + arch/arm64/boot/dts/Makefile | 1 + + arch/arm64/boot/dts/exynos/Makefile | 1 + + arch/arm64/boot/dts/exynos/axis/artpec8.dtsi | 36 +- + arch/arm64/boot/dts/exynos/axis/artpec9.dtsi | 36 +- + arch/arm64/boot/dts/exynos/exynos2200-g0s.dts | 8 +- + arch/arm64/boot/dts/exynos/exynos2200-pinctrl.dtsi | 2 - + arch/arm64/boot/dts/exynos/exynos2200.dtsi | 1 - + .../boot/dts/exynos/exynos5433-tm2-common.dtsi | 93 +- + arch/arm64/boot/dts/exynos/exynos5433.dtsi | 108 +- + .../arm64/boot/dts/exynos/exynos7870-a2corelte.dts | 39 +- + arch/arm64/boot/dts/exynos/exynos7870-j5y17lte.dts | 34 +- + arch/arm64/boot/dts/exynos/exynos7870-j6lte.dts | 34 +- + arch/arm64/boot/dts/exynos/exynos7870-j7xelte.dts | 35 +- + arch/arm64/boot/dts/exynos/exynos7870-on7xelte.dts | 39 +- + arch/arm64/boot/dts/exynos/exynos7870.dtsi | 108 +- + arch/arm64/boot/dts/exynos/exynos850-a217f.dts | 200 +++ + arch/arm64/boot/dts/exynos/exynos850.dtsi | 29 +- + arch/arm64/boot/dts/exynos/exynos8855-smdk.dts | 2 +- + arch/arm64/boot/dts/exynos/exynosautov920.dtsi | 2 +- + .../boot/dts/exynos/google/gs101-pixel-common.dtsi | 51 +- + arch/arm64/boot/dts/exynos/google/gs101.dtsi | 96 +- + arch/arm64/boot/dts/google/Makefile | 6 + + arch/arm64/boot/dts/google/lga-blazer.dts | 15 + + arch/arm64/boot/dts/google/lga-frankel.dts | 15 + + arch/arm64/boot/dts/google/lga-mustang.dts | 15 + + arch/arm64/boot/dts/google/lga-pixel-common.dtsi | 22 + + arch/arm64/boot/dts/google/lga.dtsi | 411 ++++++ + arch/arm64/boot/dts/tesla/fsd-evb.dts | 10 +- + arch/arm64/boot/dts/tesla/fsd-pinctrl.dtsi | 16 +- + arch/arm64/boot/dts/tesla/fsd.dtsi | 1162 +++++++-------- + arch/arm64/configs/defconfig | 1 + + drivers/clk/samsung/Makefile | 1 + + drivers/clk/samsung/clk-acpm.c | 61 +- + drivers/clk/samsung/clk-exynos5515.c | 1484 ++++++++++++++++++++ + drivers/clk/samsung/clk-pll.c | 1 + + drivers/clk/samsung/clk-pll.h | 1 + + drivers/input/keyboard/samsung-keypad.c | 200 ++- + drivers/soc/samsung/exynos-chipid.c | 1 + + include/dt-bindings/clock/samsung,exynos5515-cmu.h | 191 +++ + include/linux/input/samsung-keypad.h | 39 - + 65 files changed, 3922 insertions(+), 1276 deletions(-) + create mode 100644 Documentation/devicetree/bindings/clock/samsung,exynos5515-cmu.yaml + delete mode 100644 arch/arm/mach-s3c/keypad.h + delete mode 100644 arch/arm/mach-s3c/setup-keypad-s3c64xx.c + create mode 100644 arch/arm64/boot/dts/exynos/exynos850-a217f.dts + create mode 100644 arch/arm64/boot/dts/google/Makefile + create mode 100644 arch/arm64/boot/dts/google/lga-blazer.dts + create mode 100644 arch/arm64/boot/dts/google/lga-frankel.dts + create mode 100644 arch/arm64/boot/dts/google/lga-mustang.dts + create mode 100644 arch/arm64/boot/dts/google/lga-pixel-common.dtsi + create mode 100644 arch/arm64/boot/dts/google/lga.dtsi + create mode 100644 drivers/clk/samsung/clk-exynos5515.c + create mode 100644 include/dt-bindings/clock/samsung,exynos5515-cmu.h + delete mode 100644 include/linux/input/samsung-keypad.h +Merging scmi/for-linux-next (e917780f1c75d Merge branch 'for-next/vexpress/fixes', tags 'scmi-updates-7.4' and 'juno-updates-7.4' of git://git.kernel.org/pub/scm/linux/kernel/git/sudeep.holla/linux) +$ git merge -m Merge branch 'for-linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/sudeep.holla/linux.git scmi/for-linux-next +Merge made by the 'ort' strategy. + arch/arm/mach-versatile/platsmp-vexpress.c | 4 ++-- + arch/arm/mach-versatile/v2m.c | 4 ++-- + 2 files changed, 4 insertions(+), 4 deletions(-) +Merging sophgo/for-next (76acfee87c74d Merge branch 'dt/arm' into for-next) +$ git merge -m Merge branch 'for-next' of https://github.com/sophgo/linux.git sophgo/for-next +Merge made by the 'ort' strategy. +Merging sophgo-soc/soc-for-next (c8754c7deab4c soc: sophgo: cv1800: rtcsys: New driver (handling RTC only)) +$ git merge -m Merge branch 'soc-for-next' of https://github.com/sophgo/linux.git sophgo-soc/soc-for-next +Already up to date. +Merging spacemit/for-next (44034f5791c10 Merge branch 'spacemit-misc-for-next' into spacemit-for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/spacemit/linux spacemit/for-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + .../thead,c900-aclint-mswi.yaml | 1 + + .../thead,c900-aclint-sswi.yaml | 5 +- + .../devicetree/bindings/timer/sifive,clint.yaml | 1 - + .../bindings/timer/thead,c900-aclint-mtimer.yaml | 1 + + MAINTAINERS | 9 + + arch/riscv/boot/dts/spacemit/k1-musepi-pro.dts | 35 +++- + arch/riscv/boot/dts/spacemit/k1-orangepi-r2s.dts | 65 +++++- + arch/riscv/boot/dts/spacemit/k1-orangepi-rv2.dts | 128 +++++++++--- + arch/riscv/boot/dts/spacemit/k3-com260-ifx.dts | 77 +++++++ + arch/riscv/boot/dts/spacemit/k3-com260.dtsi | 18 ++ + arch/riscv/boot/dts/spacemit/k3-pico-itx.dts | 38 ++++ + arch/riscv/boot/dts/spacemit/k3-pinctrl.dtsi | 11 + + arch/riscv/boot/dts/spacemit/k3.dtsi | 229 +++++++++++++++++++-- + drivers/irqchip/irq-aclint-sswi.c | 1 + + 14 files changed, 563 insertions(+), 56 deletions(-) +Merging stm32/stm32-next (f831584128ac2 soc: st: Add STM32MP2 SYSCFG driver) +$ git merge -m Merge branch 'stm32-next' of https://git.kernel.org/pub/scm/linux/kernel/git/atorgue/stm32.git stm32/stm32-next +Auto-merging MAINTAINERS +Auto-merging arch/arm64/configs/defconfig +Auto-merging drivers/soc/Makefile +Merge made by the 'ort' strategy. + .../bindings/arm/stm32/st,stm32-syscon.yaml | 30 +- + .../devicetree/bindings/arm/stm32/stm32.yaml | 39 +- + MAINTAINERS | 3 +- + arch/arm/boot/dts/st/stm32mp131.dtsi | 24 + + arch/arm/boot/dts/st/stm32mp135.dtsi | 44 +- + arch/arm/boot/dts/st/stm32mp135f-dk.dts | 28 +- + arch/arm/boot/dts/st/stm32mp15-pinctrl.dtsi | 34 + + arch/arm/boot/dts/st/stm32mp151.dtsi | 20 +- + .../arm/boot/dts/st/stm32mp153c-lxa-fairytux2.dtsi | 2 +- + arch/arm/boot/dts/st/stm32mp157a-dk1-scmi.dts | 8 +- + arch/arm/boot/dts/st/stm32mp157c-dk2-scmi.dts | 8 +- + arch/arm/boot/dts/st/stm32mp157c-ed1-scmi.dts | 8 +- + arch/arm/boot/dts/st/stm32mp157c-ev1-scmi.dts | 8 +- + arch/arm/boot/dts/st/stm32mp157c-ev1.dts | 9 +- + arch/arm/boot/dts/st/stm32mp157c-lxa-mc1.dts | 2 +- + arch/arm/boot/dts/st/stm32mp157f-dk2-scmi.dtsi | 1 + + arch/arm/boot/dts/st/stm32mp15xc-lxa-tac.dtsi | 2 +- + arch/arm/boot/dts/st/stm32mp15xx-dhcom-som.dtsi | 2 + + arch/arm/boot/dts/st/stm32mp15xx-dkx.dtsi | 35 +- + arch/arm64/boot/dts/st/Makefile | 23 + + arch/arm64/boot/dts/st/stm32mp231.dtsi | 105 ++- + arch/arm64/boot/dts/st/stm32mp235f-dk.dts | 75 ++ + arch/arm64/boot/dts/st/stm32mp23xc.dtsi | 7 + + arch/arm64/boot/dts/st/stm32mp23xx-dhcos-bb.dts | 15 + + arch/arm64/boot/dts/st/stm32mp23xx-dhcos-som.dtsi | 51 ++ + arch/arm64/boot/dts/st/stm32mp25-pinctrl.dtsi | 829 +++++++++++++++++++++ + arch/arm64/boot/dts/st/stm32mp251.dtsi | 67 +- + arch/arm64/boot/dts/st/stm32mp253.dtsi | 16 + + ...stm32mp255c-dhcos-dhsbc-overlay-imx219-x10.dtso | 111 +++ + arch/arm64/boot/dts/st/stm32mp255c-dhcos-dhsbc.dts | 213 ++++++ + .../boot/dts/st/stm32mp257-engicam-microgea.dtsi | 63 ++ + ...microgea-rmm-overlay-am-1280800w8tzqw-t00h.dtso | 67 ++ + ...cam-microgea-rmm-overlay-rk050hr345-ct106a.dtso | 68 ++ + .../dts/st/stm32mp257d-engicam-microgea-rmm.dts | 275 +++++++ + arch/arm64/boot/dts/st/stm32mp257f-dk.dts | 93 +++ + arch/arm64/boot/dts/st/stm32mp257f-ev1.dts | 101 ++- + arch/arm64/boot/dts/st/stm32mp25xc.dtsi | 7 + + arch/arm64/boot/dts/st/stm32mp25xx-dhcos-bb.dts | 15 + + arch/arm64/boot/dts/st/stm32mp25xx-dhcos-som.dtsi | 51 ++ + arch/arm64/boot/dts/st/stm32mp2xxx-dhcos-som.dtsi | 456 ++++++++++++ + arch/arm64/configs/defconfig | 5 + + drivers/soc/Kconfig | 1 + + drivers/soc/Makefile | 1 + + drivers/soc/st/Kconfig | 12 + + drivers/soc/st/Makefile | 1 + + drivers/soc/st/stm32mp2-syscfg.c | 33 + + 46 files changed, 2924 insertions(+), 144 deletions(-) + create mode 100644 arch/arm64/boot/dts/st/stm32mp23xc.dtsi + create mode 100644 arch/arm64/boot/dts/st/stm32mp23xx-dhcos-bb.dts + create mode 100644 arch/arm64/boot/dts/st/stm32mp23xx-dhcos-som.dtsi + create mode 100644 arch/arm64/boot/dts/st/stm32mp255c-dhcos-dhsbc-overlay-imx219-x10.dtso + create mode 100644 arch/arm64/boot/dts/st/stm32mp255c-dhcos-dhsbc.dts + create mode 100644 arch/arm64/boot/dts/st/stm32mp257-engicam-microgea.dtsi + create mode 100644 arch/arm64/boot/dts/st/stm32mp257d-engicam-microgea-rmm-overlay-am-1280800w8tzqw-t00h.dtso + create mode 100644 arch/arm64/boot/dts/st/stm32mp257d-engicam-microgea-rmm-overlay-rk050hr345-ct106a.dtso + create mode 100644 arch/arm64/boot/dts/st/stm32mp257d-engicam-microgea-rmm.dts + create mode 100644 arch/arm64/boot/dts/st/stm32mp25xc.dtsi + create mode 100644 arch/arm64/boot/dts/st/stm32mp25xx-dhcos-bb.dts + create mode 100644 arch/arm64/boot/dts/st/stm32mp25xx-dhcos-som.dtsi + create mode 100644 arch/arm64/boot/dts/st/stm32mp2xxx-dhcos-som.dtsi + create mode 100644 drivers/soc/st/Kconfig + create mode 100644 drivers/soc/st/Makefile + create mode 100644 drivers/soc/st/stm32mp2-syscfg.c +Merging sunxi/sunxi/for-next (afb622bddaa6f Merge branches 'sunxi/dt-for-7.4' and 'sunxi/clk-for-7.4' into sunxi/for-next) +$ git merge -m Merge branch 'sunxi/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/sunxi/linux.git sunxi/sunxi/for-next +Auto-merging Documentation/devicetree/bindings/vendor-prefixes.yaml +Merge made by the 'ort' strategy. + Documentation/devicetree/bindings/arm/sunxi.yaml | 10 + + .../devicetree/bindings/vendor-prefixes.yaml | 2 + + arch/arm/boot/dts/allwinner/sun7i-a20.dtsi | 2 +- + arch/arm64/boot/dts/allwinner/Makefile | 1 + + .../boot/dts/allwinner/sun50i-a133-teclast-p80.dts | 205 +++++++++++++++++++++ + drivers/clk/sunxi-ng/ccu-sun55i-a523-r.c | 37 ++-- + drivers/clk/sunxi-ng/ccu-sun55i-a523.c | 128 ++++++------- + drivers/clk/sunxi-ng/ccu_common.c | 2 +- + drivers/clk/sunxi-ng/ccu_div.h | 18 +- + 9 files changed, 308 insertions(+), 97 deletions(-) + create mode 100644 arch/arm64/boot/dts/allwinner/sun50i-a133-teclast-p80.dts +Merging tee/next (b0adf8e5e1ebf Merge branches 'optee_fixes_for_v7.3', 'tee_fix_for_v7.3', 'qcomtee_fix_for_v7.3', 'qcomtee_for_v7.4', 'tee_for_v7.4' and 'optee_for_v7.4' into next) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/jenswi/linux-tee.git tee/next +Merge made by the 'ort' strategy. + drivers/tee/optee/Kconfig | 1 + + drivers/tee/optee/Makefile | 4 ++-- + drivers/tee/optee/ffa_abi.c | 8 ++------ + drivers/tee/optee/notif.c | 1 - + drivers/tee/optee/optee_private.h | 39 ++++++++++++++++++++++++++++++++++- + drivers/tee/qcomtee/async.c | 4 ++-- + drivers/tee/qcomtee/qcomtee_msg.h | 2 +- + drivers/tee/qcomtee/qcomtee_object.h | 1 + + drivers/tee/tee_shm.c | 4 ++++ + include/uapi/linux/tee.h | 40 ++++++++++++++++++------------------ + 10 files changed, 71 insertions(+), 33 deletions(-) +Merging tegra/for-next (494422afb7e18 Merge branch for-7.4/arm64/dt into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux.git tegra/for-next +Auto-merging drivers/soc/tegra/pmc.c +Merge made by the 'ort' strategy. + .../ABI/testing/sysfs-platform-tegra-bpmp | 48 + + Documentation/devicetree/bindings/arm/tegra.yaml | 6 + + .../bindings/fuse/nvidia,tegra20-fuse.yaml | 1 + + .../memory-controllers/nvidia,tegra124-emc.yaml | 10 +- + .../memory-controllers/nvidia,tegra124-mc.yaml | 1 + + arch/arm/boot/dts/nvidia/Makefile | 2 + + arch/arm/boot/dts/nvidia/tegra114-asus-tf701t.dts | 885 ++++++++++++++- + arch/arm/boot/dts/nvidia/tegra114-dalmore.dts | 4 +- + arch/arm/boot/dts/nvidia/tegra114-roth.dts | 4 +- + arch/arm/boot/dts/nvidia/tegra114-tn7.dts | 4 +- + arch/arm/boot/dts/nvidia/tegra124.dtsi | 16 +- + .../boot/dts/nvidia/tegra20-motorola-daytona.dts | 107 ++ + arch/arm/boot/dts/nvidia/tegra20-motorola-mot.dtsi | 1194 ++++++++++++++++++++ + .../boot/dts/nvidia/tegra20-motorola-olympus.dts | 108 ++ + arch/arm/boot/dts/nvidia/tegra30-lg-p880.dts | 5 +- + arch/arm/boot/dts/nvidia/tegra30-lg-p895.dts | 5 +- + arch/arm/boot/dts/nvidia/tegra30-lg-x3.dtsi | 106 +- + arch/arm64/boot/dts/nvidia/tegra132.dtsi | 20 +- + .../dts/nvidia/tegra186-p3509-0000+p3636-0001.dts | 8 +- + arch/arm64/boot/dts/nvidia/tegra186.dtsi | 76 +- + arch/arm64/boot/dts/nvidia/tegra194-p2972-0000.dts | 8 +- + .../arm64/boot/dts/nvidia/tegra194-p3509-0000.dtsi | 8 +- + arch/arm64/boot/dts/nvidia/tegra194.dtsi | 113 +- + arch/arm64/boot/dts/nvidia/tegra210-p3450-0000.dts | 10 +- + arch/arm64/boot/dts/nvidia/tegra210.dtsi | 8 +- + .../boot/dts/nvidia/tegra234-p3737-0000+p3701.dtsi | 9 +- + .../dts/nvidia/tegra234-p3740-0002+p3701-0008.dts | 45 +- + .../boot/dts/nvidia/tegra234-p3768-0000+p3767.dtsi | 12 +- + arch/arm64/boot/dts/nvidia/tegra234.dtsi | 86 +- + arch/arm64/boot/dts/nvidia/tegra264-p3834.dtsi | 57 + + .../boot/dts/nvidia/tegra264-p4071-0000+p3834.dtsi | 93 ++ + arch/arm64/boot/dts/nvidia/tegra264.dtsi | 580 +++++----- + .../boot/dts/nvidia/thermal/nvidia,tegra264-bpmp.h | 20 + + drivers/firmware/tegra/Makefile | 1 + + drivers/firmware/tegra/bpmp-debugfs.c | 9 + + drivers/firmware/tegra/bpmp-private.h | 39 + + drivers/firmware/tegra/bpmp-sysfs.c | 207 ++++ + drivers/firmware/tegra/bpmp-tegra186.c | 14 + + drivers/firmware/tegra/bpmp-tegra210.c | 13 +- + drivers/firmware/tegra/bpmp.c | 458 +++++++- + drivers/soc/tegra/fuse/fuse-tegra.c | 11 +- + drivers/soc/tegra/fuse/fuse-tegra30.c | 168 ++- + drivers/soc/tegra/fuse/fuse.h | 7 +- + drivers/soc/tegra/pmc.c | 99 ++ + include/soc/tegra/bpmp-abi.h | 165 ++- + include/soc/tegra/bpmp.h | 2 + + 46 files changed, 4280 insertions(+), 572 deletions(-) + create mode 100644 Documentation/ABI/testing/sysfs-platform-tegra-bpmp + create mode 100644 arch/arm/boot/dts/nvidia/tegra20-motorola-daytona.dts + create mode 100644 arch/arm/boot/dts/nvidia/tegra20-motorola-mot.dtsi + create mode 100644 arch/arm/boot/dts/nvidia/tegra20-motorola-olympus.dts + create mode 100644 arch/arm64/boot/dts/nvidia/thermal/nvidia,tegra264-bpmp.h + create mode 100644 drivers/firmware/tegra/bpmp-sysfs.c +Merging tenstorrent-dt/tenstorrent-dt-for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'tenstorrent-dt-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/tenstorrent/linux.git tenstorrent-dt/tenstorrent-dt-for-next +Already up to date. +Merging fustini-config/riscv-config-for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'riscv-config-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git fustini-config/riscv-config-for-next +Already up to date. +Merging thead-dt/thead-dt-for-next (2414ca8f5a0ca MAINTAINERS: Move RISC-V T-Head SoC tree to kernel.org) +$ git merge -m Merge branch 'thead-dt-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git thead-dt/thead-dt-for-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + Documentation/devicetree/bindings/riscv/thead.yaml | 6 +- + MAINTAINERS | 2 +- + arch/riscv/boot/dts/thead/Makefile | 1 + + arch/riscv/boot/dts/thead/th1520-milkv-meles.dts | 346 +++++++++++++++++++++ + 4 files changed, 352 insertions(+), 3 deletions(-) + create mode 100644 arch/riscv/boot/dts/thead/th1520-milkv-meles.dts +Merging ti/ti-next (9e2716807275d Merge branch 'ti-k3-dts-next' into ti-next) +$ git merge -m Merge branch 'ti-next' of https://git.kernel.org/pub/scm/linux/kernel/git/ti/linux.git ti/ti-next +Auto-merging MAINTAINERS +Auto-merging arch/arm64/configs/defconfig +Auto-merging drivers/firmware/ti_sci.c +Merge made by the 'ort' strategy. + Documentation/devicetree/bindings/arm/ti/k3.yaml | 7 + + .../bindings/clock/ti,am62-audio-refclk.yaml | 8 +- + .../soc/ti/ti,j721e-system-controller.yaml | 6 +- + MAINTAINERS | 1 + + arch/arm/boot/dts/ti/keystone/keystone-k2e.dtsi | 2 +- + arch/arm/boot/dts/ti/keystone/keystone-k2g.dtsi | 28 +- + arch/arm/boot/dts/ti/keystone/keystone-k2hk.dtsi | 16 +- + arch/arm/boot/dts/ti/keystone/keystone-k2l.dtsi | 16 +- + arch/arm/boot/dts/ti/keystone/keystone.dtsi | 4 +- + arch/arm64/boot/dts/ti/Makefile | 12 + + arch/arm64/boot/dts/ti/k3-am62-lp-sk.dts | 2 +- + arch/arm64/boot/dts/ti/k3-am62-phycore-som.dtsi | 2 +- + arch/arm64/boot/dts/ti/k3-am62-pocketbeagle2.dts | 2 +- + arch/arm64/boot/dts/ti/k3-am62-verdin-ivy.dtsi | 41 +- + arch/arm64/boot/dts/ti/k3-am62-verdin-zinnia.dtsi | 41 +- + arch/arm64/boot/dts/ti/k3-am62-verdin.dtsi | 49 +-- + arch/arm64/boot/dts/ti/k3-am625-beagleplay.dts | 6 +- + arch/arm64/boot/dts/ti/k3-am625-sk-common.dtsi | 3 +- + .../boot/dts/ti/k3-am625-tqma62xx-mba62xx.dts | 4 +- + arch/arm64/boot/dts/ti/k3-am625-tqma62xx.dtsi | 2 +- + .../boot/dts/ti/k3-am625-var-som-symphony.dts | 2 +- + arch/arm64/boot/dts/ti/k3-am625-var-som.dtsi | 4 +- + arch/arm64/boot/dts/ti/k3-am62a-main.dtsi | 1 + + arch/arm64/boot/dts/ti/k3-am62a-mcu.dtsi | 2 +- + arch/arm64/boot/dts/ti/k3-am62a-phycore-som.dtsi | 31 +- + .../boot/dts/ti/k3-am62a-ti-ipc-firmware.dtsi | 11 +- + arch/arm64/boot/dts/ti/k3-am62a7-sk.dts | 35 +- + arch/arm64/boot/dts/ti/k3-am62d2-evm.dts | 32 +- + arch/arm64/boot/dts/ti/k3-am62l3-evm.dts | 4 +- + .../boot/dts/ti/k3-am62p-j722s-common-main.dtsi | 12 + + .../boot/dts/ti/k3-am62p-j722s-common-mcu.dtsi | 2 +- + .../boot/dts/ti/k3-am62p-ti-ipc-firmware.dtsi | 11 +- + arch/arm64/boot/dts/ti/k3-am62p-verdin.dtsi | 29 +- + arch/arm64/boot/dts/ti/k3-am62p5-sk.dts | 29 +- + arch/arm64/boot/dts/ti/k3-am62p5-var-som.dtsi | 29 +- + arch/arm64/boot/dts/ti/k3-am62x-phyboard-lyra.dtsi | 2 +- + arch/arm64/boot/dts/ti/k3-am62x-sk-common.dtsi | 2 +- + arch/arm64/boot/dts/ti/k3-am654-base-board.dts | 1 + + arch/arm64/boot/dts/ti/k3-am67-phycore-som.dtsi | 324 ++++++++++++++++ + .../arm64/boot/dts/ti/k3-am6754-phyboard-rigel.dts | 431 +++++++++++++++++++++ + ...uila-dsi-to-lvds-v2-panel-cap-touch-10inch.dtso | 161 ++++++++ + arch/arm64/boot/dts/ti/k3-am69-aquila.dtsi | 2 +- + .../boot/dts/ti/k3-j7200-common-proc-board.dts | 1 + + .../boot/dts/ti/k3-j721s2-common-proc-board.dts | 1 + + arch/arm64/boot/dts/ti/k3-j721s2-evm-audio.dtso | 157 ++++++++ + arch/arm64/boot/dts/ti/k3-j721s2-main.dtsi | 18 + + arch/arm64/boot/dts/ti/k3-j722s-evm.dts | 26 +- + .../boot/dts/ti/k3-j784s4-j742s2-evm-common.dtsi | 2 + + arch/arm64/boot/dts/ti/k3-pinctrl.h | 10 + + arch/arm64/configs/defconfig | 15 +- + drivers/firmware/ti_sci.c | 6 +- + drivers/soc/ti/knav_qmss_queue.c | 4 +- + include/linux/soc/ti/k3-ringacc.h | 31 +- + 53 files changed, 1439 insertions(+), 241 deletions(-) + create mode 100644 arch/arm64/boot/dts/ti/k3-am67-phycore-som.dtsi + create mode 100644 arch/arm64/boot/dts/ti/k3-am6754-phyboard-rigel.dts + create mode 100644 arch/arm64/boot/dts/ti/k3-am69-aquila-dsi-to-lvds-v2-panel-cap-touch-10inch.dtso + create mode 100644 arch/arm64/boot/dts/ti/k3-j721s2-evm-audio.dtso +Merging xilinx/for-next (bf126d9aa20f2 Merge branch 'zynqmp/soc' into for-next) +$ git merge -m Merge branch 'for-next' of https://github.com/Xilinx/linux-xlnx.git xilinx/for-next +Merge made by the 'ort' strategy. + arch/arm/boot/dts/xilinx/zynq-7000.dtsi | 8 ++-- + arch/arm/boot/dts/xilinx/zynq-parallella.dts | 4 +- + drivers/firmware/xilinx/zynqmp-ufs.c | 70 ++++++++++++++++++++++++++-- + drivers/firmware/xilinx/zynqmp.c | 44 ++++++++++++++--- + drivers/ufs/host/ufs-amd-versal2.c | 46 +++++------------- + include/linux/firmware/xlnx-zynqmp-ufs.h | 8 ++-- + include/linux/firmware/xlnx-zynqmp.h | 5 -- + 7 files changed, 126 insertions(+), 59 deletions(-) +Merging socfpga/for-next (a2a94b6a6f018 arm64: dts: agilex5: add support for debug daughter card) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/dinguyen/linux.git socfpga/for-next +Merge made by the 'ort' strategy. + Documentation/devicetree/bindings/arm/altera.yaml | 2 + + .../bindings/firmware/intel,stratix10-svc.yaml | 10 ++ + arch/arm64/boot/dts/altera/socfpga_stratix10.dtsi | 4 + + arch/arm64/boot/dts/intel/Makefile | 3 + + arch/arm64/boot/dts/intel/socfpga_agilex.dtsi | 2 + + arch/arm64/boot/dts/intel/socfpga_agilex5.dtsi | 61 ++++++++++ + .../arm64/boot/dts/intel/socfpga_agilex5_socdk.dts | 38 +++++- + .../boot/dts/intel/socfpga_agilex5_socdk_debug.dts | 121 +++++++++++++++++++ + .../boot/dts/intel/socfpga_agilex5_socdk_emmc.dts | 123 +++++++++++++++++++ + .../dts/intel/socfpga_agilex5_socdk_tsn_cfg2.dts | 130 +++++++++++++++++++++ + 10 files changed, 491 insertions(+), 3 deletions(-) + create mode 100644 arch/arm64/boot/dts/intel/socfpga_agilex5_socdk_debug.dts + create mode 100644 arch/arm64/boot/dts/intel/socfpga_agilex5_socdk_emmc.dts + create mode 100644 arch/arm64/boot/dts/intel/socfpga_agilex5_socdk_tsn_cfg2.dts +Merging clk/clk-next (9832cce7217fc Merge branch 'clk-starfive' into clk-next) +$ git merge -m Merge branch 'clk-next' of https://git.kernel.org/pub/scm/linux/kernel/git/clk/linux.git clk/clk-next +Auto-merging MAINTAINERS +Auto-merging drivers/clk/at91/clk-main.c +Auto-merging drivers/clk/clk-scpi.c +Auto-merging drivers/clk/renesas/r9a06g032-clocks.c +Auto-merging drivers/clk/renesas/renesas-cpg-mssr.c +Auto-merging drivers/clk/renesas/rzg2l-cpg.c +Auto-merging drivers/clk/renesas/rzv2h-cpg.c +Auto-merging drivers/clk/samsung/clk-pll.c +Auto-merging include/linux/clk/ti.h +Merge made by the 'ort' strategy. + .../bindings/clock/airoha,en7523-scu.yaml | 24 +- + .../bindings/clock/anlogic,dr1v90-cru.yaml | 62 + + .../bindings/clock/clk-palmas-clk32kg-clocks.txt | 35 - + .../bindings/clock/clock-nexus-node.yaml | 30 + + .../bindings/clock/nuvoton,ma35d1-clk.yaml | 16 +- + .../bindings/clock/starfive,jhb100-per0crg.yaml | 70 + + .../bindings/clock/starfive,jhb100-per1crg.yaml | 70 + + .../bindings/clock/starfive,jhb100-per2crg.yaml | 76 + + .../bindings/clock/starfive,jhb100-per3crg.yaml | 76 + + .../bindings/clock/starfive,jhb100-sys0crg.yaml | 63 + + .../bindings/clock/starfive,jhb100-sys1crg.yaml | 71 + + .../bindings/clock/starfive,jhb100-sys2crg.yaml | 64 + + .../bindings/clock/ti,palmas-clk32kg.yaml | 56 + + .../devicetree/bindings/clock/ti/adpll.txt | 39 - + .../devicetree/bindings/clock/ti/davinci/pll.txt | 96 -- + .../bindings/clock/ti/davinci/ti,da850-pll.yaml | 164 ++ + .../devicetree/bindings/clock/ti/dra7-atl.txt | 94 -- + .../devicetree/bindings/clock/ti/fapll.txt | 31 - + .../bindings/clock/ti/ti,dm814-adpll-clock.yaml | 88 ++ + .../bindings/clock/ti/ti,dm816-fapll-clock.yaml | 71 + + .../bindings/clock/ti/ti,dra7-atl-clock.yaml | 38 + + .../devicetree/bindings/clock/ti/ti,dra7-atl.yaml | 100 ++ + .../bindings/clock/zte,zx297520v3-lspcrm.yaml | 101 ++ + .../bindings/clock/zte,zx297520v3-matrixcrm.yaml | 98 ++ + .../bindings/clock/zte,zx297520v3-topcrm.yaml | 122 ++ + .../soc/starfive/starfive,jhb100-syscon.yaml | 115 ++ + MAINTAINERS | 29 +- + drivers/clk/Kconfig | 15 +- + drivers/clk/Makefile | 9 +- + drivers/clk/actions/owl-composite.h | 1 - + drivers/clk/actions/owl-fixed-factor.h | 2 - + drivers/clk/actions/owl-pll.c | 2 +- + drivers/clk/anlogic/Kconfig | 21 + + drivers/clk/anlogic/Makefile | 7 + + drivers/clk/anlogic/cru-dr1v90.c | 192 +++ + drivers/clk/anlogic/cru_dr1.c | 226 +++ + drivers/clk/anlogic/cru_dr1.h | 117 ++ + drivers/clk/aspeed/Kconfig | 1 + + drivers/clk/aspeed/clk-aspeed.c | 2 +- + drivers/clk/aspeed/clk-ast2600.c | 2 +- + drivers/clk/aspeed/clk-ast2700.c | 2 +- + drivers/clk/at91/clk-audio-pll.c | 4 +- + drivers/clk/at91/clk-h32mx.c | 2 +- + drivers/clk/at91/clk-main.c | 2 +- + drivers/clk/at91/clk-pll.c | 2 +- + drivers/clk/at91/clk-plldiv.c | 2 +- + drivers/clk/at91/clk-slow.c | 2 +- + drivers/clk/at91/clk-smd.c | 2 +- + drivers/clk/at91/clk-usb.c | 6 +- + drivers/clk/at91/sckc.c | 2 +- + drivers/clk/bcm/clk-iproc-armpll.c | 2 +- + drivers/clk/bcm/clk-iproc-asiu.c | 2 +- + drivers/clk/bcm/clk-iproc-pll.c | 2 +- + drivers/clk/bcm/clk-raspberrypi.c | 2 + + drivers/clk/berlin/berlin2-avpll.c | 4 +- + drivers/clk/berlin/berlin2-pll.c | 2 +- + drivers/clk/clk-axi-clkgen.c | 2 +- + drivers/clk/clk-cdce925.c | 2 +- + drivers/clk/clk-conf.c | 12 +- + drivers/clk/clk-cs2000-cp.c | 2 +- + drivers/clk/clk-en7523.c | 218 ++- + drivers/clk/clk-eyeq.c | 31 +- + drivers/clk/clk-fractional-divider.c | 2 +- + drivers/clk/clk-gemini.c | 2 +- + drivers/clk/clk-gpio.c | 2 +- + drivers/clk/clk-highbank.c | 2 +- + drivers/clk/clk-lmk04832.c | 6 +- + drivers/clk/clk-milbeaut.c | 14 +- + drivers/clk/clk-nomadik.c | 4 +- + drivers/clk/clk-npcm7xx.c | 2 +- + drivers/clk/clk-pwm.c | 2 +- + drivers/clk/clk-renesas-pcie.c | 43 +- + drivers/clk/clk-s2mps11.c | 12 +- + drivers/clk/clk-scpi.c | 2 +- + drivers/clk/clk-si514.c | 2 +- + drivers/clk/clk-si521xx.c | 43 +- + drivers/clk/clk-si5341.c | 2 +- + drivers/clk/clk-si5351.c | 20 +- + drivers/clk/clk-si544.c | 2 +- + drivers/clk/clk-si570.c | 2 +- + drivers/clk/clk-stm32f4.c | 4 +- + drivers/clk/clk-stm32h7.c | 2 +- + drivers/clk/clk-versaclock5.c | 12 +- + drivers/clk/clk-vt8500.c | 4 +- + drivers/clk/clk-xgene.c | 6 +- + drivers/clk/clk.c | 75 +- + drivers/clk/clk.h | 5 +- + drivers/clk/clk_kunit_helpers.c | 31 + + drivers/clk/clk_test.c | 151 +- + drivers/clk/clkdev.c | 4 +- + drivers/clk/davinci/Kconfig | 10 + + drivers/clk/davinci/Makefile | 8 +- + drivers/clk/davinci/da8xx-cfgchip.c | 4 +- + drivers/clk/davinci/pll.c | 2 +- + drivers/clk/davinci/psc.c | 2 +- + drivers/clk/hisilicon/clk-hi3559a.c | 2 +- + drivers/clk/hisilicon/clk-hi3620.c | 2 +- + drivers/clk/hisilicon/clk-hi6220-stub.c | 2 +- + drivers/clk/hisilicon/clk-hisi-phase.c | 2 +- + drivers/clk/hisilicon/clk-hix5hd2.c | 2 +- + drivers/clk/hisilicon/clkdivider-hi6220.c | 2 +- + drivers/clk/hisilicon/clkgate-separated.c | 2 +- + drivers/clk/imx/clk-busy.c | 4 +- + drivers/clk/imx/clk-cpu.c | 2 +- + drivers/clk/imx/clk-divider-gate.c | 2 +- + drivers/clk/imx/clk-fixup-div.c | 2 +- + drivers/clk/imx/clk-fixup-mux.c | 2 +- + drivers/clk/imx/clk-frac-pll.c | 2 +- + drivers/clk/imx/clk-fracn-gppll.c | 2 +- + drivers/clk/imx/clk-gate-93.c | 2 +- + drivers/clk/imx/clk-gate-exclusive.c | 2 +- + drivers/clk/imx/clk-gate2.c | 2 +- + drivers/clk/imx/clk-lpcg-scu.c | 2 +- + drivers/clk/imx/clk-pfd.c | 2 +- + drivers/clk/imx/clk-pfdv2.c | 2 +- + drivers/clk/imx/clk-pll14xx.c | 2 +- + drivers/clk/imx/clk-pllv1.c | 2 +- + drivers/clk/imx/clk-pllv2.c | 2 +- + drivers/clk/imx/clk-pllv3.c | 2 +- + drivers/clk/imx/clk-pllv4.c | 2 +- + drivers/clk/imx/clk-scu.c | 4 +- + drivers/clk/imx/clk-sscg-pll.c | 2 +- + drivers/clk/ingenic/cgu.c | 2 +- + drivers/clk/keystone/gate.c | 2 +- + drivers/clk/keystone/pll.c | 2 +- + drivers/clk/keystone/syscon-clk.c | 2 +- + drivers/clk/kunit_clk_assigned_rates.h | 4 +- + drivers/clk/kunit_clk_parse_clkspec.dtso | 31 + + drivers/clk/mediatek/Kconfig | 1 + + drivers/clk/mediatek/clk-cpumux.c | 2 +- + drivers/clk/mediatek/reset.h | 6 +- + drivers/clk/mmp/Kconfig | 5 + + drivers/clk/mmp/clk-apbc.c | 2 +- + drivers/clk/mmp/clk-apmu.c | 2 +- + drivers/clk/mmp/clk-frac.c | 2 +- + drivers/clk/mmp/clk-gate.c | 2 +- + drivers/clk/mmp/clk-mix.c | 2 +- + drivers/clk/mmp/clk-pll.c | 2 +- + drivers/clk/mstar/clk-msc313-mpll.c | 2 +- + drivers/clk/mvebu/ap-cpu-clk.c | 6 +- + drivers/clk/mvebu/clk-corediv.c | 2 +- + drivers/clk/mvebu/clk-cpu.c | 2 +- + drivers/clk/mxs/clk-div.c | 2 +- + drivers/clk/mxs/clk-frac.c | 2 +- + drivers/clk/mxs/clk-pll.c | 2 +- + drivers/clk/mxs/clk-ref.c | 2 +- + drivers/clk/nuvoton/clk-ma35d1.c | 7 +- + drivers/clk/nxp/clk-lpc18xx-creg.c | 2 +- + drivers/clk/pistachio/clk-pll.c | 2 +- + drivers/clk/pistachio/clk.c | 15 +- + drivers/clk/pistachio/clk.h | 1 + + drivers/clk/renesas/Kconfig | 7 + + drivers/clk/renesas/Makefile | 1 + + drivers/clk/renesas/r8a774a3-cpg-mssr.c | 168 ++ + drivers/clk/renesas/r8a779a0-cpg-mssr.c | 6 +- + drivers/clk/renesas/r8a779f0-cpg-mssr.c | 4 +- + drivers/clk/renesas/r8a779g0-cpg-mssr.c | 4 +- + drivers/clk/renesas/r8a779h0-cpg-mssr.c | 8 +- + drivers/clk/renesas/r8a78000-cpg.c | 3 + + drivers/clk/renesas/r9a06g032-clocks.c | 2 +- + drivers/clk/renesas/r9a07g043-cpg.c | 12 +- + drivers/clk/renesas/r9a07g044-cpg.c | 12 +- + drivers/clk/renesas/r9a08g045-cpg.c | 6 +- + drivers/clk/renesas/r9a08g046-cpg.c | 117 +- + drivers/clk/renesas/r9a09g077-cpg.c | 142 ++ + drivers/clk/renesas/rcar-gen3-cpg.c | 15 +- + drivers/clk/renesas/rcar-gen3-cpg.h | 3 +- + drivers/clk/renesas/rcar-gen4-cpg.c | 25 +- + drivers/clk/renesas/rcar-gen4-cpg.h | 14 +- + drivers/clk/renesas/renesas-cpg-mssr.c | 9 + + drivers/clk/renesas/renesas-cpg-mssr.h | 2 + + drivers/clk/renesas/rzg2l-cpg.c | 554 ++++++- + drivers/clk/renesas/rzg2l-cpg.h | 51 +- + drivers/clk/renesas/rzv2h-cpg.c | 164 +- + drivers/clk/rockchip/clk-cpu.c | 2 +- + drivers/clk/rockchip/clk-ddr.c | 2 +- + drivers/clk/rockchip/clk-gate-grf.c | 2 +- + drivers/clk/rockchip/clk-inverter.c | 2 +- + drivers/clk/rockchip/clk-mmc-phase.c | 2 +- + drivers/clk/rockchip/clk-muxgrf.c | 2 +- + drivers/clk/rockchip/clk-pll.c | 2 +- + drivers/clk/rockchip/clk.c | 2 +- + drivers/clk/samsung/clk-cpu.c | 2 +- + drivers/clk/samsung/clk-exynos-clkout.c | 8 +- + drivers/clk/samsung/clk-pll.c | 2 +- + drivers/clk/socfpga/clk-gate-a10.c | 2 +- + drivers/clk/socfpga/clk-gate-s10.c | 6 +- + drivers/clk/socfpga/clk-gate.c | 2 +- + drivers/clk/socfpga/clk-periph-a10.c | 2 +- + drivers/clk/socfpga/clk-periph-s10.c | 8 +- + drivers/clk/socfpga/clk-periph.c | 2 +- + drivers/clk/socfpga/clk-pll-a10.c | 2 +- + drivers/clk/socfpga/clk-pll-s10.c | 8 +- + drivers/clk/socfpga/clk-pll.c | 2 +- + drivers/clk/spear/clk-aux-synth.c | 2 +- + drivers/clk/spear/clk-frac-synth.c | 2 +- + drivers/clk/spear/clk-gpt-synth.c | 2 +- + drivers/clk/spear/clk-vco-pll.c | 2 +- + drivers/clk/st/clk-flexgen.c | 2 +- + drivers/clk/st/clkgen-fsyn.c | 4 +- + drivers/clk/st/clkgen-pll.c | 2 +- + drivers/clk/starfive/clk-starfive-jh7110-isp.c | 2 +- + drivers/clk/starfive/clk-starfive-jh7110-sys.c | 3 + + drivers/clk/starfive/clk-starfive-jh7110-vout.c | 6 +- + drivers/clk/stm32/clk-stm32mp1.c | 4 +- + drivers/clk/sunxi/clk-sun4i-tcon-ch1.c | 2 +- + drivers/clk/tegra/clk-audio-sync.c | 2 +- + drivers/clk/tegra/clk-bpmp.c | 12 +- + drivers/clk/tegra/clk-divider.c | 2 +- + drivers/clk/tegra/clk-periph-fixed.c | 2 +- + drivers/clk/tegra/clk-periph-gate.c | 2 +- + drivers/clk/tegra/clk-periph.c | 2 +- + drivers/clk/tegra/clk-pll-out.c | 2 +- + drivers/clk/tegra/clk-pll.c | 2 +- + drivers/clk/tegra/clk-sdmmc-mux.c | 2 +- + drivers/clk/tegra/clk-super.c | 4 +- + drivers/clk/tegra/clk-tegra-super-cclk.c | 2 +- + drivers/clk/tegra/clk-tegra124-emc.c | 2 +- + drivers/clk/tegra/clk-tegra20-emc.c | 2 +- + drivers/clk/tegra/clk-tegra210-emc.c | 2 +- + drivers/clk/tegra/clk.h | 2 +- + drivers/clk/ti/adpll.c | 1 + + drivers/clk/ti/clockdomain.c | 2 +- + drivers/clk/uniphier/clk-uniphier-cpugear.c | 2 +- + drivers/clk/uniphier/clk-uniphier-fixed-factor.c | 2 +- + drivers/clk/uniphier/clk-uniphier-fixed-rate.c | 2 +- + drivers/clk/uniphier/clk-uniphier-gate.c | 2 +- + drivers/clk/uniphier/clk-uniphier-mux.c | 2 +- + drivers/clk/ux500/clk-prcc.c | 2 +- + drivers/clk/ux500/clk-prcmu.c | 4 +- + drivers/clk/ux500/clk-sysctrl.c | 2 +- + drivers/clk/versatile/clk-icst.c | 6 +- + drivers/clk/versatile/clk-sp810.c | 2 +- + drivers/clk/versatile/clk-vexpress-osc.c | 2 +- + drivers/clk/x86/clk-pmc-atom.c | 2 +- + drivers/clk/xilinx/clk-xlnx-clock-wizard.c | 18 +- + drivers/clk/xilinx/xlnx_vcu.c | 2 +- + drivers/clk/zte/Kconfig | 28 + + drivers/clk/zte/Makefile | 6 + + drivers/clk/zte/clk-regmap.c | 237 +++ + drivers/clk/zte/clk-zx.c | 192 +++ + drivers/clk/zte/clk-zx.h | 137 ++ + drivers/clk/zte/clk-zx297520v3.c | 1657 ++++++++++++++++++++ + drivers/clk/zte/pll-zx.c | 576 +++++++ + drivers/clk/zynqmp/clk-gate-zynqmp.c | 2 +- + drivers/clk/zynqmp/clk-mux-zynqmp.c | 2 +- + drivers/clk/zynqmp/divider.c | 2 +- + drivers/clk/zynqmp/pll.c | 2 +- + drivers/reset/Kconfig | 10 + + drivers/reset/Makefile | 1 + + drivers/reset/reset-dr1v90.c | 140 ++ + include/dt-bindings/clock/anlogic,dr1v90-cru.h | 46 + + include/dt-bindings/clock/nuvoton,ma35d1-clk.h | 3 +- + include/dt-bindings/clock/starfive,jhb100-crg.h | 554 +++++++ + include/dt-bindings/clock/zte,zx297520v3-clk.h | 154 ++ + include/dt-bindings/phy/zte,zx297520v3-topcrm.h | 12 + + include/dt-bindings/reset/anlogic,dr1v90-cru.h | 41 + + include/dt-bindings/reset/starfive,jhb100-crg.h | 189 +++ + include/dt-bindings/reset/zte,zx297520v3-reset.h | 62 + + include/dt-bindings/soc/airoha,scu-ssr.h | 11 + + include/kunit/clk.h | 2 + + include/linux/clk-provider.h | 25 +- + include/linux/clk/tegra.h | 2 +- + include/linux/clk/ti.h | 2 +- + rust/kernel/clk.rs | 15 + + 265 files changed, 8259 insertions(+), 831 deletions(-) + create mode 100644 Documentation/devicetree/bindings/clock/anlogic,dr1v90-cru.yaml + delete mode 100644 Documentation/devicetree/bindings/clock/clk-palmas-clk32kg-clocks.txt + create mode 100644 Documentation/devicetree/bindings/clock/clock-nexus-node.yaml + create mode 100644 Documentation/devicetree/bindings/clock/starfive,jhb100-per0crg.yaml + create mode 100644 Documentation/devicetree/bindings/clock/starfive,jhb100-per1crg.yaml + create mode 100644 Documentation/devicetree/bindings/clock/starfive,jhb100-per2crg.yaml + create mode 100644 Documentation/devicetree/bindings/clock/starfive,jhb100-per3crg.yaml + create mode 100644 Documentation/devicetree/bindings/clock/starfive,jhb100-sys0crg.yaml + create mode 100644 Documentation/devicetree/bindings/clock/starfive,jhb100-sys1crg.yaml + create mode 100644 Documentation/devicetree/bindings/clock/starfive,jhb100-sys2crg.yaml + create mode 100644 Documentation/devicetree/bindings/clock/ti,palmas-clk32kg.yaml + delete mode 100644 Documentation/devicetree/bindings/clock/ti/adpll.txt + delete mode 100644 Documentation/devicetree/bindings/clock/ti/davinci/pll.txt + create mode 100644 Documentation/devicetree/bindings/clock/ti/davinci/ti,da850-pll.yaml + delete mode 100644 Documentation/devicetree/bindings/clock/ti/dra7-atl.txt + delete mode 100644 Documentation/devicetree/bindings/clock/ti/fapll.txt + create mode 100644 Documentation/devicetree/bindings/clock/ti/ti,dm814-adpll-clock.yaml + create mode 100644 Documentation/devicetree/bindings/clock/ti/ti,dm816-fapll-clock.yaml + create mode 100644 Documentation/devicetree/bindings/clock/ti/ti,dra7-atl-clock.yaml + create mode 100644 Documentation/devicetree/bindings/clock/ti/ti,dra7-atl.yaml + create mode 100644 Documentation/devicetree/bindings/clock/zte,zx297520v3-lspcrm.yaml + create mode 100644 Documentation/devicetree/bindings/clock/zte,zx297520v3-matrixcrm.yaml + create mode 100644 Documentation/devicetree/bindings/clock/zte,zx297520v3-topcrm.yaml + create mode 100644 Documentation/devicetree/bindings/soc/starfive/starfive,jhb100-syscon.yaml + create mode 100644 drivers/clk/anlogic/Kconfig + create mode 100644 drivers/clk/anlogic/Makefile + create mode 100644 drivers/clk/anlogic/cru-dr1v90.c + create mode 100644 drivers/clk/anlogic/cru_dr1.c + create mode 100644 drivers/clk/anlogic/cru_dr1.h + create mode 100644 drivers/clk/davinci/Kconfig + create mode 100644 drivers/clk/kunit_clk_parse_clkspec.dtso + create mode 100644 drivers/clk/renesas/r8a774a3-cpg-mssr.c + create mode 100644 drivers/clk/zte/Kconfig + create mode 100644 drivers/clk/zte/Makefile + create mode 100644 drivers/clk/zte/clk-regmap.c + create mode 100644 drivers/clk/zte/clk-zx.c + create mode 100644 drivers/clk/zte/clk-zx.h + create mode 100644 drivers/clk/zte/clk-zx297520v3.c + create mode 100644 drivers/clk/zte/pll-zx.c + create mode 100644 drivers/reset/reset-dr1v90.c + create mode 100644 include/dt-bindings/clock/anlogic,dr1v90-cru.h + create mode 100644 include/dt-bindings/clock/starfive,jhb100-crg.h + create mode 100644 include/dt-bindings/clock/zte,zx297520v3-clk.h + create mode 100644 include/dt-bindings/phy/zte,zx297520v3-topcrm.h + create mode 100644 include/dt-bindings/reset/anlogic,dr1v90-cru.h + create mode 100644 include/dt-bindings/reset/starfive,jhb100-crg.h + create mode 100644 include/dt-bindings/reset/zte,zx297520v3-reset.h + create mode 100644 include/dt-bindings/soc/airoha,scu-ssr.h +Merging clk-imx/for-next (39ec460b56b26 clk: imx95-blk-ctl: Fix REFCLK rise-fall mismatch on i.MX95) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/abelvesa/linux.git clk-imx/for-next +Already up to date. +Merging clk-renesas/renesas-clk (21cbd7ec29930 clk: renesas: r8a78000: Add SGASYNCD8_PERW_BUS) +$ git merge -m Merge branch 'renesas-clk' of https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-drivers.git clk-renesas/renesas-clk +Already up to date. +Merging thead-clk/thead-clk-for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'thead-clk-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git thead-clk/thead-clk-for-next +Already up to date. +Merging tenstorrent-clk/tenstorrent-clk-for-next (c64b54ddb692c MAINTAINERS: Update RISC-V Tenstorrent SoC entry) +$ git merge -m Merge branch 'tenstorrent-clk-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/tenstorrent/linux.git tenstorrent-clk/tenstorrent-clk-for-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 5 +++-- + drivers/clk/tenstorrent/atlantis-prcm.c | 35 +++++++++++++-------------------- + 2 files changed, 17 insertions(+), 23 deletions(-) +Merging csky/linux-next (abb81e5ce7d99 csky: Fix a4/a5 restoration in syscall trace path) +$ git merge -m Merge branch 'linux-next' of https://github.com/c-sky/csky-linux.git csky/linux-next +Already up to date. +Merging loongarch/loongarch-next (a2628ce4ddb68 perf build: Add clang and rust target flags for LoongArch) +$ git merge -m Merge branch 'loongarch-next' of https://git.kernel.org/pub/scm/linux/kernel/git/chenhuacai/linux-loongson.git loongarch/loongarch-next +Already up to date. +Merging m68k/for-next (af32c3cb72529 m68k: defconfig: Enable ATARI_SVETHLANA and ETHOC) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/geert/linux-m68k.git m68k/for-next +Merge made by the 'ort' strategy. + arch/m68k/Kconfig.devices | 12 ++++++ + arch/m68k/apollo/dn_ints.c | 6 +++ + arch/m68k/atari/config.c | 75 ++++++++++++++++++++++++++++++++++++ + arch/m68k/configs/amiga_defconfig | 4 ++ + arch/m68k/configs/apollo_defconfig | 3 ++ + arch/m68k/configs/atari_defconfig | 5 +++ + arch/m68k/configs/bvme6000_defconfig | 3 ++ + arch/m68k/configs/hp300_defconfig | 3 ++ + arch/m68k/configs/mac_defconfig | 3 ++ + arch/m68k/configs/multi_defconfig | 6 +++ + arch/m68k/configs/mvme147_defconfig | 3 ++ + arch/m68k/configs/mvme16x_defconfig | 3 ++ + arch/m68k/configs/q40_defconfig | 3 ++ + arch/m68k/configs/sun3_defconfig | 3 ++ + arch/m68k/configs/sun3x_defconfig | 3 ++ + arch/m68k/include/asm/atariints.h | 2 +- + arch/m68k/include/asm/irq.h | 6 +-- + arch/m68k/include/asm/serial.h | 21 +++------- + drivers/zorro/zorro.c | 5 +-- + 19 files changed, 145 insertions(+), 24 deletions(-) +Merging m68knommu/for-next (29036c5910df0 m68knommu: fix compile breakage for 5407 cleopatra board) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/gerg/m68knommu.git m68knommu/for-next +Merge made by the 'ort' strategy. + arch/m68k/68000/entry.S | 3 +- + arch/m68k/coldfire/entry.S | 3 +- + arch/m68k/coldfire/m5407.c | 5 ++++ + arch/m68k/coldfire/reset.c | 2 +- + arch/m68k/configs/stmark2_defconfig | 2 ++ + arch/m68k/include/asm/m53xxacr.h | 2 +- + arch/m68k/include/asm/quicc_simple.h | 53 ------------------------------------ + 7 files changed, 13 insertions(+), 57 deletions(-) + delete mode 100644 arch/m68k/include/asm/quicc_simple.h +Merging microblaze/next (6b125c73aaa0d microblaze: Enable xilinx dmas and axi emac drivers) +$ git merge -m Merge branch 'next' of git://git.monstr.eu/linux-2.6-microblaze.git microblaze/next +Merge made by the 'ort' strategy. + arch/microblaze/Kconfig | 9 + + arch/microblaze/Makefile | 8 +- + arch/microblaze/configs/mmu_defconfig | 2 + + arch/microblaze/include/asm/entry.h | 13 + + arch/microblaze/include/asm/processor.h | 2 +- + arch/microblaze/include/asm/uaccess.h | 3 +- + arch/microblaze/kernel/cpu/cpuinfo-pvr-full.c | 2 +- + arch/microblaze/kernel/cpu/cpuinfo-static.c | 2 +- + arch/microblaze/kernel/cpu/cpuinfo.c | 1 + + arch/microblaze/kernel/entry.S | 348 ++++++++++++----------- + arch/microblaze/kernel/hw_exception_handler.S | 5 + + arch/microblaze/kernel/process.c | 5 +- + arch/microblaze/kernel/ptrace.c | 3 + + arch/microblaze/kernel/reset.c | 22 ++ + arch/microblaze/kernel/signal.c | 26 ++ + arch/microblaze/kernel/syscalls/syscall.tbl | 2 +- + tools/testing/kunit/qemu_configs/microblazebe.py | 24 ++ + tools/testing/kunit/qemu_configs/microblazeel.py | 27 ++ + 18 files changed, 328 insertions(+), 176 deletions(-) + create mode 100644 tools/testing/kunit/qemu_configs/microblazebe.py + create mode 100644 tools/testing/kunit/qemu_configs/microblazeel.py +Merging mips/mips-next (bfe97f12a0a64 MIPS: octeon: Add detection of UBNT Edgeroute Lite devices) +$ git merge -m Merge branch 'mips-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mips/linux.git mips/mips-next +Merge made by the 'ort' strategy. + .../ABI/testing/sysfs-firmware-lefi-boardinfo | 21 +++--- + arch/mips/Makefile | 32 +++++---- + arch/mips/alchemy/common/clock.c | 8 +-- + arch/mips/boot/dts/cavium-octeon/Makefile | 1 + + arch/mips/boot/dts/pic32/pic32mzda.dtsi | 2 - + arch/mips/cavium-octeon/Kconfig | 5 ++ + arch/mips/cavium-octeon/Makefile | 2 + + arch/mips/cavium-octeon/board-ubnt.c | 16 +++++ + arch/mips/cavium-octeon/dma-octeon.c | 2 +- + arch/mips/cavium-octeon/executive/cvmx-helper.c | 20 ++++-- + arch/mips/cavium-octeon/executive/cvmx-l2c.c | 31 ++++---- + arch/mips/cavium-octeon/flash_setup.c | 2 +- + arch/mips/cavium-octeon/octeon-memcpy.S | 2 +- + arch/mips/cavium-octeon/setup.c | 51 ++++++++++--- + arch/mips/configs/decstation_64_defconfig | 44 +++++------- + arch/mips/dec/setup.c | 4 +- + arch/mips/generic/board-jaguar2.its.S | 2 - + arch/mips/generic/board-serval.its.S | 1 - + arch/mips/include/asm/amon.h | 12 ---- + arch/mips/include/asm/cmp.h | 10 --- + arch/mips/include/asm/dec/reset.h | 1 + + arch/mips/include/asm/gt64120.h | 2 - + arch/mips/include/asm/mach-loongson64/boot_param.h | 2 +- + arch/mips/include/asm/mips-boards/sead3-addr.h | 83 ---------------------- + arch/mips/lib/memcpy.S | 2 +- + arch/mips/mm/mmap.c | 6 +- + arch/mips/mm/tlb-r4k.c | 21 ++++-- + arch/mips/n64/init.c | 2 +- + arch/mips/rb532/gpio.c | 7 +- + 29 files changed, 180 insertions(+), 214 deletions(-) + create mode 100644 arch/mips/cavium-octeon/board-ubnt.c + delete mode 100644 arch/mips/include/asm/amon.h + delete mode 100644 arch/mips/include/asm/cmp.h + delete mode 100644 arch/mips/include/asm/mips-boards/sead3-addr.h +Merging openrisc/for-next (6620f5e8c11c4 openrisc: drop unneeded semicolon) +$ git merge -m Merge branch 'for-next' of https://github.com/openrisc/linux.git openrisc/for-next +Already up to date. +Merging parisc-hd/for-next (a5f6df25261b3 parisc: fix typos in comments in parport.c) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/deller/parisc-linux.git parisc-hd/for-next +Merge made by the 'ort' strategy. + arch/parisc/include/asm/pdcpat.h | 4 ++-- + arch/parisc/include/asm/special_insns.h | 2 +- + arch/parisc/include/uapi/asm/mman.h | 2 +- + arch/parisc/lib/memcpy.c | 2 +- + drivers/parisc/ccio-dma.c | 2 +- + drivers/parisc/dino.c | 2 +- + drivers/parisc/pdc_stable.c | 2 +- + drivers/parisc/sba_iommu.c | 2 +- + drivers/parport/parport_gsc.c | 2 +- + 9 files changed, 10 insertions(+), 10 deletions(-) +Merging powerpc/next (12d238d584ba4 powerpc/sysfs: Remove redundant wait time clamps) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/powerpc/linux.git powerpc/next +Merge made by the 'ort' strategy. + Documentation/ABI/testing/ppc-memtrace | 2 +- + arch/powerpc/boot/dts/ac14xx.dts | 13 --- + arch/powerpc/boot/dts/mpc5121.dtsi | 6 +- + arch/powerpc/boot/dts/pdm360ng.dts | 8 +- + arch/powerpc/include/asm/hardirq.h | 31 +++--- + arch/powerpc/kernel/dbell.c | 2 +- + arch/powerpc/kernel/irq.c | 131 +++++++++++++------------- + arch/powerpc/kernel/sysfs.c | 4 +- + arch/powerpc/kernel/time.c | 6 +- + arch/powerpc/kernel/trace/ftrace_entry.S | 6 ++ + arch/powerpc/kernel/traps.c | 11 +-- + arch/powerpc/kernel/watchdog.c | 2 +- + arch/powerpc/kvm/guest-state-buffer.c | 2 +- + arch/powerpc/lib/sstep.c | 111 +++++++++++----------- + arch/powerpc/platforms/512x/Kconfig | 3 +- + arch/powerpc/platforms/512x/Makefile | 1 - + arch/powerpc/platforms/512x/mpc512x_generic.c | 1 + + arch/powerpc/platforms/512x/pdm360ng.c | 126 ------------------------- + arch/powerpc/platforms/ps3/interrupt.c | 22 ++--- + 19 files changed, 182 insertions(+), 306 deletions(-) + delete mode 100644 arch/powerpc/platforms/512x/pdm360ng.c +Merging risc-v/for-next (4735883c0d4bc riscv: Add support for early boot errata application on MIPS chips) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/riscv/linux.git risc-v/for-next +Auto-merging arch/riscv/Kconfig.errata +Merge made by the 'ort' strategy. + arch/riscv/Kconfig.errata | 1 + + arch/riscv/errata/mips/Makefile | 6 ++++++ + arch/riscv/errata/mips/errata.c | 38 +++++++++++++++++++++++--------------- + 3 files changed, 30 insertions(+), 15 deletions(-) +Merging riscv-dt/riscv-dt-for-next (2186ec5f59c23 riscv: dts: starfive: Correct pwm nodes) +$ git merge -m Merge branch 'riscv-dt-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux.git riscv-dt/riscv-dt-for-next +Auto-merging MAINTAINERS +Auto-merging arch/riscv/boot/dts/starfive/jh7110-common.dtsi +Auto-merging arch/riscv/boot/dts/starfive/jh7110.dtsi +Merge made by the 'ort' strategy. + MAINTAINERS | 3 +- + arch/riscv/boot/dts/starfive/jh7100-common.dtsi | 28 +++++++-- + arch/riscv/boot/dts/starfive/jh7100.dtsi | 67 +++++++++++++++++++++- + arch/riscv/boot/dts/starfive/jh7110-common.dtsi | 27 +++++++-- + arch/riscv/boot/dts/starfive/jh7110-milkv-mars.dts | 6 +- + .../boot/dts/starfive/jh7110-milkv-marscm.dtsi | 6 +- + .../boot/dts/starfive/jh7110-pine64-star64.dts | 6 +- + .../jh7110-starfive-visionfive-2-lite.dtsi | 6 +- + .../dts/starfive/jh7110-starfive-visionfive-2.dtsi | 6 +- + arch/riscv/boot/dts/starfive/jh7110.dtsi | 67 +++++++++++++++++++++- + 10 files changed, 200 insertions(+), 22 deletions(-) +Merging riscv-soc/riscv-soc-for-next (b7516f2f64fd5 Merge branch 'riscv-soc-drivers-for-next' into riscv-soc-for-next) +$ git merge -m Merge branch 'riscv-soc-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux.git riscv-soc/riscv-soc-for-next +Auto-merging drivers/soc/Kconfig +CONFLICT (content): Merge conflict in drivers/soc/Kconfig +Auto-merging drivers/soc/Makefile +CONFLICT (content): Merge conflict in drivers/soc/Makefile +Recorded preimage for 'drivers/soc/Kconfig' +Recorded preimage for 'drivers/soc/Makefile' +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +Recorded resolution for 'drivers/soc/Kconfig'. +Recorded resolution for 'drivers/soc/Makefile'. +[master 00f85fe3fbfc9] Merge branch 'riscv-soc-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux.git +$ git diff -M --stat --summary HEAD^.. + drivers/firmware/microchip/mpfs-auto-update.c | 4 +- + drivers/soc/Kconfig | 1 + + drivers/soc/Makefile | 1 + + drivers/soc/starfive/Kconfig | 6 ++ + drivers/soc/starfive/Makefile | 2 + + drivers/soc/starfive/socinfo/Kconfig | 13 ++++ + drivers/soc/starfive/socinfo/Makefile | 2 + + drivers/soc/starfive/socinfo/jhb100-socinfo.c | 105 ++++++++++++++++++++++++++ + 8 files changed, 132 insertions(+), 2 deletions(-) + create mode 100644 drivers/soc/starfive/Kconfig + create mode 100644 drivers/soc/starfive/Makefile + create mode 100644 drivers/soc/starfive/socinfo/Kconfig + create mode 100644 drivers/soc/starfive/socinfo/Makefile + create mode 100644 drivers/soc/starfive/socinfo/jhb100-socinfo.c +Merging s390/for-next (c57ae907d4f9a Merge branch 'features' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/s390/linux.git s390/for-next +Auto-merging arch/s390/kernel/traps.c +Merge made by the 'ort' strategy. + arch/s390/boot/ipl_parm.c | 13 +- + arch/s390/include/asm/ebcdic.h | 6 + + arch/s390/include/asm/entry-percpu.h | 71 ++----- + arch/s390/include/asm/idals.h | 9 +- + arch/s390/include/asm/lowcore.h | 5 +- + arch/s390/include/asm/percpu.h | 354 ++++++++++++++++++++++------------- + arch/s390/kernel/ebcdic.c | 40 +++- + arch/s390/kernel/hiperdispatch.c | 2 +- + arch/s390/kernel/irq.c | 10 +- + arch/s390/kernel/nmi.c | 4 +- + arch/s390/kernel/traps.c | 4 +- + arch/s390/lib/spinlock.c | 4 + + drivers/s390/char/keyboard.c | 6 +- + drivers/s390/char/sclp_vt220.c | 87 +++++++-- + drivers/s390/cio/chsc.c | 20 +- + drivers/s390/cio/chsc_sch.c | 300 +++++++++++------------------ + drivers/s390/cio/cmf.c | 11 +- + drivers/s390/cio/qdio.h | 2 +- + drivers/s390/cio/qdio_main.c | 23 +-- + drivers/s390/cio/qdio_setup.c | 14 +- + drivers/s390/cio/qdio_thinint.c | 2 +- + drivers/s390/cio/scm.c | 7 +- + 22 files changed, 538 insertions(+), 456 deletions(-) +Merging sh/for-next (be5a19d95030f sh: ecovec24: Use static device properties to describe the touchscreen) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/glaubitz/sh-linux.git sh/for-next +Merge made by the 'ort' strategy. + arch/sh/boards/mach-ecovec24/setup.c | 38 ++---- + arch/sh/boards/mach-x3proto/gpio.c | 15 ++- + arch/sh/boards/mach-x3proto/setup.c | 165 +++++++++++++-------------- + arch/sh/include/mach-x3proto/mach/hardware.h | 2 + + 4 files changed, 105 insertions(+), 115 deletions(-) +Merging sparc/for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/alarsson/linux-sparc.git sparc/for-next +Already up to date. +Merging uml/next (2f88f5689de1a um: fix shutdown __inittext access) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/uml/linux.git uml/next +Auto-merging arch/um/Kconfig +CONFLICT (content): Merge conflict in arch/um/Kconfig +Auto-merging lib/Kconfig.debug +Resolved 'arch/um/Kconfig' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 8ba8af1130a21] Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/uml/linux.git +$ git diff -M --stat --summary HEAD^.. + Documentation/virt/index.rst | 1 + + Documentation/virt/uml/nommu-uml.rst | 54 ++++++++ + arch/um/Kconfig | 38 +++-- + arch/um/Makefile | 7 +- + arch/um/drivers/Kconfig | 3 + + arch/um/drivers/cow_user.c | 34 ++++- + arch/um/drivers/mconsole_kern.c | 14 +- + arch/um/drivers/vector_kern.c | 37 +++-- + arch/um/drivers/vector_user.c | 3 + + arch/um/drivers/virtio_pcidev.c | 19 ++- + arch/um/drivers/virtio_uml.c | 26 ++-- + arch/um/include/asm/common.lds.S | 2 + + arch/um/include/asm/futex.h | 53 +++++++ + arch/um/include/asm/mmu.h | 8 ++ + arch/um/include/asm/mmu_context.h | 2 + + arch/um/include/asm/perf_event.h | 7 + + arch/um/include/asm/pgtable.h | 25 ++-- + arch/um/include/asm/ptrace-generic.h | 6 + + arch/um/include/asm/tlbflush.h | 20 +++ + arch/um/include/asm/uaccess.h | 6 +- + arch/um/include/shared/as-layout.h | 3 +- + arch/um/include/shared/kern_util.h | 3 +- + arch/um/include/shared/longjmp.h | 3 +- + arch/um/include/shared/os.h | 6 +- + arch/um/include/shared/skas/stub-data.h | 9 ++ + arch/um/kernel/Makefile | 6 +- + arch/um/kernel/dyn.lds.S | 9 +- + arch/um/kernel/mem-pgtable.c | 55 ++++++++ + arch/um/kernel/mem.c | 47 ++----- + arch/um/kernel/physmem.c | 7 + + arch/um/kernel/process.c | 23 +++- + arch/um/kernel/skas/Makefile | 16 ++- + arch/um/kernel/skas/mmu.c | 25 ++++ + arch/um/kernel/skas/nommu.c | 91 ++++++++++++ + arch/um/kernel/skas/process.c | 32 +---- + arch/um/kernel/skas/syscall.c | 5 +- + arch/um/kernel/skas/uaccess.c | 134 +++++++----------- + arch/um/kernel/smp.c | 3 +- + arch/um/kernel/sysrq.c | 2 +- + arch/um/kernel/time.c | 3 +- + arch/um/kernel/tlb.c | 7 - + arch/um/kernel/trap-mmu.c | 237 ++++++++++++++++++++++++++++++++ + arch/um/kernel/trap-nommu.c | 13 ++ + arch/um/kernel/trap.c | 219 ----------------------------- + arch/um/kernel/um_arch.c | 11 +- + arch/um/kernel/uml.lds.S | 11 +- + arch/um/os-Linux/main.c | 17 ++- + arch/um/os-Linux/signal.c | 2 +- + arch/um/os-Linux/skas/process.c | 11 +- + arch/x86/um/Makefile | 3 + + arch/x86/um/asm/elf.h | 8 +- + arch/x86/um/asm/required-features.h | 9 -- + arch/x86/um/vdso/vma.c | 13 +- + fs/Kconfig.binfmt | 2 +- + lib/Kconfig.debug | 10 +- + lib/kunit/Kconfig | 2 +- + 56 files changed, 906 insertions(+), 516 deletions(-) + create mode 100644 Documentation/virt/uml/nommu-uml.rst + create mode 100644 arch/um/include/asm/perf_event.h + create mode 100644 arch/um/kernel/mem-pgtable.c + create mode 100644 arch/um/kernel/skas/nommu.c + create mode 100644 arch/um/kernel/trap-mmu.c + create mode 100644 arch/um/kernel/trap-nommu.c + delete mode 100644 arch/x86/um/asm/required-features.h +Merging xtensa/xtensa-for-next (28722ed2527aa xtensa: time: Fix clk reference leak in calibrate_ccount()) +$ git merge -m Merge branch 'xtensa-for-next' of https://github.com/jcmvbkbc/linux-xtensa.git xtensa/xtensa-for-next +Merge made by the 'ort' strategy. + arch/xtensa/include/asm/atomic.h | 13 ------------- + arch/xtensa/include/uapi/asm/mman.h | 2 +- + arch/xtensa/kernel/entry.S | 2 +- + arch/xtensa/kernel/time.c | 1 + + arch/xtensa/kernel/vectors.S | 2 +- + arch/xtensa/mm/init.c | 2 +- + 6 files changed, 5 insertions(+), 17 deletions(-) +Merging fs-next (7e86d255ddfb4 Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/viro/vfs.git) +$ git merge -m Merge branch 'fs-next' of linux-next fs-next +Auto-merging CREDITS +Auto-merging Documentation/filesystems/vfs.rst +Auto-merging MAINTAINERS +Auto-merging arch/powerpc/Kconfig +Auto-merging drivers/android/binder/rust_binderfs.c +Auto-merging drivers/android/binderfs.c +Auto-merging fs/Kconfig +Auto-merging fs/buffer.c +Auto-merging fs/ceph/addr.c +Auto-merging fs/coredump.c +CONFLICT (content): Merge conflict in fs/coredump.c +Auto-merging fs/crypto/crypto.c +Auto-merging fs/f2fs/compress.c +Auto-merging fs/f2fs/data.c +Auto-merging fs/f2fs/f2fs.h +CONFLICT (content): Merge conflict in fs/f2fs/f2fs.h +Auto-merging fs/f2fs/segment.c +Auto-merging fs/fat/fat.h +Auto-merging fs/fat/file.c +Auto-merging fs/fat/misc.c +Auto-merging fs/fuse/dax.c +CONFLICT (content): Merge conflict in fs/fuse/dax.c +Auto-merging fs/hugetlbfs/inode.c +Auto-merging fs/iomap/buffered-io.c +Auto-merging fs/nfs/write.c +Auto-merging fs/ocfs2/journal.c +Auto-merging fs/ocfs2/quota_local.c +Auto-merging fs/ocfs2/xattr.c +Auto-merging fs/proc/base.c +Auto-merging fs/proc/internal.h +Auto-merging fs/proc/vmcore.c +Auto-merging fs/ubifs/file.c +Auto-merging fs/xfs/libxfs/xfs_btree.c +CONFLICT (content): Merge conflict in fs/xfs/libxfs/xfs_btree.c +Auto-merging include/linux/buffer_head.h +Auto-merging include/linux/sched.h +Auto-merging init/Kconfig +Auto-merging ipc/mqueue.c +Auto-merging kernel/Makefile +Auto-merging kernel/fork.c +Auto-merging mm/secretmem.c +Auto-merging mm/shmem.c +Auto-merging mm/userfaultfd.c +Auto-merging security/selinux/selinuxfs.c +Resolved 'fs/coredump.c' using previous resolution. +Resolved 'fs/f2fs/f2fs.h' using previous resolution. +Resolved 'fs/fuse/dax.c' using previous resolution. +Resolved 'fs/xfs/libxfs/xfs_btree.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master b70ad9cec1d20] Merge branch 'fs-next' of linux-next +$ git diff -M --stat --summary HEAD^.. + CREDITS | 2 +- + Documentation/ABI/testing/sysfs-fs-f2fs | 7 + + Documentation/admin-guide/binfmt-misc.rst | 3 + + Documentation/admin-guide/nfs/pnfs-scsi-server.rst | 2 +- + Documentation/filesystems/befs.rst | 4 +- + Documentation/filesystems/bfs.rst | 60 - + Documentation/filesystems/exfat.rst | 117 ++ + Documentation/filesystems/index.rst | 2 +- + Documentation/filesystems/locking.rst | 24 +- + Documentation/filesystems/netfs_library.rst | 33 +- + Documentation/filesystems/ntfs3.rst | 23 + + Documentation/filesystems/porting.rst | 25 +- + Documentation/filesystems/proc.rst | 4 + + Documentation/filesystems/sharedsubtree.rst | 20 +- + Documentation/filesystems/smb/ksmbd.rst | 14 +- + Documentation/filesystems/squashfs.rst | 6 +- + Documentation/filesystems/vfs.rst | 30 +- + .../filesystems/xfs/xfs-online-fsck-design.rst | 13 +- + MAINTAINERS | 20 +- + arch/mips/configs/malta_defconfig | 1 - + arch/mips/configs/malta_kvm_defconfig | 1 - + arch/mips/configs/maltaup_xpa_defconfig | 1 - + arch/mips/configs/rm200_defconfig | 1 - + arch/powerpc/Kconfig | 1 - + arch/powerpc/configs/fsl-emb-nonhw.config | 1 - + arch/powerpc/configs/ppc6xx_defconfig | 1 - + arch/powerpc/include/asm/elf.h | 6 - + arch/powerpc/include/asm/spu.h | 3 - + arch/powerpc/platforms/cell/Kconfig | 1 - + arch/powerpc/platforms/cell/spu_syscalls.c | 20 - + arch/powerpc/platforms/cell/spufs/Makefile | 1 - + arch/powerpc/platforms/cell/spufs/coredump.c | 183 -- + arch/powerpc/platforms/cell/spufs/file.c | 114 -- + arch/powerpc/platforms/cell/spufs/inode.c | 14 +- + arch/powerpc/platforms/cell/spufs/spufs.h | 12 - + arch/powerpc/platforms/cell/spufs/syscalls.c | 4 - + block/bio-integrity-fs.c | 15 +- + block/bio-integrity.c | 1 + + block/bio.c | 219 +-- + block/blk-map.c | 2 +- + block/blk-settings.c | 6 + + block/fops.c | 3 +- + block/partitions/efi.h | 8 +- + drivers/acpi/acpi_configfs.c | 2 +- + drivers/android/binder/rust_binderfs.c | 2 +- + drivers/android/binderfs.c | 2 +- + drivers/base/devtmpfs.c | 108 +- + drivers/gpio/gpiolib-cdev.c | 18 +- + drivers/gpu/drm/msm/msm_perfcntr.c | 4 +- + drivers/gpu/drm/xe/xe_configfs.c | 4 +- + drivers/media/mc/mc-request.c | 8 +- + drivers/misc/ntsync.c | 6 +- + drivers/virt/coco/guest/report.c | 6 +- + fs/9p/acl.c | 4 +- + fs/9p/acl.h | 4 +- + fs/9p/v9fs.h | 2 +- + fs/9p/v9fs_vfs.h | 2 +- + fs/9p/vfs_addr.c | 1 - + fs/9p/vfs_inode.c | 14 +- + fs/9p/vfs_inode_dotl.c | 14 +- + fs/9p/xattr.c | 2 +- + fs/Kconfig | 9 +- + fs/Makefile | 1 - + fs/adfs/adfs.h | 2 +- + fs/adfs/dir.c | 2 +- + fs/adfs/inode.c | 2 +- + fs/affs/affs.h | 10 +- + fs/affs/inode.c | 2 +- + fs/affs/namei.c | 8 +- + fs/afs/dir.c | 16 +- + fs/afs/file.c | 8 +- + fs/afs/inode.c | 4 +- + fs/afs/internal.h | 8 +- + fs/afs/security.c | 2 +- + fs/afs/xattr.c | 4 +- + fs/aio.c | 11 +- + fs/anon_inodes.c | 4 +- + fs/attr.c | 16 +- + fs/autofs/root.c | 12 +- + fs/backing-file.c | 2 +- + fs/bad_inode.c | 20 +- + fs/bfs/Kconfig | 21 - + fs/bfs/Makefile | 8 - + fs/bfs/bfs.h | 69 - + fs/bfs/dir.c | 4 +- + fs/bfs/file.c | 203 -- + fs/bfs/inode.c | 538 ------ + fs/binfmt_elf.c | 18 +- + fs/binfmt_elf_fdpic.c | 14 +- + fs/binfmt_misc.c | 12 +- + fs/bpf_fs_kfuncs.c | 4 +- + fs/btrfs/Kconfig | 1 + + fs/btrfs/acl.c | 2 +- + fs/btrfs/acl.h | 2 +- + fs/btrfs/backref.c | 2 +- + fs/btrfs/bio.c | 163 +- + fs/btrfs/bio.h | 9 +- + fs/btrfs/block-group.c | 557 ++---- + fs/btrfs/block-group.h | 16 +- + fs/btrfs/btrfs_inode.h | 30 +- + fs/btrfs/compression.c | 61 +- + fs/btrfs/delalloc-space.c | 13 +- + fs/btrfs/delayed-inode.c | 115 +- + fs/btrfs/delayed-inode.h | 22 +- + fs/btrfs/dev-replace.c | 10 +- + fs/btrfs/dir-item.c | 50 +- + fs/btrfs/dir-item.h | 5 +- + fs/btrfs/direct-io.c | 7 + + fs/btrfs/disk-io.c | 132 +- + fs/btrfs/extent-io-tree.c | 2 +- + fs/btrfs/extent-tree.c | 10 + + fs/btrfs/extent_io.c | 247 ++- + fs/btrfs/extent_map.c | 8 + + fs/btrfs/file-item.c | 46 +- + fs/btrfs/free-space-cache.c | 1344 +------------ + fs/btrfs/free-space-cache.h | 29 +- + fs/btrfs/fs.c | 2 +- + fs/btrfs/fs.h | 4 +- + fs/btrfs/inode.c | 355 ++-- + fs/btrfs/ioctl.c | 30 +- + fs/btrfs/ioctl.h | 2 +- + fs/btrfs/ordered-data.c | 27 +- + fs/btrfs/qgroup.c | 128 +- + fs/btrfs/qgroup.h | 16 +- + fs/btrfs/raid56.c | 57 +- + fs/btrfs/reflink.c | 51 +- + fs/btrfs/relocation.c | 2 +- + fs/btrfs/send.c | 2 +- + fs/btrfs/space-info.c | 2 - + fs/btrfs/space-info.h | 4 - + fs/btrfs/super.c | 48 +- + fs/btrfs/sysfs.c | 134 +- + fs/btrfs/tests/btrfs-tests.c | 9 - + fs/btrfs/tests/btrfs-tests.h | 2 - + fs/btrfs/tests/extent-io-tests.c | 129 +- + fs/btrfs/transaction.c | 153 +- + fs/btrfs/transaction.h | 58 +- + fs/btrfs/tree-checker.c | 83 +- + fs/btrfs/tree-log.c | 7 +- + fs/btrfs/tree-mod-log.c | 13 +- + fs/btrfs/verity.c | 31 +- + fs/btrfs/volumes.c | 67 +- + fs/btrfs/volumes.h | 1 + + fs/btrfs/xattr.c | 6 +- + fs/btrfs/zlib.c | 10 +- + fs/btrfs/zoned.c | 29 +- + fs/btrfs/zstd.c | 73 +- + fs/buffer.c | 131 +- + fs/cachefiles/Kconfig | 2 +- + fs/cachefiles/interface.c | 96 +- + fs/cachefiles/internal.h | 18 +- + fs/cachefiles/io.c | 472 +++-- + fs/cachefiles/namei.c | 34 +- + fs/cachefiles/xattr.c | 82 +- + fs/ceph/Kconfig | 1 + + fs/ceph/acl.c | 2 +- + fs/ceph/addr.c | 4 +- + fs/ceph/dir.c | 10 +- + fs/ceph/file.c | 2 +- + fs/ceph/inode.c | 10 +- + fs/ceph/mds_client.h | 2 +- + fs/ceph/super.h | 10 +- + fs/ceph/xattr.c | 2 +- + fs/char_dev.c | 4 +- + fs/coda/coda_linux.h | 6 +- + fs/coda/dir.c | 10 +- + fs/coda/inode.c | 4 +- + fs/coda/pioctl.c | 4 +- + fs/configfs/configfs_internal.h | 14 +- + fs/configfs/dir.c | 12 +- + fs/configfs/file.c | 6 +- + fs/configfs/inode.c | 4 +- + fs/configfs/symlink.c | 2 +- + fs/coredump.c | 558 ++++-- + fs/crypto/Kconfig | 2 +- + fs/crypto/crypto.c | 5 + + fs/crypto/fscrypt_private.h | 27 - + fs/crypto/hooks.c | 9 + + fs/crypto/keyring.c | 13 + + fs/crypto/keysetup_v1.c | 5 +- + fs/dax.c | 26 +- + fs/dcache.c | 206 +- + fs/debugfs/inode.c | 2 +- + fs/devpts/inode.c | 2 + + fs/dlm/config.c | 30 +- + fs/dlm/dlm_internal.h | 5 + + fs/dlm/lock.c | 18 +- + fs/dlm/lock.h | 4 +- + fs/dlm/lowcomms.c | 1 + + fs/dlm/midcomms.c | 1 + + fs/dlm/plock.c | 14 +- + fs/dlm/user.c | 23 + + fs/ecryptfs/inode.c | 26 +- + fs/efivarfs/inode.c | 6 +- + fs/erofs/inode.c | 2 +- + fs/erofs/internal.h | 2 +- + fs/eventfd.c | 4 +- + fs/eventpoll.c | 6 +- + fs/exec.c | 28 +- + fs/exfat/dir.c | 52 +- + fs/exfat/exfat_fs.h | 5 +- + fs/exfat/fatent.c | 19 +- + fs/exfat/file.c | 6 +- + fs/exfat/iomap.c | 28 +- + fs/exfat/misc.c | 2 +- + fs/exfat/namei.c | 13 +- + fs/exfat/super.c | 30 +- + fs/ext2/Makefile | 2 + + fs/ext2/acl.c | 2 +- + fs/ext2/acl.h | 2 +- + fs/ext2/balloc.c | 4 + + fs/ext2/ext2.h | 25 +- + fs/ext2/file.c | 7 + + fs/ext2/inode.c | 11 +- + fs/ext2/ioctl.c | 2 +- + fs/ext2/namei.c | 12 +- + fs/ext2/super.c | 33 +- + fs/ext2/xattr.c | 2 +- + fs/ext2/xattr_security.c | 2 +- + fs/ext2/xattr_trusted.c | 2 +- + fs/ext2/xattr_user.c | 2 +- + fs/ext4/acl.c | 2 +- + fs/ext4/acl.h | 2 +- + fs/ext4/ext4.h | 10 +- + fs/ext4/ext4_jbd2.c | 2 +- + fs/ext4/ialloc.c | 2 +- + fs/ext4/inode.c | 6 +- + fs/ext4/ioctl.c | 6 +- + fs/ext4/mmp.c | 2 +- + fs/ext4/namei.c | 16 +- + fs/ext4/symlink.c | 2 +- + fs/ext4/xattr_hurd.c | 2 +- + fs/ext4/xattr_security.c | 2 +- + fs/ext4/xattr_trusted.c | 2 +- + fs/ext4/xattr_user.c | 2 +- + fs/f2fs/Makefile | 2 +- + fs/f2fs/acl.c | 32 +- + fs/f2fs/acl.h | 10 +- + fs/f2fs/cache.c | 720 +++++++ + fs/f2fs/cache.h | 240 +++ + fs/f2fs/checkpoint.c | 558 +++--- + fs/f2fs/compress.c | 183 +- + fs/f2fs/data.c | 648 ++++--- + fs/f2fs/debug.c | 94 +- + fs/f2fs/dir.c | 221 ++- + fs/f2fs/extent_cache.c | 25 +- + fs/f2fs/f2fs.h | 516 +++-- + fs/f2fs/file.c | 195 +- + fs/f2fs/gc.c | 222 ++- + fs/f2fs/inline.c | 293 +-- + fs/f2fs/inode.c | 242 +-- + fs/f2fs/iostat.h | 11 + + fs/f2fs/namei.c | 138 +- + fs/f2fs/node.c | 1277 ++++++------- + fs/f2fs/node.h | 191 +- + fs/f2fs/recovery.c | 255 +-- + fs/f2fs/segment.c | 429 +++-- + fs/f2fs/segment.h | 87 +- + fs/f2fs/shrinker.c | 16 +- + fs/f2fs/super.c | 288 +-- + fs/f2fs/sysfs.c | 32 +- + fs/f2fs/verity.c | 6 +- + fs/f2fs/xattr.c | 137 +- + fs/f2fs/xattr.h | 23 +- + fs/failfs.c | 4 +- + fs/fat/fat.h | 4 +- + fs/fat/file.c | 6 +- + fs/fat/misc.c | 2 +- + fs/fat/namei_msdos.c | 6 +- + fs/fat/namei_vfat.c | 6 +- + fs/fhandle.c | 2 +- + fs/file.c | 321 +++- + fs/file_attr.c | 6 +- + fs/fs-writeback.c | 23 +- + fs/fs_pin.c | 9 +- + fs/fuse/Kconfig | 8 +- + fs/fuse/Makefile | 2 +- + fs/fuse/acl.c | 4 +- + fs/fuse/backing.c | 2 +- + fs/fuse/dax.c | 215 +-- + fs/fuse/dev.c | 50 +- + fs/fuse/dev.h | 2 +- + fs/fuse/dev_uring.c | 6 + + fs/fuse/dir.c | 55 +- + fs/fuse/file.c | 131 +- + fs/fuse/fuse_i.h | 108 +- + fs/fuse/inode.c | 82 +- + fs/fuse/ioctl.c | 5 +- + fs/fuse/iomode.c | 4 +- + fs/fuse/notify.c | 4 +- + fs/fuse/passthrough.c | 15 +- + fs/fuse/req.c | 15 +- + fs/fuse/virtio_fs.c | 28 +- + fs/fuse/xattr.c | 2 +- + fs/gfs2/acl.c | 13 +- + fs/gfs2/acl.h | 2 +- + fs/gfs2/aops.c | 17 +- + fs/gfs2/bmap.c | 93 +- + fs/gfs2/dentry.c | 6 +- + fs/gfs2/dir.c | 61 +- + fs/gfs2/export.c | 7 +- + fs/gfs2/file.c | 74 +- + fs/gfs2/glock.c | 60 +- + fs/gfs2/glops.c | 13 +- + fs/gfs2/incore.h | 8 +- + fs/gfs2/inode.c | 148 +- + fs/gfs2/inode.h | 4 +- + fs/gfs2/lock_dlm.c | 4 +- + fs/gfs2/log.c | 7 +- + fs/gfs2/lops.c | 22 +- + fs/gfs2/meta_io.c | 7 +- + fs/gfs2/ops_fstype.c | 26 +- + fs/gfs2/quota.c | 57 +- + fs/gfs2/recovery.c | 19 +- + fs/gfs2/rgrp.c | 129 +- + fs/gfs2/super.c | 98 +- + fs/gfs2/trace_gfs2.h | 6 +- + fs/gfs2/util.c | 8 +- + fs/gfs2/xattr.c | 71 +- + fs/hfs/attr.c | 2 +- + fs/hfs/dir.c | 6 +- + fs/hfs/hfs_fs.h | 2 +- + fs/hfs/inode.c | 2 +- + fs/hfsplus/dir.c | 10 +- + fs/hfsplus/hfsplus_fs.h | 4 +- + fs/hfsplus/inode.c | 6 +- + fs/hfsplus/xattr.c | 2 +- + fs/hfsplus/xattr_security.c | 2 +- + fs/hfsplus/xattr_trusted.c | 2 +- + fs/hfsplus/xattr_user.c | 2 +- + fs/hostfs/hostfs_kern.c | 14 +- + fs/hpfs/hpfs_fn.h | 2 +- + fs/hpfs/inode.c | 2 +- + fs/hpfs/namei.c | 10 +- + fs/hugetlbfs/inode.c | 14 +- + fs/inode.c | 14 +- + fs/internal.h | 30 +- + fs/iomap/bio.c | 4 +- + fs/iomap/buffered-io.c | 8 +- + fs/iomap/direct-io.c | 46 +- + fs/iomap/ioend.c | 122 +- + fs/isofs/compress.c | 77 +- + fs/isofs/dir.c | 14 +- + fs/isofs/inode.c | 70 +- + fs/isofs/isofs.h | 16 +- + fs/isofs/joliet.c | 27 +- + fs/isofs/namei.c | 8 +- + fs/isofs/rock.c | 19 +- + fs/jbd2/commit.c | 22 +- + fs/jbd2/journal.c | 31 +- + fs/jbd2/transaction.c | 2 +- + fs/jffs2/acl.c | 2 +- + fs/jffs2/acl.h | 2 +- + fs/jffs2/dir.c | 20 +- + fs/jffs2/fs.c | 2 +- + fs/jffs2/os-linux.h | 2 +- + fs/jffs2/security.c | 2 +- + fs/jffs2/xattr_trusted.c | 2 +- + fs/jffs2/xattr_user.c | 2 +- + fs/jfs/acl.c | 2 +- + fs/jfs/file.c | 2 +- + fs/jfs/ioctl.c | 2 +- + fs/jfs/jfs_acl.h | 2 +- + fs/jfs/jfs_inode.h | 4 +- + fs/jfs/namei.c | 10 +- + fs/jfs/xattr.c | 4 +- + fs/kernfs/dir.c | 231 ++- + fs/kernfs/file.c | 47 +- + fs/kernfs/inode.c | 10 +- + fs/kernfs/kernfs-internal.h | 24 +- + fs/kernfs/mount.c | 34 +- + fs/kernfs/symlink.c | 17 +- + fs/libfs.c | 8 +- + fs/lockd/svc.c | 1 - + fs/lockd/svclock.c | 33 +- + fs/lockd/trace.h | 1 - + fs/lockd/xdr.h | 2 +- + fs/minix/file.c | 2 +- + fs/minix/inode.c | 2 +- + fs/minix/minix.h | 2 +- + fs/minix/namei.c | 12 +- + fs/mnt_idmapping.c | 33 +- + fs/mount.h | 3 +- + fs/namei.c | 155 +- + fs/namespace.c | 75 +- + fs/netfs/Kconfig | 3 + + fs/netfs/Makefile | 2 +- + fs/netfs/buffered_read.c | 224 ++- + fs/netfs/buffered_write.c | 78 +- + fs/netfs/direct_read.c | 14 +- + fs/netfs/direct_write.c | 17 +- + fs/netfs/fscache_cookie.c | 8 +- + fs/netfs/fscache_internal.h | 14 - + fs/netfs/fscache_io.c | 10 +- + fs/netfs/internal.h | 72 +- + fs/netfs/iterator.c | 2 +- + fs/netfs/main.c | 1 - + fs/netfs/misc.c | 109 +- + fs/netfs/objects.c | 7 +- + fs/netfs/read_collect.c | 38 +- + fs/netfs/read_pgpriv2.c | 13 +- + fs/netfs/read_retry.c | 7 +- + fs/netfs/read_single.c | 50 +- + fs/netfs/stats.c | 4 +- + fs/netfs/write_collect.c | 157 +- + fs/netfs/write_issue.c | 141 +- + fs/netfs/write_retry.c | 8 +- + fs/nfs/Kconfig | 18 +- + fs/nfs/blocklayout/Makefile | 3 +- + fs/nfs/blocklayout/blocklayout.c | 119 +- + fs/nfs/blocklayout/dev.c | 32 +- + fs/nfs/callback_proc.c | 23 +- + fs/nfs/callback_xdr.c | 9 +- + fs/nfs/client.c | 6 +- + fs/nfs/dir.c | 12 +- + fs/nfs/direct.c | 139 +- + fs/nfs/filelayout/filelayout.c | 17 +- + fs/nfs/filelayout/filelayoutdev.c | 19 +- + fs/nfs/flexfilelayout/flexfilelayout.c | 413 ++-- + fs/nfs/flexfilelayout/flexfilelayout.h | 48 +- + fs/nfs/flexfilelayout/flexfilelayoutdev.c | 212 ++- + fs/nfs/inode.c | 4 +- + fs/nfs/internal.h | 13 +- + fs/nfs/namespace.c | 4 +- + fs/nfs/netns.h | 5 +- + fs/nfs/nfs3_fs.h | 2 +- + fs/nfs/nfs3acl.c | 2 +- + fs/nfs/nfs3client.c | 9 +- + fs/nfs/nfs4_fs.h | 3 + + fs/nfs/nfs4client.c | 8 +- + fs/nfs/nfs4file.c | 1 + + fs/nfs/nfs4proc.c | 146 +- + fs/nfs/nfs4state.c | 7 +- + fs/nfs/nfs4xdr.c | 2 +- + fs/nfs/pagelist.c | 66 +- + fs/nfs/pnfs.c | 376 +++- + fs/nfs/pnfs.h | 102 +- + fs/nfs/pnfs_dev.c | 30 +- + fs/nfs/pnfs_nfs.c | 108 +- + fs/nfs/read.c | 2 +- + fs/nfs/super.c | 25 - + fs/nfs/unlink.c | 3 + + fs/nfs/write.c | 2 +- + fs/nfs_common/nfs_ssc.c | 126 +- + fs/nfsd/blocklayout.c | 1 + + fs/nfsd/blocklayoutxdr.c | 11 + + fs/nfsd/export.c | 8 +- + fs/nfsd/export.h | 3 +- + fs/nfsd/filecache.c | 1 + + fs/nfsd/flexfilelayout.c | 24 +- + fs/nfsd/flexfilelayoutxdr.c | 9 +- + fs/nfsd/flexfilelayoutxdr.h | 8 +- + fs/nfsd/localio.c | 10 +- + fs/nfsd/lockd.c | 5 +- + fs/nfsd/netns.h | 17 +- + fs/nfsd/nfs2acl.c | 48 +- + fs/nfsd/nfs3acl.c | 1 + + fs/nfsd/nfs3proc.c | 79 +- + fs/nfsd/nfs3xdr.c | 4 + + fs/nfsd/nfs4acl.c | 1 + + fs/nfsd/nfs4callback.c | 8 +- + fs/nfsd/nfs4ctl.h | 83 + + fs/nfsd/nfs4idmap.c | 1 + + fs/nfsd/nfs4layouts.c | 1 + + fs/nfsd/nfs4proc.c | 465 +++-- + fs/nfsd/nfs4recover.c | 1 + + fs/nfsd/nfs4state.c | 587 +++++- + fs/nfsd/nfs4xdr.c | 188 +- + fs/nfsd/nfscache.c | 1 + + fs/nfsd/nfsctl.c | 36 +- + fs/nfsd/nfsd.h | 237 +-- + fs/nfsd/nfserr.h | 159 ++ + fs/nfsd/nfsfh.c | 73 +- + fs/nfsd/nfsfh.h | 28 +- + fs/nfsd/nfsproc.c | 39 +- + fs/nfsd/nfssvc.c | 8 + + fs/nfsd/nfsxdr.c | 1 + + fs/nfsd/state.h | 46 +- + fs/nfsd/trace.h | 45 +- + fs/nfsd/vfs.c | 311 ++- + fs/nfsd/vfs.h | 39 +- + fs/nfsd/xdr.h | 6 +- + fs/nfsd/xdr3.h | 2 +- + fs/nfsd/xdr4.h | 156 +- + fs/nfsd/xdr4cb.h | 20 +- + fs/nilfs2/inode.c | 4 +- + fs/nilfs2/ioctl.c | 2 +- + fs/nilfs2/namei.c | 10 +- + fs/nilfs2/nilfs.h | 6 +- + fs/nls/nls_iso8859-14.c | 26 +- + fs/notify/fanotify/fanotify_user.c | 2 +- + fs/nsfs.c | 4 +- + fs/ntfs/attrib.c | 76 +- + fs/ntfs/attrlist.c | 14 +- + fs/ntfs/bitmap.c | 19 +- + fs/ntfs/collate.c | 8 +- + fs/ntfs/dir.c | 23 +- + fs/ntfs/ea.c | 10 +- + fs/ntfs/ea.h | 6 +- + fs/ntfs/file.c | 60 +- + fs/ntfs/index.c | 2 +- + fs/ntfs/index.h | 2 +- + fs/ntfs/inode.c | 131 +- + fs/ntfs/inode.h | 4 +- + fs/ntfs/iomap.c | 16 +- + fs/ntfs/layout.h | 19 +- + fs/ntfs/mft.c | 921 ++++++--- + fs/ntfs/mft.h | 19 +- + fs/ntfs/mst.c | 4 +- + fs/ntfs/namei.c | 43 +- + fs/ntfs/ntfs.h | 7 +- + fs/ntfs/reparse.c | 2 +- + fs/ntfs/runlist.c | 5 +- + fs/ntfs/super.c | 391 +++- + fs/ntfs/unistr.c | 29 +- + fs/ntfs/volume.h | 15 + + fs/ntfs3/attrib.c | 47 +- + fs/ntfs3/dir.c | 3 + + fs/ntfs3/file.c | 9 +- + fs/ntfs3/frecord.c | 5 + + fs/ntfs3/fslog.c | 4 +- + fs/ntfs3/fsntfs.c | 3 + + fs/ntfs3/index.c | 13 +- + fs/ntfs3/inode.c | 21 +- + fs/ntfs3/namei.c | 10 +- + fs/ntfs3/ntfs_fs.h | 16 +- + fs/ntfs3/record.c | 16 +- + fs/ntfs3/super.c | 2 +- + fs/ntfs3/xattr.c | 26 +- + fs/ocfs2/acl.c | 2 +- + fs/ocfs2/acl.h | 2 +- + fs/ocfs2/buffer_head_io.c | 12 +- + fs/ocfs2/dlmfs/dlmfs.c | 6 +- + fs/ocfs2/file.c | 6 +- + fs/ocfs2/file.h | 6 +- + fs/ocfs2/ioctl.c | 2 +- + fs/ocfs2/ioctl.h | 2 +- + fs/ocfs2/journal.c | 25 +- + fs/ocfs2/namei.c | 10 +- + fs/ocfs2/quota_local.c | 4 + + fs/ocfs2/xattr.c | 6 +- + fs/omfs/dir.c | 6 +- + fs/omfs/file.c | 2 +- + fs/omfs/inode.c | 4 +- + fs/open.c | 50 +- + fs/orangefs/acl.c | 2 +- + fs/orangefs/inode.c | 29 +- + fs/orangefs/namei.c | 8 +- + fs/orangefs/orangefs-debugfs.c | 10 +- + fs/orangefs/orangefs-kernel.h | 8 +- + fs/orangefs/super.c | 18 + + fs/orangefs/xattr.c | 2 +- + fs/overlayfs/dir.c | 14 +- + fs/overlayfs/file.c | 2 +- + fs/overlayfs/inode.c | 16 +- + fs/overlayfs/overlayfs.h | 14 +- + fs/overlayfs/ovl_entry.h | 2 +- + fs/overlayfs/util.c | 4 +- + fs/overlayfs/xattrs.c | 4 +- + fs/pidfs.c | 6 +- + fs/pipe.c | 2 +- + fs/pnode.c | 108 +- + fs/posix_acl.c | 26 +- + fs/proc/base.c | 53 +- + fs/proc/fd.c | 6 +- + fs/proc/fd.h | 2 +- + fs/proc/generic.c | 4 +- + fs/proc/internal.h | 4 +- + fs/proc/proc_net.c | 2 +- + fs/proc/proc_sysctl.c | 6 +- + fs/proc/root.c | 2 +- + fs/proc/vmcore.c | 20 + + fs/quota/dquot.c | 18 +- + fs/ramfs/file-nommu.c | 4 +- + fs/ramfs/inode.c | 10 +- + fs/read_write.c | 8 +- + fs/remap_range.c | 2 +- + fs/smb/client/cifsacl.c | 4 +- + fs/smb/client/cifsfs.c | 43 +- + fs/smb/client/cifsfs.h | 21 +- + fs/smb/client/cifsglob.h | 10 +- + fs/smb/client/cifsproto.h | 4 +- + fs/smb/client/dir.c | 6 +- + fs/smb/client/file.c | 5 +- + fs/smb/client/inode.c | 32 +- + fs/smb/client/link.c | 2 +- + fs/smb/client/smb2ops.c | 130 +- + fs/smb/client/smb2pdu.c | 10 +- + fs/smb/client/transport.c | 13 +- + fs/smb/client/xattr.c | 2 +- + fs/smb/common/compress/compress.c | 16 +- + fs/smb/common/fscc.h | 5 +- + fs/smb/common/smbglob.h | 1 + + fs/smb/server/Kconfig | 2 + + fs/smb/server/Makefile | 1 + + fs/smb/server/compress.c | 1 + + fs/smb/server/connection.c | 3 +- + fs/smb/server/connection.h | 3 + + fs/smb/server/ksmbd_work.c | 3 - + fs/smb/server/ksmbd_work.h | 5 +- + fs/smb/server/mgmt/user_session.c | 4 +- + fs/smb/server/ndr.c | 2 +- + fs/smb/server/ndr.h | 2 +- + fs/smb/server/oplock.c | 2 +- + fs/smb/server/server.c | 21 + + fs/smb/server/smb2ops.c | 33 + + fs/smb/server/smb2pdu.c | 770 ++++---- + fs/smb/server/smb2pdu.h | 23 +- + fs/smb/server/smb_common.c | 6 +- + fs/smb/server/smbacl.c | 27 +- + fs/smb/server/smbacl.h | 8 +- + fs/smb/server/tests/Kconfig | 15 + + fs/smb/server/tests/Makefile | 4 + + fs/smb/server/tests/smbacl_kunit.c | 301 +++ + fs/smb/server/transport_tcp.c | 8 +- + fs/smb/server/vfs.c | 52 +- + fs/smb/server/vfs.h | 33 +- + fs/smb/server/vfs_cache.c | 66 +- + fs/smb/server/vfs_cache.h | 6 - + fs/splice.c | 70 +- + fs/stat.c | 4 +- + fs/super.c | 19 +- + fs/tests/.kunitconfig | 2 + + fs/tests/fdtable_kunit.c | 72 + + fs/tracefs/event_inode.c | 2 +- + fs/tracefs/inode.c | 8 +- + fs/ubifs/dir.c | 14 +- + fs/ubifs/file.c | 4 +- + fs/ubifs/ioctl.c | 2 +- + fs/ubifs/ubifs.h | 6 +- + fs/ubifs/xattr.c | 2 +- + fs/udf/file.c | 2 +- + fs/udf/inode.c | 59 +- + fs/udf/misc.c | 15 +- + fs/udf/namei.c | 12 +- + fs/udf/super.c | 5 +- + fs/udf/symlink.c | 2 +- + fs/ufs/dir.c | 2 +- + fs/ufs/inode.c | 2 +- + fs/ufs/namei.c | 10 +- + fs/ufs/ufs.h | 2 +- + fs/vboxsf/dir.c | 8 +- + fs/vboxsf/utils.c | 4 +- + fs/vboxsf/vfsmod.h | 4 +- + fs/xattr.c | 30 +- + fs/xfs/libxfs/xfs_attr.c | 14 +- + fs/xfs/libxfs/xfs_bmap.c | 5 +- + fs/xfs/libxfs/xfs_bmap.h | 2 +- + fs/xfs/libxfs/xfs_btree_mem.c | 6 +- + fs/xfs/libxfs/xfs_dquot_buf.c | 13 +- + fs/xfs/libxfs/xfs_errortag.h | 6 +- + fs/xfs/libxfs/xfs_exchmaps.c | 1 - + fs/xfs/libxfs/xfs_exchmaps.h | 3 +- + fs/xfs/libxfs/xfs_ialloc.c | 2 +- + fs/xfs/libxfs/xfs_ialloc_btree.c | 1 - + fs/xfs/libxfs/xfs_ialloc_btree.h | 5 +- + fs/xfs/libxfs/xfs_inode_fork.c | 4 +- + fs/xfs/libxfs/xfs_inode_util.h | 2 +- + fs/xfs/libxfs/xfs_metadir.c | 32 +- + fs/xfs/libxfs/xfs_metadir.h | 5 +- + fs/xfs/libxfs/xfs_parent.h | 1 - + fs/xfs/libxfs/xfs_refcount_btree.c | 16 +- + fs/xfs/libxfs/xfs_refcount_btree.h | 26 + + fs/xfs/libxfs/xfs_rmap.c | 13 +- + fs/xfs/libxfs/xfs_rmap_btree.c | 16 +- + fs/xfs/libxfs/xfs_rmap_btree.h | 26 + + fs/xfs/libxfs/xfs_rtrefcount_btree.c | 121 +- + fs/xfs/libxfs/xfs_rtrefcount_btree.h | 5 +- + fs/xfs/libxfs/xfs_rtrmap_btree.c | 236 +-- + fs/xfs/libxfs/xfs_rtrmap_btree.h | 3 +- + fs/xfs/libxfs/xfs_sb.c | 40 - + fs/xfs/libxfs/xfs_sb.h | 1 - + fs/xfs/libxfs/xfs_symlink_remote.c | 8 +- + fs/xfs/libxfs/xfs_symlink_remote.h | 2 +- + fs/xfs/libxfs/xfs_trans_resv.c | 45 +- + fs/xfs/libxfs/xfs_trans_space.c | 6 +- + fs/xfs/libxfs/xfs_types.c | 7 +- + fs/xfs/libxfs/xfs_types.h | 7 +- + fs/xfs/scrub/alloc_repair.c | 15 +- + fs/xfs/scrub/bmap.c | 11 +- + fs/xfs/scrub/bmap_repair.c | 4 +- + fs/xfs/scrub/inode_repair.c | 32 +- + fs/xfs/scrub/orphanage.c | 2 +- + fs/xfs/scrub/quota.c | 2 +- + fs/xfs/scrub/quota_repair.c | 9 +- + fs/xfs/scrub/quotacheck_repair.c | 54 +- + fs/xfs/scrub/rtrefcount_repair.c | 3 +- + fs/xfs/scrub/rtrmap_repair.c | 2 +- + fs/xfs/scrub/xfarray.c | 201 +- + fs/xfs/scrub/xfarray.h | 9 +- + fs/xfs/xfs_acl.c | 2 +- + fs/xfs/xfs_acl.h | 2 +- + fs/xfs/xfs_aops.c | 13 +- + fs/xfs/xfs_bmap_item.c | 2 +- + fs/xfs/xfs_buf.c | 11 + + fs/xfs/xfs_dquot.c | 20 +- + fs/xfs/xfs_dquot.h | 2 +- + fs/xfs/xfs_exchmaps_item.c | 4 +- + fs/xfs/xfs_exchrange.c | 2 +- + fs/xfs/xfs_file.c | 33 +- + fs/xfs/xfs_handle.c | 8 +- + fs/xfs/xfs_inode.c | 24 +- + fs/xfs/xfs_inode.h | 2 +- + fs/xfs/xfs_ioctl.c | 423 +++-- + fs/xfs/xfs_ioctl.h | 6 +- + fs/xfs/xfs_ioctl32.c | 188 +- + fs/xfs/xfs_ioend.c | 137 +- + fs/xfs/xfs_ioend.h | 2 + + fs/xfs/xfs_iops.c | 24 +- + fs/xfs/xfs_iops.h | 2 +- + fs/xfs/xfs_itable.c | 2 +- + fs/xfs/xfs_itable.h | 2 +- + fs/xfs/xfs_mount.h | 8 + + fs/xfs/xfs_qm.c | 6 +- + fs/xfs/xfs_rmap_item.c | 2 +- + fs/xfs/xfs_rtalloc.h | 49 +- + fs/xfs/xfs_super.c | 3 +- + fs/xfs/xfs_symlink.c | 6 +- + fs/xfs/xfs_symlink.h | 2 +- + fs/xfs/xfs_sysfs.c | 78 +- + fs/xfs/xfs_trace.h | 29 +- + fs/xfs/xfs_trans_dquot.c | 6 +- + fs/xfs/xfs_xattr.c | 2 +- + fs/xfs/xfs_zone_alloc.c | 4 + + fs/zonefs/super.c | 2 +- + include/linux/binfmts.h | 3 +- + include/linux/bio-integrity.h | 3 +- + include/linux/bio.h | 11 +- + include/linux/blkdev.h | 8 +- + include/linux/buffer_head.h | 86 +- + include/linux/capability.h | 8 +- + include/linux/cleanup.h | 7 - + include/linux/configfs.h | 73 +- + include/linux/coredump.h | 37 +- + include/linux/dax.h | 12 - + include/linux/dcache.h | 38 +- + include/linux/f2fs_fs.h | 140 +- + include/linux/fdtable.h | 15 +- + include/linux/file.h | 130 +- + include/linux/fileattr.h | 2 +- + include/linux/fs.h | 127 +- + include/linux/fs_context.h | 4 + + include/linux/fscache-cache.h | 2 +- + include/linux/fscache.h | 53 +- + include/linux/iomap.h | 38 +- + include/linux/lsm_hook_defs.h | 25 +- + include/linux/mnt_idmapping.h | 24 +- + include/linux/mount.h | 4 +- + include/linux/namei.h | 19 +- + include/linux/netfs.h | 114 +- + include/linux/nfs.h | 55 +- + include/linux/nfs3.h | 43 + + include/linux/nfs4.h | 6 + + include/linux/nfs_fh.h | 63 + + include/linux/nfs_fs.h | 6 +- + include/linux/nfs_fs_sb.h | 8 + + include/linux/nfs_page.h | 8 +- + include/linux/nfs_ssc.h | 69 +- + include/linux/nfs_xdr.h | 2 + + include/linux/nfsd_ssc.h | 38 + + include/linux/nfslocalio.h | 11 +- + include/linux/posix_acl.h | 24 +- + include/linux/quotaops.h | 6 +- + include/linux/sched.h | 2 +- + include/linux/sched/signal.h | 33 +- + include/linux/security.h | 61 +- + include/linux/splice.h | 4 +- + include/linux/sunrpc/svc_xprt.h | 5 +- + include/linux/uidgid.h | 12 +- + include/linux/user_namespace.h | 11 +- + include/linux/wait_bit.h | 26 + + include/linux/xattr.h | 20 +- + include/trace/events/cachefiles.h | 81 +- + include/trace/events/f2fs.h | 71 + + include/trace/events/fscache.h | 10 +- + include/trace/events/netfs.h | 145 +- + include/trace/misc/nfs.h | 13 +- + include/uapi/linux/btrfs_tree.h | 25 +- + include/uapi/linux/close_range.h | 31 +- + include/uapi/linux/coredump.h | 149 +- + include/uapi/linux/fs.h | 2 +- + include/uapi/linux/fuse.h | 12 +- + init/Kconfig | 11 + + init/initramfs.c | 11 +- + io_uring/io-wq.c | 2 + + io_uring/mock_file.c | 8 +- + ipc/mqueue.c | 2 +- + kernel/Makefile | 1 + + kernel/bpf/bpf_iter.c | 6 +- + kernel/bpf/inode.c | 6 +- + kernel/bpf/token.c | 6 +- + kernel/capability.c | 4 +- + kernel/exit.c | 15 +- + kernel/fork.c | 68 +- + kernel/kthread.c | 2 +- + kernel/pid_namespace.c | 3 +- + kernel/ptrace.c | 6 + + kernel/signal.c | 14 + + kernel/tests/.kunitconfig | 4 + + kernel/tests/user_ns_map_kunit.c | 98 + + kernel/user_namespace.c | 60 +- + kernel/utsname.c | 1 - + mm/secretmem.c | 2 +- + mm/shmem.c | 30 +- + mm/shmem_quota.c | 5 + + mm/userfaultfd.c | 6 +- + net/core/scm.c | 8 +- + net/handshake/netlink.c | 8 +- + net/kcm/kcmsock.c | 6 +- + net/socket.c | 6 +- + net/sunrpc/svc_xprt.c | 36 +- + net/sunrpc/svcauth_unix.c | 9 + + net/sunrpc/svcsock.c | 490 +++-- + net/sunrpc/xprtrdma/svc_rdma_recvfrom.c | 18 +- + net/sunrpc/xprtrdma/svc_rdma_transport.c | 3 +- + net/sunrpc/xprtsock.c | 69 +- + net/unix/af_unix.c | 2 +- + rust/kernel/configfs.rs | 127 +- + samples/configfs/configfs_sample.c | 137 +- + security/apparmor/apparmorfs.c | 2 +- + security/apparmor/lsm.c | 4 +- + security/commoncap.c | 10 +- + security/integrity/evm/evm_main.c | 26 +- + security/integrity/ima/ima.h | 10 +- + security/integrity/ima/ima_api.c | 2 +- + security/integrity/ima/ima_appraise.c | 12 +- + security/integrity/ima/ima_main.c | 6 +- + security/integrity/ima/ima_policy.c | 4 +- + security/security.c | 49 +- + security/selinux/hooks.c | 45 +- + security/selinux/selinuxfs.c | 2 +- + security/smack/smack_lsm.c | 14 +- + tools/include/uapi/linux/coredump.h | 149 +- + tools/testing/selftests/Makefile | 4 + + .../selftests/clone3/clone3_clear_sighand.c | 6 +- + tools/testing/selftests/core/close_range_test.c | 951 ++++++++++ + tools/testing/selftests/coredump/.gitignore | 2 + + tools/testing/selftests/coredump/Makefile | 11 +- + .../selftests/coredump/coredump_notify_signal.h | 29 + + .../coredump/coredump_notify_signal_helper.c | 46 + + .../coredump/coredump_notify_signal_test.c | 245 +++ + .../selftests/coredump/coredump_signal_test.c | 238 +++ + .../coredump/coredump_socket_protocol_test.c | 1983 +++++++++++++++----- + tools/testing/selftests/coredump/coredump_test.h | 32 +- + .../selftests/coredump/coredump_test_helpers.c | 1742 ++++++++++++++++- + .../selftests/coredump/coredump_test_helpers.h | 79 + + .../selftests/coredump/coredump_worker_test.c | 447 +++++ + tools/testing/selftests/exec/Makefile | 4 + + tools/testing/selftests/exec/binfmt_misc_delim.c | 127 ++ + tools/testing/selftests/filesystems/.gitignore | 1 - + tools/testing/selftests/filesystems/Makefile | 2 +- + tools/testing/selftests/filesystems/config | 8 + + .../selftests/filesystems/configfs/.gitignore | 2 + + .../selftests/filesystems/configfs/Makefile | 8 + + .../testing/selftests/filesystems/configfs/config | 5 + + .../selftests/filesystems/configfs/configfs_test.c | 481 +++++ + .../selftests/filesystems/file_stressor/.gitignore | 2 + + .../selftests/filesystems/file_stressor/Makefile | 6 + + .../{ => file_stressor}/file_stressor.c | 0 + .../selftests/filesystems/file_stressor/settings | 3 + + .../selftests/filesystems/fscontext_ns/.gitignore | 2 + + .../testing/selftests/filesystems/fuse/.gitignore | 1 + + tools/testing/selftests/filesystems/fuse/Makefile | 2 +- + tools/testing/selftests/filesystems/kernfs_test.c | 1203 +++++++++++- + .../filesystems/mntns_unbindable/Makefile | 6 + + .../mntns_unbindable/mntns_unbindable_test.c | 227 +++ + .../selftests/filesystems/openat2/openat2_test.c | 10 +- + .../selftests/filesystems/openat2/resolve_test.c | 7 +- + .../filesystems/statmount/statmount_test.c | 2 +- + .../filesystems/umount_propagation/Makefile | 6 + + .../umount_propagation/umount_propagation_test.c | 226 +++ + .../move_mount_set_group_test.c | 74 +- + tools/testing/selftests/pidfd/pidfd_open_test.c | 2 +- + virt/kvm/guest_memfd.c | 2 +- + 874 files changed, 28004 insertions(+), 16066 deletions(-) + delete mode 100644 Documentation/filesystems/bfs.rst + create mode 100644 Documentation/filesystems/exfat.rst + delete mode 100644 arch/powerpc/platforms/cell/spufs/coredump.c + delete mode 100644 fs/bfs/Kconfig + delete mode 100644 fs/bfs/Makefile + delete mode 100644 fs/bfs/bfs.h + delete mode 100644 fs/bfs/file.c + delete mode 100644 fs/bfs/inode.c + create mode 100644 fs/f2fs/cache.c + create mode 100644 fs/f2fs/cache.h + delete mode 100644 fs/netfs/fscache_internal.h + create mode 100644 fs/nfsd/nfs4ctl.h + create mode 100644 fs/nfsd/nfserr.h + create mode 100644 fs/smb/server/tests/Kconfig + create mode 100644 fs/smb/server/tests/Makefile + create mode 100644 fs/smb/server/tests/smbacl_kunit.c + create mode 100644 fs/tests/.kunitconfig + create mode 100644 fs/tests/fdtable_kunit.c + create mode 100644 include/linux/nfs_fh.h + create mode 100644 include/linux/nfsd_ssc.h + create mode 100644 kernel/tests/.kunitconfig + create mode 100644 kernel/tests/user_ns_map_kunit.c + create mode 100644 tools/testing/selftests/coredump/coredump_notify_signal.h + create mode 100644 tools/testing/selftests/coredump/coredump_notify_signal_helper.c + create mode 100644 tools/testing/selftests/coredump/coredump_notify_signal_test.c + create mode 100644 tools/testing/selftests/coredump/coredump_signal_test.c + create mode 100644 tools/testing/selftests/coredump/coredump_test_helpers.h + create mode 100644 tools/testing/selftests/coredump/coredump_worker_test.c + create mode 100644 tools/testing/selftests/exec/binfmt_misc_delim.c + create mode 100644 tools/testing/selftests/filesystems/config + create mode 100644 tools/testing/selftests/filesystems/configfs/.gitignore + create mode 100644 tools/testing/selftests/filesystems/configfs/Makefile + create mode 100644 tools/testing/selftests/filesystems/configfs/config + create mode 100644 tools/testing/selftests/filesystems/configfs/configfs_test.c + create mode 100644 tools/testing/selftests/filesystems/file_stressor/.gitignore + create mode 100644 tools/testing/selftests/filesystems/file_stressor/Makefile + rename tools/testing/selftests/filesystems/{ => file_stressor}/file_stressor.c (100%) + create mode 100644 tools/testing/selftests/filesystems/file_stressor/settings + create mode 100644 tools/testing/selftests/filesystems/fscontext_ns/.gitignore + create mode 100644 tools/testing/selftests/filesystems/mntns_unbindable/Makefile + create mode 100644 tools/testing/selftests/filesystems/mntns_unbindable/mntns_unbindable_test.c + create mode 100644 tools/testing/selftests/filesystems/umount_propagation/Makefile + create mode 100644 tools/testing/selftests/filesystems/umount_propagation/umount_propagation_test.c +Merging printk/for-next (3dd0c9c7109de Merge branch 'for-7.4' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/printk/linux.git printk/for-next +Auto-merging drivers/tty/serial/8250/8250_port.c +Merge made by the 'ort' strategy. + drivers/accessibility/braille/braille_console.c | 57 ++++++++++- + drivers/tty/serial/8250/8250_port.c | 5 +- + drivers/tty/serial/amba-pl011.c | 2 +- + drivers/tty/serial/imx.c | 2 +- + drivers/tty/serial/sifive.c | 2 +- + include/linux/console.h | 15 +++ + kernel/printk/nbcon.c | 79 ++++++++++++++ + kernel/printk/printk.c | 130 +++++++++++++----------- + kernel/printk/printk_ringbuffer.c | 17 ++-- + lib/vsprintf.c | 11 +- + 10 files changed, 237 insertions(+), 83 deletions(-) +Merging pci/next (a0818fd5eaf50 Merge branch 'pci/misc') +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/pci/pci.git pci/next +Auto-merging MAINTAINERS +Auto-merging arch/arm64/boot/dts/qcom/qcs6490-rb3gen2.dts +Auto-merging drivers/pci/controller/dwc/pci-imx6.c +Auto-merging drivers/pci/endpoint/pci-ep-msi.c +Merge made by the 'ort' strategy. + Documentation/ABI/testing/sysfs-bus-pci | 14 +- + Documentation/PCI/controller/index.rst | 1 - + .../PCI/controller/rcar-pcie-firmware.rst | 32 - + .../bindings/pci/nvidia,tegra20-pcie.txt | 670 --------------------- + .../bindings/pci/nvidia,tegra20-pcie.yaml | 532 ++++++++++++++++ + .../devicetree/bindings/pci/rcar-gen4-pci-ep.yaml | 9 +- + .../bindings/pci/rcar-gen4-pci-host.yaml | 81 ++- + .../bindings/pci/renesas,r9a08g045-pcie.yaml | 33 +- + .../devicetree/bindings/pci/toshiba,tc9563.yaml | 14 +- + .../devicetree/bindings/pci/xilinx-versal-cpm.yaml | 38 ++ + Documentation/driver-api/pci/p2pdma.rst | 6 +- + MAINTAINERS | 1 - + arch/arm64/boot/dts/qcom/qcs6490-rb3gen2.dts | 7 +- + drivers/gpio/Kconfig | 11 + + drivers/gpio/Makefile | 1 + + drivers/gpio/gpio-tc9563.c | 99 +++ + drivers/net/wireless/ath/ath10k/Kconfig | 2 +- + drivers/net/wireless/ath/ath10k/pci.c | 11 +- + drivers/net/wireless/ath/ath10k/pci.h | 5 +- + drivers/net/wireless/ath/ath11k/Kconfig | 2 +- + drivers/net/wireless/ath/ath11k/pci.c | 19 +- + drivers/net/wireless/ath/ath11k/pci.h | 3 +- + drivers/net/wireless/ath/ath12k/Kconfig | 2 +- + drivers/net/wireless/ath/ath12k/pci.c | 19 +- + drivers/net/wireless/ath/ath12k/pci.h | 4 +- + .../pci/controller/cadence/pcie-cadence-debugfs.c | 15 +- + drivers/pci/controller/cadence/pcie-cadence-plat.c | 6 +- + drivers/pci/controller/cadence/pcie-cadence.c | 16 +- + drivers/pci/controller/cadence/pcie-cadence.h | 2 - + drivers/pci/controller/dwc/pci-dra7xx.c | 22 +- + drivers/pci/controller/dwc/pci-imx6.c | 120 +++- + drivers/pci/controller/dwc/pci-keystone.c | 36 +- + drivers/pci/controller/dwc/pcie-designware-ep.c | 18 +- + drivers/pci/controller/dwc/pcie-designware-host.c | 18 +- + drivers/pci/controller/dwc/pcie-designware.c | 176 +++--- + drivers/pci/controller/dwc/pcie-designware.h | 252 ++++---- + drivers/pci/controller/dwc/pcie-dw-rockchip.c | 1 + + drivers/pci/controller/dwc/pcie-fu740.c | 6 +- + drivers/pci/controller/dwc/pcie-histb.c | 1 + + drivers/pci/controller/dwc/pcie-nxp-s32g.c | 18 +- + drivers/pci/controller/dwc/pcie-qcom-common.c | 28 +- + drivers/pci/controller/dwc/pcie-qcom-ep.c | 1 + + drivers/pci/controller/dwc/pcie-qcom.c | 182 ++++-- + drivers/pci/controller/dwc/pcie-rcar-gen4.c | 347 +++++++++-- + drivers/pci/controller/dwc/pcie-spacemit-k1.c | 3 + + drivers/pci/controller/dwc/pcie-tegra194-acpi.c | 22 +- + drivers/pci/controller/dwc/pcie-tegra194.c | 87 +-- + drivers/pci/controller/dwc/pcie-ultrarisc.c | 10 +- + drivers/pci/controller/pci-aardvark.c | 31 +- + drivers/pci/controller/pci-tegra.c | 1 + + drivers/pci/controller/pci-xgene.c | 11 +- + drivers/pci/controller/pcie-aspeed.c | 9 +- + drivers/pci/controller/pcie-mediatek-gen3.c | 15 +- + drivers/pci/controller/pcie-mediatek.c | 4 +- + drivers/pci/controller/pcie-rockchip-host.c | 1 + + drivers/pci/controller/pcie-rzg3s-host.c | 76 ++- + drivers/pci/controller/pcie-tegra264.c | 9 +- + drivers/pci/controller/pcie-xilinx-cpm.c | 77 ++- + drivers/pci/controller/plda/pcie-starfive.c | 1 + + drivers/pci/controller/vmd.c | 57 +- + drivers/pci/endpoint/functions/pci-epf-vntb.c | 93 ++- + drivers/pci/endpoint/pci-ep-msi.c | 40 +- + drivers/pci/hotplug/pciehp_core.c | 2 +- + drivers/pci/iov.c | 25 - + drivers/pci/p2pdma.c | 39 +- + drivers/pci/pci-label.c | 94 ++- + drivers/pci/pci.h | 13 +- + drivers/pci/pcie/aspm.c | 140 +++-- + drivers/pci/probe.c | 15 +- + drivers/pci/pwrctrl/Kconfig | 2 + + drivers/pci/pwrctrl/pci-pwrctrl-tc9563.c | 321 ++++++---- + drivers/pci/quirks.c | 126 +++- + drivers/pci/tph.c | 6 +- + include/linux/pci-epf.h | 3 +- + include/linux/pci.h | 7 +- + include/linux/soc/qcom/tc9563.h | 19 + + 76 files changed, 2608 insertions(+), 1632 deletions(-) + delete mode 100644 Documentation/PCI/controller/rcar-pcie-firmware.rst + delete mode 100644 Documentation/devicetree/bindings/pci/nvidia,tegra20-pcie.txt + create mode 100644 Documentation/devicetree/bindings/pci/nvidia,tegra20-pcie.yaml + create mode 100644 drivers/gpio/gpio-tc9563.c + create mode 100644 include/linux/soc/qcom/tc9563.h +Merging pstore/for-next/pstore (7c756181175d5 pstore: publish big_oops_buf after max_compressed_size) +$ git merge -m Merge branch 'for-next/pstore' of https://git.kernel.org/pub/scm/linux/kernel/git/kees/linux.git pstore/for-next/pstore +Merge made by the 'ort' strategy. + fs/pstore/platform.c | 12 +++++++++--- + fs/pstore/ram.c | 1 + + fs/pstore/ram_core.c | 6 ++++++ + 3 files changed, 16 insertions(+), 3 deletions(-) +Merging hid/for-next (145c2b2e9a5c0 Merge branch 'for-7.3/upstream-fixes' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/hid/hid.git hid/for-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 6 + + drivers/hid/Kconfig | 21 +- + drivers/hid/Makefile | 2 +- + drivers/hid/amd-sfh-hid/amd_sfh_common.h | 2 + + drivers/hid/amd-sfh-hid/amd_sfh_pcie.c | 10 + + drivers/hid/amd-sfh-hid/sfh1_1/amd_sfh_desc.c | 13 +- + drivers/hid/amd-sfh-hid/sfh1_1/amd_sfh_interface.c | 37 +- + drivers/hid/amd-sfh-hid/sfh1_1/amd_sfh_interface.h | 7 + + drivers/hid/bpf/hid_bpf_dispatch.c | 2 +- + .../hid/bpf/progs/Huion__Inspiroy-Frego-M.bpf.c | 11 +- + drivers/hid/hid-appletb-kbd.c | 95 ++- + drivers/hid/hid-asus.c | 14 +- + drivers/hid/hid-google-hammer.c | 2 +- + drivers/hid/hid-ids.h | 15 + + drivers/hid/hid-kysona.c | 292 -------- + drivers/hid/hid-lenovo.c | 14 + + drivers/hid/hid-logitech-dj.c | 48 +- + drivers/hid/hid-logitech-hidpp.c | 51 +- + drivers/hid/hid-multitouch.c | 2 +- + drivers/hid/hid-nintendo.c | 14 +- + drivers/hid/hid-pulsar.c | 765 +++++++++++++++++++++ + drivers/hid/hid-universal-pidff.c | 1 + + drivers/hid/i2c-hid/i2c-hid-acpi.c | 8 +- + drivers/hid/i2c-hid/i2c-hid-core.c | 11 +- + .../intel-thc-hid/intel-quicki2c/pci-quicki2c.c | 10 +- + .../intel-thc-hid/intel-quicki2c/quicki2c-dev.h | 2 + + .../intel-thc-hid/intel-quicki2c/quicki2c-hid.c | 6 +- + .../intel-thc-hid/intel-quicki2c/quicki2c-hid.h | 2 +- + .../intel-thc-hid/intel-quickspi/pci-quickspi.c | 5 +- + .../intel-thc-hid/intel-quickspi/quickspi-dev.h | 2 + + .../intel-thc-hid/intel-quickspi/quickspi-hid.c | 6 +- + .../intel-thc-hid/intel-quickspi/quickspi-hid.h | 2 +- + .../intel-quickspi/quickspi-protocol.c | 2 +- + drivers/hid/surface-hid/surface_kbd.c | 4 +- + drivers/hid/usbhid/hiddev.c | 45 +- + drivers/hid/wacom.h | 2 +- + drivers/hid/wacom_sys.c | 196 ++++-- + drivers/hid/wacom_wac.c | 72 +- + drivers/hid/wacom_wac.h | 6 +- + include/linux/hiddev.h | 2 + + tools/testing/selftests/hid/hid_bpf.c | 31 + + 41 files changed, 1270 insertions(+), 568 deletions(-) + delete mode 100644 drivers/hid/hid-kysona.c + create mode 100644 drivers/hid/hid-pulsar.c +Merging i2c/i2c/for-next (8cd9520d35a6c Linux 7.1) +$ git merge -m Merge branch 'i2c/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/wsa/linux.git i2c/i2c/for-next +Already up to date. +$ git am -3 ../patches/0001-i2c-Fix-up-the-rest-of-the-merge.patch +Applying: i2c: Fix up the rest of the merge +Using index info to reconstruct a base tree... +M drivers/i2c/i2c-core-base.c +Falling back to patching base and 3-way merge... +Auto-merging drivers/i2c/i2c-core-base.c +No changes -- Patch already applied. +Merging i2c-andi/i2c/i2c-next (91c65ba57f3d8 Merge branch 'i2c/i2c-fixes' into i2c/i2c-next) +$ git merge -m Merge branch 'i2c/i2c-next' of https://git.kernel.org/pub/scm/linux/kernel/git/andi.shyti/linux.git i2c-andi/i2c/i2c-next +Merge made by the 'ort' strategy. + .../bindings/i2c/marvell,mv64xxx-i2c.yaml | 1 + + .../devicetree/bindings/i2c/renesas,rcar-i2c.yaml | 4 +- + .../bindings/i2c/snps,designware-i2c.yaml | 15 ++++ + .../bindings/i2c/xlnx,xps-iic-2.00.a.yaml | 2 +- + drivers/i2c/Kconfig | 13 +++ + drivers/i2c/busses/i2c-bcm2835.c | 2 +- + drivers/i2c/busses/i2c-ljca.c | 9 -- + drivers/i2c/busses/i2c-qcom-geni.c | 77 +++++++++++------ + drivers/i2c/i2c-core-acpi.c | 1 + + drivers/i2c/i2c-core-base.c | 43 ++++++++++ + drivers/i2c/i2c-core-smbus.c | 24 ++++-- + drivers/i2c/i2c-dev.c | 3 + + drivers/i2c/i2c-mux.c | 98 ++++++++++++---------- + drivers/media/i2c/isl7998x.c | 3 +- + include/linux/i2c.h | 36 +++++++- + 15 files changed, 232 insertions(+), 99 deletions(-) +Merging i2c-rust/rust-i2c-next (61ddec70c9bcc i2c: rust: mark I2cAdapter methods as inline) +$ git merge -m Merge branch 'rust-i2c-next' of https://github.com/ikrtn/rust-for-linux i2c-rust/rust-i2c-next +Auto-merging drivers/i2c/busses/i2c-imx-lpi2c.c +CONFLICT (content): Merge conflict in drivers/i2c/busses/i2c-imx-lpi2c.c +Auto-merging drivers/i2c/busses/i2c-imx.c +Auto-merging drivers/i2c/busses/i2c-qcom-geni.c +Auto-merging rust/kernel/i2c.rs +Resolved 'drivers/i2c/busses/i2c-imx-lpi2c.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 7f9fc7aaf484c] Merge branch 'rust-i2c-next' of https://github.com/ikrtn/rust-for-linux +$ git diff -M --stat --summary HEAD^.. + drivers/i2c/busses/i2c-imx-lpi2c.c | 18 ++++-------------- + 1 file changed, 4 insertions(+), 14 deletions(-) +Merging i3c/i3c/next (b0cfe49a2e3b9 i3c: hub: Add support for the I3C interface in the I3C hub) +$ git merge -m Merge branch 'i3c/next' of https://git.kernel.org/pub/scm/linux/kernel/git/i3c/linux.git i3c/i3c/next +Auto-merging MAINTAINERS +Auto-merging drivers/i3c/master/amd-i3c-master.c +Merge made by the 'ort' strategy. + .../devicetree/bindings/i3c/nxp,p3h2840.yaml | 330 ++++++++ + MAINTAINERS | 9 + + drivers/i3c/Kconfig | 14 + + drivers/i3c/Makefile | 1 + + drivers/i3c/device.c | 15 +- + drivers/i3c/hub.c | 832 +++++++++++++++++++++ + drivers/i3c/internals.h | 16 + + drivers/i3c/master.c | 476 ++++++++++-- + drivers/i3c/master/adi-i3c-master.c | 4 + + drivers/i3c/master/amd-i3c-master.c | 4 +- + drivers/i3c/master/dw-i3c-master.c | 4 +- + drivers/i3c/master/mipi-i3c-hci/cmd.h | 4 +- + drivers/i3c/master/mipi-i3c-hci/cmd_v1.c | 28 +- + drivers/i3c/master/mipi-i3c-hci/cmd_v2.c | 4 +- + drivers/i3c/master/mipi-i3c-hci/core.c | 112 ++- + drivers/i3c/master/mipi-i3c-hci/dat.h | 2 + + drivers/i3c/master/mipi-i3c-hci/dat_v1.c | 13 +- + drivers/i3c/master/mipi-i3c-hci/dma.c | 136 +++- + drivers/i3c/master/mipi-i3c-hci/hci.h | 9 +- + drivers/i3c/master/mipi-i3c-hci/mipi-i3c-hci-pci.c | 5 +- + include/linux/i3c/device.h | 4 + + include/linux/i3c/hub.h | 92 +++ + include/linux/i3c/master.h | 20 + + include/linux/platform_data/mipi-i3c-hci.h | 4 + + 24 files changed, 1991 insertions(+), 147 deletions(-) + create mode 100644 Documentation/devicetree/bindings/i3c/nxp,p3h2840.yaml + create mode 100644 drivers/i3c/hub.c + create mode 100644 include/linux/i3c/hub.h +Merging dmi/dmi-for-next (1afafbaf749d8 firmware/dmi: Include product_family info to modalias) +$ git merge -m Merge branch 'dmi-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/jdelvare/staging.git dmi/dmi-for-next +Already up to date. +Merging hwmon-staging/hwmon-next (4781ca52761e6 hwmon:(pmbus/xdpe1a2g7b) Add support for xdpe1a2g7c controller) +$ git merge -m Merge branch 'hwmon-next' of https://git.kernel.org/pub/scm/linux/kernel/git/groeck/linux-staging.git hwmon-staging/hwmon-next +Auto-merging Documentation/devicetree/bindings/vendor-prefixes.yaml +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + Documentation/ABI/testing/sysfs-class-hwmon | 36 + + .../bindings/hwmon/axiado,ax3000-pwm-fan.yaml | 72 ++ + .../bindings/hwmon/axiado,ax3000-tsadc.yaml | 45 + + .../devicetree/bindings/hwmon/national,lm63.yaml | 55 ++ + .../devicetree/bindings/hwmon/national,lm90.yaml | 28 +- + .../bindings/hwmon/pmbus/adi,max20826.yaml | 102 ++ + .../bindings/hwmon/pmbus/infineon,tda38740.yaml | 55 ++ + .../bindings/hwmon/pmbus/isil,isl68137.yaml | 2 + + .../bindings/hwmon/pmbus/ti,tps25990.yaml | 8 +- + .../devicetree/bindings/hwmon/ti,tmp102.yaml | 16 +- + .../devicetree/bindings/hwmon/ti,tmp401.yaml | 3 + + .../devicetree/bindings/trivial-devices.yaml | 17 +- + .../devicetree/bindings/vendor-prefixes.yaml | 2 + + Documentation/hwmon/arctic_fan_controller.rst | 26 +- + Documentation/hwmon/asus_ec_sensors.rst | 2 + + Documentation/hwmon/asus_rog_ryujin.rst | 1 + + Documentation/hwmon/axiado-pwm-fan.rst | 38 + + Documentation/hwmon/axiado-tsadc.rst | 41 + + Documentation/hwmon/honor-fmi.rst | 32 + + Documentation/hwmon/index.rst | 6 + + Documentation/hwmon/it87.rst | 8 + + Documentation/hwmon/lm63.rst | 22 + + Documentation/hwmon/max20826.rst | 139 +++ + Documentation/hwmon/max34440.rst | 28 +- + Documentation/hwmon/minisforum-um780xtx.rst | 68 ++ + Documentation/hwmon/nct6683.rst | 4 + + Documentation/hwmon/nct6775.rst | 32 + + Documentation/hwmon/sht4x.rst | 17 +- + Documentation/hwmon/socfpga-hwmon.rst | 52 + + Documentation/hwmon/sysfs-interface.rst | 14 +- + Documentation/hwmon/tda38740.rst | 65 ++ + Documentation/hwmon/tps25990.rst | 15 +- + Documentation/hwmon/tps53679.rst | 39 +- + Documentation/hwmon/vexpress.rst | 2 +- + Documentation/hwmon/yogafan.rst | 46 +- + MAINTAINERS | 45 +- + drivers/hwmon/Kconfig | 57 +- + drivers/hwmon/Makefile | 4 + + drivers/hwmon/abituguru.c | 4 +- + drivers/hwmon/acpi_power_meter.c | 4 +- + drivers/hwmon/adm1177.c | 2 +- + drivers/hwmon/adt7x10.c | 2 +- + drivers/hwmon/arctic_fan_controller.c | 22 +- + drivers/hwmon/asus-ec-sensors.c | 4 + + drivers/hwmon/asus_atk0110.c | 4 +- + drivers/hwmon/asus_rog_ryujin.c | 3 + + drivers/hwmon/asus_wmi_sensors.c | 2 +- + drivers/hwmon/axiado-pwm-fan.c | 416 ++++++++ + drivers/hwmon/axiado-tsadc.c | 275 ++++++ + drivers/hwmon/coretemp.c | 2 +- + drivers/hwmon/dell-smm-hwmon.c | 32 + + drivers/hwmon/emc2103.c | 2 +- + drivers/hwmon/hih6130.c | 4 +- + drivers/hwmon/honor-fmi.c | 178 ++++ + drivers/hwmon/hp-wmi-sensors.c | 2 +- + drivers/hwmon/hwmon.c | 6 + + drivers/hwmon/it87.c | 308 +++++- + drivers/hwmon/k10temp.c | 2 + + drivers/hwmon/lm63.c | 426 ++++++-- + drivers/hwmon/lm85.c | 2 +- + drivers/hwmon/lm90.c | 5 + + drivers/hwmon/minisforum-um780xtx.c | 501 ++++++++++ + drivers/hwmon/nct6775-core.c | 116 +++ + drivers/hwmon/nct6775-platform.c | 33 +- + drivers/hwmon/nct6775.h | 4 +- + drivers/hwmon/pmbus/Kconfig | 41 +- + drivers/hwmon/pmbus/Makefile | 2 + + drivers/hwmon/pmbus/ibm-cffps.c | 4 +- + drivers/hwmon/pmbus/isl68137.c | 4 + + drivers/hwmon/pmbus/ltc4286.c | 96 +- + drivers/hwmon/pmbus/max20826.c | 1044 ++++++++++++++++++++ + drivers/hwmon/pmbus/max34440.c | 2 + + drivers/hwmon/pmbus/mp2888.c | 4 +- + drivers/hwmon/pmbus/pmbus.h | 7 + + drivers/hwmon/pmbus/pmbus_core.c | 29 +- + drivers/hwmon/pmbus/tda38740.c | 64 ++ + drivers/hwmon/pmbus/tps25990.c | 249 +++-- + drivers/hwmon/pmbus/tps53679.c | 146 ++- + drivers/hwmon/pmbus/xdpe1a2g7b.c | 10 +- + drivers/hwmon/pt5161l.c | 4 +- + drivers/hwmon/pwm-fan.c | 2 +- + drivers/hwmon/scmi-hwmon.c | 3 +- + drivers/hwmon/sht4x.c | 66 +- + drivers/hwmon/socfpga-hwmon.c | 22 + + drivers/hwmon/spd5118.c | 67 +- + drivers/hwmon/xgene-hwmon.c | 6 +- + drivers/hwmon/yogafan.c | 48 + + include/linux/hwmon.h | 12 + + 88 files changed, 5146 insertions(+), 391 deletions(-) + create mode 100644 Documentation/devicetree/bindings/hwmon/axiado,ax3000-pwm-fan.yaml + create mode 100644 Documentation/devicetree/bindings/hwmon/axiado,ax3000-tsadc.yaml + create mode 100644 Documentation/devicetree/bindings/hwmon/national,lm63.yaml + create mode 100644 Documentation/devicetree/bindings/hwmon/pmbus/adi,max20826.yaml + create mode 100644 Documentation/devicetree/bindings/hwmon/pmbus/infineon,tda38740.yaml + create mode 100644 Documentation/hwmon/axiado-pwm-fan.rst + create mode 100644 Documentation/hwmon/axiado-tsadc.rst + create mode 100644 Documentation/hwmon/honor-fmi.rst + create mode 100644 Documentation/hwmon/max20826.rst + create mode 100644 Documentation/hwmon/minisforum-um780xtx.rst + create mode 100644 Documentation/hwmon/tda38740.rst + create mode 100644 drivers/hwmon/axiado-pwm-fan.c + create mode 100644 drivers/hwmon/axiado-tsadc.c + create mode 100644 drivers/hwmon/honor-fmi.c + create mode 100644 drivers/hwmon/minisforum-um780xtx.c + create mode 100644 drivers/hwmon/pmbus/max20826.c + create mode 100644 drivers/hwmon/pmbus/tda38740.c +Merging jc_docs/docs-next (2d72a4c09867a docs: kdoc: parse context_lock_struct() as struct declaration) +$ git merge -m Merge branch 'docs-next' of git://git.lwn.net/linux.git jc_docs/docs-next +Auto-merging Documentation/ABI/testing/sysfs-bus-pci +Auto-merging Documentation/admin-guide/kernel-parameters.txt +Auto-merging Documentation/admin-guide/sysctl/kernel.rst +Auto-merging Documentation/filesystems/locking.rst +Auto-merging Documentation/filesystems/porting.rst +Auto-merging Documentation/filesystems/proc.rst +Auto-merging Makefile +Merge made by the 'ort' strategy. + .../ABI/stable/sysfs-driver-firmware-zynqmp | 70 +- + Documentation/ABI/testing/sysfs-bus-pci | 14 +- + Documentation/ABI/testing/sysfs-bus-usb | 3 +- + .../ABI/testing/sysfs-class-firmware-attributes | 21 +- + .../ABI/testing/sysfs-driver-aspeed-uart-routing | 7 +- + Documentation/ABI/testing/sysfs-driver-xdata | 20 +- + .../ABI/testing/sysfs-driver-xilinx-tmr-manager | 14 +- + Documentation/ABI/testing/sysfs-platform-intel-ifs | 6 +- + Documentation/Makefile | 5 + + Documentation/admin-guide/LSM/LoadPin.rst | 10 +- + Documentation/admin-guide/RAS/main.rst | 12 +- + Documentation/admin-guide/cpu-isolation.rst | 4 +- + Documentation/admin-guide/kernel-parameters.txt | 58 +- + Documentation/admin-guide/parport.rst | 2 +- + Documentation/admin-guide/sysctl/kernel.rst | 19 +- + Documentation/admin-guide/sysctl/vm.rst | 2 +- + .../verify-bugs-and-bisect-regressions.rst | 26 +- + Documentation/arch/powerpc/vas-api.rst | 4 +- + Documentation/core-api/cpu_hotplug.rst | 4 +- + Documentation/core-api/debug-objects.rst | 2 +- + Documentation/core-api/dma-attributes.rst | 2 +- + Documentation/core-api/dma-isa-lpc.rst | 11 +- + Documentation/core-api/housekeeping.rst | 2 +- + Documentation/core-api/irq/irq-affinity.rst | 2 +- + Documentation/core-api/irq/irqflags-tracing.rst | 8 +- + Documentation/core-api/maple_tree.rst | 9 +- + Documentation/core-api/real-time/differences.rst | 10 +- + Documentation/core-api/swiotlb.rst | 2 +- + Documentation/core-api/this_cpu_ops.rst | 6 +- + Documentation/core-api/xarray.rst | 4 +- + Documentation/doc-guide/kernel-doc.rst | 7 +- + Documentation/doc-guide/parse-headers.rst | 15 +- + Documentation/driver-api/reset.rst | 2 +- + Documentation/driver-api/serial/serial-rs485.rst | 5 +- + .../locking/cmpxchg-local/arch-support.txt | 2 +- + Documentation/filesystems/fuse/fuse.rst | 8 +- + Documentation/filesystems/locking.rst | 2 +- + Documentation/filesystems/porting.rst | 2 +- + Documentation/filesystems/proc.rst | 10 +- + Documentation/input/devices/yealink.rst | 7 +- + Documentation/input/notifier.rst | 2 +- + Documentation/livepatch/api.rst | 5 +- + Documentation/misc-devices/spear-pcie-gadget.rst | 10 +- + Documentation/process/1.Intro.rst | 2 +- + Documentation/process/2.Process.rst | 20 +- + Documentation/process/3.Early-stage.rst | 2 +- + Documentation/process/5.Posting.rst | 12 +- + Documentation/process/6.Followthrough.rst | 2 +- + Documentation/process/7.AdvancedTopics.rst | 36 +- + Documentation/process/backporting.rst | 18 +- + .../process/embargoed-hardware-issues.rst | 2 +- + Documentation/process/handling-regressions.rst | 2 +- + Documentation/process/howto.rst | 2 +- + Documentation/process/maintainer-pgp-guide.rst | 38 +- + Documentation/process/submitting-patches.rst | 2 +- + Documentation/scheduler/index.rst | 1 + + Documentation/scheduler/sched-preemption.rst | 87 ++ + Documentation/timers/hpet.rst | 33 +- + Documentation/timers/no_hz.rst | 4 +- + .../translations/pt_BR/admin-guide/README.rst | 382 ++++++ + .../translations/pt_BR/admin-guide/devices.rst | 278 +++++ + .../translations/pt_BR/admin-guide/index.rst | 190 +++ + Documentation/translations/pt_BR/index.rst | 10 + + .../translations/pt_BR/process/2.Process.rst | 14 +- + .../translations/pt_BR/process/3.Early-stage.rst | 10 +- + .../translations/pt_BR/process/4.Coding.rst | 6 +- + .../translations/pt_BR/process/5.Posting.rst | 12 +- + .../translations/pt_BR/process/6.Followthrough.rst | 14 +- + .../translations/pt_BR/process/8.Conclusion.rst | 3 +- + .../translations/pt_BR/process/adding-syscalls.rst | 56 +- + .../pt_BR/process/applying-patches.rst | 6 +- + .../translations/pt_BR/process/backporting.rst | 4 +- + .../process/code-of-conduct-interpretation.rst | 2 + + .../translations/pt_BR/process/code-of-conduct.rst | 4 +- + .../pt_BR/process/coding-assistants.rst | 60 + + .../translations/pt_BR/process/coding-style.rst | 1320 ++++++++++++++++++++ + Documentation/translations/pt_BR/process/cve.rst | 4 +- + .../translations/pt_BR/process/debugging/index.rst | 73 ++ + .../pt_BR/process/development-process.rst | 2 + + .../pt_BR/process/embargoed-hardware-issues.rst | 362 ++++++ + Documentation/translations/pt_BR/process/howto.rst | 19 +- + Documentation/translations/pt_BR/process/index.rst | 10 + + .../translations/pt_BR/process/kernel-docs.rst | 2 + + .../pt_BR/process/kernel-enforcement-statement.rst | 163 +++ + .../translations/pt_BR/process/license-rules.rst | 2 + + .../pt_BR/process/maintainer-devicetree.rst | 76 ++ + .../pt_BR/process/maintainer-handbooks.rst | 4 +- + .../pt_BR/process/maintainer-kvm-x86.rst | 10 +- + .../pt_BR/process/maintainer-soc-clean-dts.rst | 5 +- + .../translations/pt_BR/process/maintainer-tip.rst | 847 +++++++++++++ + .../pt_BR/process/management-style.rst | 9 +- + .../translations/pt_BR/process/security-bugs.rst | 25 +- + .../pt_BR/process/stable-api-nonsense.rst | 208 +++ + .../pt_BR/process/submit-checklist.rst | 4 +- + .../pt_BR/process/submitting-patches.rst | 963 ++++++++++++++ + .../pt_BR/process/volatile-considered-harmful.rst | 130 ++ + .../translations/zh_CN/admin-guide/README.rst | 2 +- + .../zh_CN/admin-guide/mm/damon/index.rst | 15 +- + .../zh_CN/admin-guide/mm/damon/lru_sort.rst | 68 +- + .../zh_CN/admin-guide/mm/damon/reclaim.rst | 82 +- + .../zh_CN/admin-guide/mm/damon/start.rst | 63 +- + .../zh_CN/admin-guide/mm/damon/stat.rst | 94 ++ + .../zh_CN/admin-guide/mm/damon/usage.rst | 521 ++++++-- + .../translations/zh_CN/mm/damon/design.rst | 667 +++++++++- + .../translations/zh_CN/networking/driver.rst | 138 ++ + .../translations/zh_CN/networking/index.rst | 10 +- + .../translations/zh_CN/networking/ipv6.rst | 76 ++ + .../translations/zh_CN/networking/secid.rst | 24 + + .../translations/zh_CN/networking/sriov.rst | 34 + + .../translations/zh_CN/networking/team.rst | 16 + + .../zh_CN/process/applying-patches.rst | 390 ++++++ + Documentation/translations/zh_CN/process/howto.rst | 2 +- + Documentation/translations/zh_CN/process/index.rst | 2 +- + Documentation/translations/zh_TW/glossary.rst | 168 +++ + Documentation/translations/zh_TW/index.rst | 5 +- + .../translations/zh_TW/process/1.Intro.rst | 227 ++-- + .../translations/zh_TW/process/2.Process.rst | 325 +++-- + .../translations/zh_TW/process/5.Posting.rst | 235 ++-- + .../zh_TW/process/7.AdvancedTopics.rst | 106 +- + .../translations/zh_TW/process/8.Conclusion.rst | 58 +- + .../process/code-of-conduct-interpretation.rst | 162 ++- + .../translations/zh_TW/process/coding-style.rst | 610 ++++----- + .../translations/zh_TW/process/email-clients.rst | 185 +-- + .../zh_TW/process/embargoed-hardware-issues.rst | 209 ++-- + Documentation/translations/zh_TW/process/howto.rst | 344 ++--- + Documentation/translations/zh_TW/process/index.rst | 105 +- + .../translations/zh_TW/process/license-rules.rst | 201 +-- + .../zh_TW/process/programming-language.rst | 92 +- + .../zh_TW/process/stable-kernel-rules.rst | 244 +++- + .../zh_TW/process/submitting-patches.rst | 568 +++++---- + Documentation/userspace-api/dma-buf-heaps.rst | 2 +- + Documentation/userspace-api/futex2.rst | 4 +- + Documentation/userspace-api/ioctl/ioctl-number.rst | 7 +- + Documentation/userspace-api/iommufd.rst | 2 +- + Documentation/userspace-api/vduse.rst | 33 +- + Documentation/virt/coco/sev-guest.rst | 2 +- + Makefile | 2 +- + include/uapi/linux/gpio.h | 2 +- + security/landlock/syscalls.c | 2 +- + tools/lib/python/kdoc/c_lex.py | 2 +- + tools/lib/python/kdoc/kdoc_parser.py | 11 +- + tools/lib/python/kdoc/xforms_lists.py | 3 +- + tools/unittests/test_kdoc_parser.py | 26 + + 143 files changed, 9947 insertions(+), 2187 deletions(-) + create mode 100644 Documentation/scheduler/sched-preemption.rst + create mode 100644 Documentation/translations/pt_BR/admin-guide/README.rst + create mode 100644 Documentation/translations/pt_BR/admin-guide/devices.rst + create mode 100644 Documentation/translations/pt_BR/admin-guide/index.rst + create mode 100644 Documentation/translations/pt_BR/process/coding-assistants.rst + create mode 100644 Documentation/translations/pt_BR/process/coding-style.rst + create mode 100644 Documentation/translations/pt_BR/process/debugging/index.rst + create mode 100644 Documentation/translations/pt_BR/process/embargoed-hardware-issues.rst + create mode 100644 Documentation/translations/pt_BR/process/kernel-enforcement-statement.rst + create mode 100644 Documentation/translations/pt_BR/process/maintainer-devicetree.rst + create mode 100644 Documentation/translations/pt_BR/process/maintainer-tip.rst + create mode 100644 Documentation/translations/pt_BR/process/stable-api-nonsense.rst + create mode 100644 Documentation/translations/pt_BR/process/submitting-patches.rst + create mode 100644 Documentation/translations/pt_BR/process/volatile-considered-harmful.rst + create mode 100644 Documentation/translations/zh_CN/admin-guide/mm/damon/stat.rst + create mode 100644 Documentation/translations/zh_CN/networking/driver.rst + create mode 100644 Documentation/translations/zh_CN/networking/ipv6.rst + create mode 100644 Documentation/translations/zh_CN/networking/secid.rst + create mode 100644 Documentation/translations/zh_CN/networking/sriov.rst + create mode 100644 Documentation/translations/zh_CN/networking/team.rst + create mode 100644 Documentation/translations/zh_CN/process/applying-patches.rst + create mode 100644 Documentation/translations/zh_TW/glossary.rst +Merging v4l-dvb/next (89a3d2a2237e0 media: iris: fix 64-bit division in iris_set_slice_count()) +$ git merge -m Merge branch 'next' of git://linuxtv.org/media-ci/media-pending.git v4l-dvb/next +Auto-merging MAINTAINERS +Auto-merging arch/arm/boot/dts/nvidia/tegra30-lg-p895.dts +Auto-merging arch/arm/boot/dts/nvidia/tegra30-lg-x3.dtsi +Auto-merging arch/arm64/boot/dts/freescale/imx8mq-librem5.dtsi +Auto-merging arch/arm64/boot/dts/qcom/qcm6490-fairphone-fp5.dts +Auto-merging arch/arm64/boot/dts/qcom/sc8280xp-lenovo-thinkpad-x13s.dts +Auto-merging arch/arm64/boot/dts/qcom/sdm670-google-common.dtsi +Auto-merging drivers/media/i2c/isl7998x.c +Auto-merging drivers/media/pci/intel/ipu-bridge.c +Auto-merging drivers/media/platform/nxp/imx7-media-csi.c +Auto-merging drivers/media/usb/em28xx/em28xx-video.c +Auto-merging drivers/media/v4l2-core/v4l2-ctrls-core.c +Merge made by the 'ort' strategy. + .../admin-guide/media/em28xx-cardlist.rst | 4 + + Documentation/admin-guide/media/vivid.rst | 4 +- + .../bindings/clock/nxp,imx95-blk-ctl.yaml | 71 + + .../bindings/media/fsl,imx95-csi-formatter.yaml | 88 + + .../bindings/media/i2c/himax,hm1246.yaml | 121 + + .../devicetree/bindings/media/i2c/hynix,hi846.yaml | 3 +- + .../devicetree/bindings/media/i2c/ite,it6625.yaml | 177 ++ + .../bindings/media/i2c/ovti,og0ve1b.yaml | 15 +- + .../bindings/media/i2c/ovti,os02g10.yaml | 94 + + .../bindings/media/i2c/ovti,ov08d10.yaml | 3 +- + .../devicetree/bindings/media/i2c/ovti,ov4689.yaml | 3 +- + .../devicetree/bindings/media/i2c/ovti,ov5675.yaml | 3 +- + .../devicetree/bindings/media/i2c/ovti,ov5693.yaml | 5 +- + .../bindings/media/i2c/ovti,ov64a40.yaml | 3 +- + .../bindings/media/i2c/samsung,s5kjn5.yaml | 117 + + .../devicetree/bindings/media/i2c/sony,imx111.yaml | 3 +- + .../devicetree/bindings/media/i2c/sony,imx355.yaml | 3 +- + .../devicetree/bindings/media/i2c/sony,imx415.yaml | 3 +- + .../devicetree/bindings/media/i2c/st,vd55g1.yaml | 3 +- + .../devicetree/bindings/media/i2c/st,vd56g3.yaml | 3 +- + .../bindings/media/i2c/thine,thp7312.yaml | 3 +- + .../bindings/media/i2c/toshiba,tc358743.txt | 48 - + .../bindings/media/i2c/toshiba,tc358743.yaml | 93 + + .../bindings/media/nxp,imx8mq-mipi-csi2.yaml | 4 +- + .../bindings/media/qcom,glymur-camss.yaml | 300 ++ + .../bindings/media/qcom,qcm2290-venus.yaml | 26 +- + .../bindings/media/qcom,venus-common.yaml | 5 +- + .../bindings/media/qcom,x1e80100-camss.yaml | 28 +- + .../devicetree/bindings/media/snps,dw-hdmi-rx.yaml | 13 +- + .../bindings/media/video-interface-devices.yaml | 17 +- + .../bindings/phy/qcom,x1e80100-csi2-phy.yaml | 213 ++ + Documentation/driver-api/media/tx-rx.rst | 6 +- + .../userspace-api/media/drivers/dcmipp.rst | 14 + + .../userspace-api/media/drivers/index.rst | 1 + + MAINTAINERS | 77 +- + .../nvidia/tegra30-asus-nexus7-grouper-common.dtsi | 3 +- + .../nvidia/tegra30-asus-transformer-common.dtsi | 3 +- + arch/arm/boot/dts/nvidia/tegra30-lg-p895.dts | 4 +- + arch/arm/boot/dts/nvidia/tegra30-lg-x3.dtsi | 3 +- + .../imx8mp-tqma8mpql-mba8mp-ras314-imx219.dtso | 3 +- + arch/arm64/boot/dts/freescale/imx8mq-librem5.dtsi | 3 +- + arch/arm64/boot/dts/qcom/qcm6490-fairphone-fp5.dts | 3 +- + .../dts/qcom/sc8280xp-lenovo-thinkpad-x13s.dts | 3 +- + arch/arm64/boot/dts/qcom/sdm670-google-common.dtsi | 3 +- + .../r8a779g3-sparrow-hawk-camera-j1-imx219.dtso | 3 +- + .../r8a779g3-sparrow-hawk-camera-j1-imx462.dtso | 3 +- + .../r8a779g3-sparrow-hawk-camera-j2-imx219.dtso | 3 +- + .../r8a779g3-sparrow-hawk-camera-j2-imx462.dtso | 3 +- + arch/arm64/boot/dts/rockchip/px30-pp1516.dtsi | 3 +- + .../rockchip/px30-ringneck-haikou-video-demo.dtso | 3 +- + .../boot/dts/rockchip/rk3399-pinephone-pro.dts | 5 +- + .../rk3588-rock-5b-plus-radxa-cam4k-cam0.dtso | 3 +- + .../rk3588-rock-5b-plus-radxa-cam4k-cam1.dtso | 3 +- + drivers/media/cec/platform/meson/ao-cec-g12a.c | 2 +- + .../extron-da-hd-4k-plus/extron-da-hd-4k-plus.c | 4 +- + drivers/media/common/cypress_firmware.c | 2 + + drivers/media/common/saa7146/saa7146_core.c | 23 +- + drivers/media/common/saa7146/saa7146_video.c | 11 + + drivers/media/dvb-frontends/drxd_map_firm.h | 4 +- + drivers/media/dvb-frontends/stv0900_core.c | 2 + + drivers/media/dvb-frontends/stv090x.c | 2 + + drivers/media/i2c/Kconfig | 72 +- + drivers/media/i2c/Makefile | 5 + + drivers/media/i2c/adv7170.c | 1 + + drivers/media/i2c/adv7175.c | 3 +- + drivers/media/i2c/adv7180.c | 5 +- + drivers/media/i2c/adv7183.c | 3 +- + drivers/media/i2c/adv7343.c | 10 +- + drivers/media/i2c/adv7393.c | 10 +- + drivers/media/i2c/adv748x/adv748x-afe.c | 1 + + drivers/media/i2c/adv748x/adv748x-core.c | 13 +- + drivers/media/i2c/adv748x/adv748x-csi2.c | 1 + + drivers/media/i2c/adv748x/adv748x-hdmi.c | 1 + + drivers/media/i2c/adv7511-v4l2.c | 1 + + drivers/media/i2c/adv7604.c | 7 +- + drivers/media/i2c/adv7842.c | 11 +- + drivers/media/i2c/ak881x.c | 2 +- + drivers/media/i2c/alvium-csi2.c | 4 + + drivers/media/i2c/ar0521.c | 2 +- + drivers/media/i2c/ccs/ccs-core.c | 12 +- + drivers/media/i2c/ccs/ccs-data.h | 4 +- + drivers/media/i2c/cvs/Kconfig | 1 + + drivers/media/i2c/cvs/core.c | 30 +- + drivers/media/i2c/cvs/v4l2.c | 96 +- + drivers/media/i2c/cx25840/cx25840-core.c | 39 +- + drivers/media/i2c/ds90ub913.c | 1 + + drivers/media/i2c/ds90ub953.c | 1 + + drivers/media/i2c/ds90ub960.c | 1 + + drivers/media/i2c/et8ek8/et8ek8_driver.c | 1 + + drivers/media/i2c/gc0308.c | 1 + + drivers/media/i2c/gc0310.c | 2 +- + drivers/media/i2c/gc05a2.c | 4 +- + drivers/media/i2c/gc08a3.c | 4 +- + drivers/media/i2c/gc2145.c | 2 + + drivers/media/i2c/hi556.c | 2 + + drivers/media/i2c/hi846.c | 2 + + drivers/media/i2c/hi847.c | 3 +- + drivers/media/i2c/hm1246.c | 1287 +++++++++ + drivers/media/i2c/imx111.c | 1 + + drivers/media/i2c/imx208.c | 1 + + drivers/media/i2c/imx214.c | 4 +- + drivers/media/i2c/imx219.c | 42 +- + drivers/media/i2c/imx258.c | 2 + + drivers/media/i2c/imx274.c | 3 + + drivers/media/i2c/imx283.c | 2 + + drivers/media/i2c/imx290.c | 4 +- + drivers/media/i2c/imx296.c | 7 +- + drivers/media/i2c/imx319.c | 1 + + drivers/media/i2c/imx334.c | 135 +- + drivers/media/i2c/imx335.c | 4 +- + drivers/media/i2c/imx355.c | 4 +- + drivers/media/i2c/imx412.c | 3 +- + drivers/media/i2c/imx415.c | 4 +- + drivers/media/i2c/imx471.c | 6 +- + drivers/media/i2c/imx678.c | 2 +- + drivers/media/i2c/isl7998x.c | 1 + + drivers/media/i2c/it6625.c | 2459 +++++++++++++++++ + drivers/media/i2c/lt6911uxe.c | 5 +- + drivers/media/i2c/max9286.c | 1 + + drivers/media/i2c/max96714.c | 1 + + drivers/media/i2c/max96717.c | 1 + + drivers/media/i2c/ml86v7667.c | 1 - + drivers/media/i2c/mt9m001.c | 9 +- + drivers/media/i2c/mt9m111.c | 3 + + drivers/media/i2c/mt9m114.c | 6 + + drivers/media/i2c/mt9p031.c | 3 + + drivers/media/i2c/mt9t112.c | 3 + + drivers/media/i2c/mt9v011.c | 1 + + drivers/media/i2c/mt9v032.c | 3 + + drivers/media/i2c/mt9v111.c | 1 + + drivers/media/i2c/og01a1b.c | 8 +- + drivers/media/i2c/og0ve1b.c | 455 ++- + drivers/media/i2c/os02g10.c | 934 +++++++ + drivers/media/i2c/os05b10.c | 2 + + drivers/media/i2c/ov01a10.c | 3 + + drivers/media/i2c/ov02a10.c | 3 +- + drivers/media/i2c/ov02c10.c | 5 +- + drivers/media/i2c/ov02e10.c | 1 + + drivers/media/i2c/ov05c10.c | 978 +++++++ + drivers/media/i2c/ov08d10.c | 1 + + drivers/media/i2c/ov08x40.c | 1 + + drivers/media/i2c/ov13858.c | 51 + + drivers/media/i2c/ov13b10.c | 1 + + drivers/media/i2c/ov2640.c | 2 + + drivers/media/i2c/ov2659.c | 1 + + drivers/media/i2c/ov2680.c | 3 + + drivers/media/i2c/ov2685.c | 2 + + drivers/media/i2c/ov2732.c | 4 +- + drivers/media/i2c/ov2735.c | 5 +- + drivers/media/i2c/ov2740.c | 1 + + drivers/media/i2c/ov4689.c | 2 + + drivers/media/i2c/ov5640.c | 2 + + drivers/media/i2c/ov5645.c | 4 +- + drivers/media/i2c/ov5647.c | 6 + + drivers/media/i2c/ov5648.c | 5 +- + drivers/media/i2c/ov5670.c | 2 + + drivers/media/i2c/ov5675.c | 2 + + drivers/media/i2c/ov5693.c | 28 + + drivers/media/i2c/ov5695.c | 1 + + drivers/media/i2c/ov6211.c | 3 +- + drivers/media/i2c/ov64a40.c | 2 + + drivers/media/i2c/ov7251.c | 4 +- + drivers/media/i2c/ov7670.c | 1 + + drivers/media/i2c/ov772x.c | 4 +- + drivers/media/i2c/ov7740.c | 1 + + drivers/media/i2c/ov8856.c | 33 +- + drivers/media/i2c/ov8858.c | 3 +- + drivers/media/i2c/ov8865.c | 2 + + drivers/media/i2c/ov9282.c | 4 +- + drivers/media/i2c/ov9640.c | 2 + + drivers/media/i2c/ov9650.c | 1 + + drivers/media/i2c/ov9734.c | 1 + + drivers/media/i2c/rdacm20.c | 1 - + drivers/media/i2c/rdacm21.c | 1 - + drivers/media/i2c/rj54n1cb0c.c | 3 + + drivers/media/i2c/s5c73m3/s5c73m3-core.c | 2 + + drivers/media/i2c/s5k3m5.c | 4 +- + drivers/media/i2c/s5k5baf.c | 3 + + drivers/media/i2c/s5k6a3.c | 1 + + drivers/media/i2c/s5kjn1.c | 4 +- + drivers/media/i2c/s5kjn5.c | 2897 ++++++++++++++++++++ + drivers/media/i2c/saa6752hs.c | 1 + + drivers/media/i2c/saa7115.c | 1 + + drivers/media/i2c/saa717x.c | 1 + + drivers/media/i2c/st-mipid02.c | 1 + + drivers/media/i2c/t4ka3.c | 4 +- + drivers/media/i2c/tc358743.c | 6 +- + drivers/media/i2c/tc358746.c | 1 + + drivers/media/i2c/tda1997x.c | 3 +- + drivers/media/i2c/thp7312.c | 1 + + drivers/media/i2c/ths7303.c | 10 +- + drivers/media/i2c/ths8200.c | 14 +- + drivers/media/i2c/ths8200_regs.h | 14 +- + drivers/media/i2c/tvp514x.c | 1 + + drivers/media/i2c/tvp5150.c | 3 +- + drivers/media/i2c/tvp7002.c | 1 + + drivers/media/i2c/tw9900.c | 1 + + drivers/media/i2c/tw9910.c | 2 + + drivers/media/i2c/vd55g1.c | 18 +- + drivers/media/i2c/vd56g3.c | 4 +- + drivers/media/i2c/vgxy61.c | 4 +- + drivers/media/i2c/wm8739.c | 2 +- + drivers/media/pci/bt8xx/bttv-driver.c | 8 +- + drivers/media/pci/cobalt/cobalt-driver.c | 8 +- + drivers/media/pci/cobalt/cobalt-v4l2.c | 10 +- + drivers/media/pci/cx18/cx18-av-core.c | 1 + + drivers/media/pci/cx18/cx18-controls.c | 2 +- + drivers/media/pci/cx18/cx18-driver.c | 9 +- + drivers/media/pci/cx18/cx18-ioctl.c | 2 +- + drivers/media/pci/cx18/cx18-queue.c | 5 +- + drivers/media/pci/cx18/cx23418.h | 2 +- + drivers/media/pci/cx23885/cx23885-core.c | 4 + + drivers/media/pci/cx23885/cx23885-dvb.c | 14 +- + drivers/media/pci/cx23885/cx23885-video.c | 4 +- + drivers/media/pci/cx88/cx88-input.c | 22 +- + drivers/media/pci/cx88/cx88-mpeg.c | 9 +- + drivers/media/pci/cx88/cx88-video.c | 8 +- + drivers/media/pci/hws/hws.h | 4 +- + drivers/media/pci/hws/hws_pci.c | 6 +- + drivers/media/pci/hws/hws_reg.h | 10 +- + drivers/media/pci/hws/hws_v4l2_ioctl.c | 5 +- + drivers/media/pci/hws/hws_video.c | 6 +- + drivers/media/pci/intel/ipu-bridge.c | 150 +- + drivers/media/pci/intel/ipu3/ipu3-cio2.c | 1 + + drivers/media/pci/intel/ipu6/Kconfig | 20 +- + drivers/media/pci/intel/ipu6/Makefile | 10 +- + drivers/media/pci/intel/ipu6/ipu6-bus.h | 6 +- + drivers/media/pci/intel/ipu6/ipu6-buttress.c | 631 +++-- + drivers/media/pci/intel/ipu6/ipu6-buttress.h | 50 +- + drivers/media/pci/intel/ipu6/ipu6-cpd.c | 201 +- + drivers/media/pci/intel/ipu6/ipu6-cpd.h | 43 + + drivers/media/pci/intel/ipu6/ipu6-dma.c | 24 +- + drivers/media/pci/intel/ipu6/ipu6-dma.h | 2 + + drivers/media/pci/intel/ipu6/ipu6-fw-isys.c | 618 ++++- + drivers/media/pci/intel/ipu6/ipu6-fw-isys.h | 48 +- + drivers/media/pci/intel/ipu6/ipu6-isys-csi2.c | 587 +++- + drivers/media/pci/intel/ipu6/ipu6-isys-csi2.h | 18 +- + drivers/media/pci/intel/ipu6/ipu6-isys-queue.c | 215 +- + drivers/media/pci/intel/ipu6/ipu6-isys-queue.h | 6 +- + drivers/media/pci/intel/ipu6/ipu6-isys-subdev.c | 23 +- + drivers/media/pci/intel/ipu6/ipu6-isys-subdev.h | 3 + + drivers/media/pci/intel/ipu6/ipu6-isys-video.c | 772 ++---- + drivers/media/pci/intel/ipu6/ipu6-isys-video.h | 62 +- + drivers/media/pci/intel/ipu6/ipu6-isys.c | 603 ++-- + drivers/media/pci/intel/ipu6/ipu6-isys.h | 108 +- + drivers/media/pci/intel/ipu6/ipu6-mmu-hw.c | 296 ++ + drivers/media/pci/intel/ipu6/ipu6-mmu.c | 133 +- + drivers/media/pci/intel/ipu6/ipu6-mmu.h | 158 +- + .../pci/intel/ipu6/ipu6-platform-buttress-regs.h | 117 +- + drivers/media/pci/intel/ipu6/ipu6.c | 618 +++-- + drivers/media/pci/intel/ipu6/ipu6.h | 201 +- + drivers/media/pci/intel/ipu6/ipu7-boot.c | 405 +++ + drivers/media/pci/intel/ipu6/ipu7-boot.h | 46 + + drivers/media/pci/intel/ipu6/ipu7-fw-com.c | 74 + + drivers/media/pci/intel/ipu6/ipu7-fw-com.h | 53 + + drivers/media/pci/intel/ipu6/ipu7-fw-isys.c | 875 ++++++ + drivers/media/pci/intel/ipu6/ipu7-fw-isys.h | 370 +++ + drivers/media/pci/intel/ipu6/ipu7-isys-csi-phy.c | 1074 ++++++++ + drivers/media/pci/intel/ipu6/ipu7-isys-csi-phy.h | 16 + + drivers/media/pci/intel/ipu6/ipu7-isys-csi2-regs.h | 1191 ++++++++ + drivers/media/pci/intel/ipu6/ipu7-mmu-hw.c | 1116 ++++++++ + drivers/media/pci/intel/ipu6/ipu7-mmu-hw.h | 369 +++ + drivers/media/pci/intel/ipu6/ipu7-platform-regs.h | 32 + + drivers/media/pci/intel/ivsc/mei_ace.c | 4 +- + drivers/media/pci/intel/ivsc/mei_csi.c | 8 +- + drivers/media/pci/ivtv/ivtv-controls.c | 2 +- + drivers/media/pci/ivtv/ivtv-ioctl.c | 2 +- + drivers/media/pci/saa7134/saa7134-core.c | 8 +- + drivers/media/pci/saa7134/saa7134-empress.c | 4 +- + drivers/media/pci/saa7134/saa7134-input.c | 7 +- + drivers/media/pci/saa7146/mxb.c | 4 +- + drivers/media/pci/saa7164/saa7164-core.c | 32 +- + drivers/media/pci/saa7164/saa7164.h | 1 - + drivers/media/pci/tw68/tw68-core.c | 10 +- + drivers/media/pci/tw686x/tw686x-core.c | 10 + + drivers/media/platform/amd/isp4/isp4_subdev.c | 1 + + drivers/media/platform/amd/isp4/isp4_video.c | 2 +- + .../media/platform/amlogic/c3/isp/c3-isp-core.c | 1 + + .../media/platform/amlogic/c3/isp/c3-isp-resizer.c | 3 + + .../amlogic/c3/mipi-adapter/c3-mipi-adap.c | 1 + + .../platform/amlogic/c3/mipi-csi2/c3-mipi-csi2.c | 1 + + .../media/platform/arm/mali-c55/mali-c55-core.c | 32 +- + drivers/media/platform/arm/mali-c55/mali-c55-isp.c | 4 + + .../media/platform/arm/mali-c55/mali-c55-params.c | 3 + + .../media/platform/arm/mali-c55/mali-c55-resizer.c | 15 +- + .../media/platform/arm/mali-c55/mali-c55-stats.c | 18 +- + drivers/media/platform/arm/mali-c55/mali-c55-tpg.c | 1 + + drivers/media/platform/aspeed/aspeed-video.c | 12 +- + drivers/media/platform/atmel/atmel-isi.c | 11 +- + drivers/media/platform/broadcom/bcm2835-unicam.c | 1 + + drivers/media/platform/cadence/cdns-csi2rx.c | 5 + + drivers/media/platform/cadence/cdns-csi2tx.c | 1 + + drivers/media/platform/intel/pxa_camera.c | 6 +- + drivers/media/platform/m2m-deinterlace.c | 2 +- + drivers/media/platform/marvell/Kconfig | 1 + + drivers/media/platform/marvell/cafe-driver.c | 2 +- + drivers/media/platform/marvell/mcam-core.c | 24 +- + drivers/media/platform/marvell/mmp-driver.c | 5 +- + drivers/media/platform/mediatek/mdp/mtk_mdp_ipi.h | 2 +- + .../mediatek/vcodec/decoder/vdec_ipi_msg.h | 4 +- + .../media/platform/microchip/microchip-csi2dc.c | 2 + + .../media/platform/microchip/microchip-isc-base.c | 82 +- + .../media/platform/microchip/microchip-isc-clk.c | 7 +- + .../media/platform/microchip/microchip-isc-regs.h | 16 +- + .../platform/microchip/microchip-isc-scaler.c | 2 + + drivers/media/platform/microchip/microchip-isc.h | 5 +- + .../platform/microchip/microchip-sama5d2-isc.c | 48 +- + .../platform/microchip/microchip-sama7g5-isc.c | 48 +- + drivers/media/platform/nuvoton/npcm-video.c | 18 +- + drivers/media/platform/nxp/Kconfig | 16 + + drivers/media/platform/nxp/Makefile | 1 + + drivers/media/platform/nxp/imx-jpeg/mxc-jpeg.c | 23 +- + drivers/media/platform/nxp/imx-mipi-csis.c | 3 +- + drivers/media/platform/nxp/imx7-media-csi.c | 1 + + .../platform/nxp/imx8-isi/imx8-isi-crossbar.c | 1 + + .../media/platform/nxp/imx8-isi/imx8-isi-pipe.c | 3 + + drivers/media/platform/nxp/imx8mq-mipi-csi2.c | 1 + + drivers/media/platform/nxp/imx95-csi-formatter.c | 759 +++++ + drivers/media/platform/qcom/camss/Kconfig | 2 + + drivers/media/platform/qcom/camss/camss-csid.c | 3 +- + drivers/media/platform/qcom/camss/camss-csiphy.c | 203 +- + drivers/media/platform/qcom/camss/camss-csiphy.h | 12 +- + drivers/media/platform/qcom/camss/camss-ispif.c | 3 +- + drivers/media/platform/qcom/camss/camss-tpg.c | 3 +- + drivers/media/platform/qcom/camss/camss-vfe-17x.c | 46 +- + drivers/media/platform/qcom/camss/camss-vfe.c | 26 +- + drivers/media/platform/qcom/camss/camss.c | 367 ++- + drivers/media/platform/qcom/camss/camss.h | 3 + + drivers/media/platform/qcom/iris/Makefile | 2 + + drivers/media/platform/qcom/iris/iris_core.c | 6 + + drivers/media/platform/qcom/iris/iris_core.h | 6 + + drivers/media/platform/qcom/iris/iris_ctrls.c | 96 + + drivers/media/platform/qcom/iris/iris_ctrls.h | 1 + + drivers/media/platform/qcom/iris/iris_firmware.c | 121 +- + drivers/media/platform/qcom/iris/iris_hfi_common.c | 7 +- + drivers/media/platform/qcom/iris/iris_hfi_common.h | 1 + + drivers/media/platform/qcom/iris/iris_hfi_gen1.c | 406 ++- + .../platform/qcom/iris/iris_hfi_gen1_command.c | 30 + + .../platform/qcom/iris/iris_hfi_gen1_defines.h | 15 + + .../platform/qcom/iris/iris_hfi_gen1_response.c | 82 + + drivers/media/platform/qcom/iris/iris_hfi_gen2.c | 945 +++++++ + .../platform/qcom/iris/iris_hfi_gen2_defines.h | 2 + + .../platform/qcom/iris/iris_hfi_gen2_packet.c | 3 + + .../platform/qcom/iris/iris_hfi_gen2_response.c | 1 + + drivers/media/platform/qcom/iris/iris_instance.h | 18 +- + .../platform/qcom/iris/iris_platform_common.h | 32 + + .../media/platform/qcom/iris/iris_platform_vpu2.c | 13 +- + .../media/platform/qcom/iris/iris_platform_vpu3x.c | 25 +- + .../platform/qcom/iris/iris_platform_vpu_ar50lt.c | 118 + + drivers/media/platform/qcom/iris/iris_probe.c | 68 +- + drivers/media/platform/qcom/iris/iris_resources.c | 2 + + drivers/media/platform/qcom/iris/iris_utils.c | 6 + + drivers/media/platform/qcom/iris/iris_utils.h | 1 + + drivers/media/platform/qcom/iris/iris_vb2.c | 6 +- + drivers/media/platform/qcom/iris/iris_vdec.c | 44 +- + drivers/media/platform/qcom/iris/iris_venc.c | 33 +- + drivers/media/platform/qcom/iris/iris_vidc.c | 50 +- + drivers/media/platform/qcom/iris/iris_vpu2.c | 30 +- + drivers/media/platform/qcom/iris/iris_vpu3x.c | 6 + + drivers/media/platform/qcom/iris/iris_vpu4x.c | 2 + + drivers/media/platform/qcom/iris/iris_vpu_ar50lt.c | 130 + + drivers/media/platform/qcom/iris/iris_vpu_buffer.c | 370 +++ + drivers/media/platform/qcom/iris/iris_vpu_buffer.h | 38 + + drivers/media/platform/qcom/iris/iris_vpu_common.c | 52 +- + drivers/media/platform/qcom/iris/iris_vpu_common.h | 6 + + .../platform/qcom/iris/iris_vpu_register_defines.h | 1 - + drivers/media/platform/qcom/venus/core.c | 70 +- + drivers/media/platform/qcom/venus/core.h | 6 + + drivers/media/platform/qcom/venus/hfi_msgs.c | 3 + + drivers/media/platform/raspberrypi/rp1-cfe/csi2.c | 1 + + .../media/platform/raspberrypi/rp1-cfe/pisp-fe.c | 1 + + drivers/media/platform/renesas/rcar-csi2.c | 385 ++- + drivers/media/platform/renesas/rcar-isp/core-io.c | 40 +- + drivers/media/platform/renesas/rcar-isp/csisp.c | 228 +- + .../media/platform/renesas/rcar-vin/rcar-core.c | 27 +- + drivers/media/platform/renesas/rcar-vin/rcar-dma.c | 2 +- + .../media/platform/renesas/rcar-vin/rcar-v4l2.c | 70 +- + drivers/media/platform/renesas/rcar_drif.c | 2 +- + drivers/media/platform/renesas/renesas-ceu.c | 7 +- + .../media/platform/renesas/rzg2l-cru/rzg2l-core.c | 3 +- + .../media/platform/renesas/rzg2l-cru/rzg2l-cru.h | 2 +- + .../media/platform/renesas/rzg2l-cru/rzg2l-csi2.c | 3 +- + .../media/platform/renesas/rzg2l-cru/rzg2l-ip.c | 3 +- + .../media/platform/renesas/rzg2l-cru/rzg2l-video.c | 13 +- + .../platform/renesas/rzv2h-ivc/rzv2h-ivc-dev.c | 11 +- + .../platform/renesas/rzv2h-ivc/rzv2h-ivc-subdev.c | 1 + + drivers/media/platform/renesas/sh_vou.c | 6 +- + drivers/media/platform/renesas/vsp1/vsp1_brx.c | 3 + + drivers/media/platform/renesas/vsp1/vsp1_dl.c | 2 +- + drivers/media/platform/renesas/vsp1/vsp1_drm.c | 18 +- + drivers/media/platform/renesas/vsp1/vsp1_entity.c | 4 +- + drivers/media/platform/renesas/vsp1/vsp1_entity.h | 1 + + drivers/media/platform/renesas/vsp1/vsp1_histo.c | 3 + + drivers/media/platform/renesas/vsp1/vsp1_hsit.c | 1 + + drivers/media/platform/renesas/vsp1/vsp1_rwpf.c | 3 + + drivers/media/platform/renesas/vsp1/vsp1_sru.c | 1 + + drivers/media/platform/renesas/vsp1/vsp1_uds.c | 1 + + drivers/media/platform/renesas/vsp1/vsp1_uif.c | 2 + + drivers/media/platform/renesas/vsp1/vsp1_vspx.c | 3 +- + .../platform/rockchip/rkcif/rkcif-interface.c | 3 + + .../media/platform/rockchip/rkisp1/rkisp1-csi.c | 1 + + .../media/platform/rockchip/rkisp1/rkisp1-dev.c | 4 +- + .../media/platform/rockchip/rkisp1/rkisp1-isp.c | 3 + + .../platform/rockchip/rkisp1/rkisp1-resizer.c | 3 + + .../platform/samsung/exynos4-is/fimc-capture.c | 8 +- + .../media/platform/samsung/exynos4-is/fimc-isp.c | 1 + + .../media/platform/samsung/exynos4-is/fimc-lite.c | 6 +- + .../media/platform/samsung/exynos4-is/mipi-csis.c | 1 + + .../platform/samsung/s3c-camif/camif-capture.c | 3 + + .../media/platform/samsung/s3c-camif/camif-core.c | 2 +- + drivers/media/platform/st/stm32/stm32-csi.c | 6 +- + drivers/media/platform/st/stm32/stm32-dcmi.c | 31 +- + .../media/platform/st/stm32/stm32-dcmipp/Makefile | 3 +- + .../st/stm32/stm32-dcmipp/dcmipp-byteproc.c | 30 +- + .../{dcmipp-bytecap.c => dcmipp-capture.c} | 600 ++-- + .../platform/st/stm32/stm32-dcmipp/dcmipp-common.h | 99 +- + .../platform/st/stm32/stm32-dcmipp/dcmipp-core.c | 124 +- + .../platform/st/stm32/stm32-dcmipp/dcmipp-input.c | 127 +- + .../platform/st/stm32/stm32-dcmipp/dcmipp-isp.c | 493 ++++ + .../st/stm32/stm32-dcmipp/dcmipp-pixelcommon.c | 181 ++ + .../st/stm32/stm32-dcmipp/dcmipp-pixelcommon.h | 42 + + .../st/stm32/stm32-dcmipp/dcmipp-pixelproc.c | 942 +++++++ + .../media/platform/sunxi/sun4i-csi/sun4i_v4l2.c | 1 + + .../platform/sunxi/sun6i-csi/sun6i_csi_bridge.c | 156 +- + .../platform/sunxi/sun6i-csi/sun6i_csi_bridge.h | 9 - + .../platform/sunxi/sun6i-csi/sun6i_csi_capture.c | 27 +- + .../sunxi/sun6i-mipi-csi2/sun6i_mipi_csi2.c | 108 +- + .../sunxi/sun6i-mipi-csi2/sun6i_mipi_csi2.h | 2 - + .../sun8i-a83t-mipi-csi2/sun8i_a83t_mipi_csi2.c | 113 +- + .../sun8i-a83t-mipi-csi2/sun8i_a83t_mipi_csi2.h | 2 - + drivers/media/platform/synopsys/dw-mipi-csi2rx.c | 1 + + .../media/platform/synopsys/hdmirx/snps_hdmirx.c | 423 ++- + .../media/platform/synopsys/hdmirx/snps_hdmirx.h | 8 + + drivers/media/platform/ti/Kconfig | 11 - + drivers/media/platform/ti/am437x/am437x-vpfe.c | 2 +- + drivers/media/platform/ti/cal/cal-camerarx.c | 1 + + drivers/media/platform/ti/cal/cal-video.c | 19 +- + drivers/media/platform/ti/cal/cal.c | 10 +- + drivers/media/platform/ti/davinci/vpif_capture.c | 5 +- + drivers/media/platform/ti/davinci/vpif_display.c | 3 + + .../media/platform/ti/j721e-csi2rx/j721e-csi2rx.c | 25 + + drivers/media/platform/ti/omap3isp/ispccdc.c | 5 +- + drivers/media/platform/ti/omap3isp/ispccp2.c | 3 +- + drivers/media/platform/ti/omap3isp/ispcsi2.c | 3 +- + drivers/media/platform/ti/omap3isp/isppreview.c | 5 +- + drivers/media/platform/ti/omap3isp/ispresizer.c | 5 +- + drivers/media/platform/ti/omap3isp/ispvideo.c | 4 +- + drivers/media/platform/ti/vpe/vip.c | 4 +- + drivers/media/platform/via/via-camera.c | 4 +- + drivers/media/platform/video-mux.c | 1 + + drivers/media/platform/xilinx/xilinx-csi2rxss.c | 1 + + drivers/media/platform/xilinx/xilinx-tpg.c | 1 + + drivers/media/radio/si4713/radio-usb-si4713.c | 4 +- + drivers/media/radio/si4713/si4713.c | 2 +- + drivers/media/rc/bpf-lirc.c | 18 +- + drivers/media/rc/ene_ir.c | 4 +- + drivers/media/rc/fintek-cir.c | 17 +- + drivers/media/rc/fintek-cir.h | 22 - + drivers/media/rc/imon.c | 180 +- + drivers/media/rc/ir-hix5hd2.c | 5 +- + drivers/media/rc/ir-mce_kbd-decoder.c | 5 +- + drivers/media/rc/ir_toy.c | 4 +- + drivers/media/rc/ite-cir.c | 1 - + drivers/media/rc/ite-cir.h | 1 - + drivers/media/rc/lirc_dev.c | 5 + + drivers/media/rc/mceusb.c | 1 - + drivers/media/rc/meson-ir-tx.c | 20 +- + drivers/media/rc/nuvoton-cir.h | 3 - + drivers/media/rc/rc-ir-raw.c | 86 +- + drivers/media/rc/rc-loopback.c | 5 - + drivers/media/rc/rc-main.c | 206 +- + drivers/media/rc/redrat3.c | 32 +- + drivers/media/rc/serial_ir.c | 2 +- + drivers/media/rc/streamzap.c | 1 + + drivers/media/rc/sunxi-cir.c | 2 +- + drivers/media/spi/Kconfig | 4 - + drivers/media/test-drivers/vicodec/codec-fwht.h | 2 +- + drivers/media/test-drivers/vicodec/vicodec-core.c | 14 +- + drivers/media/test-drivers/vidtv/vidtv_bridge.c | 75 +- + drivers/media/test-drivers/vidtv/vidtv_demod.c | 9 - + drivers/media/test-drivers/vim2m.c | 33 +- + drivers/media/test-drivers/vimc/vimc-debayer.c | 1 + + drivers/media/test-drivers/vimc/vimc-scaler.c | 3 + + drivers/media/test-drivers/vimc/vimc-sensor.c | 9 +- + drivers/media/test-drivers/vivid/vivid-cec.c | 10 +- + drivers/media/test-drivers/vivid/vivid-core.c | 1 + + drivers/media/test-drivers/vivid/vivid-vid-cap.c | 13 + + drivers/media/tuners/tda18250.c | 3 +- + drivers/media/usb/airspy/airspy.c | 6 +- + drivers/media/usb/au0828/au0828-core.c | 4 + + drivers/media/usb/au0828/au0828-dvb.c | 20 +- + drivers/media/usb/cx231xx/cx231xx-417.c | 2 +- + drivers/media/usb/cx231xx/cx231xx-audio.c | 15 +- + drivers/media/usb/cx231xx/cx231xx-avcore.c | 103 - + drivers/media/usb/cx231xx/cx231xx-cards.c | 2 +- + drivers/media/usb/cx231xx/cx231xx-video.c | 4 +- + drivers/media/usb/cx231xx/cx231xx.h | 2 +- + drivers/media/usb/dvb-usb-v2/mxl111sf-i2c.c | 4 +- + drivers/media/usb/dvb-usb-v2/rtl28xxu.c | 14 +- + drivers/media/usb/dvb-usb/cxusb-analog.c | 6 +- + drivers/media/usb/dvb-usb/dib0700_core.c | 2 +- + drivers/media/usb/dvb-usb/dvb-usb-firmware.c | 2 + + drivers/media/usb/em28xx/em28xx-audio.c | 7 + + drivers/media/usb/em28xx/em28xx-camera.c | 2 +- + drivers/media/usb/em28xx/em28xx-cards.c | 38 + + drivers/media/usb/em28xx/em28xx-core.c | 3 - + drivers/media/usb/em28xx/em28xx-video.c | 46 +- + drivers/media/usb/em28xx/em28xx.h | 2 + + drivers/media/usb/go7007/go7007-driver.c | 5 +- + drivers/media/usb/go7007/go7007-usb.c | 8 + + drivers/media/usb/go7007/go7007-v4l2.c | 2 +- + drivers/media/usb/go7007/s2250-board.c | 1 + + drivers/media/usb/gspca/gspca.c | 3 +- + drivers/media/usb/gspca/ov519.c | 2 +- + drivers/media/usb/gspca/w996Xcf.c | 2 +- + drivers/media/usb/hackrf/hackrf.c | 7 +- + drivers/media/usb/pvrusb2/pvrusb2-hdw.c | 14 +- + drivers/media/usb/pvrusb2/pvrusb2-sysfs.c | 1 + + drivers/media/usb/usbtv/usbtv-core.c | 2 - + drivers/media/usb/usbtv/usbtv-video.c | 4 + + drivers/media/usb/uvc/uvc_ctrl.c | 117 +- + drivers/media/usb/uvc/uvc_driver.c | 45 +- + drivers/media/usb/uvc/uvc_status.c | 2 +- + drivers/media/usb/uvc/uvc_video.c | 4 +- + drivers/media/usb/uvc/uvcvideo.h | 4 +- + drivers/media/v4l2-core/v4l2-common.c | 21 +- + drivers/media/v4l2-core/v4l2-ctrls-core.c | 19 +- + drivers/media/v4l2-core/v4l2-isp.c | 15 +- + drivers/media/v4l2-core/v4l2-mc.c | 5 +- + drivers/media/v4l2-core/v4l2-subdev.c | 163 +- + drivers/phy/phy-core.c | 166 +- + drivers/phy/qualcomm/Kconfig | 15 + + drivers/phy/qualcomm/Makefile | 5 + + drivers/phy/qualcomm/phy-qcom-mipi-csi2-3ph-dphy.c | 385 +++ + drivers/phy/qualcomm/phy-qcom-mipi-csi2-core.c | 463 ++++ + drivers/phy/qualcomm/phy-qcom-mipi-csi2.h | 97 + + drivers/platform/x86/intel/int3472/discrete.c | 82 +- + drivers/platform/x86/intel/int3472/tps68470.c | 2 +- + drivers/staging/media/atomisp/i2c/atomisp-gc2235.c | 1 + + drivers/staging/media/atomisp/i2c/atomisp-ov2722.c | 1 + + drivers/staging/media/atomisp/pci/atomisp_cmd.c | 16 +- + drivers/staging/media/atomisp/pci/atomisp_csi2.c | 1 + + drivers/staging/media/atomisp/pci/atomisp_subdev.c | 3 + + drivers/staging/media/atomisp/pci/atomisp_v4l2.c | 8 +- + .../pci/isp/kernels/s3a/s3a_1.0/ia_css_s3a_types.h | 4 +- + drivers/staging/media/av7110/av7110.c | 2 +- + drivers/staging/media/av7110/av7110_ir.c | 2 - + drivers/staging/media/av7110/sp8870.c | 9 +- + drivers/staging/media/imx/imx-ic-prp.c | 2 +- + drivers/staging/media/imx/imx-ic-prpencvf.c | 4 +- + drivers/staging/media/imx/imx-media-capture.c | 1 - + drivers/staging/media/imx/imx-media-csc-scaler.c | 3 +- + drivers/staging/media/imx/imx-media-csi.c | 3 + + drivers/staging/media/imx/imx-media-dev.c | 1 - + drivers/staging/media/imx/imx-media-vdic.c | 1 + + drivers/staging/media/imx/imx6-mipi-csi2.c | 1 + + drivers/staging/media/ipu3/ipu3-css.c | 1 - + drivers/staging/media/ipu3/ipu3-v4l2.c | 3 + + drivers/staging/media/ipu7/TODO | 28 +- + drivers/staging/media/ipu7/ipu7-isys-csi2.c | 2 + + drivers/staging/media/ipu7/ipu7-isys-subdev.c | 1 + + drivers/staging/media/ipu7/ipu7-isys-subdev.h | 1 + + drivers/staging/media/ipu7/ipu7-isys.c | 1 + + drivers/staging/media/ipu7/ipu7.c | 7 + + drivers/staging/media/max96712/max96712.c | 1 - + .../staging/media/sunxi/sun6i-isp/sun6i_isp_proc.c | 1 + + drivers/staging/media/tegra-video/csi.c | 1 + + drivers/staging/media/tegra-video/tegra20.c | 7 +- + drivers/staging/media/tegra-video/vi.c | 113 +- + .../dt-bindings/media/video-interface-devices.h | 13 + + include/linux/phy/phy.h | 13 + + include/linux/platform_data/x86/int3472.h | 2 + + include/linux/property.h | 5 + + include/media/ipu-bridge.h | 52 +- + include/media/ipu6-pci-table.h | 6 + + include/media/rc-map.h | 2 - + include/media/v4l2-common.h | 75 +- + include/media/v4l2-ctrls.h | 6 + + include/media/v4l2-dv-timings.h | 14 + + include/media/v4l2-subdev.h | 34 +- + include/uapi/linux/it6625.h | 25 + + include/uapi/linux/media/arm/mali-c55-config.h | 2 + + include/uapi/linux/media/st/dcmipp_config.h | 16 + + include/uapi/linux/v4l2-controls.h | 21 + + 584 files changed, 30643 insertions(+), 5067 deletions(-) + create mode 100644 Documentation/devicetree/bindings/media/fsl,imx95-csi-formatter.yaml + create mode 100644 Documentation/devicetree/bindings/media/i2c/himax,hm1246.yaml + create mode 100644 Documentation/devicetree/bindings/media/i2c/ite,it6625.yaml + create mode 100644 Documentation/devicetree/bindings/media/i2c/ovti,os02g10.yaml + create mode 100644 Documentation/devicetree/bindings/media/i2c/samsung,s5kjn5.yaml + delete mode 100644 Documentation/devicetree/bindings/media/i2c/toshiba,tc358743.txt + create mode 100644 Documentation/devicetree/bindings/media/i2c/toshiba,tc358743.yaml + create mode 100644 Documentation/devicetree/bindings/media/qcom,glymur-camss.yaml + create mode 100644 Documentation/devicetree/bindings/phy/qcom,x1e80100-csi2-phy.yaml + create mode 100644 Documentation/userspace-api/media/drivers/dcmipp.rst + create mode 100644 drivers/media/i2c/hm1246.c + create mode 100644 drivers/media/i2c/it6625.c + create mode 100644 drivers/media/i2c/os02g10.c + create mode 100644 drivers/media/i2c/ov05c10.c + create mode 100644 drivers/media/i2c/s5kjn5.c + create mode 100644 drivers/media/pci/intel/ipu6/ipu6-mmu-hw.c + create mode 100644 drivers/media/pci/intel/ipu6/ipu7-boot.c + create mode 100644 drivers/media/pci/intel/ipu6/ipu7-boot.h + create mode 100644 drivers/media/pci/intel/ipu6/ipu7-fw-com.c + create mode 100644 drivers/media/pci/intel/ipu6/ipu7-fw-com.h + create mode 100644 drivers/media/pci/intel/ipu6/ipu7-fw-isys.c + create mode 100644 drivers/media/pci/intel/ipu6/ipu7-fw-isys.h + create mode 100644 drivers/media/pci/intel/ipu6/ipu7-isys-csi-phy.c + create mode 100644 drivers/media/pci/intel/ipu6/ipu7-isys-csi-phy.h + create mode 100644 drivers/media/pci/intel/ipu6/ipu7-isys-csi2-regs.h + create mode 100644 drivers/media/pci/intel/ipu6/ipu7-mmu-hw.c + create mode 100644 drivers/media/pci/intel/ipu6/ipu7-mmu-hw.h + create mode 100644 drivers/media/pci/intel/ipu6/ipu7-platform-regs.h + create mode 100644 drivers/media/platform/nxp/imx95-csi-formatter.c + create mode 100644 drivers/media/platform/qcom/iris/iris_platform_vpu_ar50lt.c + create mode 100644 drivers/media/platform/qcom/iris/iris_vpu_ar50lt.c + rename drivers/media/platform/st/stm32/stm32-dcmipp/{dcmipp-bytecap.c => dcmipp-capture.c} (53%) + create mode 100644 drivers/media/platform/st/stm32/stm32-dcmipp/dcmipp-isp.c + create mode 100644 drivers/media/platform/st/stm32/stm32-dcmipp/dcmipp-pixelcommon.c + create mode 100644 drivers/media/platform/st/stm32/stm32-dcmipp/dcmipp-pixelcommon.h + create mode 100644 drivers/media/platform/st/stm32/stm32-dcmipp/dcmipp-pixelproc.c + create mode 100644 drivers/phy/qualcomm/phy-qcom-mipi-csi2-3ph-dphy.c + create mode 100644 drivers/phy/qualcomm/phy-qcom-mipi-csi2-core.c + create mode 100644 drivers/phy/qualcomm/phy-qcom-mipi-csi2.h + create mode 100644 include/dt-bindings/media/video-interface-devices.h + create mode 100644 include/uapi/linux/it6625.h + create mode 100644 include/uapi/linux/media/st/dcmipp_config.h +Merging v4l-dvb-next/master (adc218676eef2 Linux 6.12) +$ git merge -m Merge branch 'master' of git://linuxtv.org/mchehab/media-next.git v4l-dvb-next/master +Already up to date. +Merging pm/linux-next (1729fbce1c728 Merge branch 'acpi-driver' into linux-next) +$ git merge -m Merge branch 'linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/rafael/linux-pm.git pm/linux-next +Auto-merging Documentation/admin-guide/kernel-parameters.txt +Auto-merging arch/arm64/kernel/topology.c +Auto-merging sound/soc/intel/boards/bytcr_rt5651.c +Merge made by the 'ort' strategy. + Documentation/ABI/testing/sysfs-devices-system-cpu | 14 +- + Documentation/admin-guide/kernel-parameters.txt | 7 - + Documentation/admin-guide/pm/amd-pstate.rst | 6 - + Documentation/admin-guide/pm/cpuidle.rst | 2 +- + .../driver-api/thermal/cpu-idle-cooling.rst | 2 +- + Documentation/firmware-guide/acpi/apei/einj.rst | 6 +- + Documentation/power/runtime_pm.rst | 477 +---- + Documentation/power/userland-swsusp.rst | 4 +- + arch/arm64/kernel/topology.c | 14 +- + arch/x86/kernel/acpi/cppc.c | 14 + + arch/x86/power/hibernate.c | 51 +- + drivers/acpi/acpi_extlog.c | 52 +- + drivers/acpi/acpi_mrrm.c | 14 +- + drivers/acpi/acpi_pcc.c | 55 +- + drivers/acpi/acpi_video.c | 43 +- + drivers/acpi/acpica/dsfield.c | 3 +- + drivers/acpi/acpica/exfield.c | 11 +- + drivers/acpi/apei/ghes.c | 79 +- + drivers/acpi/apei/ghes_helpers.c | 18 +- + drivers/acpi/arm64/amba.c | 3 +- + drivers/acpi/battery.c | 131 +- + drivers/acpi/bus.c | 28 +- + drivers/acpi/button.c | 12 + + drivers/acpi/cppc_acpi.c | 2223 +++++++++++++++++--- + drivers/acpi/device_pm.c | 77 +- + drivers/acpi/fan.h | 2 +- + drivers/acpi/fan_core.c | 147 +- + drivers/acpi/fan_hwmon.c | 2 +- + drivers/acpi/glue.c | 145 +- + drivers/acpi/internal.h | 1 + + drivers/acpi/numa/hmat.c | 2 +- + drivers/acpi/numa/srat.c | 2 +- + drivers/acpi/osl.c | 2 +- + drivers/acpi/pfr_update.c | 2 +- + drivers/acpi/power.c | 13 + + drivers/acpi/processor_driver.c | 6 +- + drivers/acpi/processor_thermal.c | 80 +- + drivers/acpi/riscv/cppc.c | 26 +- + drivers/acpi/sbs.c | 8 +- + drivers/acpi/scan.c | 54 +- + drivers/acpi/sysfs.c | 5 +- + drivers/acpi/tables.c | 3 + + drivers/acpi/thermal.c | 19 +- + drivers/acpi/utils.c | 13 +- + drivers/acpi/video_detect.c | 8 + + drivers/acpi/x86/s2idle.c | 29 + + drivers/base/base.h | 2 + + drivers/base/bus.c | 67 +- + drivers/base/power/clock_ops.c | 2 +- + drivers/base/power/main.c | 32 +- + drivers/base/power/runtime-test.c | 57 + + drivers/base/power/runtime.c | 109 +- + drivers/clocksource/timer-ti-dm.c | 9 +- + drivers/cpufreq/amd-pstate-ut.c | 23 +- + drivers/cpufreq/amd-pstate.c | 257 ++- + drivers/cpufreq/amd-pstate.h | 16 + + drivers/cpufreq/cppc_cpufreq.c | 73 +- + drivers/cpufreq/cpufreq_conservative.c | 8 +- + drivers/cpufreq/cpufreq_governor.c | 29 +- + drivers/cpufreq/cpufreq_governor.h | 2 + + drivers/cpuidle/cpuidle-tegra.c | 2 +- + drivers/cpuidle/governors/menu.c | 8 +- + drivers/cpuidle/governors/teo.c | 11 + + drivers/cxl/core/ras.c | 3 +- + drivers/cxl/core/ras_rch.c | 27 +- + drivers/firmware/efi/cper.c | 51 +- + drivers/firmware/efi/dev-path-parser.c | 4 +- + drivers/idle/intel_idle.c | 12 +- + drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.c | 2 +- + drivers/pci/pcie/tlp.c | 88 + + drivers/platform/x86/lenovo/yogabook.c | 4 +- + drivers/platform/x86/serdev_helpers.h | 3 +- + drivers/platform/x86/x86-android-tablets/core.c | 4 +- + drivers/pnp/driver.c | 4 +- + drivers/powercap/intel_rapl_msr.c | 5 +- + drivers/thermal/cpufreq_cooling.c | 14 +- + drivers/thermal/devfreq_cooling.c | 4 +- + drivers/thermal/gov_power_allocator.c | 7 +- + .../int340x_thermal/processor_thermal_device.c | 14 +- + .../int340x_thermal/processor_thermal_soc_slider.c | 28 +- + drivers/thermal/intel/intel_powerclamp.c | 64 +- + drivers/thermal/intel/intel_tcc_cooling.c | 1 + + drivers/thermal/qcom/qcom-spmi-adc-tm5.c | 4 +- + drivers/thermal/thermal_core.c | 30 +- + drivers/thermal/thermal_core.h | 3 +- + drivers/thermal/thermal_of.c | 17 +- + drivers/thermal/ti-soc-thermal/ti-bandgap.c | 12 +- + drivers/thunderbolt/acpi.c | 5 +- + include/acpi/acpi_bus.h | 43 +- + include/acpi/cppc_acpi.h | 18 +- + include/acpi/ghes.h | 4 + + include/acpi/processor.h | 6 +- + include/cxl/event.h | 6 +- + include/linux/acpi.h | 37 +- + include/linux/aer.h | 9 + + include/linux/apple-gmux.h | 2 +- + include/linux/device/bus.h | 1 + + include/linux/pm.h | 93 + + include/linux/pm_runtime.h | 409 ++-- + include/linux/thermal.h | 19 +- + include/uapi/linux/thermal.h | 8 +- + kernel/cpu_pm.c | 9 +- + kernel/power/em_netlink.c | 126 +- + kernel/power/em_netlink.h | 8 +- + kernel/power/energy_model.c | 8 +- + sound/hda/codecs/side-codecs/aw88399_hda.c | 3 +- + sound/hda/codecs/side-codecs/cs35l41_hda.c | 3 +- + sound/hda/codecs/side-codecs/tas2781_hda_i2c.c | 3 +- + sound/hda/codecs/side-codecs/tas2781_hda_spi.c | 3 +- + sound/soc/amd/acp-es8336.c | 2 +- + sound/soc/amd/acp/acp3x-es83xx/acp3x-es83xx.c | 2 +- + sound/soc/intel/boards/bytcht_es8316.c | 4 +- + sound/soc/intel/boards/bytcr_rt5640.c | 4 +- + sound/soc/intel/boards/bytcr_rt5651.c | 4 +- + sound/soc/intel/boards/cht_bsw_rt5645.c | 5 +- + sound/soc/intel/boards/sof_cirrus_common.c | 2 +- + sound/soc/intel/boards/sof_es8336.c | 4 +- + sound/soc/loongson/loongson_card.c | 3 +- + tools/power/pm-graph/sleepgraph.py | 2 +- + 119 files changed, 4031 insertions(+), 1851 deletions(-) +Merging cpufreq-arm/cpufreq/arm/linux-next (d82d896f00e7b cpufreq: Use %pe to print error pointers symbolically) +$ git merge -m Merge branch 'cpufreq/arm/linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/vireshk/pm.git cpufreq-arm/cpufreq/arm/linux-next +Auto-merging drivers/cpufreq/cppc_cpufreq.c +CONFLICT (content): Merge conflict in drivers/cpufreq/cppc_cpufreq.c +Resolved 'drivers/cpufreq/cppc_cpufreq.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 3a8d75b0bb0e0] Merge branch 'cpufreq/arm/linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/vireshk/pm.git +$ git diff -M --stat --summary HEAD^.. + drivers/cpufreq/airoha-cpufreq.c | 2 +- + drivers/cpufreq/bmips-cpufreq.c | 4 ++-- + drivers/cpufreq/cppc_cpufreq.c | 4 ++-- + drivers/cpufreq/qoriq-cpufreq.c | 3 +-- + drivers/cpufreq/rcpufreq_dt.rs | 1 + + drivers/cpufreq/s3c64xx-cpufreq.c | 5 ++--- + drivers/cpufreq/sparc-us2e-cpufreq.c | 11 +++++------ + drivers/cpufreq/sti-cpufreq.c | 4 ++-- + drivers/cpufreq/tegra194-cpufreq.c | 2 +- + rust/kernel/cpufreq.rs | 18 ++++++++++++++---- + 10 files changed, 31 insertions(+), 23 deletions(-) +Merging cpupower/cpupower (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'cpupower' of https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux.git cpupower/cpupower +Already up to date. +Merging devfreq/devfreq-next (9a222650d9e70 PM / devfreq: Fix governor_store() failing when device has no current governor) +$ git merge -m Merge branch 'devfreq-next' of https://git.kernel.org/pub/scm/linux/kernel/git/chanwoo/linux.git devfreq/devfreq-next +Auto-merging drivers/devfreq/event/rockchip-dfi.c +Merge made by the 'ort' strategy. + drivers/devfreq/devfreq.c | 50 +++++++----------------------------- + drivers/devfreq/event/rockchip-dfi.c | 4 ++- + 2 files changed, 12 insertions(+), 42 deletions(-) +Merging pmdomain/next (d1a93cf1d3bed pmdomain: Merge branch dt into next) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/ulfh/linux-pm.git pmdomain/next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + .../devicetree/bindings/power/qcom,rpmpd.yaml | 1 + + MAINTAINERS | 8 + + drivers/cpuidle/cpuidle-psci-domain.c | 7 +- + drivers/cpuidle/cpuidle-psci.c | 2 +- + drivers/pmdomain/Kconfig | 1 + + drivers/pmdomain/Makefile | 1 + + drivers/pmdomain/core.c | 163 ++++++- + drivers/pmdomain/core.h | 18 + + drivers/pmdomain/governor.c | 45 ++ + drivers/pmdomain/imx/gpcv2.c | 16 +- + drivers/pmdomain/imx/imx8m-blk-ctrl.c | 15 +- + drivers/pmdomain/imx/imx93-blk-ctrl.c | 3 +- + drivers/pmdomain/imx/scu-pd.c | 10 +- + drivers/pmdomain/qcom/rpmhpd.c | 41 ++ + drivers/pmdomain/renesas/Kconfig | 4 + + drivers/pmdomain/renesas/Makefile | 1 + + drivers/pmdomain/renesas/r8a774a3-sysc.c | 45 ++ + drivers/pmdomain/renesas/r8a7795-sysc.c | 2 +- + drivers/pmdomain/renesas/r8a779a0-sysc.c | 2 +- + drivers/pmdomain/renesas/r8a779f0-sysc.c | 2 +- + drivers/pmdomain/renesas/r8a779g0-sysc.c | 2 +- + drivers/pmdomain/renesas/r8a779h0-sysc.c | 2 +- + drivers/pmdomain/renesas/r8a78000-mdlc.c | 6 +- + drivers/pmdomain/renesas/rcar-sysc.c | 3 + + drivers/pmdomain/renesas/rcar-sysc.h | 13 +- + drivers/pmdomain/riscv/Kconfig | 15 + + drivers/pmdomain/riscv/Makefile | 3 + + drivers/pmdomain/riscv/riscv-rpmi-device-power.c | 485 +++++++++++++++++++++ + drivers/pmdomain/rockchip/pm-domains.c | 4 +- + include/linux/mailbox/riscv-rpmi-message.h | 11 + + include/linux/pm_domain.h | 9 + + include/linux/pm_qos.h | 9 + + kernel/power/qos.c | 4 +- + 33 files changed, 886 insertions(+), 67 deletions(-) + create mode 100644 drivers/pmdomain/core.h + create mode 100644 drivers/pmdomain/renesas/r8a774a3-sysc.c + create mode 100644 drivers/pmdomain/riscv/Kconfig + create mode 100644 drivers/pmdomain/riscv/Makefile + create mode 100644 drivers/pmdomain/riscv/riscv-rpmi-device-power.c +Merging opp/opp/linux-next (706081c6b9327 OPP: Fix the return value in dev_pm_opp_get_of_node() kernel-doc) +$ git merge -m Merge branch 'opp/linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/vireshk/pm.git opp/opp/linux-next +Merge made by the 'ort' strategy. + drivers/opp/of.c | 2 +- + 1 file changed, 1 insertion(+), 1 deletion(-) +Merging thermal/thermal/linux-next (e856ca3013b30 thermal/drivers/renesas/rzg3s: Add RZ/G3L TSU support) +$ git merge -m Merge branch 'thermal/linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/thermal/linux.git thermal/thermal/linux-next +Merge made by the 'ort' strategy. + .../bindings/thermal/qcom,pm8775-mbg-tm.yaml | 7 +- + .../bindings/thermal/renesas,r9a08g045-tsu.yaml | 4 +- + .../bindings/thermal/ti,omap-bandgap.yaml | 177 +++++++++++++++++++++ + .../devicetree/bindings/thermal/ti_soc_thermal.txt | 88 ---------- + drivers/thermal/airoha_thermal.c | 24 +-- + drivers/thermal/imx91_thermal.c | 2 +- + drivers/thermal/qcom/tsens-v2.c | 4 +- + drivers/thermal/renesas/rzg3e_thermal.c | 35 ++-- + drivers/thermal/renesas/rzg3s_thermal.c | 38 +++-- + tools/thermal/tmon/sysfs.c | 2 +- + tools/thermal/tmon/tmon.h | 2 +- + tools/thermal/tmon/tui.c | 2 +- + 12 files changed, 258 insertions(+), 127 deletions(-) + create mode 100644 Documentation/devicetree/bindings/thermal/ti,omap-bandgap.yaml + delete mode 100644 Documentation/devicetree/bindings/thermal/ti_soc_thermal.txt +Merging rdma/for-next (485effd117d03 RDMA/ionic: Fix double removal of CMB mmap entries in create QP error path) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/rdma/rdma.git rdma/for-next +Auto-merging MAINTAINERS +Auto-merging drivers/infiniband/core/nldev.c +Auto-merging drivers/infiniband/core/verbs.c +Auto-merging drivers/infiniband/hw/bnxt_re/main.c +Auto-merging drivers/infiniband/hw/erdma/erdma_main.c +Auto-merging drivers/infiniband/hw/erdma/erdma_verbs.c +Auto-merging drivers/infiniband/hw/hns/hns_roce_debugfs.c +Auto-merging drivers/infiniband/hw/irdma/verbs.c +Auto-merging drivers/infiniband/hw/mlx5/main.c +Auto-merging drivers/infiniband/sw/rxe/rxe_mr.c +Auto-merging drivers/infiniband/sw/rxe/rxe_odp.c +Auto-merging drivers/infiniband/sw/rxe/rxe_verbs.c +CONFLICT (content): Merge conflict in drivers/infiniband/sw/rxe/rxe_verbs.c +Auto-merging drivers/infiniband/ulp/isert/ib_isert.c +Auto-merging drivers/infiniband/ulp/rtrs/rtrs-clt.c +Auto-merging drivers/net/ethernet/microsoft/mana/gdma_main.c +Resolved 'drivers/infiniband/sw/rxe/rxe_verbs.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 9805f3548d387] Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/rdma/rdma.git +$ git diff -M --stat --summary HEAD^.. + Documentation/ABI/stable/sysfs-class-infiniband | 118 ----- + MAINTAINERS | 2 +- + drivers/infiniband/core/cma.c | 34 +- + drivers/infiniband/core/cma_trace.h | 1 + + drivers/infiniband/core/device.c | 4 +- + drivers/infiniband/core/iwcm.c | 2 +- + drivers/infiniband/core/lag.c | 2 +- + drivers/infiniband/core/mad_rmpp.c | 7 +- + drivers/infiniband/core/multicast.c | 7 +- + drivers/infiniband/core/nldev.c | 9 +- + drivers/infiniband/core/restrack.c | 3 +- + drivers/infiniband/core/umem.c | 77 ++- + drivers/infiniband/core/uverbs_std_types_device.c | 1 + + drivers/infiniband/core/uverbs_std_types_wq.c | 2 +- + drivers/infiniband/core/uverbs_uapi.c | 2 +- + drivers/infiniband/core/verbs.c | 2 +- + drivers/infiniband/hw/bnxt_re/bnxt_re.h | 4 + + drivers/infiniband/hw/bnxt_re/hw_counters.c | 6 +- + drivers/infiniband/hw/bnxt_re/ib_verbs.c | 69 ++- + drivers/infiniband/hw/bnxt_re/main.c | 26 +- + drivers/infiniband/hw/bnxt_re/qplib_res.c | 9 +- + drivers/infiniband/hw/bnxt_re/qplib_sp.c | 2 +- + drivers/infiniband/hw/cxgb4/cm.c | 5 +- + drivers/infiniband/hw/cxgb4/cq.c | 2 +- + drivers/infiniband/hw/efa/efa_admin_cmds_defs.h | 5 +- + drivers/infiniband/hw/efa/efa_com_cmd.c | 9 +- + drivers/infiniband/hw/efa/efa_com_cmd.h | 8 +- + drivers/infiniband/hw/efa/efa_verbs.c | 5 +- + drivers/infiniband/hw/erdma/erdma_cq.c | 18 +- + drivers/infiniband/hw/erdma/erdma_main.c | 2 +- + drivers/infiniband/hw/erdma/erdma_qp.c | 38 +- + drivers/infiniband/hw/erdma/erdma_verbs.c | 583 ++++++++++++---------- + drivers/infiniband/hw/erdma/erdma_verbs.h | 66 ++- + drivers/infiniband/hw/hfi1/affinity.c | 2 +- + drivers/infiniband/hw/hfi1/firmware.c | 2 +- + drivers/infiniband/hw/hfi1/pcie.c | 14 +- + drivers/infiniband/hw/hns/hns_roce_bond.c | 4 +- + drivers/infiniband/hw/hns/hns_roce_cq.c | 3 +- + drivers/infiniband/hw/hns/hns_roce_debugfs.c | 55 ++ + drivers/infiniband/hw/hns/hns_roce_debugfs.h | 1 + + drivers/infiniband/hw/hns/hns_roce_device.h | 6 +- + drivers/infiniband/hw/hns/hns_roce_hw_v2.c | 63 ++- + drivers/infiniband/hw/hns/hns_roce_mr.c | 22 +- + drivers/infiniband/hw/hns/hns_roce_qp.c | 2 +- + drivers/infiniband/hw/hns/hns_roce_srq.c | 4 +- + drivers/infiniband/hw/ionic/ionic_controlpath.c | 41 +- + drivers/infiniband/hw/ionic/ionic_fw.h | 2 + + drivers/infiniband/hw/ionic/ionic_ibdev.c | 4 +- + drivers/infiniband/hw/ionic/ionic_lif_cfg.c | 1 + + drivers/infiniband/hw/ionic/ionic_lif_cfg.h | 1 + + drivers/infiniband/hw/irdma/ctrl.c | 23 +- + drivers/infiniband/hw/irdma/hw.c | 15 +- + drivers/infiniband/hw/irdma/icrdma_if.c | 9 +- + drivers/infiniband/hw/irdma/main.c | 2 + + drivers/infiniband/hw/irdma/pble.c | 8 +- + drivers/infiniband/hw/irdma/protos.h | 3 +- + drivers/infiniband/hw/irdma/type.h | 7 +- + drivers/infiniband/hw/irdma/verbs.c | 7 +- + drivers/infiniband/hw/mana/cq.c | 396 +++++++++++---- + drivers/infiniband/hw/mana/device.c | 3 +- + drivers/infiniband/hw/mana/main.c | 43 +- + drivers/infiniband/hw/mana/mana_ib.h | 134 ++++- + drivers/infiniband/hw/mana/qp.c | 150 ++++-- + drivers/infiniband/hw/mana/shadow_queue.h | 61 +-- + drivers/infiniband/hw/mana/wq.c | 3 +- + drivers/infiniband/hw/mana/wr.c | 170 ++++--- + drivers/infiniband/hw/mlx4/cq.c | 17 +- + drivers/infiniband/hw/mlx4/mcg.c | 4 +- + drivers/infiniband/hw/mlx5/cq.c | 9 +- + drivers/infiniband/hw/mlx5/data_direct.c | 10 +- + drivers/infiniband/hw/mlx5/fs.c | 45 +- + drivers/infiniband/hw/mlx5/main.c | 54 +- + drivers/infiniband/hw/mlx5/odp.c | 2 +- + drivers/infiniband/hw/mlx5/qp.c | 32 +- + drivers/infiniband/hw/mlx5/qpc.c | 20 +- + drivers/infiniband/hw/mlx5/srq.c | 3 +- + drivers/infiniband/hw/mlx5/wr.c | 1 + + drivers/infiniband/hw/mthca/mthca_cq.c | 3 +- + drivers/infiniband/hw/ocrdma/ocrdma_verbs.c | 6 +- + drivers/infiniband/hw/qedr/verbs.c | 25 +- + drivers/infiniband/hw/usnic/usnic_abi.h | 2 +- + drivers/infiniband/hw/usnic/usnic_ib_verbs.c | 2 +- + drivers/infiniband/hw/usnic/usnic_transport.h | 2 +- + drivers/infiniband/hw/vmw_pvrdma/pvrdma_qp.c | 6 +- + drivers/infiniband/hw/vmw_pvrdma/pvrdma_srq.c | 3 +- + drivers/infiniband/sw/rdmavt/qp.c | 2 +- + drivers/infiniband/sw/rdmavt/srq.c | 2 +- + drivers/infiniband/sw/rxe/rxe_loc.h | 1 + + drivers/infiniband/sw/rxe/rxe_mr.c | 6 + + drivers/infiniband/sw/rxe/rxe_net.c | 71 ++- + drivers/infiniband/sw/rxe/rxe_odp.c | 4 +- + drivers/infiniband/sw/rxe/rxe_req.c | 2 +- + drivers/infiniband/sw/rxe/rxe_resp.c | 6 +- + drivers/infiniband/sw/rxe/rxe_verbs.c | 6 + + drivers/infiniband/sw/siw/siw_main.c | 5 +- + drivers/infiniband/sw/siw/siw_qp_tx.c | 1 - + drivers/infiniband/ulp/ipoib/ipoib_vlan.c | 2 +- + drivers/infiniband/ulp/iser/iscsi_iser.c | 6 +- + drivers/infiniband/ulp/iser/iser_verbs.c | 2 +- + drivers/infiniband/ulp/isert/ib_isert.c | 2 +- + drivers/infiniband/ulp/rtrs/rtrs-clt.c | 27 +- + drivers/infiniband/ulp/srpt/ib_srpt.c | 2 +- + drivers/infiniband/ulp/srpt/ib_srpt.h | 2 +- + drivers/net/ethernet/microsoft/mana/gdma_main.c | 51 +- + include/net/mana/gdma.h | 12 +- + include/rdma/ib_umem.h | 2 + + include/rdma/ib_verbs.h | 4 +- + include/rdma/uverbs_ioctl.h | 2 +- + include/uapi/rdma/ionic-abi.h | 9 +- + include/uapi/rdma/mana-abi.h | 12 + + 110 files changed, 1854 insertions(+), 1024 deletions(-) +Merging net-next/main (cfb7793d1bc0f Merge branch 'net-lan966x-add-support-for-pcie-fdma') +$ git merge -m Merge branch 'main' of https://git.kernel.org/pub/scm/linux/kernel/git/netdev/net-next.git net-next/main +Auto-merging .mailmap +Auto-merging MAINTAINERS +Auto-merging drivers/net/amt.c +Auto-merging drivers/net/ethernet/microsoft/Kconfig +Auto-merging drivers/net/ethernet/microsoft/mana/gdma_main.c +Auto-merging drivers/net/ethernet/stmicro/stmmac/stmmac_main.c +CONFLICT (content): Merge conflict in drivers/net/ethernet/stmicro/stmmac/stmmac_main.c +Auto-merging drivers/net/wireless/ath/ath11k/Kconfig +Auto-merging drivers/net/wireless/ath/ath12k/pci.c +Auto-merging drivers/net/wireless/intel/iwlwifi/mvm/rx.c +Auto-merging include/net/mana/gdma.h +Auto-merging net/ipv4/tcp.c +Auto-merging net/packet/af_packet.c +Auto-merging net/socket.c +Auto-merging net/unix/af_unix.c +Resolved 'drivers/net/ethernet/stmicro/stmmac/stmmac_main.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master e147a5c7cc166] Merge branch 'main' of https://git.kernel.org/pub/scm/linux/kernel/git/netdev/net-next.git +$ git diff -M --stat --summary HEAD^.. + .mailmap | 13 +- + .../ABI/testing/sysfs-devices-platform-soc-ipa | 14 - + Documentation/ABI/testing/sysfs-timecard | 37 +- + Documentation/admin-guide/sysctl/net.rst | 14 + + .../bindings/net/allwinner,sun8i-a83t-emac.yaml | 13 + + .../bindings/net/altr,socfpga-stmmac.yaml | 2 - + .../devicetree/bindings/net/brcm,asp-v2.0.yaml | 2 +- + .../bindings/net/cortina,gemini-ethernet.yaml | 52 +- + .../devicetree/bindings/net/dsa/microchip,ksz.yaml | 1 + + .../bindings/net/dsa/motorcomm,yt921x.yaml | 23 + + .../devicetree/bindings/net/dsa/realtek.yaml | 33 + + .../bindings/net/dsa/renesas,rzn1-a5psw.yaml | 4 +- + .../devicetree/bindings/net/ethernet-phy.yaml | 16 + + .../devicetree/bindings/net/fsl,fman-dtsec.yaml | 4 - + .../devicetree/bindings/net/intel,dwmac-plat.yaml | 6 +- + .../devicetree/bindings/net/mdio-gpio.yaml | 16 +- + .../devicetree/bindings/net/microchip,lan8650.yaml | 5 + + .../bindings/net/mscc,vsc7514-switch.yaml | 8 +- + .../bindings/net/nvidia,tegra234-mgbe.yaml | 8 +- + .../devicetree/bindings/net/qcom,ipa.yaml | 10 +- + .../devicetree/bindings/net/realtek,rtl82xx.yaml | 6 +- + .../bindings/net/realtek,rtl9301-mdio.yaml | 14 +- + .../devicetree/bindings/net/renesas,etheravb.yaml | 11 + + .../devicetree/bindings/net/snps,dwmac-common.yaml | 66 + + .../devicetree/bindings/net/snps,dwmac.yaml | 59 +- + .../bindings/net/socionext,uniphier-ave4.yaml | 32 +- + .../devicetree/bindings/net/ti,cpsw-switch.yaml | 70 +- + .../bindings/net/ultrarisc,dp1000-gmac.yaml | 74 + + .../bindings/net/wireless/marvell,sd8787.yaml | 6 + + .../bindings/net/wireless/mediatek,mt76.yaml | 32 +- + .../bindings/net/wireless/qcom,ath10k.yaml | 18 +- + .../bindings/net/wireless/qcom,ath12k-wsi.yaml | 7 - + .../bindings/net/wireless/silabs,wfx.yaml | 3 + + .../devicetree/bindings/net/wireless/ti,wl1251.txt | 64 - + .../bindings/net/wireless/ti,wl1251.yaml | 85 + + .../devicetree/bindings/net/wiznet,w5100.yaml | 88 + + .../devicetree/bindings/net/wiznet,w5x00.txt | 50 - + .../bindings/net/x-powers,acx00-ephy-package.yaml | 170 + + Documentation/netlink/netlink-raw.yaml | 31 +- + Documentation/netlink/specs/conntrack.yaml | 60 +- + Documentation/netlink/specs/devlink.yaml | 171 +- + Documentation/netlink/specs/dpll.yaml | 22 +- + Documentation/netlink/specs/ethtool.yaml | 4 +- + Documentation/netlink/specs/fou.yaml | 4 + + Documentation/netlink/specs/handshake.yaml | 9 +- + Documentation/netlink/specs/netdev.yaml | 5 +- + Documentation/netlink/specs/nl80211.yaml | 4 +- + Documentation/netlink/specs/nlctrl.yaml | 20 +- + Documentation/netlink/specs/psp.yaml | 2 + + Documentation/netlink/specs/rt-link.yaml | 44 +- + Documentation/netlink/specs/tc.yaml | 2 +- + Documentation/netlink/specs/tcp_metrics.yaml | 18 +- + .../networking/device_drivers/ethernet/index.rst | 1 + + .../device_drivers/ethernet/intel/i40e.rst | 2 +- + .../device_drivers/ethernet/intel/iavf.rst | 2 +- + .../device_drivers/ethernet/microsoft/netvsc.rst | 2 - + .../device_drivers/ethernet/nebula-matrix/nbl.rst | 28 + + .../device_drivers/ethernet/stmicro/stmmac.rst | 1 - + Documentation/networking/devlink/index.rst | 1 + + Documentation/networking/devlink/ptp_ocp.rst | 75 + + Documentation/networking/dsa/dsa.rst | 2 +- + Documentation/networking/ethtool-netlink.rst | 2 +- + Documentation/networking/ip-sysctl.rst | 28 +- + Documentation/networking/mctp.rst | 4 +- + .../networking/net_cachelines/net_device.rst | 1 + + Documentation/networking/netconsole.rst | 5 + + Documentation/networking/netdevices.rst | 21 +- + Documentation/networking/nf_conntrack-sysctl.rst | 2 +- + Documentation/networking/page_pool.rst | 2 +- + Documentation/networking/phonet.rst | 2 +- + Documentation/networking/phy.rst | 17 +- + Documentation/networking/proc_net_tcp.rst | 20 +- + Documentation/networking/radiotap-headers.rst | 2 +- + Documentation/networking/scaling.rst | 9 + + Documentation/process/maintainer-netdev.rst | 32 + + MAINTAINERS | 69 +- + arch/loongarch/configs/loongson32_defconfig | 1 - + arch/loongarch/configs/loongson64_defconfig | 1 - + arch/mips/configs/gpr_defconfig | 1 - + arch/mips/configs/mtx1_defconfig | 1 - + arch/mips/configs/rb532_defconfig | 1 - + arch/parisc/configs/generic-32bit_defconfig | 1 - + drivers/dibs/dibs_loopback.c | 5 +- + drivers/dpll/dpll_netlink.c | 2 +- + drivers/dpll/dpll_nl.c | 20 +- + drivers/infiniband/ulp/ipoib/ipoib_main.c | 27 +- + drivers/misc/lan966x_pci.dtso | 5 +- + drivers/net/amt.c | 8 +- + drivers/net/bonding/bond_alb.c | 63 +- + drivers/net/can/bxcan.c | 2 +- + drivers/net/can/c_can/c_can_platform.c | 4 +- + drivers/net/can/flexcan/flexcan-core.c | 2 +- + drivers/net/can/ifi_canfd/ifi_canfd.c | 2 +- + drivers/net/can/m_can/m_can_platform.c | 2 +- + drivers/net/can/sja1000/f81601.c | 2 +- + drivers/net/can/sja1000/sja1000_platform.c | 2 +- + drivers/net/can/slcan/slcan-core.c | 5 +- + drivers/net/dsa/Kconfig | 19 +- + drivers/net/dsa/Makefile | 3 +- + drivers/net/dsa/b53/b53_mdio.c | 2 +- + drivers/net/dsa/b53/b53_srab.c | 2 +- + drivers/net/dsa/bcm_sf2.c | 2 +- + drivers/net/dsa/hirschmann/hellcreek.c | 2 +- + drivers/net/dsa/ks8995.c | 857 ---- + drivers/net/dsa/lan9303_i2c.c | 2 +- + drivers/net/dsa/lan9303_mdio.c | 2 +- + drivers/net/dsa/lantiq/lantiq_gswip_common.c | 9 +- + drivers/net/dsa/lantiq/mxl-gsw1xx.c | 31 +- + drivers/net/dsa/microchip/Kconfig | 1 + + drivers/net/dsa/microchip/ksz8.c | 254 +- + drivers/net/dsa/microchip/ksz8.h | 2 + + drivers/net/dsa/microchip/ksz8_reg.h | 2 + + drivers/net/dsa/microchip/ksz9477.c | 1 + + drivers/net/dsa/microchip/ksz_common.c | 136 +- + drivers/net/dsa/microchip/ksz_common.h | 18 +- + drivers/net/dsa/microchip/ksz_dcb.c | 68 +- + drivers/net/dsa/microchip/ksz_ptp.c | 514 +- + drivers/net/dsa/microchip/ksz_ptp.h | 9 +- + drivers/net/dsa/microchip/ksz_ptp_reg.h | 16 + + drivers/net/dsa/microchip/ksz_spi.c | 33 +- + drivers/net/dsa/microchip/lan937x_main.c | 1 + + drivers/net/dsa/motorcomm/Kconfig | 17 + + drivers/net/dsa/motorcomm/Makefile | 7 + + drivers/net/dsa/{yt921x.c => motorcomm/chip.c} | 752 +-- + drivers/net/dsa/{yt921x.h => motorcomm/chip.h} | 119 +- + drivers/net/dsa/motorcomm/leds.c | 641 +++ + drivers/net/dsa/motorcomm/leds.h | 118 + + drivers/net/dsa/motorcomm/mdio_bus.c | 301 ++ + drivers/net/dsa/motorcomm/mdio_bus.h | 54 + + drivers/net/dsa/motorcomm/pcs-921x.c | 247 + + drivers/net/dsa/motorcomm/pcs.h | 13 + + drivers/net/dsa/motorcomm/smi.c | 180 + + drivers/net/dsa/motorcomm/smi.h | 61 + + drivers/net/dsa/mt7530-mdio.c | 14 +- + drivers/net/dsa/mt7530-mmio.c | 2 +- + drivers/net/dsa/mt7530.c | 834 ++-- + drivers/net/dsa/mt7530.h | 221 +- + drivers/net/dsa/mv88e6060.c | 3 +- + drivers/net/dsa/mv88e6xxx/chip.c | 9 +- + drivers/net/dsa/qca/qca8k-8xxx.c | 2 +- + drivers/net/dsa/qca/qca8k-common.c | 15 +- + drivers/net/dsa/realtek/realtek.h | 7 + + drivers/net/dsa/realtek/rtl8365mb_main.c | 145 +- + drivers/net/dsa/realtek/rtl8366rb.c | 2 +- + drivers/net/dsa/realtek/rtl83xx.c | 43 +- + drivers/net/dsa/rzn1_a5psw.c | 2 +- + drivers/net/dsa/sja1105/sja1105_main.c | 2 +- + drivers/net/dsa/vitesse-vsc73xx-spi.c | 9 +- + drivers/net/ethernet/8390/pcnet_cs.c | 9 +- + drivers/net/ethernet/Kconfig | 1 + + drivers/net/ethernet/Makefile | 1 + + drivers/net/ethernet/adaptec/starfire.c | 6 +- + drivers/net/ethernet/adi/adin1140.c | 4 +- + drivers/net/ethernet/airoha/airoha_eth.c | 363 +- + drivers/net/ethernet/airoha/airoha_eth.h | 29 +- + drivers/net/ethernet/airoha/airoha_regs.h | 23 +- + drivers/net/ethernet/alacritech/slicoss.c | 1 + + drivers/net/ethernet/amd/xgbe/xgbe-mdio.c | 5 +- + drivers/net/ethernet/aquantia/atlantic/aq_nic.c | 5 +- + .../net/ethernet/aquantia/atlantic/aq_pci_func.c | 2 - + drivers/net/ethernet/atheros/alx/main.c | 13 +- + drivers/net/ethernet/atheros/atl1c/atl1c_hw.h | 2 +- + drivers/net/ethernet/atheros/atl1e/atl1e.h | 2 +- + drivers/net/ethernet/atheros/atl1e/atl1e_hw.c | 2 +- + drivers/net/ethernet/atheros/atl1e/atl1e_main.c | 2 +- + drivers/net/ethernet/atheros/atlx/atl1.c | 2 +- + drivers/net/ethernet/atheros/atlx/atl2.c | 2 +- + drivers/net/ethernet/atheros/atlx/atlx.c | 1 + + drivers/net/ethernet/broadcom/asp2/bcmasp.c | 4 +- + drivers/net/ethernet/broadcom/asp2/bcmasp.h | 5 - + drivers/net/ethernet/broadcom/bcm63xx_enet.c | 2 +- + drivers/net/ethernet/broadcom/bcm63xx_enet.h | 2 +- + drivers/net/ethernet/broadcom/bcmsysport.c | 25 +- + drivers/net/ethernet/broadcom/bcmsysport.h | 2 +- + drivers/net/ethernet/broadcom/bnx2x/bnx2x.h | 4 +- + drivers/net/ethernet/broadcom/bnx2x/bnx2x_dcb.c | 4 +- + drivers/net/ethernet/broadcom/bnx2x/bnx2x_hsi.h | 2 +- + .../net/ethernet/broadcom/bnx2x/bnx2x_init_ops.h | 2 +- + drivers/net/ethernet/broadcom/bnx2x/bnx2x_link.c | 8 +- + drivers/net/ethernet/broadcom/bnx2x/bnx2x_main.c | 2 +- + drivers/net/ethernet/broadcom/bnx2x/bnx2x_reg.h | 4 +- + drivers/net/ethernet/broadcom/bnx2x/bnx2x_sriov.c | 18 +- + drivers/net/ethernet/broadcom/bnxt/bnxt.c | 4 +- + drivers/net/ethernet/broadcom/genet/bcmgenet.c | 11 +- + drivers/net/ethernet/cadence/macb.h | 7 +- + drivers/net/ethernet/cadence/macb_main.c | 178 +- + drivers/net/ethernet/cadence/macb_ptp.c | 62 +- + .../net/ethernet/cavium/liquidio/cn66xx_device.c | 2 +- + drivers/net/ethernet/cavium/liquidio/lio_core.c | 2 +- + .../net/ethernet/cavium/liquidio/octeon_config.h | 4 +- + drivers/net/ethernet/cavium/liquidio/octeon_iq.h | 2 +- + drivers/net/ethernet/cavium/liquidio/octeon_main.h | 2 +- + drivers/net/ethernet/cavium/liquidio/octeon_nic.h | 2 +- + .../net/ethernet/cavium/liquidio/request_manager.c | 2 +- + drivers/net/ethernet/cavium/octeon/octeon_mgmt.c | 1 + + drivers/net/ethernet/cavium/thunder/nic.h | 2 +- + drivers/net/ethernet/cavium/thunder/nic_main.c | 2 +- + drivers/net/ethernet/cisco/enic/enic_main.c | 19 +- + drivers/net/ethernet/cortina/gemini.c | 32 +- + drivers/net/ethernet/cortina/gemini.h | 2 +- + drivers/net/ethernet/davicom/dm9000.c | 6 +- + drivers/net/ethernet/dec/tulip/de2104x.c | 2 +- + drivers/net/ethernet/dec/tulip/interrupt.c | 4 +- + drivers/net/ethernet/dec/tulip/media.c | 2 +- + drivers/net/ethernet/dec/tulip/tulip.h | 4 +- + drivers/net/ethernet/dec/tulip/winbond-840.c | 2 +- + .../ethernet/freescale/dpaa2/dpaa2-eth-devlink.c | 45 +- + drivers/net/ethernet/freescale/dpaa2/dpaa2-eth.c | 2 + + drivers/net/ethernet/freescale/dpaa2/dpaa2-eth.h | 3 + + drivers/net/ethernet/freescale/dpaa2/dpaa2-ptp.c | 2 + + drivers/net/ethernet/freescale/enetc/Kconfig | 1 + + drivers/net/ethernet/freescale/enetc/enetc.c | 68 +- + drivers/net/ethernet/freescale/enetc/enetc.h | 11 +- + .../net/ethernet/freescale/enetc/enetc4_debugfs.c | 51 +- + drivers/net/ethernet/freescale/enetc/enetc4_hw.h | 1 + + drivers/net/ethernet/freescale/enetc/enetc4_pf.c | 85 +- + .../net/ethernet/freescale/enetc/enetc_ethtool.c | 6 + + drivers/net/ethernet/freescale/enetc/enetc_hw.h | 26 + + .../net/ethernet/freescale/enetc/enetc_mailbox.h | 109 +- + drivers/net/ethernet/freescale/enetc/enetc_mdio.c | 2 +- + drivers/net/ethernet/freescale/enetc/enetc_msg.c | 641 ++- + drivers/net/ethernet/freescale/enetc/enetc_pf.c | 64 +- + drivers/net/ethernet/freescale/enetc/enetc_pf.h | 21 +- + .../net/ethernet/freescale/enetc/enetc_pf_common.c | 150 +- + .../net/ethernet/freescale/enetc/enetc_pf_common.h | 19 + + drivers/net/ethernet/freescale/enetc/enetc_vf.c | 453 +- + drivers/net/ethernet/freescale/fec.h | 2 +- + drivers/net/ethernet/freescale/fec_main.c | 18 +- + drivers/net/ethernet/freescale/fec_ptp.c | 24 +- + .../net/ethernet/freescale/fs_enet/fs_enet-main.c | 1 + + .../net/ethernet/hisilicon/hibmcge/hbg_common.h | 2 +- + drivers/net/ethernet/hisilicon/hibmcge/hbg_hw.c | 4 +- + drivers/net/ethernet/hisilicon/hibmcge/hbg_irq.c | 4 +- + drivers/net/ethernet/hisilicon/hibmcge/hbg_reg.h | 4 +- + drivers/net/ethernet/hisilicon/hibmcge/hbg_txrx.c | 12 +- + drivers/net/ethernet/hisilicon/hip04_eth.c | 5 +- + drivers/net/ethernet/hisilicon/hns/hns_dsaf_misc.c | 2 +- + drivers/net/ethernet/hisilicon/hns/hns_dsaf_ppe.c | 2 +- + drivers/net/ethernet/hisilicon/hns/hns_enet.c | 4 +- + drivers/net/ethernet/hisilicon/hns3/hns3_ethtool.c | 10 +- + .../ethernet/hisilicon/hns3/hns3pf/hclge_main.c | 11 +- + .../net/ethernet/hisilicon/hns3/hns3pf/hclge_mbx.c | 7 +- + .../net/ethernet/hisilicon/hns3/hns3pf/hclge_tm.c | 2 +- + .../ethernet/hisilicon/hns3/hns3vf/hclgevf_main.c | 2 +- + drivers/net/ethernet/huawei/hinic3/hinic3_hwdev.c | 7 +- + drivers/net/ethernet/intel/fm10k/fm10k_pci.c | 2 - + drivers/net/ethernet/intel/i40e/i40e_common.c | 2 +- + drivers/net/ethernet/intel/i40e/i40e_ethtool.c | 12 +- + drivers/net/ethernet/intel/i40e/i40e_main.c | 70 +- + drivers/net/ethernet/intel/i40e/i40e_txrx.c | 5 +- + drivers/net/ethernet/intel/i40e/i40e_txrx.h | 7 +- + drivers/net/ethernet/intel/i40e/i40e_type.h | 5 + + drivers/net/ethernet/intel/i40e/i40e_virtchnl_pf.c | 5 +- + drivers/net/ethernet/intel/i40e/i40e_xsk.c | 12 + + drivers/net/ethernet/intel/ice/Makefile | 2 +- + drivers/net/ethernet/intel/ice/ice.h | 17 +- + drivers/net/ethernet/intel/ice/ice_arfs.c | 8 +- + drivers/net/ethernet/intel/ice/ice_arfs.h | 2 +- + drivers/net/ethernet/intel/ice/ice_dpll.c | 495 +- + drivers/net/ethernet/intel/ice/ice_dpll.h | 4 + + drivers/net/ethernet/intel/ice/ice_ethtool.c | 14 +- + .../{ice_ethtool_fdir.c => ice_ethtool_ntuple.c} | 88 +- + drivers/net/ethernet/intel/ice/ice_fdir.c | 38 +- + drivers/net/ethernet/intel/ice/ice_fdir.h | 14 +- + drivers/net/ethernet/intel/ice/ice_lib.c | 343 +- + drivers/net/ethernet/intel/ice/ice_main.c | 11 +- + drivers/net/ethernet/intel/ice/ice_ptp.c | 82 + + drivers/net/ethernet/intel/ice/ice_ptp.h | 11 + + drivers/net/ethernet/intel/ice/ice_sriov.c | 2 +- + drivers/net/ethernet/intel/ice/ice_tspll.c | 120 +- + drivers/net/ethernet/intel/ice/ice_tspll.h | 6 + + drivers/net/ethernet/intel/ice/ice_txclk.c | 20 +- + drivers/net/ethernet/intel/ice/ice_type.h | 4 +- + drivers/net/ethernet/intel/ice/virt/fdir.c | 28 +- + drivers/net/ethernet/intel/idpf/idpf.h | 12 + + drivers/net/ethernet/intel/idpf/idpf_lib.c | 19 + + drivers/net/ethernet/intel/idpf/idpf_txrx.c | 71 +- + drivers/net/ethernet/intel/idpf/idpf_txrx.h | 8 +- + drivers/net/ethernet/intel/idpf/idpf_virtchnl.c | 60 +- + drivers/net/ethernet/intel/ixgbevf/ixgbevf_main.c | 2 +- + drivers/net/ethernet/marvell/mvmdio.c | 4 +- + drivers/net/ethernet/marvell/mvneta.c | 2 +- + drivers/net/ethernet/marvell/mvpp2/mvpp2.h | 2 +- + drivers/net/ethernet/marvell/mvpp2/mvpp2_main.c | 2 +- + drivers/net/ethernet/marvell/mvpp2/mvpp2_prs.c | 8 +- + .../ethernet/marvell/octeon_ep/octep_pfvf_mbox.c | 1 - + drivers/net/ethernet/marvell/octeontx2/af/cgx.c | 25 +- + drivers/net/ethernet/marvell/octeontx2/af/cgx.h | 2 + + .../net/ethernet/marvell/octeontx2/af/cn20k/npa.c | 1 - + .../net/ethernet/marvell/octeontx2/af/cn20k/npc.c | 5 +- + .../ethernet/marvell/octeontx2/af/lmac_common.h | 2 + + .../net/ethernet/marvell/octeontx2/af/mcs_reg.h | 2 - + drivers/net/ethernet/marvell/octeontx2/af/npc.h | 4 +- + drivers/net/ethernet/marvell/octeontx2/af/rpm.c | 40 +- + drivers/net/ethernet/marvell/octeontx2/af/rpm.h | 2 + + drivers/net/ethernet/marvell/octeontx2/af/rvu.c | 4 +- + drivers/net/ethernet/marvell/octeontx2/af/rvu.h | 5 +- + .../net/ethernet/marvell/octeontx2/af/rvu_cgx.c | 12 + + .../ethernet/marvell/octeontx2/af/rvu_debugfs.c | 8 + + .../ethernet/marvell/octeontx2/af/rvu_devlink.c | 46 +- + .../net/ethernet/marvell/octeontx2/af/rvu_nix.c | 33 +- + .../net/ethernet/marvell/octeontx2/af/rvu_npa.c | 10 +- + .../net/ethernet/marvell/octeontx2/af/rvu_npc.c | 5 +- + .../net/ethernet/marvell/octeontx2/af/rvu_npc_fs.c | 1 - + drivers/net/ethernet/marvell/octeontx2/nic/cn10k.c | 2 +- + .../ethernet/marvell/octeontx2/nic/otx2_common.c | 8 +- + .../ethernet/marvell/octeontx2/nic/otx2_common.h | 1 - + .../ethernet/marvell/octeontx2/nic/otx2_dcbnl.c | 7 +- + .../ethernet/marvell/octeontx2/nic/otx2_ethtool.c | 18 + + .../ethernet/marvell/octeontx2/nic/otx2_flows.c | 3 - + .../net/ethernet/marvell/octeontx2/nic/otx2_pf.c | 101 +- + .../net/ethernet/marvell/octeontx2/nic/otx2_reg.h | 2 + + .../net/ethernet/marvell/octeontx2/nic/otx2_tc.c | 1 - + .../net/ethernet/marvell/octeontx2/nic/otx2_vf.c | 9 +- + drivers/net/ethernet/marvell/octeontx2/nic/qos.h | 2 - + .../ethernet/marvell/prestera/prestera_router.c | 10 +- + drivers/net/ethernet/marvell/pxa168_eth.c | 4 +- + drivers/net/ethernet/mellanox/mlx4/main.c | 8 +- + drivers/net/ethernet/mellanox/mlx5/core/en/ptp.c | 2 +- + .../net/ethernet/mellanox/mlx5/core/en/rep/neigh.c | 29 +- + .../net/ethernet/mellanox/mlx5/core/en/tc_priv.h | 14 +- + .../ethernet/mellanox/mlx5/core/en/tc_tun_encap.c | 23 +- + .../ethernet/mellanox/mlx5/core/en/tc_tun_vxlan.c | 11 +- + drivers/net/ethernet/mellanox/mlx5/core/en/xdp.c | 2 +- + .../ethernet/mellanox/mlx5/core/en_accel/ipsec.c | 6 +- + .../net/ethernet/mellanox/mlx5/core/en_ethtool.c | 6 +- + drivers/net/ethernet/mellanox/mlx5/core/en_main.c | 3 +- + drivers/net/ethernet/mellanox/mlx5/core/en_rx.c | 2 +- + drivers/net/ethernet/mellanox/mlx5/core/en_tc.c | 75 +- + drivers/net/ethernet/mellanox/mlx5/core/en_tx.c | 2 +- + drivers/net/ethernet/mellanox/mlx5/core/eswitch.c | 20 +- + drivers/net/ethernet/mellanox/mlx5/core/eswitch.h | 2 +- + .../ethernet/mellanox/mlx5/core/eswitch_offloads.c | 37 +- + .../net/ethernet/mellanox/mlx5/core/fpga/conn.c | 2 +- + drivers/net/ethernet/mellanox/mlx5/core/fw_reset.c | 6 +- + .../net/ethernet/mellanox/mlx5/core/lag/debugfs.c | 14 +- + drivers/net/ethernet/mellanox/mlx5/core/lag/lag.c | 154 +- + drivers/net/ethernet/mellanox/mlx5/core/lag/lag.h | 2 +- + .../ethernet/mellanox/mlx5/core/lag/shared_fdb.c | 3 +- + drivers/net/ethernet/mellanox/mlx5/core/lib/aso.c | 2 +- + drivers/net/ethernet/mellanox/mlx5/core/main.c | 8 +- + .../ethernet/mellanox/mlx5/core/steering/hws/bwc.c | 4 +- + .../mellanox/mlx5/core/steering/hws/send.c | 34 +- + drivers/net/ethernet/mellanox/mlxsw/pci.c | 7 +- + .../ethernet/mellanox/mlxsw/spectrum_nve_vxlan.c | 19 +- + .../net/ethernet/mellanox/mlxsw/spectrum_router.c | 31 +- + .../net/ethernet/mellanox/mlxsw/spectrum_span.c | 10 +- + .../ethernet/mellanox/mlxsw/spectrum_switchdev.c | 57 +- + drivers/net/ethernet/meta/Kconfig | 13 + + drivers/net/ethernet/meta/Makefile | 1 + + drivers/net/ethernet/meta/fbnic/fbnic.h | 9 +- + drivers/net/ethernet/meta/fbnic/fbnic_csr.h | 34 +- + drivers/net/ethernet/meta/fbnic/fbnic_debugfs.c | 5 +- + drivers/net/ethernet/meta/fbnic/fbnic_devlink.c | 2 + + drivers/net/ethernet/meta/fbnic/fbnic_ethtool.c | 15 +- + drivers/net/ethernet/meta/fbnic/fbnic_fw.c | 87 +- + drivers/net/ethernet/meta/fbnic/fbnic_fw.h | 19 +- + drivers/net/ethernet/meta/fbnic/fbnic_mac.c | 9 +- + drivers/net/ethernet/meta/fbnic/fbnic_netdev.c | 4 + + drivers/net/ethernet/meta/fbnic/fbnic_pci.c | 2 - + drivers/net/ethernet/meta/fbnic/fbnic_rpc.c | 129 +- + drivers/net/ethernet/meta/fbnic/fbnic_tlv.c | 13 +- + drivers/net/ethernet/meta/fbnic/fbnic_txrx.c | 191 +- + drivers/net/ethernet/meta/fbnic/fbnic_txrx.h | 17 + + drivers/net/ethernet/meta/mpnic/Makefile | 16 + + drivers/net/ethernet/meta/mpnic/mpnic.h | 66 + + drivers/net/ethernet/meta/mpnic/mpnic_csr.h | 319 ++ + drivers/net/ethernet/meta/mpnic/mpnic_init.c | 556 +++ + drivers/net/ethernet/meta/mpnic/mpnic_irq.c | 64 + + drivers/net/ethernet/meta/mpnic/mpnic_netdev.c | 186 + + drivers/net/ethernet/meta/mpnic/mpnic_netdev.h | 35 + + drivers/net/ethernet/meta/mpnic/mpnic_pci.c | 187 + + drivers/net/ethernet/meta/mpnic/mpnic_txrx.c | 1395 ++++++ + drivers/net/ethernet/meta/mpnic/mpnic_txrx.h | 151 + + drivers/net/ethernet/microchip/fdma/Makefile | 4 + + drivers/net/ethernet/microchip/fdma/fdma_api.c | 65 +- + drivers/net/ethernet/microchip/fdma/fdma_api.h | 42 +- + drivers/net/ethernet/microchip/fdma/fdma_pci.c | 208 + + drivers/net/ethernet/microchip/fdma/fdma_pci.h | 57 + + drivers/net/ethernet/microchip/lan865x/lan865x.c | 4 +- + drivers/net/ethernet/microchip/lan966x/Makefile | 4 + + .../net/ethernet/microchip/lan966x/lan966x_fdma.c | 97 +- + .../ethernet/microchip/lan966x/lan966x_fdma_pci.c | 731 +++ + .../net/ethernet/microchip/lan966x/lan966x_main.c | 121 +- + .../net/ethernet/microchip/lan966x/lan966x_main.h | 69 + + .../net/ethernet/microchip/lan966x/lan966x_regs.h | 25 + + .../net/ethernet/microchip/lan966x/lan966x_xdp.c | 6 + + drivers/net/ethernet/microchip/vcap/Kconfig | 1 - + drivers/net/ethernet/microsoft/Kconfig | 26 +- + drivers/net/ethernet/microsoft/Makefile | 2 +- + drivers/net/ethernet/microsoft/mana/Makefile | 9 +- + drivers/net/ethernet/microsoft/mana/gdma_cdx.c | 340 ++ + drivers/net/ethernet/microsoft/mana/gdma_main.c | 1067 +---- + drivers/net/ethernet/microsoft/mana/gdma_pci.c | 917 ++++ + drivers/net/ethernet/microsoft/mana/hw_channel.c | 24 +- + drivers/net/ethernet/microsoft/mana/mana_bpf.c | 3 - + drivers/net/ethernet/microsoft/mana/mana_en.c | 18 +- + drivers/net/ethernet/nebula-matrix/Kconfig | 32 + + drivers/net/ethernet/nebula-matrix/Makefile | 6 + + drivers/net/ethernet/nebula-matrix/nbl/Makefile | 7 + + drivers/net/ethernet/nebula-matrix/nbl/nbl_core.h | 31 + + .../nbl/nbl_hw/nbl_hw_leonis/nbl_hw_leonis.c | 153 + + .../nbl/nbl_hw/nbl_hw_leonis/nbl_hw_leonis.h | 15 + + .../ethernet/nebula-matrix/nbl/nbl_hw/nbl_hw_reg.h | 31 + + .../nebula-matrix/nbl/nbl_include/nbl_def_common.h | 31 + + .../nebula-matrix/nbl/nbl_include/nbl_def_hw.h | 17 + + .../nebula-matrix/nbl/nbl_include/nbl_include.h | 23 + + drivers/net/ethernet/nebula-matrix/nbl/nbl_main.c | 192 + + .../ethernet/netronome/nfp/flower/tunnel_conf.c | 14 +- + drivers/net/ethernet/oa_tc6.c | 46 +- + .../net/ethernet/oki-semi/pch_gbe/pch_gbe_main.c | 10 +- + .../net/ethernet/pensando/ionic/ionic_bus_pci.c | 1 + + .../net/ethernet/pensando/ionic/ionic_ethtool.c | 10 +- + drivers/net/ethernet/pensando/ionic/ionic_if.h | 31 +- + .../net/ethernet/qlogic/netxen/netxen_nic_ctx.c | 2 +- + drivers/net/ethernet/qlogic/qed/qed_cxt.c | 2 +- + drivers/net/ethernet/qlogic/qed/qed_mcp.h | 2 +- + drivers/net/ethernet/qlogic/qede/qede_ethtool.c | 6 +- + drivers/net/ethernet/qlogic/qla3xxx.c | 19 +- + drivers/net/ethernet/qlogic/qlcnic/qlcnic.h | 2 +- + .../net/ethernet/qlogic/qlcnic/qlcnic_83xx_init.c | 4 +- + drivers/net/ethernet/qlogic/qlcnic/qlcnic_ctx.c | 2 +- + drivers/net/ethernet/qlogic/qlcnic/qlcnic_dcb.c | 5 +- + drivers/net/ethernet/qlogic/qlcnic/qlcnic_main.c | 4 +- + .../ethernet/qlogic/qlcnic/qlcnic_sriov_common.c | 6 +- + .../net/ethernet/qlogic/qlcnic/qlcnic_sriov_pf.c | 6 +- + drivers/net/ethernet/qualcomm/rmnet/rmnet_config.c | 75 +- + drivers/net/ethernet/qualcomm/rmnet/rmnet_config.h | 2 +- + .../net/ethernet/qualcomm/rmnet/rmnet_handlers.c | 34 +- + drivers/net/ethernet/qualcomm/rmnet/rmnet_map.h | 9 +- + .../ethernet/qualcomm/rmnet/rmnet_map_command.c | 9 +- + .../net/ethernet/qualcomm/rmnet/rmnet_map_data.c | 14 +- + drivers/net/ethernet/qualcomm/rmnet/rmnet_vnd.c | 13 +- + drivers/net/ethernet/realtek/Kconfig | 2 +- + drivers/net/ethernet/realtek/r8169_main.c | 706 ++- + drivers/net/ethernet/renesas/ravb.h | 35 +- + drivers/net/ethernet/renesas/ravb_main.c | 253 +- + drivers/net/ethernet/renesas/ravb_ptp.c | 38 +- + drivers/net/ethernet/renesas/rswitch_main.c | 7 +- + drivers/net/ethernet/rocker/rocker_main.c | 2 +- + drivers/net/ethernet/rocker/rocker_ofdpa.c | 2 +- + drivers/net/ethernet/sfc/Kconfig | 2 +- + drivers/net/ethernet/sfc/ptp.c | 1 - + drivers/net/ethernet/sfc/siena/ptp.c | 1 - + drivers/net/ethernet/sfc/tc_counters.c | 8 +- + drivers/net/ethernet/sfc/tc_encap_actions.c | 4 +- + drivers/net/ethernet/smsc/Kconfig | 3 +- + drivers/net/ethernet/smsc/smc9194.h | 241 - + drivers/net/ethernet/spacemit/k1_emac.c | 4 +- + drivers/net/ethernet/stmicro/stmmac/Kconfig | 11 + + drivers/net/ethernet/stmicro/stmmac/Makefile | 1 + + drivers/net/ethernet/stmicro/stmmac/common.h | 5 + + .../net/ethernet/stmicro/stmmac/dwmac-eic7700.c | 13 +- + drivers/net/ethernet/stmicro/stmmac/dwmac-sophgo.c | 10 +- + drivers/net/ethernet/stmicro/stmmac/dwmac-sun8i.c | 77 +- + .../net/ethernet/stmicro/stmmac/dwmac-ultrarisc.c | 52 + + drivers/net/ethernet/stmicro/stmmac/dwmac4.h | 6 +- + drivers/net/ethernet/stmicro/stmmac/dwmac4_core.c | 79 +- + drivers/net/ethernet/stmicro/stmmac/dwmac4_dma.c | 2 + + .../net/ethernet/stmicro/stmmac/dwxgmac2_core.c | 18 - + drivers/net/ethernet/stmicro/stmmac/hwif.h | 3 - + drivers/net/ethernet/stmicro/stmmac/stmmac.h | 7 + + .../net/ethernet/stmicro/stmmac/stmmac_ethtool.c | 3 +- + drivers/net/ethernet/stmicro/stmmac/stmmac_main.c | 192 +- + .../net/ethernet/stmicro/stmmac/stmmac_platform.c | 17 +- + .../net/ethernet/stmicro/stmmac/stmmac_selftests.c | 112 - + drivers/net/ethernet/sun/niu.c | 10 +- + drivers/net/ethernet/ti/am65-cpsw-nuss.c | 2 +- + drivers/net/ethernet/ti/cpsw.c | 2 +- + drivers/net/ethernet/ti/cpsw_new.c | 2 +- + drivers/net/ethernet/ti/davinci_mdio.c | 4 +- + drivers/net/ethernet/wangxun/libwx/wx_hw.c | 2 +- + drivers/net/ethernet/wangxun/libwx/wx_ptp.c | 5 +- + drivers/net/ethernet/wangxun/libwx/wx_type.h | 6 +- + drivers/net/ethernet/wangxun/txgbe/txgbe_phy.c | 6 +- + drivers/net/ethernet/wiznet/w5100.c | 234 +- + drivers/net/ethernet/xilinx/ll_temac_main.c | 7 +- + drivers/net/ethernet/xilinx/xilinx_axienet_main.c | 34 +- + drivers/net/ethernet/xscale/ptp_ixp46x.c | 1 + + drivers/net/fddi/skfp/cfm.c | 2 +- + drivers/net/fddi/skfp/ecm.c | 2 +- + drivers/net/fddi/skfp/fplustm.c | 4 +- + drivers/net/fddi/skfp/h/hwmtm.h | 2 +- + drivers/net/fddi/skfp/h/sba.h | 2 +- + drivers/net/fddi/skfp/pcmplc.c | 2 +- + drivers/net/fddi/skfp/rmt.c | 2 +- + drivers/net/gtp.c | 2 +- + drivers/net/hyperv/netvsc.c | 10 +- + drivers/net/ieee802154/at86rf230.c | 2 +- + drivers/net/macsec.c | 32 + + drivers/net/mdio/Kconfig | 4 +- + drivers/net/mdio/mdio-bcm-iproc.c | 2 +- + drivers/net/mdio/mdio-bcm-unimac.c | 2 +- + drivers/net/mdio/mdio-mux-meson-g12a.c | 2 +- + drivers/net/mdio/mdio-realtek-rtl9300.c | 426 +- + drivers/net/netconsole.c | 132 +- + drivers/net/netdevsim/bus.c | 7 +- + drivers/net/netdevsim/ethtool.c | 53 + + drivers/net/netdevsim/ipsec.c | 10 +- + drivers/net/netdevsim/netdev.c | 10 +- + drivers/net/netdevsim/netdevsim.h | 8 +- + drivers/net/netdevsim/psp.c | 33 - + drivers/net/netkit.c | 17 +- + drivers/net/pcs/pcs-lynx.c | 59 +- + drivers/net/pcs/pcs-xpcs-plat.c | 2 +- + drivers/net/phy/Kconfig | 11 + + drivers/net/phy/Makefile | 3 +- + drivers/net/phy/air_en8811h.c | 84 +- + drivers/net/phy/broadcom.c | 7 + + drivers/net/phy/dp83848.c | 3 + + drivers/net/phy/dp83867.c | 109 +- + drivers/net/phy/mediatek/mtk-phy-lib.c | 26 +- + drivers/net/phy/microchip_t1.c | 16 +- + drivers/net/phy/motorcomm.c | 84 +- + drivers/net/phy/mxl-gpy.c | 2 +- + drivers/net/phy/nxp-c45-tja11xx.c | 2 +- + drivers/net/phy/phy_device.c | 360 +- + drivers/net/phy/phy_fixup.c | 98 + + drivers/net/phy/phy_led_triggers.c | 11 +- + drivers/net/phy/phylib-internal.h | 1 + + drivers/net/phy/phylink.c | 176 +- + drivers/net/phy/qcom/qca808x.c | 15 +- + drivers/net/phy/realtek/realtek_main.c | 103 +- + drivers/net/phy/sfp.c | 21 +- + drivers/net/phy/xpowers/Makefile | 3 + + drivers/net/phy/xpowers/ac200.c | 314 ++ + drivers/net/phy/xpowers/ac300.c | 387 ++ + drivers/net/phy/xpowers/acx00.c | 633 +++ + drivers/net/phy/xpowers/acx00.h | 28 + + drivers/net/ppp/ppp_synctty.c | 20 +- + drivers/net/usb/asix_common.c | 5 - + drivers/net/usb/ax88172a.c | 3 +- + drivers/net/usb/ax88179_178a.c | 9 +- + drivers/net/usb/ch9200.c | 14 +- + drivers/net/usb/qmi_wwan.c | 1 + + drivers/net/usb/r8152.c | 51 +- + drivers/net/virtio_net.c | 33 +- + drivers/net/vrf.c | 2 +- + drivers/net/vxlan/vxlan_core.c | 780 ++-- + drivers/net/vxlan/vxlan_mdb.c | 45 +- + drivers/net/vxlan/vxlan_multicast.c | 94 +- + drivers/net/vxlan/vxlan_private.h | 34 +- + drivers/net/vxlan/vxlan_vnifilter.c | 206 +- + drivers/net/wan/slic_ds26522.c | 2 +- + drivers/net/wireless/ath/ath10k/debug.c | 10 +- + drivers/net/wireless/ath/ath10k/snoc.c | 6 +- + drivers/net/wireless/ath/ath11k/Kconfig | 3 + + drivers/net/wireless/ath/ath11k/ahb.c | 16 +- + drivers/net/wireless/ath/ath11k/ce.c | 2 +- + drivers/net/wireless/ath/ath11k/core.c | 10 + + drivers/net/wireless/ath/ath11k/dp.c | 2 + + drivers/net/wireless/ath/ath11k/dp_rx.c | 21 +- + drivers/net/wireless/ath/ath11k/htc.c | 5 +- + drivers/net/wireless/ath/ath11k/wmi.c | 9 + + drivers/net/wireless/ath/ath12k/ahb.c | 166 +- + drivers/net/wireless/ath/ath12k/ahb.h | 9 + + drivers/net/wireless/ath/ath12k/ce.c | 2 +- + drivers/net/wireless/ath/ath12k/core.c | 23 + + drivers/net/wireless/ath/ath12k/core.h | 9 - + drivers/net/wireless/ath/ath12k/debugfs.c | 78 +- + drivers/net/wireless/ath/ath12k/dp.c | 92 +- + drivers/net/wireless/ath/ath12k/dp.h | 114 +- + drivers/net/wireless/ath/ath12k/dp_mon.c | 17 +- + drivers/net/wireless/ath/ath12k/dp_mon.h | 2 +- + drivers/net/wireless/ath/ath12k/dp_rx.c | 53 +- + drivers/net/wireless/ath/ath12k/dp_rx.h | 3 - + drivers/net/wireless/ath/ath12k/dp_stats.h | 25 + + drivers/net/wireless/ath/ath12k/htc.c | 5 +- + drivers/net/wireless/ath/ath12k/hw.h | 21 +- + drivers/net/wireless/ath/ath12k/mac.c | 50 +- + drivers/net/wireless/ath/ath12k/pci.c | 55 +- + drivers/net/wireless/ath/ath12k/qmi.c | 106 +- + drivers/net/wireless/ath/ath12k/qmi.h | 4 + + drivers/net/wireless/ath/ath12k/wifi7/ahb.c | 2 + + drivers/net/wireless/ath/ath12k/wifi7/dp_mon.c | 45 +- + drivers/net/wireless/ath/ath12k/wifi7/dp_rx.c | 72 +- + drivers/net/wireless/ath/ath12k/wifi7/dp_rx.h | 12 +- + drivers/net/wireless/ath/ath12k/wifi7/dp_tx.c | 56 +- + drivers/net/wireless/ath/ath12k/wifi7/hal_rx.c | 5 +- + drivers/net/wireless/ath/ath12k/wifi7/hal_tx.c | 2 +- + drivers/net/wireless/ath/ath12k/wifi7/hw.c | 16 +- + drivers/net/wireless/ath/ath12k/wmi.c | 22 +- + drivers/net/wireless/ath/ath6kl/cfg80211.c | 10 +- + drivers/net/wireless/ath/ath9k/ar9003_calib.c | 5 +- + drivers/net/wireless/ath/ath9k/mci.c | 5 +- + drivers/net/wireless/ath/ath9k/xmit.c | 2 +- + drivers/net/wireless/ath/wil6210/cfg80211.c | 13 +- + drivers/net/wireless/broadcom/b43legacy/dma.c | 24 +- + drivers/net/wireless/broadcom/b43legacy/dma.h | 4 +- + drivers/net/wireless/broadcom/b43legacy/main.c | 3 +- + .../broadcom/brcm80211/brcmfmac/cfg80211.c | 140 +- + .../broadcom/brcm80211/brcmfmac/cfg80211.h | 5 +- + .../wireless/broadcom/brcm80211/brcmfmac/feature.c | 2 + + .../wireless/broadcom/brcm80211/brcmfmac/feature.h | 4 +- + drivers/net/wireless/intel/iwlwifi/mld/agg.c | 17 +- + drivers/net/wireless/intel/iwlwifi/mld/agg.h | 4 +- + drivers/net/wireless/intel/iwlwifi/mld/d3.c | 6 +- + drivers/net/wireless/intel/iwlwifi/mld/rx.c | 50 +- + drivers/net/wireless/intel/iwlwifi/mld/rx.h | 2 +- + drivers/net/wireless/intel/iwlwifi/mld/tests/agg.c | 7 +- + drivers/net/wireless/intel/iwlwifi/mvm/d3.c | 6 +- + drivers/net/wireless/intel/iwlwifi/mvm/rx.c | 2 +- + drivers/net/wireless/intel/iwlwifi/mvm/rxmq.c | 6 +- + drivers/net/wireless/marvell/libertas/cfg.c | 6 +- + drivers/net/wireless/marvell/libertas/if_usb.c | 10 +- + drivers/net/wireless/marvell/libertas/mesh.c | 506 -- + drivers/net/wireless/marvell/libertas/rx.c | 9 + + drivers/net/wireless/marvell/libertas_tf/if_usb.c | 10 +- + .../net/wireless/marvell/mwifiex/11n_rxreorder.c | 20 +- + drivers/net/wireless/marvell/mwifiex/cfg80211.c | 27 +- + drivers/net/wireless/marvell/mwifiex/main.h | 2 +- + drivers/net/wireless/marvell/mwifiex/sdio.c | 3 + + drivers/net/wireless/mediatek/mt76/mac80211.c | 25 +- + .../net/wireless/mediatek/mt76/mt76_connac_mcu.c | 5 +- + drivers/net/wireless/mediatek/mt76/mt7925/mcu.c | 5 +- + drivers/net/wireless/microchip/wilc1000/cfg80211.c | 13 +- + drivers/net/wireless/morsemicro/mm81x/core.h | 7 +- + drivers/net/wireless/morsemicro/mm81x/fw.c | 9 +- + drivers/net/wireless/morsemicro/mm81x/mac.c | 181 +- + drivers/net/wireless/morsemicro/mm81x/sdio.c | 9 +- + drivers/net/wireless/morsemicro/mm81x/skbq.c | 7 +- + drivers/net/wireless/nxp/nxpwifi/cfg80211.c | 11 +- + drivers/net/wireless/quantenna/qtnfmac/cfg80211.c | 8 +- + drivers/net/wireless/ralink/rt2x00/rt2x00pci.c | 104 +- + drivers/net/wireless/ralink/rt2x00/rt2x00usb.c | 67 +- + drivers/net/wireless/realtek/rtw88/debug.c | 5 +- + drivers/net/wireless/realtek/rtw89/core.c | 12 +- + drivers/net/wireless/realtek/rtw89/mac.c | 10 +- + drivers/net/wireless/realtek/rtw89/regd.c | 5 +- + drivers/net/wireless/realtek/rtw89/wow.c | 6 +- + drivers/net/wireless/silabs/wfx/bh.c | 6 +- + drivers/net/wireless/silabs/wfx/bus_sdio.c | 1 + + drivers/net/wireless/silabs/wfx/main.c | 37 +- + drivers/net/wireless/st/cw1200/txrx.c | 2 +- + drivers/net/wireless/ti/wlcore/rx.c | 5 +- + drivers/net/wireless/virtual/mac80211_hwsim.h | 5 +- + drivers/net/wireless/virtual/mac80211_hwsim_main.c | 133 +- + drivers/net/wireless/virtual/mac80211_hwsim_nan.c | 22 + + drivers/net/wireless/virtual/mac80211_hwsim_nan.h | 3 + + drivers/ptp/ptp_idt82p33.c | 25 +- + drivers/ptp/ptp_idt82p33.h | 1 + + drivers/ptp/ptp_ocp.c | 939 +++- + drivers/s390/net/ism_drv.c | 3 +- + drivers/staging/rtl8723bs/os_dep/ioctl_cfg80211.c | 9 +- + drivers/usb/atm/cxacru.c | 18 +- + drivers/usb/atm/speedtch.c | 10 +- + include/linux/bnge/hsi.h | 136 +- + include/linux/dibs.h | 6 +- + include/linux/ethtool.h | 4 +- + include/linux/ieee80211-eht.h | 18 +- + include/linux/ieee80211-uhr.h | 101 +- + include/linux/ieee80211.h | 25 +- + include/linux/igmp.h | 3 +- + include/linux/llc.h | 5 - + include/linux/mdio-mux.h | 2 +- + include/linux/mlx5/eswitch.h | 3 + + include/linux/net.h | 34 +- + include/linux/netdevice.h | 23 +- + include/linux/netpoll.h | 5 - + include/linux/phy.h | 28 +- + include/linux/phylink.h | 2 + + include/linux/platform_data/microchip-ksz.h | 2 + + include/linux/skbuff.h | 7 +- + include/linux/stmmac.h | 2 + + include/net/arp.h | 10 +- + include/net/bond_alb.h | 14 +- + include/net/cfg80211.h | 51 +- + include/net/cfg802154.h | 9 +- + include/net/dropreason-core.h | 6 + + include/net/dsa.h | 2 + + include/net/ieee80211_radiotap.h | 113 +- + include/net/inet_connection_sock.h | 1 + + include/net/ip6_tunnel.h | 1 + + include/net/ip_tunnels.h | 7 +- + include/net/llc.h | 108 +- + include/net/llc_c_ac.h | 180 - + include/net/llc_c_ev.h | 217 - + include/net/llc_c_st.h | 46 - + include/net/llc_conn.h | 114 - + include/net/llc_if.h | 63 - + include/net/llc_pdu.h | 379 +- + include/net/llc_s_ac.h | 35 - + include/net/llc_s_ev.h | 61 - + include/net/llc_s_st.h | 32 - + include/net/llc_sap.h | 26 - + include/net/mac80211.h | 79 +- + include/net/mana/gdma.h | 106 +- + include/net/mana/hw_channel.h | 8 +- + include/net/ndisc.h | 20 +- + include/net/neighbour.h | 22 +- + include/net/net_namespace.h | 4 + + include/net/netdev_lock.h | 20 +- + include/net/netdev_queues.h | 56 +- + include/net/netfilter/nf_conntrack_seqadj.h | 1 + + include/net/netmem.h | 27 + + include/net/netns/netfilter.h | 2 + + include/net/page_pool/helpers.h | 12 +- + include/net/phy/realtek_phy.h | 7 - + include/net/psp/functions.h | 5 - + include/net/psp/types.h | 4 + + include/net/route.h | 7 +- + include/net/seg6_hmac.h | 3 +- + include/net/tcp.h | 17 +- + include/net/vxlan.h | 12 +- + include/trace/events/bridge.h | 7 +- + include/uapi/linux/atm.h | 2 +- + include/uapi/linux/dpll.h | 3 +- + include/uapi/linux/handshake.h | 5 +- + include/uapi/linux/if_link.h | 1 + + include/uapi/linux/mii.h | 2 +- + include/uapi/linux/netdev.h | 2 +- + include/uapi/linux/netfilter/nfnetlink_conntrack.h | 1 + + include/uapi/linux/netlink.h | 14 + + include/uapi/linux/nl80211.h | 82 +- + lib/ref_tracker.c | 2 +- + net/6lowpan/debugfs.c | 10 +- + net/802/garp.c | 4 +- + net/802/psnap.c | 2 +- + net/8021q/vlan.c | 8 +- + net/8021q/vlan.h | 22 +- + net/8021q/vlan_core.c | 2 +- + net/8021q/vlan_dev.c | 37 +- + net/8021q/vlan_netlink.c | 52 +- + net/8021q/vlanproc.c | 18 +- + net/Kconfig | 1 - + net/atm/proc.c | 7 +- + net/batman-adv/bat_iv_ogm.c | 29 +- + net/batman-adv/bat_v.c | 42 +- + net/batman-adv/bridge_loop_avoidance.c | 6 +- + net/batman-adv/distributed-arp-table.c | 10 +- + net/batman-adv/hash.h | 2 +- + net/batman-adv/mesh-interface.c | 12 +- + net/batman-adv/multicast.c | 6 +- + net/batman-adv/translation-table.c | 1118 +++-- + net/batman-adv/types.h | 49 +- + net/bridge/br.c | 9 +- + net/bridge/br_arp_nd_proxy.c | 61 +- + net/bridge/br_device.c | 17 +- + net/bridge/br_fdb.c | 272 +- + net/bridge/br_forward.c | 294 +- + net/bridge/br_if.c | 13 +- + net/bridge/br_input.c | 30 +- + net/bridge/br_mrp.c | 6 +- + net/bridge/br_mst.c | 28 +- + net/bridge/br_multicast.c | 18 +- + net/bridge/br_netlink.c | 12 +- + net/bridge/br_netlink_tunnel.c | 14 +- + net/bridge/br_private.h | 251 +- + net/bridge/br_stp_bpdu.c | 3 +- + net/bridge/br_stp_if.c | 2 +- + net/bridge/br_switchdev.c | 2 +- + net/bridge/br_sysfs_if.c | 13 +- + net/bridge/br_vlan.c | 281 +- + net/bridge/br_vlan_options.c | 20 +- + net/bridge/br_vlan_tunnel.c | 2 +- + net/bridge/netfilter/nft_reject_bridge.c | 8 +- + net/core/dev.c | 90 +- + net/core/dev.h | 9 + + net/core/devmem.c | 229 +- + net/core/devmem.h | 44 +- + net/core/gro_cells.c | 5 +- + net/core/neighbour.c | 428 +- + net/core/net-sysfs.c | 112 +- + net/core/netdev-genl.c | 26 +- + net/core/netdev_config.c | 75 +- + net/core/netdev_queues.c | 31 +- + net/core/netdev_work.c | 15 +- + net/core/netpoll.c | 15 + + net/core/page_pool_priv.h | 14 +- + net/core/page_pool_user.c | 8 +- + net/core/pktgen.c | 17 +- + net/core/rtnetlink.c | 22 + + net/core/skbuff.c | 32 +- + net/core/sock.c | 4 +- + net/core/stream.c | 2 +- + net/core/sysctl_net_core.c | 17 +- + net/devlink/dev.c | 4 +- + net/devlink/netlink_gen.c | 9 +- + net/devlink/netlink_gen.h | 2 +- + net/devlink/port.c | 19 +- + net/dsa/Kconfig | 6 + + net/dsa/Makefile | 1 + + net/dsa/tag_ks8995.c | 180 + + net/dsa/tag_ksz.c | 5 +- + net/ethtool/common.c | 182 + + net/ethtool/common.h | 2 + + net/ethtool/ioctl.c | 33 +- + net/ethtool/netlink.c | 15 +- + net/ethtool/rings.c | 13 +- + net/ethtool/ts.h | 8 +- + net/handshake/genl.c | 4 +- + net/handshake/handshake-test.c | 2 +- + net/handshake/request.c | 2 +- + net/hsr/hsr_device.c | 2 +- + net/hsr/hsr_forward.c | 2 +- + net/hsr/hsr_netlink.c | 37 +- + net/ieee802154/6lowpan/tx.c | 3 +- + net/ieee802154/socket.c | 5 +- + net/ipv4/af_inet.c | 3 - + net/ipv4/arp.c | 137 +- + net/ipv4/bpf_tcp_ca.c | 5 +- + net/ipv4/devinet.c | 18 +- + net/ipv4/fib_semantics.c | 7 +- + net/ipv4/fou_nl.c | 6 +- + net/ipv4/igmp.c | 23 +- + net/ipv4/inet_connection_sock.c | 1 + + net/ipv4/inet_hashtables.c | 11 + + net/ipv4/inet_timewait_sock.c | 6 +- + net/ipv4/ip_forward.c | 2 +- + net/ipv4/ip_gre.c | 6 +- + net/ipv4/ip_sockglue.c | 10 +- + net/ipv4/ip_tunnel.c | 146 +- + net/ipv4/ip_vti.c | 2 +- + net/ipv4/ipip.c | 2 +- + net/ipv4/ipmr.c | 17 +- + net/ipv4/netfilter/arp_tables.c | 2 +- + net/ipv4/ping.c | 5 +- + net/ipv4/raw.c | 4 +- + net/ipv4/route.c | 2 +- + net/ipv4/tcp.c | 21 +- + net/ipv4/tcp_bbr.c | 13 +- + net/ipv4/tcp_bpf.c | 10 +- + net/ipv4/tcp_input.c | 26 +- + net/ipv4/tcp_ipv4.c | 13 +- + net/ipv4/tcp_output.c | 28 +- + net/ipv4/udp.c | 26 +- + net/ipv6/addrconf.c | 17 +- + net/ipv6/af_inet6.c | 4 - + net/ipv6/ah6.c | 2 +- + net/ipv6/datagram.c | 5 +- + net/ipv6/exthdrs.c | 8 +- + net/ipv6/ip6_fib.c | 14 +- + net/ipv6/ip6_gre.c | 268 +- + net/ipv6/ip6_output.c | 6 +- + net/ipv6/ip6_vti.c | 2 +- + net/ipv6/ip6mr.c | 5 +- + net/ipv6/ipv6_sockglue.c | 123 +- + net/ipv6/ndisc.c | 193 +- + net/ipv6/route.c | 63 +- + net/ipv6/seg6_hmac.c | 5 +- + net/ipv6/seg6_iptunnel.c | 10 +- + net/ipv6/seg6_local.c | 6 +- + net/ipv6/sit.c | 480 +- + net/ipv6/tcp_ipv6.c | 12 +- + net/ipv6/udp.c | 8 +- + net/iucv/af_iucv.c | 20 +- + net/key/af_key.c | 5 +- + net/llc/Kconfig | 7 - + net/llc/Makefile | 10 +- + net/llc/af_llc.c | 1304 ------ + net/llc/llc_c_ac.c | 1448 ------ + net/llc/llc_c_ev.c | 742 --- + net/llc/llc_c_st.c | 4940 -------------------- + net/llc/llc_conn.c | 1025 ---- + net/llc/llc_core.c | 28 +- + net/llc/llc_if.c | 151 - + net/llc/llc_input.c | 128 +- + net/llc/llc_output.c | 3 +- + net/llc/llc_pdu.c | 366 -- + net/llc/llc_proc.c | 245 - + net/llc/llc_s_ac.c | 220 - + net/llc/llc_s_ev.c | 109 - + net/llc/llc_s_st.c | 177 - + net/llc/llc_sap.c | 437 -- + net/llc/llc_station.c | 120 - + net/llc/sysctl_net_llc.c | 71 - + net/mac80211/agg-tx.c | 9 +- + net/mac80211/ap.c | 3 - + net/mac80211/cfg.c | 54 +- + net/mac80211/debugfs.c | 2 +- + net/mac80211/debugfs_key.c | 18 +- + net/mac80211/debugfs_netdev.c | 8 +- + net/mac80211/ht.c | 2 +- + net/mac80211/ieee80211_i.h | 37 +- + net/mac80211/iface.c | 12 +- + net/mac80211/key.c | 96 +- + net/mac80211/key.h | 3 +- + net/mac80211/main.c | 13 +- + net/mac80211/mesh_pathtbl.c | 15 +- + net/mac80211/mesh_plink.c | 4 +- + net/mac80211/mlme.c | 132 +- + net/mac80211/nan.c | 33 + + net/mac80211/offchannel.c | 23 +- + net/mac80211/parse.c | 4 + + net/mac80211/rate.c | 1 - + net/mac80211/rx.c | 470 +- + net/mac80211/s1g.c | 8 - + net/mac80211/scan.c | 10 +- + net/mac80211/sta_info.h | 2 + + net/mac80211/tdls.c | 4 +- + net/mac80211/tests/chan-mode.c | 6 +- + net/mac80211/tests/util.c | 6 +- + net/mac80211/tx.c | 429 +- + net/mac80211/util.c | 43 +- + net/mac802154/iface.c | 10 +- + net/mptcp/options.c | 6 +- + net/mptcp/protocol.c | 14 +- + net/mptcp/protocol.h | 45 +- + net/mptcp/subflow.c | 22 +- + net/ncsi/ncsi-rsp.c | 10 +- + net/netfilter/core.c | 18 +- + net/netfilter/ipset/ip_set_core.c | 2 +- + net/netfilter/ipvs/ip_vs_sync.c | 2 +- + net/netfilter/nf_conncount.c | 2 +- + net/netfilter/nf_conntrack_core.c | 5 +- + net/netfilter/nf_conntrack_netlink.c | 19 +- + net/netfilter/nf_conntrack_ovs.c | 2 +- + net/netfilter/nf_conntrack_seqadj.c | 28 +- + net/netfilter/nf_conntrack_standalone.c | 2 +- + net/netfilter/nf_flow_table_core.c | 17 +- + net/netfilter/nf_hooks_lwtunnel.c | 2 +- + net/netfilter/nf_log.c | 2 +- + net/netfilter/nf_nat_core.c | 16 +- + net/netfilter/nf_synproxy_core.c | 6 +- + net/netfilter/nfnetlink_acct.c | 2 +- + net/netfilter/nfnetlink_cthelper.c | 3 +- + net/netfilter/nfnetlink_cttimeout.c | 8 +- + net/netfilter/nfnetlink_hook.c | 74 +- + net/netfilter/nfnetlink_osf.c | 3 +- + net/netfilter/nft_ct.c | 5 +- + net/netfilter/nft_set_pipapo.c | 6 +- + net/netfilter/xt_CT.c | 6 +- + net/netfilter/xt_IDLETIMER.c | 8 +- + net/netfilter/xt_LED.c | 5 +- + net/netfilter/xt_RATEEST.c | 2 +- + net/netfilter/xt_TEE.c | 2 +- + net/netfilter/xt_hashlimit.c | 4 +- + net/netfilter/xt_limit.c | 2 +- + net/netfilter/xt_quota.c | 2 +- + net/netfilter/xt_recent.c | 3 +- + net/netfilter/xt_statistic.c | 2 +- + net/netfilter/xt_string.c | 2 +- + net/netlink/af_netlink.c | 8 +- + net/netlink/policy.c | 16 +- + net/openvswitch/datapath.c | 14 +- + net/openvswitch/datapath.h | 2 +- + net/openvswitch/flow.c | 1 - + net/openvswitch/flow_netlink.c | 2 - + net/openvswitch/flow_table.c | 2 - + net/openvswitch/vport-internal_dev.c | 2 + + net/openvswitch/vport-netdev.c | 1 - + net/packet/af_packet.c | 7 +- + net/phonet/socket.c | 5 +- + net/psp/psp.h | 13 + + net/psp/psp_main.c | 29 +- + net/psp/psp_nl.c | 51 +- + net/psp/psp_sock.c | 115 +- + net/rds/ib.c | 14 +- + net/rds/ib.h | 4 +- + net/rds/ib_cm.c | 41 +- + net/rds/rdma_transport.c | 13 +- + net/sched/act_api.c | 15 +- + net/sched/act_connmark.c | 6 + + net/sched/act_ct.c | 2 +- + net/sched/act_mirred.c | 3 +- + net/sched/act_mpls.c | 11 + + net/sched/act_nat.c | 6 + + net/sched/act_simple.c | 7 + + net/sched/act_skbmod.c | 9 + + net/sched/cls_flower.c | 39 +- + net/sched/ematch.c | 9 +- + net/sched/sch_api.c | 92 +- + net/sched/sch_fq.c | 71 +- + net/sctp/proc.c | 6 +- + net/smc/af_smc.c | 12 +- + net/smc/smc_close.c | 18 +- + net/smc/smc_core.c | 4 +- + net/smc/smc_tx.c | 18 +- + net/socket.c | 32 +- + net/tipc/bearer.c | 6 +- + net/tipc/bearer.h | 3 +- + net/tipc/monitor.c | 3 +- + net/tls/tls_main.c | 2 - + net/unix/af_unix.c | 3 +- + net/vmw_vsock/af_vsock.c | 35 + + net/vmw_vsock/af_vsock_tap.c | 4 - + net/wireless/ap.c | 7 +- + net/wireless/chan.c | 44 +- + net/wireless/core.c | 87 +- + net/wireless/core.h | 11 +- + net/wireless/ibss.c | 3 +- + net/wireless/nl80211.c | 295 +- + net/wireless/rdev-ops.h | 40 +- + net/wireless/reg.c | 13 +- + net/wireless/sme.c | 10 +- + net/wireless/sysfs.c | 12 +- + net/wireless/trace.h | 62 +- + net/wireless/util.c | 40 +- + net/wireless/wext-compat.c | 8 +- + net/x25/af_x25.c | 5 +- + rust/kernel/net/netlink.rs | 8 +- + tools/include/uapi/linux/netdev.h | 2 +- + tools/net/ynl/Makefile | 1 + + tools/net/ynl/Makefile.deps | 3 +- + tools/net/ynl/generated/Makefile | 3 +- + tools/net/ynl/lib/Makefile | 3 +- + tools/net/ynl/pyynl/lib/ynl.py | 8 +- + tools/net/ynl/pyynl/ynl_gen_c.py | 15 +- + tools/net/ynl/tests/Makefile | 2 +- + tools/net/ynl/ynltool/Makefile | 10 +- + tools/testing/selftests/bpf/progs/tcp_ca_kfunc.c | 8 +- + tools/testing/selftests/drivers/net/Makefile | 11 +- + tools/testing/selftests/drivers/net/README.rst | 9 + + .../testing/selftests/drivers/net/bonding/Makefile | 3 +- + tools/testing/selftests/drivers/net/dsa/Makefile | 1 + + tools/testing/selftests/drivers/net/gro_hw.py | 13 + + .../selftests/drivers/net/{gro.py => gro_lib.py} | 190 +- + tools/testing/selftests/drivers/net/gro_lro.py | 14 + + tools/testing/selftests/drivers/net/gro_sw.py | 13 + + tools/testing/selftests/drivers/net/hds.py | 29 - + tools/testing/selftests/drivers/net/hw/Makefile | 18 +- + tools/testing/selftests/drivers/net/hw/config | 1 + + .../selftests/drivers/net/hw/devlink_rate_tc_bw.py | 5 +- + .../testing/selftests/drivers/net/hw/devmem_lib.py | 4 +- + .../drivers/net/hw/{gro_hw.py => gro_stats.py} | 0 + tools/testing/selftests/drivers/net/hw/iou-zcrx.c | 422 +- + tools/testing/selftests/drivers/net/hw/iou-zcrx.py | 56 +- + .../selftests/drivers/net/hw/lib/py/__init__.py | 7 +- + tools/testing/selftests/drivers/net/hw/toeplitz.py | 10 +- + tools/testing/selftests/drivers/net/hw/tso.py | 161 + + tools/testing/selftests/drivers/net/hw/vlan.py | 127 + + .../selftests/drivers/net/lib/py/__init__.py | 7 +- + tools/testing/selftests/drivers/net/lib/py/env.py | 19 + + tools/testing/selftests/drivers/net/lib/py/feat.py | 43 + + .../selftests/drivers/net/netconsole/Makefile | 3 +- + .../selftests/drivers/net/netdevsim/Makefile | 8 + + tools/testing/selftests/drivers/net/pppoe_gro.py | 45 + + tools/testing/selftests/drivers/net/psp.py | 180 +- + .../testing/selftests/drivers/net/ring_reconfig.py | 26 +- + tools/testing/selftests/drivers/net/rss_key.py | 343 ++ + tools/testing/selftests/drivers/net/ruff.toml | 4 + + tools/testing/selftests/drivers/net/settings | 2 +- + tools/testing/selftests/drivers/net/so_txtime.c | 161 +- + tools/testing/selftests/drivers/net/so_txtime.py | 80 +- + tools/testing/selftests/drivers/net/team/Makefile | 4 +- + .../selftests/drivers/net/virtio_net/Makefile | 1 + + tools/testing/selftests/net/Makefile | 1 + + tools/testing/selftests/net/bind_wildcard.c | 5 + + tools/testing/selftests/net/drop_monitor_tests.sh | 13 +- + tools/testing/selftests/net/fcnal-test.sh | 29 +- + tools/testing/selftests/net/fdb_flush.sh | 35 +- + tools/testing/selftests/net/fib-onlink-tests.sh | 18 +- + .../selftests/net/fib_nexthop_multiprefix.sh | 21 +- + tools/testing/selftests/net/fib_nexthop_nongw.sh | 21 +- + tools/testing/selftests/net/fib_rule_tests.sh | 19 +- + tools/testing/selftests/net/fib_tests.sh | 66 +- + tools/testing/selftests/net/forwarding/Makefile | 2 + + tools/testing/selftests/net/fou_mcast_encap.sh | 11 +- + tools/testing/selftests/net/gre_gso.sh | 26 +- + tools/testing/selftests/net/icmp_redirect.sh | 29 +- + tools/testing/selftests/net/ip_local_port_range.c | 52 +- + tools/testing/selftests/net/l2tp.sh | 19 +- + tools/testing/selftests/net/lib.sh | 34 + + tools/testing/selftests/net/lib/csum.c | 55 +- + tools/testing/selftests/net/lib/gro.c | 136 +- + .../selftests/net/lib/ksft_setup_loopback.sh | 2 +- + tools/testing/selftests/net/lib/py/__init__.py | 4 +- + tools/testing/selftests/net/lib/py/utils.py | 22 + + tools/testing/selftests/net/mptcp/config | 2 +- + .../selftests/net/ndisc_unsolicited_na_test.sh | 223 +- + tools/testing/selftests/net/netdev_lock.py | 42 + + tools/testing/selftests/net/netfilter/Makefile | 2 + + tools/testing/selftests/net/netfilter/config | 2 +- + .../selftests/net/netfilter/conntrack_dump_flush.c | 145 +- + .../net/netfilter/conntrack_icmp_related.sh | 2 +- + tools/testing/selftests/net/openvswitch/config | 3 + + .../selftests/net/openvswitch/openvswitch.sh | 202 + + .../packetdrill/tcp_rcv_ssthresh_scaling_ratio.pkt | 46 + + tools/testing/selftests/net/rtnetlink.py | 113 +- + tools/testing/selftests/net/rtnetlink.sh | 8 +- + tools/testing/selftests/net/ruff.toml | 4 + + tools/testing/selftests/net/so_incoming_cpu.c | 7 +- + .../selftests/net/srv6_encap_lookup_l3vpn_test.sh | 19 +- + .../selftests/net/srv6_end_dt46_l3vpn_test.sh | 19 +- + .../selftests/net/srv6_end_dt4_l3vpn_test.sh | 40 +- + .../selftests/net/srv6_end_dt6_l3vpn_test.sh | 40 +- + .../selftests/net/srv6_end_dx4_netfilter_test.sh | 23 +- + .../selftests/net/srv6_end_dx6_netfilter_test.sh | 23 +- + .../testing/selftests/net/srv6_end_flavors_test.sh | 23 +- + .../selftests/net/srv6_end_next_csid_l3vpn_test.sh | 19 +- + .../net/srv6_end_x_next_csid_l3vpn_test.sh | 19 +- + .../selftests/net/srv6_hencap_red_l3vpn_test.sh | 19 +- + .../selftests/net/srv6_hl2encap_red_l2vpn_test.sh | 19 +- + .../selftests/net/test_bridge_backup_port.sh | 32 +- + .../selftests/net/test_bridge_neigh_suppress.sh | 34 +- + tools/testing/selftests/net/test_neigh.sh | 66 +- + tools/testing/selftests/net/test_vxlan_mdb.sh | 32 +- + .../selftests/net/test_vxlan_nolocalbypass.sh | 32 +- + .../selftests/net/test_vxlan_vnifiltering.sh | 26 +- + tools/testing/selftests/net/tun.c | 9 + + tools/testing/selftests/net/vrf-xfrm-tests.sh | 19 +- + tools/testing/selftests/net/vrf_route_leaking.sh | 19 +- + .../testing/selftests/net/vrf_strict_mode_test.sh | 19 +- + tools/testing/selftests/net/xfrm_policy.sh | 2 +- + tools/testing/vsock/util.c | 9 + + tools/testing/vsock/util.h | 1 + + tools/testing/vsock/vsock_uring_test.c | 391 ++ + 1097 files changed, 33893 insertions(+), 26120 deletions(-) + create mode 100644 Documentation/devicetree/bindings/net/snps,dwmac-common.yaml + create mode 100644 Documentation/devicetree/bindings/net/ultrarisc,dp1000-gmac.yaml + delete mode 100644 Documentation/devicetree/bindings/net/wireless/ti,wl1251.txt + create mode 100644 Documentation/devicetree/bindings/net/wireless/ti,wl1251.yaml + create mode 100644 Documentation/devicetree/bindings/net/wiznet,w5100.yaml + delete mode 100644 Documentation/devicetree/bindings/net/wiznet,w5x00.txt + create mode 100644 Documentation/devicetree/bindings/net/x-powers,acx00-ephy-package.yaml + create mode 100644 Documentation/networking/device_drivers/ethernet/nebula-matrix/nbl.rst + create mode 100644 Documentation/networking/devlink/ptp_ocp.rst + delete mode 100644 drivers/net/dsa/ks8995.c + create mode 100644 drivers/net/dsa/motorcomm/Kconfig + create mode 100644 drivers/net/dsa/motorcomm/Makefile + rename drivers/net/dsa/{yt921x.c => motorcomm/chip.c} (88%) + rename drivers/net/dsa/{yt921x.h => motorcomm/chip.h} (94%) + create mode 100644 drivers/net/dsa/motorcomm/leds.c + create mode 100644 drivers/net/dsa/motorcomm/leds.h + create mode 100644 drivers/net/dsa/motorcomm/mdio_bus.c + create mode 100644 drivers/net/dsa/motorcomm/mdio_bus.h + create mode 100644 drivers/net/dsa/motorcomm/pcs-921x.c + create mode 100644 drivers/net/dsa/motorcomm/pcs.h + create mode 100644 drivers/net/dsa/motorcomm/smi.c + create mode 100644 drivers/net/dsa/motorcomm/smi.h + rename drivers/net/ethernet/intel/ice/{ice_ethtool_fdir.c => ice_ethtool_ntuple.c} (96%) + create mode 100644 drivers/net/ethernet/meta/mpnic/Makefile + create mode 100644 drivers/net/ethernet/meta/mpnic/mpnic.h + create mode 100644 drivers/net/ethernet/meta/mpnic/mpnic_csr.h + create mode 100644 drivers/net/ethernet/meta/mpnic/mpnic_init.c + create mode 100644 drivers/net/ethernet/meta/mpnic/mpnic_irq.c + create mode 100644 drivers/net/ethernet/meta/mpnic/mpnic_netdev.c + create mode 100644 drivers/net/ethernet/meta/mpnic/mpnic_netdev.h + create mode 100644 drivers/net/ethernet/meta/mpnic/mpnic_pci.c + create mode 100644 drivers/net/ethernet/meta/mpnic/mpnic_txrx.c + create mode 100644 drivers/net/ethernet/meta/mpnic/mpnic_txrx.h + create mode 100644 drivers/net/ethernet/microchip/fdma/fdma_pci.c + create mode 100644 drivers/net/ethernet/microchip/fdma/fdma_pci.h + create mode 100644 drivers/net/ethernet/microchip/lan966x/lan966x_fdma_pci.c + create mode 100644 drivers/net/ethernet/microsoft/mana/gdma_cdx.c + create mode 100644 drivers/net/ethernet/microsoft/mana/gdma_pci.c + create mode 100644 drivers/net/ethernet/nebula-matrix/Kconfig + create mode 100644 drivers/net/ethernet/nebula-matrix/Makefile + create mode 100644 drivers/net/ethernet/nebula-matrix/nbl/Makefile + create mode 100644 drivers/net/ethernet/nebula-matrix/nbl/nbl_core.h + create mode 100644 drivers/net/ethernet/nebula-matrix/nbl/nbl_hw/nbl_hw_leonis/nbl_hw_leonis.c + create mode 100644 drivers/net/ethernet/nebula-matrix/nbl/nbl_hw/nbl_hw_leonis/nbl_hw_leonis.h + create mode 100644 drivers/net/ethernet/nebula-matrix/nbl/nbl_hw/nbl_hw_reg.h + create mode 100644 drivers/net/ethernet/nebula-matrix/nbl/nbl_include/nbl_def_common.h + create mode 100644 drivers/net/ethernet/nebula-matrix/nbl/nbl_include/nbl_def_hw.h + create mode 100644 drivers/net/ethernet/nebula-matrix/nbl/nbl_include/nbl_include.h + create mode 100644 drivers/net/ethernet/nebula-matrix/nbl/nbl_main.c + delete mode 100644 drivers/net/ethernet/smsc/smc9194.h + create mode 100644 drivers/net/ethernet/stmicro/stmmac/dwmac-ultrarisc.c + create mode 100644 drivers/net/phy/phy_fixup.c + create mode 100644 drivers/net/phy/xpowers/Makefile + create mode 100644 drivers/net/phy/xpowers/ac200.c + create mode 100644 drivers/net/phy/xpowers/ac300.c + create mode 100644 drivers/net/phy/xpowers/acx00.c + create mode 100644 drivers/net/phy/xpowers/acx00.h + create mode 100644 drivers/net/wireless/ath/ath12k/dp_stats.h + delete mode 100644 include/net/llc_c_ac.h + delete mode 100644 include/net/llc_c_ev.h + delete mode 100644 include/net/llc_c_st.h + delete mode 100644 include/net/llc_conn.h + delete mode 100644 include/net/llc_if.h + delete mode 100644 include/net/llc_s_ac.h + delete mode 100644 include/net/llc_s_ev.h + delete mode 100644 include/net/llc_s_st.h + delete mode 100644 include/net/llc_sap.h + delete mode 100644 include/net/phy/realtek_phy.h + create mode 100644 net/dsa/tag_ks8995.c + delete mode 100644 net/llc/af_llc.c + delete mode 100644 net/llc/llc_c_ac.c + delete mode 100644 net/llc/llc_c_ev.c + delete mode 100644 net/llc/llc_c_st.c + delete mode 100644 net/llc/llc_conn.c + delete mode 100644 net/llc/llc_if.c + delete mode 100644 net/llc/llc_pdu.c + delete mode 100644 net/llc/llc_proc.c + delete mode 100644 net/llc/llc_s_ac.c + delete mode 100644 net/llc/llc_s_ev.c + delete mode 100644 net/llc/llc_s_st.c + delete mode 100644 net/llc/llc_sap.c + delete mode 100644 net/llc/llc_station.c + delete mode 100644 net/llc/sysctl_net_llc.c + create mode 100755 tools/testing/selftests/drivers/net/gro_hw.py + rename tools/testing/selftests/drivers/net/{gro.py => gro_lib.py} (74%) + mode change 100755 => 100644 + create mode 100755 tools/testing/selftests/drivers/net/gro_lro.py + create mode 100755 tools/testing/selftests/drivers/net/gro_sw.py + rename tools/testing/selftests/drivers/net/hw/{gro_hw.py => gro_stats.py} (100%) + create mode 100755 tools/testing/selftests/drivers/net/hw/vlan.py + create mode 100644 tools/testing/selftests/drivers/net/lib/py/feat.py + create mode 100755 tools/testing/selftests/drivers/net/pppoe_gro.py + create mode 100755 tools/testing/selftests/drivers/net/rss_key.py + create mode 100644 tools/testing/selftests/drivers/net/ruff.toml + create mode 100755 tools/testing/selftests/net/netdev_lock.py + create mode 100644 tools/testing/selftests/net/packetdrill/tcp_rcv_ssthresh_scaling_ratio.pkt + create mode 100644 tools/testing/selftests/net/ruff.toml +Merging bpf-next/for-next (2b5440b31cafa Merge git://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf 7.3-rc5) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf-next.git bpf-next/for-next +Auto-merging Documentation/admin-guide/kernel-parameters.txt +Auto-merging MAINTAINERS +Auto-merging arch/arm64/net/bpf_jit_comp.c +Auto-merging arch/x86/Kconfig +Auto-merging block/fops.c +Auto-merging drivers/nvme/host/pci.c +Auto-merging drivers/pci/pci.c +CONFLICT (content): Merge conflict in drivers/pci/pci.c +Auto-merging fs/bpf_fs_kfuncs.c +Auto-merging fs/exec.c +Auto-merging include/linux/mm.h +Auto-merging include/net/tcp.h +Auto-merging init/Kconfig +CONFLICT (content): Merge conflict in init/Kconfig +Auto-merging kernel/bpf/arena.c +Auto-merging mm/internal.h +CONFLICT (content): Merge conflict in mm/internal.h +Auto-merging mm/memory.c +Auto-merging mm/nommu.c +Auto-merging mm/util.c +Auto-merging net/ipv4/af_inet.c +Auto-merging net/ipv4/bpf_tcp_ca.c +Auto-merging net/ipv4/tcp.c +Auto-merging net/ipv4/tcp_bpf.c +Auto-merging net/ipv4/tcp_input.c +Auto-merging net/ipv4/tcp_output.c +Recorded preimage for 'drivers/pci/pci.c' +Recorded preimage for 'init/Kconfig' +Resolved 'mm/internal.h' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +Recorded resolution for 'drivers/pci/pci.c'. +Recorded resolution for 'init/Kconfig'. +[master d84a21a6b7bb4] Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf-next.git +$ git diff -M --stat --summary HEAD^.. + Documentation/admin-guide/kernel-parameters.txt | 16 + + Documentation/bpf/bpf_design_QA.rst | 11 +- + Documentation/bpf/btf.rst | 85 +- + Documentation/bpf/clang-notes.rst | 7 +- + Documentation/bpf/kfuncs.rst | 126 +- + Documentation/bpf/linux-notes.rst | 31 +- + Documentation/bpf/signing.rst | 283 +- + MAINTAINERS | 2 - + arch/arm64/kernel/vdso/Makefile | 2 +- + arch/arm64/net/bpf_jit_comp.c | 172 +- + arch/loongarch/vdso/Makefile | 1 + + arch/mips/include/asm/uasm.h | 8 + + arch/mips/mm/uasm-mips.c | 12 + + arch/mips/mm/uasm.c | 24 +- + arch/mips/net/bpf_jit_comp.c | 57 +- + arch/mips/net/bpf_jit_comp.h | 6 +- + arch/mips/net/bpf_jit_comp32.c | 109 +- + arch/mips/net/bpf_jit_comp64.c | 139 +- + arch/parisc/net/bpf_jit.h | 2 + + arch/parisc/net/bpf_jit_comp32.c | 29 +- + arch/parisc/net/bpf_jit_comp64.c | 35 +- + arch/parisc/net/bpf_jit_core.c | 14 + + arch/powerpc/kernel/vdso/Makefile | 1 + + arch/riscv/kernel/vdso/Makefile | 1 + + arch/riscv/net/bpf_jit.h | 4 + + arch/riscv/net/bpf_jit_comp64.c | 109 +- + arch/riscv/net/bpf_jit_core.c | 8 + + arch/riscv/net/bpf_timed_may_goto.S | 8 +- + arch/s390/kernel/vdso/Makefile | 1 + + arch/x86/Kconfig | 1 + + arch/x86/entry/vdso/vdso64/Makefile | 2 +- + arch/x86/net/bpf_jit_comp.c | 501 ++- + arch/x86/net/bpf_timed_may_goto.S | 10 +- + block/blk-cgroup.c | 5 +- + block/blk-mq.c | 63 +- + block/fops.c | 5 + + block/mq-deadline.c | 2 +- + crypto/Makefile | 3 - + crypto/bpf_crypto_skcipher.c | 83 - + drivers/block/drbd/drbd_nl_gen.c | 4 - + drivers/block/virtio_blk.c | 1 + + drivers/char/tpm/tpm-interface.c | 2 +- + drivers/char/tpm/tpm2-cmd.c | 6 +- + drivers/char/tpm/tpm2-sessions.c | 1 + + drivers/net/ethernet/netronome/nfp/bpf/verifier.c | 2 +- + drivers/nvme/host/core.c | 63 +- + drivers/nvme/host/fc.c | 2 +- + drivers/nvme/host/multipath.c | 19 +- + drivers/nvme/host/nvme.h | 9 +- + drivers/nvme/host/pci.c | 2 + + drivers/nvme/host/sysfs.c | 4 +- + drivers/nvme/host/tcp.c | 39 +- + drivers/nvme/target/configfs.c | 9 +- + drivers/nvme/target/core.c | 35 +- + drivers/nvme/target/fabrics-cmd-auth.c | 2 +- + drivers/nvme/target/nvmet.h | 2 + + drivers/nvme/target/pci-epf.c | 3 + + drivers/virt/vmgenid.c | 11 +- + fs/bpf_fs_kfuncs.c | 99 + + fs/exec.c | 7 +- + include/linux/bpf-cgroup-defs.h | 1 + + include/linux/bpf-cgroup.h | 28 + + include/linux/bpf.h | 179 +- + include/linux/bpf_crypto.h | 24 - + include/linux/bpf_lsm.h | 11 +- + include/linux/bpf_verifier.h | 312 +- + include/linux/btf.h | 23 +- + include/linux/filter.h | 45 + + include/linux/key.h | 2 + + include/linux/mm.h | 8 +- + include/linux/siphash.h | 8 +- + include/net/tcp.h | 156 +- + include/uapi/linux/bpf.h | 82 +- + include/uapi/linux/btf.h | 83 +- + include/uapi/linux/keyctl.h | 1 + + include/uapi/linux/random.h | 2 +- + include/vdso/getrandom.h | 2 +- + io_uring/bpf_filter.c | 1 + + io_uring/cmd_net.c | 19 +- + io_uring/io_uring.c | 16 +- + io_uring/loop.c | 5 + + io_uring/tw.c | 3 + + io_uring/zcrx.c | 9 +- + kernel/bpf/Kconfig | 26 + + kernel/bpf/Makefile | 7 +- + kernel/bpf/arena.c | 233 +- + kernel/bpf/arraymap.c | 2 +- + kernel/bpf/backtrack.c | 153 +- + kernel/bpf/bpf_insn_array.c | 19 +- + kernel/bpf/bpf_local_storage.c | 15 +- + kernel/bpf/bpf_lsm.c | 26 +- + kernel/bpf/bpf_lsm_proto.c | 19 +- + kernel/bpf/bpf_struct_ops.c | 142 +- + kernel/bpf/btf.c | 645 ++- + kernel/bpf/cfg.c | 311 +- + kernel/bpf/cgroup.c | 500 ++- + kernel/bpf/cnum_defs.h | 37 +- + kernel/bpf/const_fold.c | 6 +- + kernel/bpf/core.c | 137 +- + kernel/bpf/crypto.c | 288 +- + kernel/bpf/diagnostics.c | 71 +- + kernel/bpf/diagnostics.h | 1 + + kernel/bpf/disasm.c | 5 +- + kernel/bpf/fixups.c | 411 +- + kernel/bpf/hashtab.c | 65 +- + kernel/bpf/helpers.c | 259 +- + kernel/bpf/keys.c | 73 + + kernel/bpf/liveness.c | 633 ++- + kernel/bpf/log.c | 13 +- + kernel/bpf/map_in_map.c | 4 + + kernel/bpf/map_iter.c | 6 + + kernel/bpf/queue_stack_maps.c | 3 + + kernel/bpf/range_tree.c | 52 +- + kernel/bpf/states.c | 108 +- + kernel/bpf/stream.c | 271 +- + kernel/bpf/syscall.c | 89 +- + kernel/bpf/trampoline.c | 8 +- + kernel/bpf/verifier.c | 4614 +++++++++++++------- + kernel/trace/bpf_trace.c | 6 +- + mm/bpf_memcontrol.c | 65 +- + mm/internal.h | 13 +- + mm/memory.c | 43 +- + mm/nommu.c | 43 +- + mm/util.c | 60 + + net/core/filter.c | 33 +- + net/ipv4/Makefile | 1 + + net/ipv4/af_inet.c | 1 + + net/ipv4/bpf_tcp_ca.c | 16 + + net/ipv4/bpf_tcp_ops.c | 326 ++ + net/ipv4/tcp.c | 1 + + net/ipv4/tcp_bpf.c | 1 - + net/ipv4/tcp_input.c | 17 + + net/ipv4/tcp_output.c | 104 +- + net/ipv4/tcp_timer.c | 1 + + net/sched/bpf_qdisc.c | 2 - + samples/bpf/Makefile | 1 + + samples/bpf/hash_func01.h | 55 - + security/bpf/hooks.c | 1 + + security/keys/process_keys.c | 25 + + tools/bpf/bpftool/Documentation/bpftool-btf.rst | 7 +- + tools/bpf/bpftool/Documentation/bpftool-cgroup.rst | 25 +- + tools/bpf/bpftool/Documentation/bpftool-map.rst | 14 +- + tools/bpf/bpftool/Documentation/bpftool-prog.rst | 25 +- + tools/bpf/bpftool/bash-completion/bpftool | 65 +- + tools/bpf/bpftool/btf.c | 337 +- + tools/bpf/bpftool/cgroup.c | 99 +- + tools/bpf/bpftool/common.c | 40 + + tools/bpf/bpftool/gen.c | 26 +- + tools/bpf/bpftool/main.c | 61 +- + tools/bpf/bpftool/main.h | 4 +- + tools/bpf/bpftool/map.c | 95 +- + tools/bpf/bpftool/map_perf_ring.c | 87 +- + tools/bpf/bpftool/prog.c | 302 +- + tools/bpf/bpftool/sign.c | 24 +- + tools/bpf/bpftool/skeleton/profiler.bpf.c | 29 +- + tools/include/uapi/linux/bpf.h | 82 +- + tools/include/uapi/linux/btf.h | 83 +- + tools/lib/bpf/bpf.c | 21 + + tools/lib/bpf/bpf.h | 27 +- + tools/lib/bpf/bpf_gen_internal.h | 13 +- + tools/lib/bpf/btf.c | 424 +- + tools/lib/bpf/btf.h | 50 + + tools/lib/bpf/btf_dump.c | 217 +- + tools/lib/bpf/btf_iter.c | 18 + + tools/lib/bpf/elf.c | 2 +- + tools/lib/bpf/gen_loader.c | 188 +- + tools/lib/bpf/libbpf.c | 670 ++- + tools/lib/bpf/libbpf.h | 41 +- + tools/lib/bpf/libbpf.map | 10 + + tools/lib/bpf/libbpf_internal.h | 6 +- + tools/lib/bpf/libbpf_probes.c | 11 +- + tools/lib/bpf/linker.c | 31 +- + tools/lib/bpf/skel_internal.h | 48 +- + tools/lib/bpf/usdt.c | 4 + + tools/testing/selftests/bpf/.gitignore | 1 - + tools/testing/selftests/bpf/Makefile | 758 +--- + tools/testing/selftests/bpf/Makefile.buildvars | 162 + + tools/testing/selftests/bpf/Makefile.runner | 125 + + tools/testing/selftests/bpf/Makefile.skel | 140 + + tools/testing/selftests/bpf/README.rst | 2 +- + tools/testing/selftests/bpf/bench.c | 6 + + .../testing/selftests/bpf/benchs/bench_libarena.c | 210 + + .../selftests/bpf/benchs/run_bench_libarena.sh | 31 + + tools/testing/selftests/bpf/bpf_kfuncs.h | 19 +- + .../selftests/bpf/bpftool_btf_dump_sorted.expected | 48 + + .../bpf/bpftool_btf_dump_unsorted.expected | 48 + + tools/testing/selftests/bpf/bpftool_helpers.c | 9 +- + tools/testing/selftests/bpf/bpftool_helpers.h | 1 + + tools/testing/selftests/bpf/btf_helpers.c | 45 +- + tools/testing/selftests/bpf/cgroup_helpers.c | 67 + + tools/testing/selftests/bpf/cgroup_helpers.h | 4 + + tools/testing/selftests/bpf/config | 6 +- + tools/testing/selftests/bpf/gen_bpf_skel.sh | 99 + + tools/testing/selftests/bpf/libarena/Makefile | 31 +- + .../bpf/libarena/benchs/bench_malloc.bpf.c | 51 + + .../bpf/libarena/include/bpf_arena_spin_lock.h | 2 +- + .../selftests/bpf/libarena/include/bpf_atomic.h | 2 +- + .../selftests/bpf/libarena/include/bpf_may_goto.h | 1 + + .../bpf/libarena/include/libarena/bitmap.h | 30 +- + .../bpf/libarena/include/libarena/common.h | 71 + + .../bpf/libarena/selftests/test_bitmap.bpf.c | 3 + + .../testing/selftests/bpf/libarena/src/asan.bpf.c | 21 +- + .../selftests/bpf/libarena/src/bitmap.bpf.c | 18 - + .../testing/selftests/bpf/libarena/src/buddy.bpf.c | 44 +- + .../selftests/bpf/libarena/src/common.bpf.c | 27 +- + tools/testing/selftests/bpf/network_helpers.c | 46 +- + tools/testing/selftests/bpf/network_helpers.h | 15 +- + .../selftests/bpf/prog_tests/aggregate_arg.c | 11 + + .../selftests/bpf/prog_tests/aggregate_ret.c | 53 + + .../testing/selftests/bpf/prog_tests/arena_memcg.c | 196 + + .../selftests/bpf/prog_tests/attach_probe.c | 4 +- + .../selftests/bpf/prog_tests/bpf_insn_array.c | 650 ++- + tools/testing/selftests/bpf/prog_tests/bpf_nf.c | 14 +- + .../testing/selftests/bpf/prog_tests/bpf_tcp_ops.c | 560 +++ + .../selftests/bpf/prog_tests/bpf_tcp_ops_hdr.c | 77 + + .../selftests/bpf/prog_tests/bpf_verif_scale.c | 2 +- + .../selftests/bpf/prog_tests/bpftool_btf_dump.c | 297 ++ + .../selftests/bpf/prog_tests/bpftool_ringbuf.c | 417 ++ + tools/testing/selftests/bpf/prog_tests/btf.c | 126 + + .../selftests/bpf/prog_tests/btf_dedup_split.c | 116 + + .../testing/selftests/bpf/prog_tests/btf_distill.c | 106 + + tools/testing/selftests/bpf/prog_tests/btf_dump.c | 400 +- + .../selftests/bpf/prog_tests/btf_field_iter.c | 31 +- + .../bpf/prog_tests/btf_module_allowlist.c | 162 + + tools/testing/selftests/bpf/prog_tests/btf_write.c | 34 + + tools/testing/selftests/bpf/prog_tests/call_rcu.c | 275 ++ + .../selftests/bpf/prog_tests/callx_func_ptr_map.c | 207 + + .../selftests/bpf/prog_tests/callx_rodata_lskel.c | 74 + + tools/testing/selftests/bpf/prog_tests/cb_refs.c | 48 - + .../bpf/prog_tests/cgroup_getset_retval.c | 2 +- + .../selftests/bpf/prog_tests/cgroup_mprog_opts.c | 71 +- + .../selftests/bpf/prog_tests/copy_from_user_bprm.c | 65 + + .../testing/selftests/bpf/prog_tests/core_reloc.c | 10 +- + .../testing/selftests/bpf/prog_tests/exceptions.c | 7 + + .../selftests/bpf/prog_tests/fexit_bpf2bpf.c | 22 +- + .../testing/selftests/bpf/prog_tests/file_reader.c | 15 + + .../bpf/prog_tests/flow_dissector_classification.c | 2 +- + .../selftests/bpf/prog_tests/global_data_init.c | 96 +- + tools/testing/selftests/bpf/prog_tests/kasan.c | 479 ++ + .../testing/selftests/bpf/prog_tests/kernel_flag.c | 22 +- + .../testing/selftests/bpf/prog_tests/kfunc_call.c | 2 +- + .../selftests/bpf/prog_tests/kprobe_multi_test.c | 3 +- + .../selftests/bpf/prog_tests/linked_externs.c | 120 + + .../testing/selftests/bpf/prog_tests/lirc_mode2.c | 334 ++ + .../bpf/prog_tests/lsm_inode_init_xattr.c | 399 ++ + .../selftests/bpf/prog_tests/lwt_ip_encap.c | 53 + + .../selftests/bpf/prog_tests/may_goto_far.c | 92 + + .../selftests/bpf/prog_tests/may_goto_priv_stack.c | 41 + + .../bpf/prog_tests/prog_tests_framework.c | 23 + + .../selftests/bpf/prog_tests/queue_stack_map.c | 29 + + .../selftests/bpf/prog_tests/signed_loader.c | 600 ++- + tools/testing/selftests/bpf/prog_tests/snprintf.c | 4 +- + tools/testing/selftests/bpf/prog_tests/stream.c | 553 +++ + .../bpf/prog_tests/struct_ops_private_stack.c | 31 + + tools/testing/selftests/bpf/prog_tests/tailcalls.c | 116 + + .../bpf/prog_tests/test_struct_ops_multi_args.c | 57 +- + .../selftests/bpf/prog_tests/test_task_work.c | 2 +- + tools/testing/selftests/bpf/prog_tests/timer.c | 33 + + tools/testing/selftests/bpf/prog_tests/usdt.c | 54 + + tools/testing/selftests/bpf/prog_tests/verifier.c | 29 + + .../selftests/bpf/prog_tests/verify_pkcs7_sig.c | 4 +- + .../selftests/bpf/prog_tests/xdp_adjust_frags.c | 3 + + .../selftests/bpf/prog_tests/xdp_devmap_attach.c | 5 + + .../selftests/bpf/prog_tests/xdp_metadata.c | 4 +- + tools/testing/selftests/bpf/prog_tests/xsk.c | 2 +- + .../selftests/bpf/progs/aggregate_arg_func.c | 188 + + .../selftests/bpf/progs/aggregate_arg_kfunc.c | 168 + + .../selftests/bpf/progs/aggregate_ret_func.c | 383 ++ + .../selftests/bpf/progs/aggregate_ret_kfunc.c | 203 + + .../bpf/progs/aggregate_ret_kfunc_arena.c | 121 + + .../selftests/bpf/progs/aggregate_ret_target.c | 29 + + tools/testing/selftests/bpf/progs/arena_kfunc.c | 20 +- + tools/testing/selftests/bpf/progs/arena_memcg.c | 23 + + .../selftests/bpf/progs/async_stack_depth.c | 75 + + .../testing/selftests/bpf/progs/bpf_iter_netlink.c | 4 +- + tools/testing/selftests/bpf/progs/bpf_iter_tcp4.c | 11 +- + tools/testing/selftests/bpf/progs/bpf_iter_tcp6.c | 11 +- + tools/testing/selftests/bpf/progs/bpf_iter_udp4.c | 4 +- + tools/testing/selftests/bpf/progs/bpf_iter_udp6.c | 4 +- + tools/testing/selftests/bpf/progs/bpf_iter_unix.c | 5 +- + tools/testing/selftests/bpf/progs/bpf_misc.h | 22 +- + tools/testing/selftests/bpf/progs/bpf_tcp_ops.c | 141 + + .../testing/selftests/bpf/progs/bpf_tcp_ops_hdr.c | 86 + + .../testing/selftests/bpf/progs/bpftool_ringbuf.c | 47 + + .../selftests/bpf/progs/btf__stack_arg_precision.c | 3 +- + .../bpf/progs/btf__verifier_stack_arg_order.c | 3 +- + .../selftests/bpf/progs/btf_module_allowlist.c | 13 + + tools/testing/selftests/bpf/progs/call_rcu.c | 110 + + tools/testing/selftests/bpf/progs/call_rcu_fail.c | 114 + + tools/testing/selftests/bpf/progs/callx_rodata.c | 75 + + tools/testing/selftests/bpf/progs/cb_refs.c | 5 + + .../bpf/{ => progs}/cgroup_getset_retval_hooks.h | 0 + .../selftests/bpf/progs/cgrp_kfunc_failure.c | 8 +- + .../selftests/bpf/progs/compute_live_registers.c | 36 + + .../selftests/bpf/progs/copy_from_user_bprm.c | 69 + + .../testing/selftests/bpf/progs/cpumask_failure.c | 4 +- + tools/testing/selftests/bpf/progs/dynptr_fail.c | 14 +- + .../selftests/bpf/progs/exceptions_dead_subprog.c | 40 + + .../testing/selftests/bpf/progs/exceptions_fail.c | 2 +- + tools/testing/selftests/bpf/progs/file_reader.c | 129 + + .../selftests/bpf/progs/freplace_ret_pair.c | 12 + + tools/testing/selftests/bpf/progs/irq.c | 17 +- + tools/testing/selftests/bpf/progs/iters.c | 388 +- + .../selftests/bpf/progs/iters_state_safety.c | 3 + + tools/testing/selftests/bpf/progs/iters_testmod.c | 7 +- + tools/testing/selftests/bpf/progs/kasan.c | 502 +++ + tools/testing/selftests/bpf/progs/kasan_harden.c | 52 + + tools/testing/selftests/bpf/progs/linked_arena1.c | 42 + + tools/testing/selftests/bpf/progs/linked_arena2.c | 22 + + tools/testing/selftests/bpf/progs/lirc_mode2.c | 32 + + tools/testing/selftests/bpf/progs/lsm.c | 5 +- + .../selftests/bpf/progs/lsm_inode_init_xattr.c | 98 + + .../bpf/progs/lsm_inode_init_xattr_budget.c | 37 + + .../bpf/progs/lsm_inode_init_xattr_value.c | 47 + + tools/testing/selftests/bpf/progs/map_kptr_fail.c | 10 +- + .../selftests/bpf/progs/may_goto_priv_stack.c | 52 + + .../selftests/bpf/progs/mem_rdonly_untrusted.c | 3 +- + .../selftests/bpf/progs/percpu_alloc_fail.c | 35 + + tools/testing/selftests/bpf/progs/preempt_lock.c | 26 + + tools/testing/selftests/bpf/progs/rbtree_fail.c | 4 +- + .../selftests/bpf/progs/refcounted_kptr_fail.c | 9 +- + .../selftests/bpf/progs/res_spin_lock_fail.c | 2 +- + tools/testing/selftests/bpf/progs/stack_arg.c | 3 +- + tools/testing/selftests/bpf/progs/stack_arg_fail.c | 13 +- + .../testing/selftests/bpf/progs/stack_arg_kfunc.c | 3 +- + .../selftests/bpf/progs/stack_arg_precision.c | 4 +- + tools/testing/selftests/bpf/progs/stream.c | 48 + + tools/testing/selftests/bpf/progs/stream_fail.c | 2 +- + .../selftests/bpf/progs/struct_ops_multi_args.c | 14 +- + .../bpf/progs/struct_ops_private_stack_fail.c | 47 +- + .../bpf/progs/struct_ops_private_stack_large.c | 51 + + .../selftests/bpf/progs/tailcall_freplace_multi.c | 27 + + .../selftests/bpf/progs/tailcall_large_stack.c | 62 + + .../selftests/bpf/progs/task_kfunc_failure.c | 10 +- + tools/testing/selftests/bpf/progs/task_work_fail.c | 2 +- + .../selftests/bpf/progs/test_global_func1.c | 65 + + .../selftests/bpf/progs/test_global_func5.c | 2 +- + .../bpf/progs/test_global_func_deep_stack.c | 33 +- + .../selftests/bpf/progs/test_global_percpu_data.c | 10 +- + .../selftests/bpf/progs/test_kfunc_dynptr_param.c | 2 +- + .../selftests/bpf/progs/test_lirc_mode2_kern.c | 26 - + tools/testing/selftests/bpf/progs/test_snprintf.c | 4 +- + tools/testing/selftests/bpf/progs/timer.c | 60 +- + tools/testing/selftests/bpf/progs/timer_failure.c | 29 + + .../selftests/bpf/progs/verifier_aggregate_arg.c | 200 + + .../selftests/bpf/progs/verifier_aggregate_ret.c | 179 + + tools/testing/selftests/bpf/progs/verifier_arena.c | 48 + + .../selftests/bpf/progs/verifier_arena_large.c | 64 +- + .../selftests/bpf/progs/verifier_bpf_fastcall.c | 169 + + tools/testing/selftests/bpf/progs/verifier_bswap.c | 3 +- + tools/testing/selftests/bpf/progs/verifier_callx.c | 1085 +++++ + .../selftests/bpf/progs/verifier_callx_rodata.c | 902 ++++ + tools/testing/selftests/bpf/progs/verifier_cfg.c | 79 + + tools/testing/selftests/bpf/progs/verifier_ctx.c | 2 +- + .../selftests/bpf/progs/verifier_global_ptr_args.c | 9 +- + .../selftests/bpf/progs/verifier_global_subprogs.c | 33 +- + tools/testing/selftests/bpf/progs/verifier_gotol.c | 1 + + tools/testing/selftests/bpf/progs/verifier_gotox.c | 127 +- + .../bpf/progs/verifier_helper_access_var_len.c | 336 +- + .../bpf/progs/verifier_helper_packet_access.c | 4 +- + .../selftests/bpf/progs/verifier_jit_inline.c | 2 +- + .../bpf/progs/verifier_kfunc_packet_access.c | 47 + + .../selftests/bpf/progs/verifier_kfunc_uninit.c | 236 + + .../bpf/progs/verifier_kfunc_uninit_multi.c | 113 + + .../selftests/bpf/progs/verifier_large_stack.c | 425 ++ + tools/testing/selftests/bpf/progs/verifier_ldsx.c | 36 +- + .../selftests/bpf/progs/verifier_live_stack.c | 137 +- + .../selftests/bpf/progs/verifier_load_acquire.c | 1 + + tools/testing/selftests/bpf/progs/verifier_lsm.c | 13 + + .../selftests/bpf/progs/verifier_lsm_init_xattr.c | 245 ++ + .../selftests/bpf/progs/verifier_map_in_map.c | 3 +- + .../bpf/progs/verifier_map_lookup_refine.c | 2 +- + .../selftests/bpf/progs/verifier_may_goto_1.c | 121 + + tools/testing/selftests/bpf/progs/verifier_movsx.c | 3 +- + tools/testing/selftests/bpf/progs/verifier_mtu.c | 88 + + .../selftests/bpf/progs/verifier_percpu_addr.c | 1 + + .../selftests/bpf/progs/verifier_private_stack.c | 1 + + .../selftests/bpf/progs/verifier_raw_stack.c | 25 + + .../selftests/bpf/progs/verifier_ref_tracking.c | 6 +- + tools/testing/selftests/bpf/progs/verifier_sdiv.c | 3 +- + tools/testing/selftests/bpf/progs/verifier_sock.c | 10 +- + .../selftests/bpf/progs/verifier_spill_fill.c | 43 + + .../selftests/bpf/progs/verifier_stack_arg.c | 4 +- + .../selftests/bpf/progs/verifier_stack_arg_order.c | 8 +- + .../selftests/bpf/progs/verifier_stack_ptr.c | 53 + + .../selftests/bpf/progs/verifier_store_release.c | 1 + + .../selftests/bpf/progs/verifier_tailcall.c | 57 + + .../testing/selftests/bpf/progs/verifier_unpriv.c | 37 + + .../testing/selftests/bpf/progs/verifier_var_off.c | 32 + + .../selftests/bpf/progs/verifier_vfs_reject.c | 6 +- + tools/testing/selftests/bpf/progs/verifier_xadd.c | 27 + + .../selftests/bpf/progs/wakeup_source_fail.c | 2 +- + tools/testing/selftests/bpf/progs/wq_failures.c | 4 +- + tools/testing/selftests/bpf/test_btf.h | 2 +- + .../testing/selftests/bpf/test_kmods/bpf_testmod.c | 303 +- + .../testing/selftests/bpf/test_kmods/bpf_testmod.h | 5 + + .../selftests/bpf/test_kmods/bpf_testmod_kfunc.h | 136 + + tools/testing/selftests/bpf/test_lirc_mode2.sh | 41 - + tools/testing/selftests/bpf/test_lirc_mode2_user.c | 177 - + tools/testing/selftests/bpf/test_loader.c | 86 +- + tools/testing/selftests/bpf/test_progs.h | 3 + + tools/testing/selftests/bpf/testing_helpers.c | 88 +- + tools/testing/selftests/bpf/testing_helpers.h | 4 + + tools/testing/selftests/bpf/unpriv_helpers.c | 35 +- + tools/testing/selftests/bpf/unpriv_helpers.h | 4 + + tools/testing/selftests/bpf/usdt_2.c | 16 + + tools/testing/selftests/bpf/verifier/basic_call.c | 2 +- + tools/testing/selftests/bpf/verifier/calls.c | 132 +- + tools/testing/selftests/bpf/verifier/map_kptr.c | 4 +- + tools/testing/selftests/bpf/verify_sig_setup.sh | 63 +- + tools/testing/selftests/bpf/veristat.c | 4 +- + tools/testing/selftests/bpf/vmtest.sh | 16 +- + tools/testing/selftests/bpf/xdp_hw_metadata.c | 2 +- + tools/testing/selftests/ublk/kublk.c | 2 +- + 414 files changed, 30643 insertions(+), 5567 deletions(-) + delete mode 100644 crypto/bpf_crypto_skcipher.c + delete mode 100644 include/linux/bpf_crypto.h + create mode 100644 kernel/bpf/keys.c + create mode 100644 net/ipv4/bpf_tcp_ops.c + delete mode 100644 samples/bpf/hash_func01.h + create mode 100644 tools/testing/selftests/bpf/Makefile.buildvars + create mode 100644 tools/testing/selftests/bpf/Makefile.runner + create mode 100644 tools/testing/selftests/bpf/Makefile.skel + create mode 100644 tools/testing/selftests/bpf/benchs/bench_libarena.c + create mode 100755 tools/testing/selftests/bpf/benchs/run_bench_libarena.sh + create mode 100644 tools/testing/selftests/bpf/bpftool_btf_dump_sorted.expected + create mode 100644 tools/testing/selftests/bpf/bpftool_btf_dump_unsorted.expected + create mode 100755 tools/testing/selftests/bpf/gen_bpf_skel.sh + create mode 100644 tools/testing/selftests/bpf/libarena/benchs/bench_malloc.bpf.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/aggregate_arg.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/aggregate_ret.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/arena_memcg.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/bpf_tcp_ops.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/bpf_tcp_ops_hdr.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/bpftool_btf_dump.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/bpftool_ringbuf.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/btf_module_allowlist.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/call_rcu.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/callx_func_ptr_map.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/callx_rodata_lskel.c + delete mode 100644 tools/testing/selftests/bpf/prog_tests/cb_refs.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/copy_from_user_bprm.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/kasan.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/linked_externs.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/lirc_mode2.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/lsm_inode_init_xattr.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/may_goto_far.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/may_goto_priv_stack.c + create mode 100644 tools/testing/selftests/bpf/progs/aggregate_arg_func.c + create mode 100644 tools/testing/selftests/bpf/progs/aggregate_arg_kfunc.c + create mode 100644 tools/testing/selftests/bpf/progs/aggregate_ret_func.c + create mode 100644 tools/testing/selftests/bpf/progs/aggregate_ret_kfunc.c + create mode 100644 tools/testing/selftests/bpf/progs/aggregate_ret_kfunc_arena.c + create mode 100644 tools/testing/selftests/bpf/progs/aggregate_ret_target.c + create mode 100644 tools/testing/selftests/bpf/progs/arena_memcg.c + create mode 100644 tools/testing/selftests/bpf/progs/bpf_tcp_ops.c + create mode 100644 tools/testing/selftests/bpf/progs/bpf_tcp_ops_hdr.c + create mode 100644 tools/testing/selftests/bpf/progs/bpftool_ringbuf.c + create mode 100644 tools/testing/selftests/bpf/progs/btf_module_allowlist.c + create mode 100644 tools/testing/selftests/bpf/progs/call_rcu.c + create mode 100644 tools/testing/selftests/bpf/progs/call_rcu_fail.c + create mode 100644 tools/testing/selftests/bpf/progs/callx_rodata.c + rename tools/testing/selftests/bpf/{ => progs}/cgroup_getset_retval_hooks.h (100%) + create mode 100644 tools/testing/selftests/bpf/progs/copy_from_user_bprm.c + create mode 100644 tools/testing/selftests/bpf/progs/exceptions_dead_subprog.c + create mode 100644 tools/testing/selftests/bpf/progs/freplace_ret_pair.c + create mode 100644 tools/testing/selftests/bpf/progs/kasan.c + create mode 100644 tools/testing/selftests/bpf/progs/kasan_harden.c + create mode 100644 tools/testing/selftests/bpf/progs/linked_arena1.c + create mode 100644 tools/testing/selftests/bpf/progs/linked_arena2.c + create mode 100644 tools/testing/selftests/bpf/progs/lirc_mode2.c + create mode 100644 tools/testing/selftests/bpf/progs/lsm_inode_init_xattr.c + create mode 100644 tools/testing/selftests/bpf/progs/lsm_inode_init_xattr_budget.c + create mode 100644 tools/testing/selftests/bpf/progs/lsm_inode_init_xattr_value.c + create mode 100644 tools/testing/selftests/bpf/progs/may_goto_priv_stack.c + create mode 100644 tools/testing/selftests/bpf/progs/struct_ops_private_stack_large.c + create mode 100644 tools/testing/selftests/bpf/progs/tailcall_freplace_multi.c + create mode 100644 tools/testing/selftests/bpf/progs/tailcall_large_stack.c + delete mode 100644 tools/testing/selftests/bpf/progs/test_lirc_mode2_kern.c + create mode 100644 tools/testing/selftests/bpf/progs/verifier_aggregate_arg.c + create mode 100644 tools/testing/selftests/bpf/progs/verifier_aggregate_ret.c + create mode 100644 tools/testing/selftests/bpf/progs/verifier_callx.c + create mode 100644 tools/testing/selftests/bpf/progs/verifier_callx_rodata.c + create mode 100644 tools/testing/selftests/bpf/progs/verifier_kfunc_packet_access.c + create mode 100644 tools/testing/selftests/bpf/progs/verifier_kfunc_uninit.c + create mode 100644 tools/testing/selftests/bpf/progs/verifier_kfunc_uninit_multi.c + create mode 100644 tools/testing/selftests/bpf/progs/verifier_large_stack.c + create mode 100644 tools/testing/selftests/bpf/progs/verifier_lsm_init_xattr.c + delete mode 100755 tools/testing/selftests/bpf/test_lirc_mode2.sh + delete mode 100644 tools/testing/selftests/bpf/test_lirc_mode2_user.c +$ git am -3 ../patches/0001-perf-Fixup-for-btf_vlan-API-change.patch +Applying: perf: Fixup for btf_vlan() API change +Using index info to reconstruct a base tree... +M tools/perf/builtin-trace.c +M tools/perf/util/btf.c +Falling back to patching base and 3-way merge... +Auto-merging tools/perf/builtin-trace.c +No changes -- Patch already applied. +Merging ipsec-next/master (014d795c73837 idpf: fix kernel-doc parameter descriptions) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/klassert/ipsec-next.git ipsec-next/master +Already up to date. +Merging mlx5-next/mlx5-next (00290b9ff59ef mlx5: Move data direct implementation to mlx5_core) +$ git merge -m Merge branch 'mlx5-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mellanox/linux.git mlx5-next/mlx5-next +CONFLICT (modify/delete): drivers/infiniband/hw/mlx5/data_direct.c deleted in mlx5-next/mlx5-next and modified in HEAD. Version HEAD of drivers/infiniband/hw/mlx5/data_direct.c left in tree. +Auto-merging drivers/infiniband/hw/mlx5/main.c +Auto-merging drivers/infiniband/hw/mlx5/mlx5_ib.h +Auto-merging drivers/infiniband/hw/mlx5/mr.c +Auto-merging drivers/infiniband/hw/mlx5/odp.c +Auto-merging drivers/infiniband/hw/mlx5/umr.c +Auto-merging drivers/net/ethernet/mellanox/mlx5/core/main.c +Auto-merging drivers/net/ethernet/mellanox/mlx5/core/mlx5_core.h +Auto-merging include/linux/mlx5/driver.h +Automatic merge failed; fix conflicts and then commit the result. +$ git rm -f drivers/infiniband/hw/mlx5/data_direct.c +rm 'drivers/infiniband/hw/mlx5/data_direct.c' +$ git commit --no-edit -v -a +[master 247e5713b737f] Merge branch 'mlx5-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mellanox/linux.git +$ git diff -M --stat --summary HEAD^.. + drivers/infiniband/hw/mlx5/Makefile | 1 - + drivers/infiniband/hw/mlx5/cmd.c | 21 -- + drivers/infiniband/hw/mlx5/cmd.h | 2 - + drivers/infiniband/hw/mlx5/data_direct.c | 223 ------------ + drivers/infiniband/hw/mlx5/data_direct.h | 23 -- + drivers/infiniband/hw/mlx5/main.c | 144 ++------ + drivers/infiniband/hw/mlx5/mlx5_ib.h | 18 +- + drivers/infiniband/hw/mlx5/mr.c | 7 +- + drivers/infiniband/hw/mlx5/odp.c | 2 +- + drivers/infiniband/hw/mlx5/std_types.c | 3 +- + drivers/infiniband/hw/mlx5/umr.c | 11 +- + drivers/net/ethernet/mellanox/mlx5/core/Makefile | 2 +- + .../net/ethernet/mellanox/mlx5/core/data_direct.c | 400 +++++++++++++++++++++ + drivers/net/ethernet/mellanox/mlx5/core/main.c | 15 + + .../net/ethernet/mellanox/mlx5/core/mlx5_core.h | 5 + + include/linux/mlx5/data_direct.h | 47 +++ + include/linux/mlx5/driver.h | 8 + + 17 files changed, 515 insertions(+), 417 deletions(-) + delete mode 100644 drivers/infiniband/hw/mlx5/data_direct.c + delete mode 100644 drivers/infiniband/hw/mlx5/data_direct.h + create mode 100644 drivers/net/ethernet/mellanox/mlx5/core/data_direct.c + create mode 100644 include/linux/mlx5/data_direct.h +Merging netfilter-next/main (87b80c2f6b05c net: txgbe: free the fixed-rate clock on cleanup) +$ git merge -m Merge branch 'main' of https://git.kernel.org/pub/scm/linux/kernel/git/netfilter/nf-next.git netfilter-next/main +Already up to date. +Merging ipvs-next/main (6ebcf5074cff0 net: openvswitch: don't schedule rebalancing if there are no datapaths) +$ git merge -m Merge branch 'main' of https://git.kernel.org/pub/scm/linux/kernel/git/horms/ipvs-next.git ipvs-next/main +Already up to date. +Merging bluetooth/master (036d4119079a7 Bluetooth: btusb: add ASUS 0b05:1825 to QCA Rome quirks) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/bluetooth/bluetooth-next.git bluetooth/master +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + .../bindings/net/bluetooth/brcm,bluetooth.yaml | 1 + + MAINTAINERS | 2 + + drivers/bluetooth/Kconfig | 15 + + drivers/bluetooth/Makefile | 3 + + drivers/bluetooth/btbcm.c | 7 +- + drivers/bluetooth/btintel.c | 57 +- + drivers/bluetooth/btintel.h | 14 + + drivers/bluetooth/btintel_pcie.c | 1543 ++++++- + drivers/bluetooth/btintel_pcie.h | 257 +- + drivers/bluetooth/btmrvl_main.c | 4 +- + drivers/bluetooth/btmtk.c | 104 +- + drivers/bluetooth/btmtk.h | 9 + + drivers/bluetooth/btmtksdio.c | 57 +- + drivers/bluetooth/btnxpuart.c | 4 +- + drivers/bluetooth/{btusb.c => btusb_main.c} | 371 +- + drivers/bluetooth/btusb_qcom.c | 4557 ++++++++++++++++++++ + drivers/bluetooth/btusb_qcom.h | 99 + + drivers/bluetooth/hci_bcm.c | 1 + + drivers/bluetooth/hci_bcsp.c | 26 +- + drivers/bluetooth/hci_h4.c | 123 +- + drivers/bluetooth/hci_h5.c | 11 +- + drivers/bluetooth/hci_ll.c | 2 +- + drivers/bluetooth/hci_serdev.c | 46 +- + drivers/bluetooth/hci_uart.h | 39 +- + drivers/bluetooth/virtio_bt.c | 21 +- + include/net/bluetooth/hci_core.h | 67 +- + include/net/bluetooth/hci_h4.h | 60 + + include/net/bluetooth/hci_mon.h | 2 + + include/net/bluetooth/l2cap.h | 102 +- + include/net/bluetooth/mgmt.h | 14 + + net/bluetooth/6lowpan.c | 44 +- + net/bluetooth/Makefile | 2 +- + net/bluetooth/bnep/core.c | 16 +- + net/bluetooth/hci_conn.c | 28 + + net/bluetooth/hci_core.c | 47 +- + net/bluetooth/hci_event.c | 186 +- + net/bluetooth/hci_h4.c | 160 + + net/bluetooth/hci_sock.c | 8 + + net/bluetooth/hci_sync.c | 64 +- + net/bluetooth/iso.c | 15 +- + net/bluetooth/l2cap_core.c | 284 +- + net/bluetooth/l2cap_sock.c | 118 +- + net/bluetooth/mgmt.c | 80 +- + net/bluetooth/rfcomm/core.c | 6 +- + net/bluetooth/rfcomm/sock.c | 5 +- + net/bluetooth/rfcomm/tty.c | 2 +- + net/bluetooth/sco.c | 10 +- + net/bluetooth/smp.c | 4 +- + scripts/context-analysis-suppression.txt | 1 + + 49 files changed, 7917 insertions(+), 781 deletions(-) + rename drivers/bluetooth/{btusb.c => btusb_main.c} (94%) + create mode 100644 drivers/bluetooth/btusb_qcom.c + create mode 100644 drivers/bluetooth/btusb_qcom.h + create mode 100644 include/net/bluetooth/hci_h4.h + create mode 100644 net/bluetooth/hci_h4.c +Merging wireless-next/for-next (f49defea7668d Merge git://git.kernel.org/pub/scm/linux/kernel/git/netdev/net) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/wireless/wireless-next.git wireless-next/for-next +Already up to date. +Merging ath-next/for-next (21b4248bfa0f4 Merge tag 'ath-next-20260927' of git://git.kernel.org/pub/scm/linux/kernel/git/ath/ath) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/ath/ath.git ath-next/for-next +Already up to date. +Merging iwlwifi-next/next (f57fea3e69a06 wifi: iwlwifi: mld: remove unused fields from the firmware API) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/iwlwifi/iwlwifi-next.git iwlwifi-next/next +Auto-merging drivers/net/wireless/intel/iwlwifi/iwl-nvm-parse.c +Auto-merging drivers/net/wireless/intel/iwlwifi/iwl-trans.c +Auto-merging drivers/net/wireless/intel/iwlwifi/mvm/debugfs.c +CONFLICT (content): Merge conflict in drivers/net/wireless/intel/iwlwifi/mvm/debugfs.c +Auto-merging drivers/net/wireless/intel/iwlwifi/mvm/fw.c +Resolved 'drivers/net/wireless/intel/iwlwifi/mvm/debugfs.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 7e3ec0733c571] Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/iwlwifi/iwlwifi-next.git +$ git diff -M --stat --summary HEAD^.. + drivers/net/wireless/intel/iwlwifi/Makefile | 7 +- + drivers/net/wireless/intel/iwlwifi/cfg/bz.c | 8 +- + drivers/net/wireless/intel/iwlwifi/cfg/dr.c | 2 +- + drivers/net/wireless/intel/iwlwifi/cfg/rf-fm.c | 4 + + drivers/net/wireless/intel/iwlwifi/cfg/sc.c | 2 +- + drivers/net/wireless/intel/iwlwifi/dvm/debugfs.c | 5 +- + drivers/net/wireless/intel/iwlwifi/dvm/devices.c | 32 +- + drivers/net/wireless/intel/iwlwifi/dvm/eeprom.c | 88 +- + drivers/net/wireless/intel/iwlwifi/dvm/led.c | 11 +- + drivers/net/wireless/intel/iwlwifi/dvm/lib.c | 2 +- + drivers/net/wireless/intel/iwlwifi/dvm/mac80211.c | 11 +- + drivers/net/wireless/intel/iwlwifi/dvm/main.c | 36 +- + drivers/net/wireless/intel/iwlwifi/dvm/power.c | 3 +- + drivers/net/wireless/intel/iwlwifi/dvm/rx.c | 19 +- + drivers/net/wireless/intel/iwlwifi/dvm/tt.c | 14 +- + drivers/net/wireless/intel/iwlwifi/dvm/tx.c | 3 +- + drivers/net/wireless/intel/iwlwifi/dvm/ucode.c | 3 +- + drivers/net/wireless/intel/iwlwifi/fw/acpi.c | 306 +++++-- + drivers/net/wireless/intel/iwlwifi/fw/acpi.h | 20 +- + drivers/net/wireless/intel/iwlwifi/fw/api/d3.h | 3 +- + .../net/wireless/intel/iwlwifi/fw/api/datapath.h | 31 +- + .../net/wireless/intel/iwlwifi/fw/api/mac-cfg.h | 38 +- + drivers/net/wireless/intel/iwlwifi/fw/api/power.h | 14 +- + drivers/net/wireless/intel/iwlwifi/fw/api/rx.h | 23 +- + drivers/net/wireless/intel/iwlwifi/fw/api/scan.h | 68 +- + drivers/net/wireless/intel/iwlwifi/fw/api/stats.h | 60 +- + drivers/net/wireless/intel/iwlwifi/fw/api/tx.h | 11 +- + drivers/net/wireless/intel/iwlwifi/fw/dbg-old.c | 120 +-- + drivers/net/wireless/intel/iwlwifi/fw/dbg.c | 147 ++-- + drivers/net/wireless/intel/iwlwifi/fw/dbg.h | 2 +- + drivers/net/wireless/intel/iwlwifi/fw/dump.c | 16 +- + drivers/net/wireless/intel/iwlwifi/fw/file.h | 2 + + drivers/net/wireless/intel/iwlwifi/fw/init.c | 6 +- + drivers/net/wireless/intel/iwlwifi/fw/pnvm.c | 6 +- + drivers/net/wireless/intel/iwlwifi/fw/regulatory.c | 76 +- + drivers/net/wireless/intel/iwlwifi/fw/regulatory.h | 26 +- + drivers/net/wireless/intel/iwlwifi/fw/runtime.h | 13 +- + drivers/net/wireless/intel/iwlwifi/fw/uefi.c | 412 ++++++++-- + drivers/net/wireless/intel/iwlwifi/fw/uefi.h | 26 + + drivers/net/wireless/intel/iwlwifi/iwl-csr.h | 13 +- + drivers/net/wireless/intel/iwlwifi/iwl-dbg-tlv.c | 54 +- + .../net/wireless/intel/iwlwifi/iwl-devtrace-io.h | 18 +- + drivers/net/wireless/intel/iwlwifi/iwl-drv.c | 9 +- + drivers/net/wireless/intel/iwlwifi/iwl-io.c | 448 ---------- + drivers/net/wireless/intel/iwlwifi/iwl-io.h | 107 --- + drivers/net/wireless/intel/iwlwifi/iwl-nvm-parse.c | 32 +- + drivers/net/wireless/intel/iwlwifi/iwl-scd.h | 22 +- + drivers/net/wireless/intel/iwlwifi/iwl-trans.c | 171 +++- + drivers/net/wireless/intel/iwlwifi/iwl-trans.h | 227 +++++- + drivers/net/wireless/intel/iwlwifi/mld/agg.c | 7 +- + drivers/net/wireless/intel/iwlwifi/mld/constants.h | 4 + + drivers/net/wireless/intel/iwlwifi/mld/d3.c | 5 + + drivers/net/wireless/intel/iwlwifi/mld/debugfs.c | 4 +- + .../net/wireless/intel/iwlwifi/mld/ftm-initiator.c | 2 +- + drivers/net/wireless/intel/iwlwifi/mld/fw.c | 19 +- + drivers/net/wireless/intel/iwlwifi/mld/hcmd.h | 2 + + drivers/net/wireless/intel/iwlwifi/mld/key.c | 6 + + drivers/net/wireless/intel/iwlwifi/mld/link.c | 14 +- + drivers/net/wireless/intel/iwlwifi/mld/mac80211.c | 6 + + drivers/net/wireless/intel/iwlwifi/mld/mld.c | 20 + + drivers/net/wireless/intel/iwlwifi/mld/mld.h | 6 + + drivers/net/wireless/intel/iwlwifi/mld/nan.c | 35 +- + drivers/net/wireless/intel/iwlwifi/mld/notif.c | 6 +- + drivers/net/wireless/intel/iwlwifi/mld/power.c | 2 +- + drivers/net/wireless/intel/iwlwifi/mld/ptp.c | 12 +- + .../net/wireless/intel/iwlwifi/mld/regulatory.c | 83 +- + .../net/wireless/intel/iwlwifi/mld/regulatory.h | 4 +- + drivers/net/wireless/intel/iwlwifi/mld/rx.c | 26 +- + drivers/net/wireless/intel/iwlwifi/mld/scan.c | 20 +- + drivers/net/wireless/intel/iwlwifi/mld/scan.h | 4 +- + drivers/net/wireless/intel/iwlwifi/mld/sta.c | 7 +- + drivers/net/wireless/intel/iwlwifi/mld/stats.c | 35 +- + .../net/wireless/intel/iwlwifi/mld/tests/Makefile | 2 +- + .../net/wireless/intel/iwlwifi/mld/tests/tx-gp2.c | 154 ++++ + drivers/net/wireless/intel/iwlwifi/mld/tlc.c | 184 +++-- + drivers/net/wireless/intel/iwlwifi/mld/tx.c | 131 +++ + drivers/net/wireless/intel/iwlwifi/mld/tx.h | 41 +- + drivers/net/wireless/intel/iwlwifi/mvm/d3.c | 89 +- + drivers/net/wireless/intel/iwlwifi/mvm/debugfs.c | 144 +--- + .../net/wireless/intel/iwlwifi/mvm/ftm-initiator.c | 2 +- + drivers/net/wireless/intel/iwlwifi/mvm/fw.c | 44 +- + drivers/net/wireless/intel/iwlwifi/mvm/led.c | 8 +- + drivers/net/wireless/intel/iwlwifi/mvm/mac-ctxt.c | 2 +- + drivers/net/wireless/intel/iwlwifi/mvm/mac80211.c | 6 +- + drivers/net/wireless/intel/iwlwifi/mvm/mvm.h | 3 +- + drivers/net/wireless/intel/iwlwifi/mvm/nvm.c | 15 + + drivers/net/wireless/intel/iwlwifi/mvm/ops.c | 9 +- + drivers/net/wireless/intel/iwlwifi/mvm/rxmq.c | 26 +- + drivers/net/wireless/intel/iwlwifi/mvm/scan.c | 6 +- + drivers/net/wireless/intel/iwlwifi/mvm/sta.c | 8 +- + drivers/net/wireless/intel/iwlwifi/mvm/tdls.c | 4 +- + .../net/wireless/intel/iwlwifi/mvm/time-event.c | 1 - + drivers/net/wireless/intel/iwlwifi/mvm/utils.c | 4 +- + .../net/wireless/intel/iwlwifi/pcie/ctxt-info-v2.c | 30 +- + .../net/wireless/intel/iwlwifi/pcie/ctxt-info.c | 5 +- + drivers/net/wireless/intel/iwlwifi/pcie/drv.c | 14 +- + .../intel/iwlwifi/pcie/{gen1_2 => }/internal.h | 136 ++-- + .../wireless/intel/iwlwifi/pcie/{gen1_2 => }/rx.c | 198 ++--- + .../intel/iwlwifi/pcie/{gen1_2 => }/trans-gen2.c | 96 +-- + .../intel/iwlwifi/pcie/{gen1_2 => }/trans.c | 899 ++++++++++----------- + .../intel/iwlwifi/pcie/{gen1_2 => }/tx-gen2.c | 6 +- + .../wireless/intel/iwlwifi/pcie/{gen1_2 => }/tx.c | 167 ++-- + drivers/net/wireless/intel/iwlwifi/pcie/utils.c | 91 ++- + drivers/net/wireless/intel/iwlwifi/pcie/utils.h | 46 +- + 104 files changed, 3372 insertions(+), 2305 deletions(-) + delete mode 100644 drivers/net/wireless/intel/iwlwifi/iwl-io.c + delete mode 100644 drivers/net/wireless/intel/iwlwifi/iwl-io.h + create mode 100644 drivers/net/wireless/intel/iwlwifi/mld/tests/tx-gp2.c + rename drivers/net/wireless/intel/iwlwifi/pcie/{gen1_2 => }/internal.h (92%) + rename drivers/net/wireless/intel/iwlwifi/pcie/{gen1_2 => }/rx.c (93%) + rename drivers/net/wireless/intel/iwlwifi/pcie/{gen1_2 => }/trans-gen2.c (87%) + rename drivers/net/wireless/intel/iwlwifi/pcie/{gen1_2 => }/trans.c (84%) + rename drivers/net/wireless/intel/iwlwifi/pcie/{gen1_2 => }/tx-gen2.c (99%) + rename drivers/net/wireless/intel/iwlwifi/pcie/{gen1_2 => }/tx.c (94%) +Merging wpan-next/master (a6bfdfcc6711d ieee802154: allow legacy LLSEC ADD/DEL ops to pass strict validation) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/wpan/wpan-next.git wpan-next/master +Already up to date. +Merging wpan-staging/staging (a6bfdfcc6711d ieee802154: allow legacy LLSEC ADD/DEL ops to pass strict validation) +$ git merge -m Merge branch 'staging' of https://git.kernel.org/pub/scm/linux/kernel/git/wpan/wpan-next.git wpan-staging/staging +Already up to date. +Merging mtd/mtd/next (112a666bd82e9 mtd: virt-concat: unlink discarded items before freeing) +$ git merge -m Merge branch 'mtd/next' of https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git mtd/mtd/next +Auto-merging drivers/mtd/chips/cfi_cmdset_0001.c +Auto-merging drivers/mtd/mtd_virt_concat.c +Auto-merging drivers/mtd/mtdconcat.c +Auto-merging drivers/mtd/mtdcore.c +Auto-merging drivers/mtd/nand/raw/cadence-nand-controller.c +Auto-merging drivers/mtd/nand/raw/gpmi-nand/gpmi-nand.c +Merge made by the 'ort' strategy. + drivers/mtd/chips/cfi_cmdset_0001.c | 5 +- + drivers/mtd/chips/cfi_cmdset_0002.c | 15 ++++ + drivers/mtd/devices/docg3.c | 20 ++--- + drivers/mtd/devices/powernv_flash.c | 2 +- + drivers/mtd/hyperbus/hbmc-am654.c | 7 +- + drivers/mtd/maps/Kconfig | 17 ++++ + drivers/mtd/maps/Makefile | 1 + + drivers/mtd/maps/amd76xrom.c | 2 + + drivers/mtd/maps/int0800.c | 115 ++++++++++++++++++++++++ + drivers/mtd/maps/l440gx.c | 2 - + drivers/mtd/maps/pci.c | 2 +- + drivers/mtd/mtd_blkdevs.c | 5 ++ + drivers/mtd/mtd_virt_concat.c | 2 + + drivers/mtd/mtdconcat.c | 2 +- + drivers/mtd/mtdcore.c | 2 +- + drivers/mtd/mtdsuper.c | 4 +- + drivers/mtd/nand/onenand/onenand_base.c | 2 +- + drivers/mtd/nand/raw/cadence-nand-controller.c | 2 +- + drivers/mtd/nand/raw/fsmc_nand.c | 2 +- + drivers/mtd/nand/raw/gpmi-nand/gpmi-nand.c | 2 +- + drivers/mtd/nand/raw/intel-nand-controller.c | 6 +- + drivers/mtd/nand/raw/loongson-nand-controller.c | 4 +- + drivers/mtd/nand/raw/lpc32xx_mlc.c | 10 +-- + drivers/mtd/nand/raw/lpc32xx_slc.c | 10 +-- + drivers/mtd/nand/raw/marvell_nand.c | 7 +- + drivers/mtd/nand/raw/mtk_nand.c | 2 +- + drivers/mtd/nand/raw/nand_base.c | 2 +- + drivers/mtd/nand/raw/nand_legacy.c | 4 +- + drivers/mtd/nand/raw/omap2.c | 10 +-- + drivers/mtd/nand/raw/sh_flctl.c | 8 +- + drivers/mtd/spi-nor/otp.c | 2 +- + include/linux/mtd/bbm.h | 2 +- + include/linux/mtd/rawnand.h | 2 +- + include/linux/mtd/sh_flctl.h | 4 +- + 34 files changed, 215 insertions(+), 69 deletions(-) + create mode 100644 drivers/mtd/maps/int0800.c +Merging nand/nand/next (55c5b6d5f59f5 mtd: nand: qpic_common: drop stray empty line from struct 'bam_transaction') +$ git merge -m Merge branch 'nand/next' of https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git nand/nand/next +Auto-merging drivers/mtd/nand/raw/lpc32xx_mlc.c +Auto-merging drivers/mtd/nand/raw/lpc32xx_slc.c +Auto-merging drivers/mtd/nand/raw/omap2.c +Auto-merging drivers/mtd/nand/spi/core.c +Merge made by the 'ort' strategy. + drivers/mtd/nand/raw/atmel/nand-controller.c | 6 +- + drivers/mtd/nand/raw/brcmnand/brcmnand.c | 2 + + drivers/mtd/nand/raw/fsl_ifc_nand.c | 34 +--- + drivers/mtd/nand/raw/internals.h | 1 + + drivers/mtd/nand/raw/lpc32xx_mlc.c | 21 +-- + drivers/mtd/nand/raw/lpc32xx_slc.c | 21 +-- + drivers/mtd/nand/raw/nand_esmt.c | 34 ++++ + drivers/mtd/nand/raw/nand_ids.c | 2 +- + drivers/mtd/nand/raw/ndfc.c | 28 ++- + .../mtd/nand/raw/nuvoton-ma35d1-nand-controller.c | 1 - + drivers/mtd/nand/raw/omap2.c | 4 +- + drivers/mtd/nand/raw/stm32_fmc2_nand.c | 8 +- + drivers/mtd/nand/raw/sunxi_nand.c | 37 ++-- + drivers/mtd/nand/spi/Makefile | 2 +- + drivers/mtd/nand/spi/core.c | 1 + + drivers/mtd/nand/spi/issi.c | 187 +++++++++++++++++++++ + drivers/mtd/nand/spi/winbond.c | 87 +++++++++- + include/linux/mtd/lpc32xx_mlc.h | 17 -- + include/linux/mtd/lpc32xx_slc.h | 17 -- + include/linux/mtd/nand-qpic-common.h | 1 - + include/linux/mtd/spinand.h | 3 +- + 21 files changed, 370 insertions(+), 144 deletions(-) + create mode 100644 drivers/mtd/nand/spi/issi.c + delete mode 100644 include/linux/mtd/lpc32xx_mlc.h + delete mode 100644 include/linux/mtd/lpc32xx_slc.h +Merging spi-nor/spi-nor/next (68d7115d77a87 mtd: spi-nor: sfdp: get the 1-1-8 and 1-8-8 page programs from 4BAIT) +$ git merge -m Merge branch 'spi-nor/next' of https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git spi-nor/spi-nor/next +Auto-merging drivers/mtd/spi-nor/core.c +Auto-merging drivers/mtd/spi-nor/otp.c +Merge made by the 'ort' strategy. + Documentation/driver-api/mtd/spi-nor.rst | 2 +- + drivers/mtd/spi-nor/atmel.c | 75 ++- + drivers/mtd/spi-nor/core.c | 1035 ++++++++++++------------------ + drivers/mtd/spi-nor/core.h | 144 +++-- + drivers/mtd/spi-nor/debugfs.c | 31 +- + drivers/mtd/spi-nor/everspin.c | 7 +- + drivers/mtd/spi-nor/gigadevice.c | 15 +- + drivers/mtd/spi-nor/issi.c | 29 +- + drivers/mtd/spi-nor/macronix.c | 73 ++- + drivers/mtd/spi-nor/micron-st.c | 70 +- + drivers/mtd/spi-nor/otp.c | 16 +- + drivers/mtd/spi-nor/sfdp.c | 255 ++++++-- + drivers/mtd/spi-nor/sfdp.h | 23 +- + drivers/mtd/spi-nor/spansion.c | 109 ++-- + drivers/mtd/spi-nor/sst.c | 16 +- + drivers/mtd/spi-nor/swp.c | 133 ++-- + drivers/mtd/spi-nor/sysfs.c | 4 +- + drivers/mtd/spi-nor/winbond.c | 320 +++++++-- + drivers/mtd/spi-nor/xmc.c | 1 + + include/linux/mtd/spi-nor.h | 10 +- + 20 files changed, 1310 insertions(+), 1058 deletions(-) +Merging crypto/master (4c3ca7c8c1522 hwrng: intel - return -ENOMEM when ioremap() fails) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/herbert/cryptodev-2.6.git crypto/master +Auto-merging Documentation/devicetree/bindings/trivial-devices.yaml +Auto-merging MAINTAINERS +Auto-merging arch/arm64/crypto/aes-neonbs-glue.c +Auto-merging drivers/crypto/hisilicon/sec2/sec_main.c +Auto-merging lib/Kconfig.debug +Auto-merging lib/tests/Makefile +Merge made by the 'ort' strategy. + Documentation/ABI/testing/sysfs-driver-qat_kpt | 3 +- + .../bindings/crypto/hisilicon,hip06-sec.yaml | 134 -- + .../devicetree/bindings/trivial-devices.yaml | 2 + + MAINTAINERS | 12 +- + arch/arm/crypto/aes-neonbs-glue.c | 2 +- + arch/arm64/boot/dts/hisilicon/hip07.dtsi | 280 ----- + arch/arm64/crypto/aes-neonbs-glue.c | 2 +- + arch/powerpc/crypto/vmx.c | 14 +- + crypto/algapi.c | 23 +- + crypto/asymmetric_keys/Kconfig | 12 + + crypto/asymmetric_keys/Makefile | 2 + + crypto/asymmetric_keys/pkcs7_parser.c | 6 + + crypto/asymmetric_keys/restrict.c | 8 +- + crypto/asymmetric_keys/verify_pefile.c | 5 + + crypto/asymmetric_keys/verify_pefile_test.c | 100 ++ + crypto/asymmetric_keys/x509_public_key.c | 6 +- + crypto/cast5_generic.c | 4 +- + crypto/cast6_generic.c | 3 +- + crypto/ecc.c | 33 +- + crypto/lskcipher.c | 6 +- + crypto/rsassa-pkcs1.c | 4 +- + crypto/tcrypt.c | 182 +-- + crypto/testmgr.c | 11 +- + crypto/testmgr.h | 327 ++++- + crypto/zstd.c | 40 +- + drivers/char/hw_random/amd-rng.c | 2 +- + drivers/char/hw_random/ba431-rng.c | 2 +- + drivers/char/hw_random/cctrng.c | 7 +- + drivers/char/hw_random/core.c | 1 + + drivers/char/hw_random/imx-rngc.c | 8 +- + drivers/char/hw_random/intel-rng.c | 15 +- + drivers/char/hw_random/jh7110-trng.c | 39 +- + drivers/char/hw_random/omap-rng.c | 8 +- + drivers/char/hw_random/omap3-rom-rng.c | 7 +- + drivers/char/hw_random/virtio-rng.c | 6 +- + drivers/crypto/allwinner/sun8i-ss/sun8i-ss-hash.c | 3 +- + drivers/crypto/amcc/crypto4xx_core.c | 35 +- + drivers/crypto/amcc/crypto4xx_trng.c | 15 +- + drivers/crypto/amcc/crypto4xx_trng.h | 6 +- + drivers/crypto/amlogic/amlogic-gxl-cipher.c | 25 +- + drivers/crypto/amlogic/amlogic-gxl-core.c | 48 +- + drivers/crypto/aspeed/aspeed-hace-crypto.c | 3 +- + drivers/crypto/atmel-aes.c | 4 +- + drivers/crypto/atmel-ecc.c | 12 +- + drivers/crypto/atmel-sha.c | 21 +- + drivers/crypto/atmel-tdes.c | 74 +- + drivers/crypto/caam/caampkc.c | 6 + + drivers/crypto/cavium/cpt/cptpf_main.c | 12 +- + drivers/crypto/cavium/cpt/cptvf_main.c | 4 +- + drivers/crypto/cavium/cpt/cptvf_mbox.c | 4 +- + drivers/crypto/cavium/nitrox/nitrox_main.c | 2 +- + drivers/crypto/ccp/ccp-dev-v3.c | 4 + + drivers/crypto/ccp/ccp-dev-v5.c | 4 + + drivers/crypto/ccp/ccp-dev.c | 2 +- + drivers/crypto/ccp/dbc.c | 4 +- + drivers/crypto/ccp/sfs.c | 9 +- + drivers/crypto/ccp/sp-platform.c | 4 +- + drivers/crypto/ccree/cc_aead.c | 4 +- + drivers/crypto/ccree/cc_buffer_mgr.c | 4 +- + drivers/crypto/ccree/cc_pm.c | 8 +- + drivers/crypto/chelsio/chcr_algo.c | 2 +- + drivers/crypto/gemini/sl3516-ce-cipher.c | 3 +- + drivers/crypto/gemini/sl3516-ce-rng.c | 6 +- + drivers/crypto/hisilicon/Kconfig | 14 - + drivers/crypto/hisilicon/Makefile | 1 - + drivers/crypto/hisilicon/hpre/hpre.h | 1 - + drivers/crypto/hisilicon/hpre/hpre_crypto.c | 28 +- + drivers/crypto/hisilicon/hpre/hpre_main.c | 49 +- + drivers/crypto/hisilicon/qm.c | 2 +- + drivers/crypto/hisilicon/sec/Makefile | 3 - + drivers/crypto/hisilicon/sec/sec_algs.c | 1122 ----------------- + drivers/crypto/hisilicon/sec/sec_drv.c | 1307 -------------------- + drivers/crypto/hisilicon/sec/sec_drv.h | 428 ------- + drivers/crypto/hisilicon/sec2/sec_main.c | 2 +- + drivers/crypto/hisilicon/zip/dae_main.c | 23 +- + drivers/crypto/hisilicon/zip/zip_main.c | 2 +- + drivers/crypto/inside-secure/eip93/eip93-aead.c | 3 +- + drivers/crypto/inside-secure/eip93/eip93-regs.h | 2 +- + drivers/crypto/inside-secure/safexcel_cipher.c | 16 +- + drivers/crypto/inside-secure/safexcel_hash.c | 6 +- + drivers/crypto/intel/ixp4xx/ixp4xx_crypto.c | 3 +- + drivers/crypto/intel/qat/qat_common/adf_cfg.c | 84 +- + .../crypto/intel/qat/qat_common/adf_gen4_vf_mig.c | 5 + + drivers/crypto/intel/qat/qat_common/adf_gen6_ras.c | 20 +- + drivers/crypto/intel/qat/qat_common/adf_gen6_ras.h | 24 +- + drivers/crypto/intel/qat/qat_common/adf_init.c | 24 +- + .../crypto/intel/qat/qat_common/adf_sysfs_kpt.c | 24 +- + drivers/crypto/intel/qat/qat_common/qat_algs.c | 18 +- + .../crypto/intel/qat/qat_common/qat_asym_algs.c | 38 +- + drivers/crypto/marvell/octeontx/otx_cptvf_algs.c | 10 +- + drivers/crypto/marvell/octeontx/otx_cptvf_mbox.c | 4 +- + drivers/crypto/marvell/octeontx2/otx2_cptvf_algs.c | 10 +- + drivers/crypto/mxs-dcp.c | 3 + + drivers/crypto/nx/nx-842.c | 2 + + drivers/crypto/omap-des.c | 2 + + drivers/crypto/omap-sham.c | 5 +- + drivers/crypto/padlock-aes.c | 22 +- + drivers/crypto/qce/aead.c | 11 +- + drivers/crypto/qce/sha.c | 9 +- + drivers/crypto/qce/skcipher.c | 9 +- + drivers/crypto/rockchip/rk3288_crypto.c | 4 +- + drivers/crypto/s5p-sss.c | 7 +- + drivers/crypto/sa2ul.c | 121 +- + drivers/crypto/starfive/jh7110-cryp.c | 20 +- + drivers/crypto/talitos.c | 21 +- + drivers/crypto/xilinx/zynqmp-aes-gcm.c | 14 +- + drivers/crypto/xilinx/zynqmp-sha.c | 2 +- + include/crypto/aes.h | 13 + + include/linux/ccp.h | 58 +- + include/linux/hisi_acc_qm.h | 6 +- + include/linux/psp-sev.h | 86 +- + include/linux/rhashtable-types.h | 74 +- + include/linux/rhashtable.h | 39 +- + include/uapi/linux/psp-sev.h | 26 +- + kernel/padata.c | 2 +- + lib/842/842_decompress.c | 7 +- + lib/Kconfig.debug | 15 + + lib/rhashtable.c | 75 +- + lib/tests/842_decompress_kunit.c | 144 +++ + lib/tests/Makefile | 1 + + 120 files changed, 1622 insertions(+), 4093 deletions(-) + delete mode 100644 Documentation/devicetree/bindings/crypto/hisilicon,hip06-sec.yaml + create mode 100644 crypto/asymmetric_keys/verify_pefile_test.c + delete mode 100644 drivers/crypto/hisilicon/sec/Makefile + delete mode 100644 drivers/crypto/hisilicon/sec/sec_algs.c + delete mode 100644 drivers/crypto/hisilicon/sec/sec_drv.c + delete mode 100644 drivers/crypto/hisilicon/sec/sec_drv.h + create mode 100644 lib/tests/842_decompress_kunit.c +Merging libcrypto/libcrypto-next (da4d5933e3034 lib/crypto: x86/aes-ctr: Remove some unreachable code) +$ git merge -m Merge branch 'libcrypto-next' of https://git.kernel.org/pub/scm/linux/kernel/git/ebiggers/linux.git libcrypto/libcrypto-next +Auto-merging MAINTAINERS +Auto-merging include/crypto/aes.h +Merge made by the 'ort' strategy. + Documentation/crypto/libcrypto-zeroization.rst | 150 +++ + Documentation/crypto/libcrypto.rst | 14 + + MAINTAINERS | 1 + + arch/riscv/crypto/Kconfig | 15 - + arch/riscv/crypto/Makefile | 4 - + arch/riscv/crypto/aes-riscv64-glue.c | 566 --------- + arch/riscv/crypto/aes-riscv64-zvkned.S | 312 ----- + arch/x86/crypto/Kconfig | 18 +- + arch/x86/crypto/Makefile | 10 +- + arch/x86/crypto/aesni-intel_asm.S | 1338 -------------------- + arch/x86/crypto/aesni-intel_glue.c | 792 +----------- + arch/x86/purgatory/Makefile | 3 +- + crypto/aes.c | 22 +- + fs/smb/client/smb2transport.c | 3 +- + include/crypto/aes-ccm.h | 22 +- + include/crypto/aes-gcm.h | 22 +- + include/crypto/aes-xts.h | 13 +- + include/crypto/aes.h | 18 + + include/crypto/blake2b.h | 9 + + include/crypto/blake2s.h | 9 + + include/crypto/md5.h | 19 + + include/crypto/sha1.h | 19 + + include/crypto/sha2.h | 73 ++ + include/crypto/sm3.h | 10 + + lib/crypto/Makefile | 14 + + lib/crypto/aes.c | 34 +- + lib/crypto/blake2b.c | 2 +- + lib/crypto/blake2s.c | 2 +- + lib/crypto/md5.c | 2 +- + .../riscv/crypto => lib/crypto/riscv}/aes-macros.S | 25 +- + .../crypto/riscv}/aes-riscv64-zvkned-zvbb-zvkg.S | 99 +- + .../crypto/riscv}/aes-riscv64-zvkned-zvkb.S | 23 +- + lib/crypto/riscv/aes-riscv64-zvkned.S | 305 ++++- + lib/crypto/riscv/aes.h | 244 +++- + lib/crypto/sha1.c | 2 +- + lib/crypto/sm3.c | 2 +- + lib/crypto/x86/aes-aesni.S | 831 ++++++++++-- + .../crypto => lib/crypto/x86}/aes-ctr-avx-x86_64.S | 117 +- + .../crypto => lib/crypto/x86}/aes-xts-avx-x86_64.S | 179 +-- + lib/crypto/x86/aes.h | 381 +++++- + security/keys/trusted-keys/trusted_tpm1.c | 2 +- + 41 files changed, 2233 insertions(+), 3493 deletions(-) + create mode 100644 Documentation/crypto/libcrypto-zeroization.rst + delete mode 100644 arch/riscv/crypto/aes-riscv64-glue.c + delete mode 100644 arch/riscv/crypto/aes-riscv64-zvkned.S + delete mode 100644 arch/x86/crypto/aesni-intel_asm.S + rename {arch/riscv/crypto => lib/crypto/riscv}/aes-macros.S (90%) + rename {arch/riscv/crypto => lib/crypto/riscv}/aes-riscv64-zvkned-zvbb-zvkg.S (74%) + rename {arch/riscv/crypto => lib/crypto/riscv}/aes-riscv64-zvkned-zvkb.S (93%) + rename {arch/x86/crypto => lib/crypto/x86}/aes-ctr-avx-x86_64.S (88%) + rename {arch/x86/crypto => lib/crypto/x86}/aes-xts-avx-x86_64.S (81%) +Merging drm/drm-next (845c5cc3697c5 Merge tag 'drm-xe-next-2026-10-01' of https://gitlab.freedesktop.org/drm/xe/kernel into drm-next) +$ git merge -m Merge branch 'drm-next' of https://gitlab.freedesktop.org/drm/kernel.git drm/drm-next +Auto-merging MAINTAINERS +Auto-merging drivers/accel/ivpu/ivpu_drv.c +Auto-merging drivers/accel/ivpu/ivpu_drv.h +Auto-merging drivers/accel/ivpu/ivpu_job.c +Auto-merging drivers/accel/ivpu/ivpu_mmu.c +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_debugfs.c +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_ring.c +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_userq.c +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_vm.c +Auto-merging drivers/gpu/drm/amd/amdgpu/vcn_v4_0_3.c +Auto-merging drivers/gpu/drm/amd/amdgpu/vcn_v5_0_1.c +Auto-merging drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c +Auto-merging drivers/gpu/drm/amd/display/dc/dml2_0/Makefile +Auto-merging drivers/gpu/drm/aspeed/aspeed_gfx_drv.c +Auto-merging drivers/gpu/drm/bridge/samsung-dsim.c +Auto-merging drivers/gpu/drm/clients/drm_fbdev_client.c +Auto-merging drivers/gpu/drm/drm_bridge.c +Auto-merging drivers/gpu/drm/drm_gpusvm.c +Auto-merging drivers/gpu/drm/i915/display/intel_cursor.c +Auto-merging drivers/gpu/drm/i915/display/intel_display_types.h +Auto-merging drivers/gpu/drm/i915/display/intel_dp_mst.c +Auto-merging drivers/gpu/drm/i915/display/intel_psr.c +Auto-merging drivers/gpu/drm/i915/display/intel_vrr.c +Auto-merging drivers/gpu/drm/i915/display/skl_universal_plane.c +Auto-merging drivers/gpu/drm/nouveau/nouveau_connector.c +CONFLICT (content): Merge conflict in drivers/gpu/drm/nouveau/nouveau_connector.c +Auto-merging drivers/gpu/drm/nouveau/nouveau_drm.c +Auto-merging drivers/gpu/drm/panthor/panthor_sched.c +Auto-merging drivers/gpu/drm/sti/sti_cursor.c +Auto-merging drivers/gpu/drm/sti/sti_hqvdp.c +Auto-merging drivers/gpu/drm/vc4/vc4_drv.c +Auto-merging drivers/gpu/drm/vc4/vc4_v3d.c +CONFLICT (content): Merge conflict in drivers/gpu/drm/vc4/vc4_v3d.c +Auto-merging drivers/gpu/drm/virtio/virtgpu_plane.c +Auto-merging drivers/gpu/drm/vmwgfx/vmwgfx_kms.c +Auto-merging drivers/gpu/drm/xe/regs/xe_gt_regs.h +Auto-merging drivers/gpu/drm/xe/xe_bo.c +Auto-merging drivers/gpu/drm/xe/xe_bo.h +Auto-merging drivers/gpu/drm/xe/xe_configfs.c +Auto-merging drivers/gpu/drm/xe/xe_guc_ads.c +Auto-merging drivers/gpu/drm/xe/xe_tlb_inval.c +Auto-merging drivers/gpu/drm/xe/xe_vm.c +Auto-merging drivers/gpu/drm/xe/xe_wa_oob.rules +CONFLICT (content): Merge conflict in drivers/gpu/drm/xe/xe_wa_oob.rules +Auto-merging rust/bindings/bindings_helper.h +Resolved 'drivers/gpu/drm/nouveau/nouveau_connector.c' using previous resolution. +Resolved 'drivers/gpu/drm/vc4/vc4_v3d.c' using previous resolution. +Resolved 'drivers/gpu/drm/xe/xe_wa_oob.rules' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 5062f2639114f] Merge branch 'drm-next' of https://gitlab.freedesktop.org/drm/kernel.git +$ git diff -M --stat --summary HEAD^.. + .../ABI/testing/sysfs-driver-intel-xe-gpu | 10 + + .../bindings/display/brcm,bcm2835-v3d.yaml | 3 + + .../bindings/display/bridge/analogix,dp.yaml | 19 +- + .../display/bridge/renesas,r8a779g0-dsc.yaml | 96 + + .../bindings/display/panel/ilitek,ili7836a.yaml | 58 + + .../bindings/display/panel/novatek,nt36532.yaml | 83 + + .../panel/samsung,s6e8aa5x01-ams561ra01.yaml | 2 +- + .../display/rockchip/rockchip,analogix-dp.yaml | 1 + + .../bindings/display/solomon,ssd1351.yaml | 42 + + Documentation/gpu/amdgpu/index.rst | 1 + + Documentation/gpu/amdgpu/ualink.rst | 75 + + Documentation/gpu/drm-kms-helpers.rst | 11 +- + Documentation/gpu/drm-kms.rst | 3 + + Documentation/gpu/drm-ras.rst | 39 + + Documentation/gpu/drm-uapi.rst | 93 +- + Documentation/gpu/todo.rst | 15 - + Documentation/gpu/xe/index.rst | 1 + + Documentation/gpu/xe/xe_migrate.rst | 3 + + Documentation/gpu/xe/xe_sigid.rst | 14 + + Documentation/netlink/specs/drm_ras.yaml | 80 + + MAINTAINERS | 55 +- + arch/arm/boot/dts/broadcom/bcm2835-common.dtsi | 1 + + drivers/accel/amdxdna/aie2_ctx.c | 9 +- + drivers/accel/amdxdna/aie2_message.c | 1 + + drivers/accel/amdxdna/amdxdna_gem.c | 460 +- + drivers/accel/amdxdna/amdxdna_gem.h | 6 +- + drivers/accel/ethosu/ethosu_job.c | 4 +- + drivers/accel/ivpu/ivpu_drv.c | 13 + + drivers/accel/ivpu/ivpu_drv.h | 5 +- + drivers/accel/ivpu/ivpu_hw.c | 22 + + drivers/accel/ivpu/ivpu_hw_ip.c | 6 +- + drivers/accel/ivpu/ivpu_job.c | 92 +- + drivers/accel/ivpu/ivpu_job.h | 5 +- + drivers/accel/ivpu/ivpu_mmu.c | 12 +- + drivers/accel/qaic/qaic_data.c | 8 +- + drivers/accel/qaic/qaic_debugfs.c | 1 + + drivers/accel/qaic/qaic_drv.c | 1 + + drivers/accel/qaic/qaic_ras.c | 1 + + drivers/accel/qaic/qaic_ssr.c | 1 + + drivers/accel/qaic/qaic_timesync.c | 1 + + drivers/accel/qaic/sahara.c | 1 + + drivers/dma-buf/dma-buf.c | 2 +- + drivers/dma-buf/dma-resv.c | 2 +- + drivers/firmware/efi/sysfb_efi.c | 9 - + drivers/gpu/buddy.c | 1342 +- + drivers/gpu/drm/Kconfig | 9 +- + drivers/gpu/drm/Kconfig.debug | 1 + + drivers/gpu/drm/Makefile | 3 +- + drivers/gpu/drm/adp/adp_drv.c | 2 +- + drivers/gpu/drm/amd/amdgpu/Makefile | 5 +- + drivers/gpu/drm/amd/amdgpu/amdgpu.h | 12 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_acp.c | 6 - + drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd.c | 31 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd.h | 8 +- + .../gpu/drm/amd/amdgpu/amdgpu_amdkfd_aldebaran.c | 3 +- + .../gpu/drm/amd/amdgpu/amdgpu_amdkfd_gc_9_4_3.c | 1 + + drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v10.c | 151 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v10.h | 6 +- + .../gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v10_3.c | 1 + + .../gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v12_1.c | 36 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v9.c | 14 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v9.h | 4 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gpuvm.c | 2 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_cs.c | 4 - + drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c | 2 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_debugfs.c | 15 + + drivers/gpu/drm/amd/amdgpu/amdgpu_dev_coredump.c | 8 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_device.c | 130 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_discovery.c | 340 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_discovery.h | 13 + + drivers/gpu/drm/amd/amdgpu/amdgpu_dma_buf.c | 5 + + drivers/gpu/drm/amd/amdgpu/amdgpu_drv.c | 31 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_gem.c | 9 + + drivers/gpu/drm/amd/amdgpu/amdgpu_gfx.c | 11 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_gfx.h | 4 + + drivers/gpu/drm/amd/amdgpu/amdgpu_gmc.c | 17 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_gmc.h | 3 + + drivers/gpu/drm/amd/amdgpu/amdgpu_ib.c | 124 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_ids.c | 37 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_ids.h | 27 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_ih.h | 2 + + drivers/gpu/drm/amd/amdgpu/amdgpu_imu.h | 1 + + drivers/gpu/drm/amd/amdgpu/amdgpu_ip.c | 68 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_ip.h | 20 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_irq.c | 35 + + drivers/gpu/drm/amd/amdgpu/amdgpu_irq.h | 11 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_isp.c | 6 - + drivers/gpu/drm/amd/amdgpu/amdgpu_job.c | 5 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_job.h | 2 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_jpeg.c | 39 + + drivers/gpu/drm/amd/amdgpu/amdgpu_jpeg.h | 2 + + drivers/gpu/drm/amd/amdgpu/amdgpu_kms.c | 11 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_lsdma.c | 20 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_lsdma.h | 4 + + drivers/gpu/drm/amd/amdgpu/amdgpu_mca.c | 6 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c | 99 + + drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h | 11 + + drivers/gpu/drm/amd/amdgpu/amdgpu_mmhub.h | 3 + + drivers/gpu/drm/amd/amdgpu/amdgpu_object.c | 11 + + drivers/gpu/drm/amd/amdgpu/amdgpu_object.h | 5 + + drivers/gpu/drm/amd/amdgpu/amdgpu_psp.c | 588 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_psp.h | 86 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_ras.c | 110 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_ras.h | 3 + + drivers/gpu/drm/amd/amdgpu/amdgpu_ras_eeprom.c | 8 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_res_cursor.h | 1 + + drivers/gpu/drm/amd/amdgpu/amdgpu_reset.c | 7 + + drivers/gpu/drm/amd/amdgpu/amdgpu_ring.c | 19 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_ring.h | 50 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_sdma.c | 169 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_sdma.h | 165 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_sdma_types.h | 163 + + drivers/gpu/drm/amd/amdgpu/amdgpu_trace.h | 45 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_trace_points.c | 1 + + drivers/gpu/drm/amd/amdgpu/amdgpu_ttm.c | 58 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_ttm.h | 4 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_ualink.c | 6539 ++ + drivers/gpu/drm/amd/amdgpu/amdgpu_ualink.h | 427 + + drivers/gpu/drm/amd/amdgpu/amdgpu_umc.c | 46 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_umc.h | 8 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_userq.c | 46 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_userq_fence.c | 68 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_virt.c | 42 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_virt.h | 2 + + drivers/gpu/drm/amd/amdgpu/amdgpu_vkms.c | 10 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_vm.c | 105 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_vm.h | 122 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_vm_cpu.c | 1 + + drivers/gpu/drm/amd/amdgpu/amdgpu_vm_internal.h | 146 + + drivers/gpu/drm/amd/amdgpu/amdgpu_vm_pt.c | 35 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_vm_sdma.c | 6 + + drivers/gpu/drm/amd/amdgpu/amdgpu_vm_tlb_fence.c | 4 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_vpe.c | 34 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_xcp.c | 18 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_xcp.h | 7 + + drivers/gpu/drm/amd/amdgpu/amdgv_sriovmsg.h | 49 +- + drivers/gpu/drm/amd/amdgpu/aqua_vanjaram.c | 42 +- + drivers/gpu/drm/amd/amdgpu/atom.c | 29 +- + drivers/gpu/drm/amd/amdgpu/cik.c | 7 - + drivers/gpu/drm/amd/amdgpu/cik_ih.c | 12 - + drivers/gpu/drm/amd/amdgpu/cik_sdma.c | 68 +- + drivers/gpu/drm/amd/amdgpu/cz_ih.c | 12 - + drivers/gpu/drm/amd/amdgpu/dce_v10_0.c | 6 - + drivers/gpu/drm/amd/amdgpu/dce_v6_0.c | 6 - + drivers/gpu/drm/amd/amdgpu/dce_v8_0.c | 6 - + drivers/gpu/drm/amd/amdgpu/gfx_v10_0.c | 44 +- + drivers/gpu/drm/amd/amdgpu/gfx_v11_0.c | 44 +- + drivers/gpu/drm/amd/amdgpu/gfx_v12_0.c | 44 +- + drivers/gpu/drm/amd/amdgpu/gfx_v12_1.c | 797 +- + drivers/gpu/drm/amd/amdgpu/gfx_v12_1_pkt.h | 39 - + drivers/gpu/drm/amd/amdgpu/gfx_v6_0.c | 26 +- + drivers/gpu/drm/amd/amdgpu/gfx_v7_0.c | 39 +- + drivers/gpu/drm/amd/amdgpu/gfx_v8_0.c | 153 +- + drivers/gpu/drm/amd/amdgpu/gfx_v9_0.c | 146 +- + drivers/gpu/drm/amd/amdgpu/gfx_v9_4_2.c | 111 +- + drivers/gpu/drm/amd/amdgpu/gfx_v9_4_2.h | 1 + + drivers/gpu/drm/amd/amdgpu/gfx_v9_4_3.c | 41 +- + drivers/gpu/drm/amd/amdgpu/gfxhub_v11_5_0.c | 9 +- + drivers/gpu/drm/amd/amdgpu/gfxhub_v12_0.c | 9 +- + drivers/gpu/drm/amd/amdgpu/gfxhub_v12_1.c | 41 +- + drivers/gpu/drm/amd/amdgpu/gfxhub_v1_0.c | 9 +- + drivers/gpu/drm/amd/amdgpu/gfxhub_v1_2.c | 3 + + drivers/gpu/drm/amd/amdgpu/gfxhub_v2_0.c | 9 +- + drivers/gpu/drm/amd/amdgpu/gfxhub_v2_1.c | 9 +- + drivers/gpu/drm/amd/amdgpu/gfxhub_v3_0.c | 9 +- + drivers/gpu/drm/amd/amdgpu/gfxhub_v3_0_3.c | 9 +- + drivers/gpu/drm/amd/amdgpu/gmc_v10_0.c | 12 +- + drivers/gpu/drm/amd/amdgpu/gmc_v11_0.c | 18 +- + drivers/gpu/drm/amd/amdgpu/gmc_v12_0.c | 59 +- + drivers/gpu/drm/amd/amdgpu/gmc_v12_1.c | 184 +- + drivers/gpu/drm/amd/amdgpu/gmc_v12_1.h | 2 + + drivers/gpu/drm/amd/amdgpu/gmc_v6_0.c | 7 +- + drivers/gpu/drm/amd/amdgpu/gmc_v7_0.c | 7 +- + drivers/gpu/drm/amd/amdgpu/gmc_v8_0.c | 19 +- + drivers/gpu/drm/amd/amdgpu/gmc_v9_0.c | 14 +- + drivers/gpu/drm/amd/amdgpu/iceland_ih.c | 12 - + drivers/gpu/drm/amd/amdgpu/ih_v6_0.c | 39 +- + drivers/gpu/drm/amd/amdgpu/ih_v6_1.c | 16 +- + drivers/gpu/drm/amd/amdgpu/ih_v7_0.c | 43 +- + drivers/gpu/drm/amd/amdgpu/imu_v12_1.c | 91 +- + drivers/gpu/drm/amd/amdgpu/jpeg_v2_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/jpeg_v2_5.c | 2 - + drivers/gpu/drm/amd/amdgpu/jpeg_v3_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/jpeg_v4_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/jpeg_v4_0_3.c | 2 +- + drivers/gpu/drm/amd/amdgpu/jpeg_v4_0_5.c | 2 +- + drivers/gpu/drm/amd/amdgpu/jpeg_v5_0_0.c | 2 +- + drivers/gpu/drm/amd/amdgpu/jpeg_v5_0_1.c | 4 +- + drivers/gpu/drm/amd/amdgpu/jpeg_v5_0_1.h | 8 - + drivers/gpu/drm/amd/amdgpu/jpeg_v5_0_2.c | 265 +- + drivers/gpu/drm/amd/amdgpu/jpeg_v5_3_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/lsdma_v7_1.c | 27 +- + drivers/gpu/drm/amd/amdgpu/mes_userqueue.c | 62 +- + drivers/gpu/drm/amd/amdgpu/mes_v11_0.c | 35 +- + drivers/gpu/drm/amd/amdgpu/mes_v12_0.c | 29 +- + drivers/gpu/drm/amd/amdgpu/mes_v12_1.c | 282 +- + drivers/gpu/drm/amd/amdgpu/mmhub_v4_2_0.c | 372 +- + drivers/gpu/drm/amd/amdgpu/navi10_ih.c | 11 +- + drivers/gpu/drm/amd/amdgpu/nbif_v6_3_1.c | 155 +- + drivers/gpu/drm/amd/amdgpu/nbio_v4_3.c | 2 +- + drivers/gpu/drm/amd/amdgpu/nbio_v6_3_2.c | 18 +- + drivers/gpu/drm/amd/amdgpu/nbio_v7_11_5.c | 351 + + drivers/gpu/drm/amd/amdgpu/nbio_v7_11_5.h | 32 + + drivers/gpu/drm/amd/amdgpu/nbio_v7_9.c | 1 + + drivers/gpu/drm/amd/amdgpu/nv.c | 6 - + drivers/gpu/drm/amd/amdgpu/psp_gfx_if.h | 162 + + drivers/gpu/drm/amd/amdgpu/psp_v15_0_8.c | 146 + + drivers/gpu/drm/amd/amdgpu/sdma_v2_4.c | 105 +- + drivers/gpu/drm/amd/amdgpu/sdma_v3_0.c | 113 +- + drivers/gpu/drm/amd/amdgpu/sdma_v4_0.c | 124 +- + drivers/gpu/drm/amd/amdgpu/sdma_v4_4_2.c | 250 +- + drivers/gpu/drm/amd/amdgpu/sdma_v5_0.c | 141 +- + drivers/gpu/drm/amd/amdgpu/sdma_v5_2.c | 133 +- + drivers/gpu/drm/amd/amdgpu/sdma_v6_0.c | 143 +- + drivers/gpu/drm/amd/amdgpu/sdma_v7_0.c | 142 +- + drivers/gpu/drm/amd/amdgpu/sdma_v7_1.c | 114 +- + drivers/gpu/drm/amd/amdgpu/si.c | 6 - + drivers/gpu/drm/amd/amdgpu/si_dma.c | 38 +- + drivers/gpu/drm/amd/amdgpu/si_ih.c | 1 - + drivers/gpu/drm/amd/amdgpu/soc15.c | 6 - + drivers/gpu/drm/amd/amdgpu/soc15_common.h | 6 +- + drivers/gpu/drm/amd/amdgpu/soc21.c | 6 - + drivers/gpu/drm/amd/amdgpu/soc24.c | 6 - + drivers/gpu/drm/amd/amdgpu/soc_v1_0.c | 493 +- + drivers/gpu/drm/amd/amdgpu/soc_v1_0.h | 3 + + drivers/gpu/drm/amd/amdgpu/ta_ras_if.h | 1 + + drivers/gpu/drm/amd/amdgpu/tonga_ih.c | 12 - + drivers/gpu/drm/amd/amdgpu/ualink_v1_0.c | 143 + + drivers/gpu/drm/amd/amdgpu/ualink_v1_0.h | 30 + + drivers/gpu/drm/amd/amdgpu/uvd_v3_1.c | 8 - + drivers/gpu/drm/amd/amdgpu/uvd_v4_2.c | 8 - + drivers/gpu/drm/amd/amdgpu/uvd_v5_0.c | 8 - + drivers/gpu/drm/amd/amdgpu/uvd_v6_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/vce_v1_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/vce_v2_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/vce_v3_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/vcn_v1_0.c | 3 +- + drivers/gpu/drm/amd/amdgpu/vcn_v2_0.c | 2 +- + drivers/gpu/drm/amd/amdgpu/vcn_v2_5.c | 3 +- + drivers/gpu/drm/amd/amdgpu/vcn_v3_0.c | 17 +- + drivers/gpu/drm/amd/amdgpu/vcn_v4_0.c | 24 +- + drivers/gpu/drm/amd/amdgpu/vcn_v4_0_3.c | 22 +- + drivers/gpu/drm/amd/amdgpu/vcn_v4_0_5.c | 24 +- + drivers/gpu/drm/amd/amdgpu/vcn_v5_0_0.c | 24 +- + drivers/gpu/drm/amd/amdgpu/vcn_v5_0_1.c | 20 +- + drivers/gpu/drm/amd/amdgpu/vcn_v5_0_2.c | 351 +- + drivers/gpu/drm/amd/amdgpu/vega10_ih.c | 7 - + drivers/gpu/drm/amd/amdgpu/vega20_ih.c | 7 - + drivers/gpu/drm/amd/amdgpu/vi.c | 6 - + drivers/gpu/drm/amd/amdkfd/cwsr_trap_handler.h | 90 +- + .../gpu/drm/amd/amdkfd/cwsr_trap_handler_gfx12.asm | 122 +- + drivers/gpu/drm/amd/amdkfd/kfd_device.c | 18 +- + .../gpu/drm/amd/amdkfd/kfd_device_queue_manager.c | 151 +- + .../gpu/drm/amd/amdkfd/kfd_device_queue_manager.h | 3 + + drivers/gpu/drm/amd/amdkfd/kfd_flat_memory.c | 6 + + drivers/gpu/drm/amd/amdkfd/kfd_int_process_v9.c | 23 + + drivers/gpu/drm/amd/amdkfd/kfd_migrate.c | 53 +- + drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_v12_1.c | 31 +- + drivers/gpu/drm/amd/amdkfd/kfd_packet_manager_v9.c | 3 +- + drivers/gpu/drm/amd/amdkfd/kfd_process.c | 46 +- + .../gpu/drm/amd/amdkfd/kfd_process_queue_manager.c | 10 +- + drivers/gpu/drm/amd/amdkfd/kfd_queue.c | 10 +- + drivers/gpu/drm/amd/amdkfd/kfd_svm.c | 7 +- + drivers/gpu/drm/amd/display/Kconfig | 12 +- + drivers/gpu/drm/amd/display/Makefile | 1 + + drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c | 454 +- + drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.h | 182 +- + .../amd/display/amdgpu_dm/amdgpu_dm_backlight.c | 33 +- + .../amd/display/amdgpu_dm/amdgpu_dm_backlight.h | 1 + + .../drm/amd/display/amdgpu_dm/amdgpu_dm_color.c | 85 +- + .../drm/amd/display/amdgpu_dm/amdgpu_dm_colorop.c | 137 +- + .../drm/amd/display/amdgpu_dm/amdgpu_dm_colorop.h | 29 + + .../amd/display/amdgpu_dm/amdgpu_dm_connector.c | 379 +- + .../amd/display/amdgpu_dm/amdgpu_dm_connector.h | 11 +- + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_crtc.c | 87 +- + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_crtc.h | 9 +- + .../drm/amd/display/amdgpu_dm/amdgpu_dm_cursor.c | 63 +- + .../drm/amd/display/amdgpu_dm/amdgpu_dm_cursor.h | 6 + + .../drm/amd/display/amdgpu_dm/amdgpu_dm_debugfs.c | 88 + + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_dmub.c | 47 +- + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_dmub.h | 15 + + .../drm/amd/display/amdgpu_dm/amdgpu_dm_freesync.c | 47 + + .../drm/amd/display/amdgpu_dm/amdgpu_dm_helpers.c | 32 + + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_irq.c | 8 + + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_irq.h | 9 + + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_ism.c | 21 +- + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_ism.h | 2 +- + .../amd/display/amdgpu_dm/amdgpu_dm_mst_types.c | 73 +- + .../amd/display/amdgpu_dm/amdgpu_dm_mst_types.h | 25 +- + .../drm/amd/display/amdgpu_dm/amdgpu_dm_plane.c | 109 +- + .../drm/amd/display/amdgpu_dm/amdgpu_dm_plane.h | 14 +- + .../drm/amd/display/amdgpu_dm/amdgpu_dm_services.c | 35 +- + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_wb.c | 72 +- + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_wb.h | 16 + + .../drm/amd/display/amdgpu_dm/tests/.kunitconfig | 1 - + .../gpu/drm/amd/display/amdgpu_dm/tests/Makefile | 4 +- + .../amdgpu_dm/tests/amdgpu_dm_backlight_test.c | 53 +- + .../amdgpu_dm/tests/amdgpu_dm_colorop_test.c | 232 +- + .../amdgpu_dm/tests/amdgpu_dm_connector_test.c | 4632 +- + .../display/amdgpu_dm/tests/amdgpu_dm_crtc_test.c | 813 +- + .../amdgpu_dm/tests/amdgpu_dm_cursor_test.c | 736 + + .../display/amdgpu_dm/tests/amdgpu_dm_dmub_test.c | 578 +- + .../amdgpu_dm/tests/amdgpu_dm_freesync_test.c | 429 + + .../amdgpu_dm/tests/amdgpu_dm_helpers_test.c | 1129 +- + .../display/amdgpu_dm/tests/amdgpu_dm_irq_test.c | 2308 +- + .../display/amdgpu_dm/tests/amdgpu_dm_ism_test.c | 436 +- + .../amdgpu_dm/tests/amdgpu_dm_mst_types_test.c | 2791 +- + .../display/amdgpu_dm/tests/amdgpu_dm_plane_test.c | 2660 +- + .../amdgpu_dm/tests/amdgpu_dm_services_test.c | 228 +- + .../amd/display/amdgpu_dm/tests/amdgpu_dm_test.c | 4282 +- + .../display/amdgpu_dm/tests/amdgpu_dm_wb_test.c | 334 + + drivers/gpu/drm/amd/display/dc/Makefile | 5 +- + .../drm/amd/display/dc/bios/command_table_helper.c | 7 +- + .../amd/display/dc/bios/command_table_helper2.c | 2 + + drivers/gpu/drm/amd/display/dc/clk_mgr/Makefile | 8 + + drivers/gpu/drm/amd/display/dc/clk_mgr/clk_mgr.c | 4 + + .../amd/display/dc/clk_mgr/dce112/dce112_clk_mgr.c | 12 +- + .../amd/display/dc/clk_mgr/dcn10/dcn10_clk_mgr.c | 181 + + .../amd/display/dc/clk_mgr/dcn10/dcn10_clk_mgr.h | 29 + + .../drm/amd/display/dc/clk_mgr/dcn10/rv1_clk_mgr.c | 9 +- + .../drm/amd/display/dc/clk_mgr/dcn10/rv2_clk_mgr.c | 6 +- + .../amd/display/dc/clk_mgr/dcn20/dcn20_clk_mgr.c | 12 +- + .../amd/display/dc/clk_mgr/dcn201/dcn201_clk_mgr.c | 6 +- + .../drm/amd/display/dc/clk_mgr/dcn21/rn_clk_mgr.c | 5 +- + .../amd/display/dc/clk_mgr/dcn30/dcn30_clk_mgr.c | 8 +- + .../drm/amd/display/dc/clk_mgr/dcn301/vg_clk_mgr.c | 8 +- + .../amd/display/dc/clk_mgr/dcn31/dcn31_clk_mgr.c | 10 +- + .../amd/display/dc/clk_mgr/dcn314/dcn314_clk_mgr.c | 10 +- + .../amd/display/dc/clk_mgr/dcn315/dcn315_clk_mgr.c | 14 +- + .../amd/display/dc/clk_mgr/dcn316/dcn316_clk_mgr.c | 10 +- + .../amd/display/dc/clk_mgr/dcn32/dcn32_clk_mgr.c | 18 +- + .../amd/display/dc/clk_mgr/dcn35/dcn35_clk_mgr.c | 12 +- + .../amd/display/dc/clk_mgr/dcn401/dcn401_clk_mgr.c | 19 +- + .../amd/display/dc/clk_mgr/dcn42/dcn42_clk_mgr.c | 33 +- + .../amd/display/dc/clk_mgr/dcn42/dcn42_clk_mgr.h | 5 + + .../amd/display/dc/clk_mgr/dcn42b/dcn42b_clk_mgr.c | 54 +- + .../gpu/drm/amd/display/dc/clk_mgr/dcn60/dalsmc.h | 345 +- + .../amd/display/dc/clk_mgr/dcn60/dcn60_clk_mgr.c | 341 +- + .../amd/display/dc/clk_mgr/dcn60/dcn60_clk_mgr.h | 17 +- + .../dc/clk_mgr/dcn60/dcn60_clk_mgr_smu_msg.c | 184 +- + .../dc/clk_mgr/dcn60/dcn60_clk_mgr_smu_msg.h | 167 +- + .../display/dc/clk_mgr/dcn60/dcn60_smu_driver_if.h | 78 + + drivers/gpu/drm/amd/display/dc/core/dc.c | 291 +- + drivers/gpu/drm/amd/display/dc/core/dc_debug.c | 2 + + .../gpu/drm/amd/display/dc/core/dc_hw_sequencer.c | 1035 +- + .../gpu/drm/amd/display/dc/core/dc_link_exports.c | 25 +- + drivers/gpu/drm/amd/display/dc/core/dc_resource.c | 243 +- + drivers/gpu/drm/amd/display/dc/core/dc_state.c | 6 +- + drivers/gpu/drm/amd/display/dc/core/dc_stream.c | 168 +- + drivers/gpu/drm/amd/display/dc/dc.h | 64 +- + drivers/gpu/drm/amd/display/dc/dc_dmub_srv.c | 281 +- + drivers/gpu/drm/amd/display/dc/dc_dmub_srv.h | 58 +- + drivers/gpu/drm/amd/display/dc/dc_edid_parser.c | 80 - + drivers/gpu/drm/amd/display/dc/dc_hw_types.h | 21 +- + drivers/gpu/drm/amd/display/dc/dc_memory_pool.c | 275 + + drivers/gpu/drm/amd/display/dc/dc_memory_pool.h | 107 + + drivers/gpu/drm/amd/display/dc/dc_probe.h | 1 + + drivers/gpu/drm/amd/display/dc/dc_stream.h | 14 +- + drivers/gpu/drm/amd/display/dc/dc_types.h | 6 +- + .../gpu/drm/amd/display/dc/dccg/dcn20/dcn20_dccg.c | 1 + + .../gpu/drm/amd/display/dc/dccg/dcn20/dcn20_dccg.h | 4 + + .../drm/amd/display/dc/dccg/dcn201/dcn201_dccg.c | 1 + + .../gpu/drm/amd/display/dc/dccg/dcn21/dcn21_dccg.c | 1 + + .../gpu/drm/amd/display/dc/dccg/dcn30/dcn30_dccg.c | 2 + + .../drm/amd/display/dc/dccg/dcn301/dcn301_dccg.c | 1 + + .../gpu/drm/amd/display/dc/dccg/dcn31/dcn31_dccg.c | 1 + + .../drm/amd/display/dc/dccg/dcn314/dcn314_dccg.c | 1 + + .../gpu/drm/amd/display/dc/dccg/dcn32/dcn32_dccg.c | 1 + + .../gpu/drm/amd/display/dc/dccg/dcn35/dcn35_dccg.c | 2 + + .../drm/amd/display/dc/dccg/dcn401/dcn401_dccg.c | 1 + + .../gpu/drm/amd/display/dc/dccg/dcn42/dcn42_dccg.c | 15 +- + .../gpu/drm/amd/display/dc/dccg/dcn42/dcn42_dccg.h | 2 +- + .../gpu/drm/amd/display/dc/dccg/dcn60/dcn60_dccg.c | 5 + + .../gpu/drm/amd/display/dc/dccg/dcn60/dcn60_dccg.h | 4 + + drivers/gpu/drm/amd/display/dc/dce/dce_abm.c | 2 + + drivers/gpu/drm/amd/display/dc/dce/dce_aux.c | 12 +- + .../gpu/drm/amd/display/dc/dce/dce_clock_source.c | 11 +- + drivers/gpu/drm/amd/display/dc/dce/dce_dmcu.c | 121 - + drivers/gpu/drm/amd/display/dc/dce/dmub_abm.c | 1 + + .../gpu/drm/amd/display/dc/dce/dmub_hw_lock_mgr.c | 10 +- + drivers/gpu/drm/amd/display/dc/dce/dmub_psr.c | 1 + + drivers/gpu/drm/amd/display/dc/dce/dmub_psr.h | 1 + + drivers/gpu/drm/amd/display/dc/dce/dmub_replay.c | 28 +- + drivers/gpu/drm/amd/display/dc/dce/dmub_replay.h | 3 + + drivers/gpu/drm/amd/display/dc/dcn20/dcn20_vmid.c | 2 +- + .../gpu/drm/amd/display/dc/dcn30/dcn30_cm_common.c | 8 +- + .../drm/amd/display/dc/dcn301/dcn301_panel_cntl.c | 1 + + .../drm/amd/display/dc/dcn31/dcn31_panel_cntl.c | 1 + + .../gpu/drm/amd/display/dc/dio/dcn10/dcn10_dio.c | 1 + + .../amd/display/dc/dio/dcn10/dcn10_link_encoder.c | 12 +- + .../display/dc/dio/dcn10/dcn10_stream_encoder.c | 2 + + .../amd/display/dc/dio/dcn20/dcn20_link_encoder.c | 2 + + .../display/dc/dio/dcn20/dcn20_stream_encoder.c | 2 + + .../display/dc/dio/dcn30/dcn30_dio_link_encoder.c | 2 + + .../dc/dio/dcn30/dcn30_dio_stream_encoder.c | 2 + + .../dc/dio/dcn301/dcn301_dio_link_encoder.c | 2 + + .../display/dc/dio/dcn31/dcn31_dio_link_encoder.c | 2 + + .../dc/dio/dcn314/dcn314_dio_stream_encoder.c | 2 + + .../display/dc/dio/dcn32/dcn32_dio_link_encoder.c | 2 + + .../dc/dio/dcn32/dcn32_dio_stream_encoder.c | 2 + + .../dc/dio/dcn321/dcn321_dio_link_encoder.c | 2 + + .../display/dc/dio/dcn35/dcn35_dio_link_encoder.c | 2 + + .../dc/dio/dcn35/dcn35_dio_stream_encoder.c | 2 + + .../dc/dio/dcn401/dcn401_dio_link_encoder.c | 2 + + .../dc/dio/dcn401/dcn401_dio_stream_encoder.c | 1 + + .../display/dc/dio/dcn42/dcn42_dio_link_encoder.c | 6 +- + .../dc/dio/dcn42/dcn42_dio_stream_encoder.c | 2 + + .../display/dc/dio/dcn60/dcn60_dio_link_encoder.c | 2 + + .../dc/dio/dcn60/dcn60_dio_stream_encoder.c | 2 + + .../display/dc/dio/virtual/virtual_link_encoder.c | 3 + + .../dc/dio/virtual/virtual_stream_encoder.c | 1 + + .../gpu/drm/amd/display/dc/dml/dcn10/dcn10_fpu.c | 1 - + .../gpu/drm/amd/display/dc/dml/dcn30/dcn30_fpu.c | 18 +- + .../amd/display/dc/dml/dcn30/display_mode_vba_30.c | 2 +- + .../amd/display/dc/dml/dcn31/display_mode_vba_31.c | 2 +- + .../gpu/drm/amd/display/dc/dml/dml1_frl_cap_chk.c | 8 +- + .../gpu/drm/amd/display/dc/dml/dsc/rc_calc_fpu.c | 1 - + drivers/gpu/drm/amd/display/dc/dml2_0/Makefile | 17 +- + .../amd/display/dc/dml2_0/dml21/dml21_wrapper.h | 106 - + .../dml2_0/dml21/inc/bounding_boxes/dcn42_soc_bb.h | 39 +- + .../dml21/inc/bounding_boxes/dcn42b_soc_bb.h | 37 +- + .../dml2_0/dml21/inc/bounding_boxes/dcn4_soc_bb.h | 27 +- + .../dml2_0/dml21/inc/bounding_boxes/dcn6_soc_bb.h | 12 + + .../dc/dml2_0/dml21/inc/dml_top_dchub_registers.h | 1 + + .../dml2_0/dml21/inc/dml_top_display_cfg_types.h | 1 + + .../dml2_0/dml21/inc/dml_top_soc_parameter_types.h | 28 +- + .../display/dc/dml2_0/dml21/inc/dml_top_types.h | 14 + + .../dc/dml2_0/dml21/src/dml2_cga/dml2_cga_dcn6.c | 4 +- + .../dc/dml2_0/dml21/src/dml2_core/dml2_core_dcn4.c | 4 + + .../dml21/src/dml2_core/dml2_core_dcn4_calcs.c | 222 +- + .../dml21/src/dml2_core/dml2_core_dcn4_calcs.h | 1 + + .../src/dml2_core/dml2_core_dcn5_calcs_dchub.c | 18 +- + .../src/dml2_core/dml2_core_dcn5_calcs_dchub.h | 23 + + .../dml2_core/dml2_core_dcn5_calcs_display_pipe.c | 2 +- + .../dml2_core/dml2_core_dcn5_funcs_initialize.c | 2 +- + .../dml2_core_dcn5_funcs_mode_programming.c | 4 + + .../dml2_core/dml2_core_dcn5_funcs_mode_support.c | 32 +- + .../dml21/src/dml2_core/dml2_core_dcn6_calcs.c | 63 +- + .../dml21/src/dml2_core/dml2_core_dcn6_calcs.h | 468 + + .../src/dml2_core/dml2_core_dcn6_calcs_dchub.c | 86 +- + .../dml2_core/dml2_core_dcn6_funcs_initialize.c | 4 + + .../dml2_core_dcn6_funcs_mode_programming.c | 52 +- + .../dml2_core_dcn6_funcs_mode_programming.h | 31 + + .../dml2_core/dml2_core_dcn6_funcs_mode_support.c | 350 +- + .../dml2_core/dml2_core_dcn6_funcs_mode_support.h | 324 + + .../dml21/src/dml2_core/dml2_core_shared_types.h | 11 +- + .../dc/dml2_0/dml21/src/dml2_dpmm/dml2_dpmm_dcn4.c | 106 +- + .../dml21/src/dml2_pmo/dml2_pmo_dcn4_fams2.c | 11 +- + .../alternate_pstate_shared_lib.c | 22 +- + .../dc/dml2_0/dml21/src/dml2_top/dml2_top_utm.c | 4 +- + .../src/dml2_utm_soc_bb/dml2_utm_soc_bb_dcn6.c | 15 +- + .../dml21/src/inc/dml2_internal_shared_types.h | 19 + + .../gpu/drm/amd/display/dc/dml2_wrapper/Makefile | 61 + + .../dml21_wrapper}/dml21_translation_helper.c | 46 +- + .../dml21_wrapper}/dml21_translation_helper.h | 7 +- + .../dml21_wrapper}/dml21_utils.c | 24 +- + .../dml21_wrapper}/dml21_utils.h | 7 +- + .../dml21_wrapper}/dml21_wrapper.c | 2 +- + .../dc/dml2_wrapper/dml21_wrapper/dml21_wrapper.h | 107 + + .../dml21_wrapper}/dml21_wrapper_fpu.c | 19 +- + .../dml21_wrapper}/dml21_wrapper_fpu.h | 7 +- + .../dml2_dc_resource_mgmt.c | 60 +- + .../dml2_dc_resource_mgmt.h | 0 + .../dc/{dml2_0 => dml2_wrapper}/dml2_dc_types.h | 0 + .../{dml2_0 => dml2_wrapper}/dml2_internal_types.h | 2 +- + .../{dml2_0 => dml2_wrapper}/dml2_mall_phantom.c | 70 +- + .../{dml2_0 => dml2_wrapper}/dml2_mall_phantom.h | 0 + .../dc/{dml2_0 => dml2_wrapper}/dml2_policy.c | 2 +- + .../dc/{dml2_0 => dml2_wrapper}/dml2_policy.h | 0 + .../dml2_translation_helper.c | 20 +- + .../dml2_translation_helper.h | 0 + .../dc/{dml2_0 => dml2_wrapper}/dml2_utils.c | 28 +- + .../dc/{dml2_0 => dml2_wrapper}/dml2_utils.h | 0 + .../dc/{dml2_0 => dml2_wrapper}/dml2_wrapper.c | 6 +- + .../dc/{dml2_0 => dml2_wrapper}/dml2_wrapper.h | 0 + .../dc/{dml2_0 => dml2_wrapper}/dml2_wrapper_fpu.c | 45 +- + .../dc/{dml2_0 => dml2_wrapper}/dml2_wrapper_fpu.h | 0 + .../drm/amd/display/dc/dpp/dcn30/dcn30_dpp_cm.c | 2 + + .../gpu/drm/amd/display/dc/dpp/dcn401/dcn401_dpp.h | 2 + + .../drm/amd/display/dc/dpp/dcn401/dcn401_dpp_cm.c | 19 + + .../gpu/drm/amd/display/dc/dpp/dcn50/dcn50_dpp.c | 20 +- + .../gpu/drm/amd/display/dc/dpp/dcn50/dcn50_dpp.h | 14 +- + .../gpu/drm/amd/display/dc/dpp/dcn60/dcn60_dpp.c | 1 + + .../gpu/drm/amd/display/dc/dpp/dcn60/dcn60_dpp.h | 7 +- + drivers/gpu/drm/amd/display/dc/dsc/dc_dsc.c | 8 + + .../gpu/drm/amd/display/dc/dsc/dcn60/dcn60_dsc.c | 4 +- + drivers/gpu/drm/amd/display/dc/gpio/hw_ddc.c | 7 +- + drivers/gpu/drm/amd/display/dc/gpio/hw_factory.c | 4 + + drivers/gpu/drm/amd/display/dc/gpio/hw_translate.c | 4 + + .../dc/hpo/dcn30/dcn30_hpo_frl_stream_encoder.c | 8 +- + .../dc/hpo/dcn401/dcn401_hpo_frl_stream_encoder.c | 2 + + .../dc/hpo/dcn42/dcn42_hpo_frl_stream_encoder.c | 11 +- + .../dc/hpo/dcn60/dcn60_hpo_frl_stream_encoder.c | 142 +- + .../drm/amd/display/dc/hubbub/dcn10/dcn10_hubbub.c | 2 + + .../drm/amd/display/dc/hubbub/dcn10/dcn10_hubbub.h | 11 +- + .../drm/amd/display/dc/hubbub/dcn20/dcn20_hubbub.c | 7 +- + .../amd/display/dc/hubbub/dcn201/dcn201_hubbub.c | 2 + + .../drm/amd/display/dc/hubbub/dcn21/dcn21_hubbub.c | 2 + + .../drm/amd/display/dc/hubbub/dcn30/dcn30_hubbub.c | 2 + + .../amd/display/dc/hubbub/dcn301/dcn301_hubbub.c | 2 + + .../drm/amd/display/dc/hubbub/dcn31/dcn31_hubbub.c | 6 +- + .../drm/amd/display/dc/hubbub/dcn32/dcn32_hubbub.c | 2 + + .../drm/amd/display/dc/hubbub/dcn35/dcn35_hubbub.c | 16 +- + .../amd/display/dc/hubbub/dcn401/dcn401_hubbub.c | 2 + + .../drm/amd/display/dc/hubbub/dcn42/dcn42_hubbub.c | 15 +- + .../drm/amd/display/dc/hubbub/dcn60/dcn60_hubbub.c | 241 +- + .../drm/amd/display/dc/hubbub/dcn60/dcn60_hubbub.h | 19 +- + .../gpu/drm/amd/display/dc/hubp/dcn10/dcn10_hubp.c | 4 +- + .../gpu/drm/amd/display/dc/hubp/dcn10/dcn10_hubp.h | 3 +- + .../gpu/drm/amd/display/dc/hubp/dcn20/dcn20_hubp.c | 5 +- + .../gpu/drm/amd/display/dc/hubp/dcn20/dcn20_hubp.h | 3 +- + .../gpu/drm/amd/display/dc/hubp/dcn21/dcn21_hubp.c | 11 +- + .../gpu/drm/amd/display/dc/hubp/dcn21/dcn21_hubp.h | 2 +- + .../gpu/drm/amd/display/dc/hubp/dcn30/dcn30_hubp.c | 7 +- + .../gpu/drm/amd/display/dc/hubp/dcn30/dcn30_hubp.h | 3 +- + .../gpu/drm/amd/display/dc/hubp/dcn31/dcn31_hubp.c | 1 + + .../drm/amd/display/dc/hubp/dcn401/dcn401_hubp.c | 114 +- + .../drm/amd/display/dc/hubp/dcn401/dcn401_hubp.h | 5 +- + .../gpu/drm/amd/display/dc/hubp/dcn42/dcn42_hubp.c | 6 +- + .../gpu/drm/amd/display/dc/hubp/dcn50/dcn50_hubp.c | 28 +- + .../gpu/drm/amd/display/dc/hubp/dcn50/dcn50_hubp.h | 3 +- + .../gpu/drm/amd/display/dc/hubp/dcn60/dcn60_hubp.c | 4 + + drivers/gpu/drm/amd/display/dc/hwss/Makefile | 6 + + .../gpu/drm/amd/display/dc/hwss/dce/dce_hwseq.c | 46 +- + .../gpu/drm/amd/display/dc/hwss/dce/dce_hwseq.h | 12 +- + .../drm/amd/display/dc/hwss/dce110/dce110_hwseq.c | 35 +- + .../drm/amd/display/dc/hwss/dce60/dce60_hwseq.c | 8 +- + .../drm/amd/display/dc/hwss/dce80/dce80_hwseq.c | 3 +- + .../drm/amd/display/dc/hwss/dcn10/dcn10_hwseq.c | 280 +- + .../drm/amd/display/dc/hwss/dcn10/dcn10_hwseq.h | 22 +- + .../gpu/drm/amd/display/dc/hwss/dcn10/dcn10_init.c | 6 +- + .../drm/amd/display/dc/hwss/dcn20/dcn20_hwseq.c | 291 +- + .../drm/amd/display/dc/hwss/dcn20/dcn20_hwseq.h | 26 +- + .../gpu/drm/amd/display/dc/hwss/dcn20/dcn20_init.c | 5 +- + .../drm/amd/display/dc/hwss/dcn201/dcn201_hwseq.c | 61 +- + .../drm/amd/display/dc/hwss/dcn201/dcn201_hwseq.h | 7 +- + .../drm/amd/display/dc/hwss/dcn201/dcn201_init.c | 6 +- + .../gpu/drm/amd/display/dc/hwss/dcn21/dcn21_init.c | 5 +- + .../drm/amd/display/dc/hwss/dcn30/dcn30_hwseq.c | 57 +- + .../drm/amd/display/dc/hwss/dcn30/dcn30_hwseq.h | 8 +- + .../gpu/drm/amd/display/dc/hwss/dcn30/dcn30_init.c | 5 +- + .../drm/amd/display/dc/hwss/dcn301/dcn301_init.c | 5 +- + .../gpu/drm/amd/display/dc/hwss/dcn31/dcn31_init.c | 5 +- + .../drm/amd/display/dc/hwss/dcn314/dcn314_hwseq.c | 8 +- + .../drm/amd/display/dc/hwss/dcn314/dcn314_init.c | 5 +- + .../drm/amd/display/dc/hwss/dcn32/dcn32_hwseq.c | 104 +- + .../drm/amd/display/dc/hwss/dcn32/dcn32_hwseq.h | 10 +- + .../gpu/drm/amd/display/dc/hwss/dcn32/dcn32_init.c | 5 +- + .../drm/amd/display/dc/hwss/dcn35/dcn35_hwseq.c | 128 +- + .../drm/amd/display/dc/hwss/dcn35/dcn35_hwseq.h | 14 +- + .../gpu/drm/amd/display/dc/hwss/dcn35/dcn35_init.c | 5 +- + .../drm/amd/display/dc/hwss/dcn351/dcn351_init.c | 5 +- + .../drm/amd/display/dc/hwss/dcn401/dcn401_hwseq.c | 314 +- + .../drm/amd/display/dc/hwss/dcn401/dcn401_hwseq.h | 24 +- + .../drm/amd/display/dc/hwss/dcn401/dcn401_init.c | 5 +- + .../drm/amd/display/dc/hwss/dcn42/dcn42_hwseq.c | 330 +- + .../drm/amd/display/dc/hwss/dcn42/dcn42_hwseq.h | 21 +- + .../gpu/drm/amd/display/dc/hwss/dcn42/dcn42_init.c | 10 +- + .../drm/amd/display/dc/hwss/dcn50/dcn50_hwseq.c | 19 +- + .../drm/amd/display/dc/hwss/dcn60/dcn60_hwseq.c | 244 +- + .../drm/amd/display/dc/hwss/dcn60/dcn60_hwseq.h | 5 +- + .../gpu/drm/amd/display/dc/hwss/dcn60/dcn60_init.c | 12 +- + drivers/gpu/drm/amd/display/dc/hwss/hw_sequencer.h | 259 +- + .../drm/amd/display/dc/hwss/hw_sequencer_private.h | 28 +- + drivers/gpu/drm/amd/display/dc/inc/clock_source.h | 4 +- + drivers/gpu/drm/amd/display/dc/inc/core_status.h | 2 + + drivers/gpu/drm/amd/display/dc/inc/core_types.h | 13 +- + drivers/gpu/drm/amd/display/dc/inc/custom_float.h | 21 +- + drivers/gpu/drm/amd/display/dc/inc/hw/abm.h | 1 + + drivers/gpu/drm/amd/display/dc/inc/hw/clk_mgr.h | 1 + + drivers/gpu/drm/amd/display/dc/inc/hw/dccg.h | 1 + + drivers/gpu/drm/amd/display/dc/inc/hw/dchubbub.h | 2 + + drivers/gpu/drm/amd/display/dc/inc/hw/dio.h | 1 + + drivers/gpu/drm/amd/display/dc/inc/hw/dmcu.h | 10 - + drivers/gpu/drm/amd/display/dc/inc/hw/dpp.h | 5 +- + drivers/gpu/drm/amd/display/dc/inc/hw/hubp.h | 6 +- + drivers/gpu/drm/amd/display/dc/inc/hw/hw_shared.h | 10 - + .../gpu/drm/amd/display/dc/inc/hw/link_encoder.h | 1 + + drivers/gpu/drm/amd/display/dc/inc/hw/mpc.h | 65 +- + drivers/gpu/drm/amd/display/dc/inc/hw/opp.h | 14 +- + drivers/gpu/drm/amd/display/dc/inc/hw/pg_cntl.h | 1 + + drivers/gpu/drm/amd/display/dc/inc/hw/rmcm.h | 140 + + .../gpu/drm/amd/display/dc/inc/hw/stream_encoder.h | 2 + + .../drm/amd/display/dc/inc/hw/timing_generator.h | 1 + + drivers/gpu/drm/amd/display/dc/inc/link_service.h | 9 +- + drivers/gpu/drm/amd/display/dc/inc/resource.h | 20 +- + drivers/gpu/drm/amd/display/dc/irq/Makefile | 2 + + .../amd/display/dc/irq/dce110/irq_service_dce110.c | 21 - + .../amd/display/dc/irq/dce110/irq_service_dce110.h | 9 - + drivers/gpu/drm/amd/display/dc/irq/irq_service.c | 28 + + drivers/gpu/drm/amd/display/dc/irq/irq_service.h | 9 + + .../amd/display/dc/link/hwss/link_hwss_hpo_dp.c | 23 +- + .../gpu/drm/amd/display/dc/link/link_detection.c | 4 +- + drivers/gpu/drm/amd/display/dc/link/link_factory.c | 7 +- + .../display/dc/link/protocols/link_dp_capability.c | 3 +- + .../dc/link/protocols/link_dp_panel_replay.c | 36 +- + .../dc/link/protocols/link_dp_panel_replay.h | 2 +- + .../dc/link/protocols/link_edp_panel_control.c | 103 +- + .../dc/link/protocols/link_edp_panel_control.h | 6 + + .../amd/display/dc/link/protocols/link_hdmi_frl.c | 32 +- + .../gpu/drm/amd/display/dc/mpc/dcn10/dcn10_mpc.c | 1 + + .../gpu/drm/amd/display/dc/mpc/dcn20/dcn20_mpc.c | 1 + + .../gpu/drm/amd/display/dc/mpc/dcn30/dcn30_mpc.c | 1 + + .../gpu/drm/amd/display/dc/mpc/dcn32/dcn32_mpc.c | 1 + + .../gpu/drm/amd/display/dc/mpc/dcn401/dcn401_mpc.c | 1 + + .../gpu/drm/amd/display/dc/mpc/dcn42/dcn42_mpc.c | 680 +- + .../gpu/drm/amd/display/dc/mpc/dcn42/dcn42_mpc.h | 714 +- + .../gpu/drm/amd/display/dc/mpc/dcn60/dcn60_mpc.c | 41 +- + .../gpu/drm/amd/display/dc/mpc/dcn60/dcn60_mpc.h | 238 - + .../gpu/drm/amd/display/dc/opp/dcn20/dcn20_opp.c | 2 + + .../gpu/drm/amd/display/dc/optc/dcn31/dcn31_optc.c | 10 + + .../gpu/drm/amd/display/dc/optc/dcn31/dcn31_optc.h | 1 + + .../drm/amd/display/dc/optc/dcn314/dcn314_optc.c | 1 + + .../gpu/drm/amd/display/dc/optc/dcn35/dcn35_optc.c | 1 + + .../gpu/drm/amd/display/dc/optc/dcn42/dcn42_optc.c | 1 + + drivers/gpu/drm/amd/display/dc/os_types.h | 19 + + .../drm/amd/display/dc/pg/dcn35/dcn35_pg_cntl.c | 1 + + .../drm/amd/display/dc/pg/dcn42/dcn42_pg_cntl.c | 5 + + drivers/gpu/drm/amd/display/dc/resource/Makefile | 4 + + .../display/dc/resource/dce112/dce112_resource.c | 62 - + .../amd/display/dc/resource/dcn32/dcn32_resource.c | 2 +- + .../amd/display/dc/resource/dcn35/dcn35_resource.c | 2 +- + .../display/dc/resource/dcn351/dcn351_resource.c | 2 +- + .../amd/display/dc/resource/dcn36/dcn36_resource.c | 2 +- + .../display/dc/resource/dcn401/dcn401_resource.c | 2 +- + .../amd/display/dc/resource/dcn42/dcn42_resource.c | 62 +- + .../display/dc/resource/dcn42b/dcn42b_resource.c | 129 +- + .../amd/display/dc/resource/dcn60/dcn60_resource.c | 64 +- + .../amd/display/dc/resource/dcn60/dcn60_resource.h | 9 +- + drivers/gpu/drm/amd/display/dc/rmcm/Makefile | 44 + + .../gpu/drm/amd/display/dc/rmcm/dcn42/dcn42_rmcm.c | 867 + + .../gpu/drm/amd/display/dc/rmcm/dcn42/dcn42_rmcm.h | 717 + + .../gpu/drm/amd/display/dc/rmcm/dcn60/dcn60_rmcm.c | 251 + + .../{dc_edid_parser.h => rmcm/dcn60/dcn60_rmcm.h} | 23 +- + .../dcn401/dcn401_soc_and_ip_translator.c | 6 +- + .../dcn60/dcn60_soc_and_ip_translator.c | 6 +- + drivers/gpu/drm/amd/display/dc/sspl/Makefile | 5 +- + drivers/gpu/drm/amd/display/dc/sspl/dc_spl.c | 130 +- + .../amd/display/dc/sspl/dc_spl_isharp_filters.c | 10 +- + .../amd/display/dc/sspl/dc_spl_scl_easf_filters.c | 88 +- + .../drm/amd/display/dc/sspl/dc_spl_scl_filters.c | 24 +- + drivers/gpu/drm/amd/display/dc/sspl/dc_spl_types.h | 60 +- + .../gpu/drm/amd/display/dc/sspl/spl_custom_float.c | 152 - + .../gpu/drm/amd/display/dc/sspl/spl_custom_float.h | 29 - + .../gpu/drm/amd/display/dc/sspl/spl_fixpt31_32.c | 495 - + .../gpu/drm/amd/display/dc/sspl/spl_fixpt31_32.h | 526 - + .../gpu/drm/amd/display/dc/sspl/spl_namespace.h | 17 + + drivers/gpu/drm/amd/display/dc/sspl/spl_os_types.h | 43 +- + drivers/gpu/drm/amd/display/dmub/dmub_srv.h | 60 +- + drivers/gpu/drm/amd/display/dmub/inc/dmub_cmd.h | 189 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn20.c | 74 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn20.h | 2 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn31.c | 82 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn31.h | 2 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn32.c | 82 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn32.h | 2 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn35.c | 82 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn35.h | 2 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn401.c | 86 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn401.h | 2 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn42.c | 86 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn42.h | 2 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn60.c | 86 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn60.h | 2 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_srv.c | 63 +- + drivers/gpu/drm/amd/display/include/dal_asic_id.h | 5 + + .../drm/amd/display/modules/color/color_gamma.c | 3 +- + .../drm/amd/display/modules/inc/mod_info_packet.h | 4 + + .../gpu/drm/amd/display/modules/inc/mod_power.h | 4 + + .../amd/display/modules/info_packet/info_packet.c | 109 + + .../drm/amd/display/modules/power/power_replay.c | 13 + + drivers/gpu/drm/amd/include/amd_shared.h | 6 +- + .../drm/amd/include/asic_reg/gc/gc_12_1_0_offset.h | 2 + + .../amd/include/asic_reg/gc/gc_12_1_0_sh_mask.h | 4 +- + .../amd/include/asic_reg/nbio/nbio_7_11_5_offset.h | 11074 ++++ + .../include/asic_reg/nbio/nbio_7_11_5_sh_mask.h | 63248 +++++++++++++++++++ + drivers/gpu/drm/amd/include/discovery.h | 179 +- + .../amd/include/ivsrcid/mpnht/irqsrcs_mpnht_15_0.h | 30 + + drivers/gpu/drm/amd/include/kgd_kfd_interface.h | 4 +- + drivers/gpu/drm/amd/include/kgd_pp_interface.h | 17 + + drivers/gpu/drm/amd/include/v12_structs.h | 2 +- + drivers/gpu/drm/amd/pm/amdgpu_dpm.c | 26 +- + drivers/gpu/drm/amd/pm/amdgpu_pm.c | 56 +- + drivers/gpu/drm/amd/pm/inc/amdgpu_dpm.h | 1 + + drivers/gpu/drm/amd/pm/legacy-dpm/kv_dpm.c | 6 - + drivers/gpu/drm/amd/pm/legacy-dpm/si_dpm.c | 7 - + drivers/gpu/drm/amd/pm/powerplay/amd_powerplay.c | 6 - + .../gpu/drm/amd/pm/powerplay/hwmgr/ppatomctrl.c | 3 + + drivers/gpu/drm/amd/pm/swsmu/amdgpu_smu.c | 49 +- + drivers/gpu/drm/amd/pm/swsmu/inc/amdgpu_smu.h | 23 +- + .../pm/swsmu/inc/pmfw_if/smu15_driver_if_v15_0_0.h | 46 - + .../amd/pm/swsmu/inc/pmfw_if/smu_v13_0_12_ppsmc.h | 1 + + drivers/gpu/drm/amd/pm/swsmu/inc/smu_types.h | 1 + + drivers/gpu/drm/amd/pm/swsmu/inc/smu_v15_0.h | 6 - + drivers/gpu/drm/amd/pm/swsmu/smu11/smu_v11_0.c | 2 - + drivers/gpu/drm/amd/pm/swsmu/smu13/smu_v13_0.c | 6 +- + .../gpu/drm/amd/pm/swsmu/smu13/smu_v13_0_0_ppt.c | 2 +- + .../gpu/drm/amd/pm/swsmu/smu13/smu_v13_0_12_ppt.c | 24 + + .../gpu/drm/amd/pm/swsmu/smu13/smu_v13_0_6_ppt.c | 10 +- + .../gpu/drm/amd/pm/swsmu/smu13/smu_v13_0_6_ppt.h | 1 + + drivers/gpu/drm/amd/pm/swsmu/smu14/smu_v14_0.c | 8 +- + drivers/gpu/drm/amd/pm/swsmu/smu15/smu_v15_0.c | 174 +- + .../gpu/drm/amd/pm/swsmu/smu15/smu_v15_0_0_ppt.c | 270 +- + .../gpu/drm/amd/pm/swsmu/smu15/smu_v15_0_8_ppt.c | 16 +- + drivers/gpu/drm/amd/ras/core/Makefile | 9 +- + drivers/gpu/drm/amd/ras/core/aca.c | 353 +- + drivers/gpu/drm/amd/ras/core/aca.h | 44 +- + drivers/gpu/drm/amd/ras/core/aca_v1_0.c | 41 +- + drivers/gpu/drm/amd/ras/core/aca_v5_0.c | 448 + + drivers/gpu/drm/amd/ras/core/aca_v5_0.h | 62 + + drivers/gpu/drm/amd/ras/core/cmd.c | 147 +- + drivers/gpu/drm/amd/ras/core/cmd.h | 32 +- + drivers/gpu/drm/amd/ras/core/core.c | 309 +- + drivers/gpu/drm/amd/ras/core/eeprom.c | 504 +- + drivers/gpu/drm/amd/ras/core/eeprom.h | 27 +- + drivers/gpu/drm/amd/ras/core/eeprom_fw.c | 640 +- + drivers/gpu/drm/amd/ras/core/eeprom_fw.h | 67 +- + drivers/gpu/drm/amd/ras/core/log_ring.c | 56 +- + drivers/gpu/drm/amd/ras/core/log_ring.h | 46 +- + drivers/gpu/drm/amd/ras/core/ras.h | 138 +- + drivers/gpu/drm/amd/ras/core/ras_bert.c | 546 + + drivers/gpu/drm/amd/ras/core/ras_bert.h | 33 + + drivers/gpu/drm/amd/ras/core/ras_cper.c | 844 +- + drivers/gpu/drm/amd/ras/core/ras_cper.h | 177 +- + drivers/gpu/drm/amd/ras/core/ras_eeprom_mgr.c | 418 + + drivers/gpu/drm/amd/ras/core/ras_eeprom_mgr.h | 124 + + drivers/gpu/drm/amd/ras/core/ras_gfx.c | 10 +- + drivers/gpu/drm/amd/ras/core/ras_mce.c | 151 + + drivers/gpu/drm/amd/ras/core/ras_mce.h | 51 + + drivers/gpu/drm/amd/ras/core/ras_mp1.c | 169 +- + drivers/gpu/drm/amd/ras/core/ras_mp1.h | 87 +- + drivers/gpu/drm/amd/ras/core/ras_mp1_v13_0.c | 175 +- + drivers/gpu/drm/amd/ras/core/ras_mp1_v15_0.c | 249 + + drivers/gpu/drm/amd/ras/core/ras_mp1_v15_0.h | 30 + + drivers/gpu/drm/amd/ras/core/ras_nbio.c | 9 +- + drivers/gpu/drm/amd/ras/core/ras_process.c | 36 +- + drivers/gpu/drm/amd/ras/core/ras_psp.c | 599 +- + drivers/gpu/drm/amd/ras/core/ras_psp.h | 111 +- + drivers/gpu/drm/amd/ras/core/ras_psp_v13_0.c | 76 + + drivers/gpu/drm/amd/ras/core/ras_psp_v15_0.c | 133 + + drivers/gpu/drm/amd/ras/core/ras_psp_v15_0.h | 31 + + drivers/gpu/drm/amd/ras/core/ras_umc.c | 635 +- + drivers/gpu/drm/amd/ras/core/ras_umc.h | 52 +- + drivers/gpu/drm/amd/ras/core/ras_umc_v12_0.c | 143 +- + drivers/gpu/drm/amd/ras/core/ras_umc_v12_0.h | 3 - + drivers/gpu/drm/amd/ras/core/ras_umc_v15_0.c | 220 + + drivers/gpu/drm/amd/ras/core/ras_umc_v15_0.h | 69 + + drivers/gpu/drm/amd/ras/core/ta_if.h | 35 + + drivers/gpu/drm/amd/ras/ras_mgr/Makefile | 6 +- + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_bert.c | 191 + + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_bert.h | 30 + + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_cmd.c | 38 + + .../drm/amd/ras/ras_mgr/amdgpu_ras_eeprom_i2c.c | 141 +- + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mce.c | 267 + + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mce.h | 31 + + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mgr.c | 357 +- + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mgr.h | 8 +- + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mp1.c | 105 + + .../{amdgpu_ras_mp1_v13_0.h => amdgpu_ras_mp1.h} | 8 +- + .../gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mp1_v13_0.c | 154 - + .../gpu/drm/amd/ras/ras_mgr/amdgpu_ras_process.c | 9 +- + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_sys.c | 100 +- + .../gpu/drm/amd/ras/ras_mgr/amdgpu_virt_ras_cmd.c | 42 +- + drivers/gpu/drm/amd/ras/ras_mgr/ras_sys.h | 9 +- + .../gpu/drm/arm/display/include/malidp_product.h | 10 +- + drivers/gpu/drm/arm/display/komeda/komeda_crtc.c | 27 +- + drivers/gpu/drm/arm/display/komeda/komeda_plane.c | 18 +- + .../drm/arm/display/komeda/komeda_wb_connector.c | 33 + + drivers/gpu/drm/arm/hdlcd_crtc.c | 4 +- + drivers/gpu/drm/arm/malidp_crtc.c | 19 +- + drivers/gpu/drm/arm/malidp_planes.c | 18 +- + drivers/gpu/drm/armada/armada_crtc.c | 2 +- + drivers/gpu/drm/armada/armada_overlay.c | 39 +- + drivers/gpu/drm/armada/armada_plane.c | 15 +- + drivers/gpu/drm/armada/armada_plane.h | 2 +- + drivers/gpu/drm/aspeed/aspeed_gfx.h | 11 +- + drivers/gpu/drm/aspeed/aspeed_gfx_crtc.c | 203 +- + drivers/gpu/drm/aspeed/aspeed_gfx_drv.c | 3 +- + drivers/gpu/drm/ast/ast_dp.c | 24 +- + drivers/gpu/drm/ast/ast_mode.c | 20 +- + drivers/gpu/drm/atmel-hlcdc/atmel_hlcdc_crtc.c | 19 +- + drivers/gpu/drm/atmel-hlcdc/atmel_hlcdc_plane.c | 33 +- + drivers/gpu/drm/bridge/Kconfig | 7 +- + drivers/gpu/drm/bridge/analogix/analogix_dp_core.c | 88 +- + drivers/gpu/drm/bridge/analogix/analogix_dp_core.h | 4 +- + drivers/gpu/drm/bridge/analogix/analogix_dp_reg.c | 15 +- + drivers/gpu/drm/bridge/analogix/analogix_dp_reg.h | 4 + + .../gpu/drm/bridge/cadence/cdns-mhdp8546-core.c | 1 - + drivers/gpu/drm/bridge/chipone-icn6211.c | 8 +- + drivers/gpu/drm/bridge/ite-it6505.c | 4 +- + drivers/gpu/drm/bridge/lontium-lt8713sx.c | 2 +- + drivers/gpu/drm/bridge/lontium-lt9211.c | 2 +- + drivers/gpu/drm/bridge/lontium-lt9611.c | 6 +- + drivers/gpu/drm/bridge/lontium-lt9611uxc.c | 2 +- + drivers/gpu/drm/bridge/samsung-dsim.c | 4 +- + drivers/gpu/drm/bridge/sil-sii8620.c | 3 +- + drivers/gpu/drm/bridge/synopsys/dw-dp.c | 4 +- + drivers/gpu/drm/bridge/synopsys/dw-hdmi-qp.c | 209 +- + drivers/gpu/drm/bridge/synopsys/dw-hdmi.c | 2 +- + drivers/gpu/drm/bridge/tc358767.c | 4 +- + drivers/gpu/drm/bridge/ti-sn65dsi83.c | 79 +- + drivers/gpu/drm/clients/drm_fbdev_client.c | 23 +- + drivers/gpu/drm/clients/drm_log.c | 54 +- + drivers/gpu/drm/display/drm_bridge_connector.c | 186 +- + drivers/gpu/drm/display/drm_hdmi_helper.c | 304 + + drivers/gpu/drm/display/drm_hdmi_state_helper.c | 239 +- + drivers/gpu/drm/display/drm_scdc_helper.c | 359 +- + drivers/gpu/drm/drm_atomic.c | 36 +- + drivers/gpu/drm/drm_atomic_helper.c | 10 +- + drivers/gpu/drm/drm_atomic_state_helper.c | 86 - + drivers/gpu/drm/drm_atomic_uapi.c | 7 + + drivers/gpu/drm/drm_bridge.c | 27 +- + drivers/gpu/drm/drm_bridge_helper.c | 2 - + drivers/gpu/drm/drm_buddy.c | 3 +- + drivers/gpu/drm/drm_client_event.c | 18 + + drivers/gpu/drm/drm_colorop.c | 107 + + drivers/gpu/drm/drm_connector.c | 153 +- + drivers/gpu/drm/drm_crtc_internal.h | 2 - + drivers/gpu/drm/drm_debugfs.c | 157 - + drivers/gpu/drm/drm_drv.c | 34 +- + drivers/gpu/drm/drm_edid.c | 237 +- + drivers/gpu/drm/drm_gem.c | 19 +- + drivers/gpu/drm/drm_gem_atomic_helper.c | 100 +- + drivers/gpu/drm/drm_gpusvm.c | 465 +- + drivers/gpu/drm/drm_kms_helper_common.c | 14 + + drivers/gpu/drm/drm_mode_config.c | 21 +- + drivers/gpu/drm/drm_modeset_helper.c | 2 + + drivers/gpu/drm/drm_of.c | 38 +- + drivers/gpu/drm/drm_pagemap.c | 18 +- + drivers/gpu/drm/drm_panic.c | 927 +- + drivers/gpu/drm/drm_panic_helper.c | 919 + + .../{drm_panic_qr.rs => drm_panic_helper_qr.rs} | 4 +- + drivers/gpu/drm/drm_panic_internal.h | 69 + + drivers/gpu/drm/drm_probe_helper.c | 15 +- + drivers/gpu/drm/drm_ras.c | 278 +- + drivers/gpu/drm/drm_ras_nl.c | 33 + + drivers/gpu/drm/drm_ras_nl.h | 8 + + drivers/gpu/drm/drm_simple_kms_helper.c | 37 +- + drivers/gpu/drm/drm_sysfs.c | 32 - + drivers/gpu/drm/exynos/exynos_drm_crtc.c | 2 +- + drivers/gpu/drm/exynos/exynos_drm_dpi.c | 3 +- + drivers/gpu/drm/exynos/exynos_drm_fimd.c | 3 +- + drivers/gpu/drm/exynos/exynos_drm_plane.c | 22 +- + drivers/gpu/drm/exynos/exynos_drm_vidi.c | 15 +- + drivers/gpu/drm/exynos/exynos_hdmi.c | 3 +- + drivers/gpu/drm/fsl-dcu/fsl_dcu_drm_crtc.c | 2 +- + drivers/gpu/drm/fsl-dcu/fsl_dcu_drm_plane.c | 2 +- + drivers/gpu/drm/fsl-dcu/fsl_dcu_drm_rgb.c | 10 +- + drivers/gpu/drm/gma500/psb_intel_display.c | 34 +- + drivers/gpu/drm/gud/gud_drv.c | 2 +- + drivers/gpu/drm/hisilicon/hibmc/hibmc_drm_de.c | 18 +- + drivers/gpu/drm/hisilicon/hibmc/hibmc_drm_drv.c | 5 +- + drivers/gpu/drm/hisilicon/kirin/dw_drm_dsi.c | 9 +- + drivers/gpu/drm/hisilicon/kirin/kirin_drm_ade.c | 4 +- + drivers/gpu/drm/hyperv/hyperv_drm.h | 1 - + drivers/gpu/drm/hyperv/hyperv_drm_modeset.c | 4 +- + drivers/gpu/drm/hyperv/hyperv_drm_proto.c | 44 +- + drivers/gpu/drm/i915/display/i9xx_plane.c | 3 + + drivers/gpu/drm/i915/display/intel_atomic.c | 1 + + drivers/gpu/drm/i915/display/intel_audio.c | 153 + + drivers/gpu/drm/i915/display/intel_audio_regs.h | 16 +- + drivers/gpu/drm/i915/display/intel_cdclk.c | 326 +- + drivers/gpu/drm/i915/display/intel_cmtg.c | 2 +- + .../gpu/drm/i915/display/intel_crtc_state_dump.c | 3 + + drivers/gpu/drm/i915/display/intel_cursor.c | 296 +- + drivers/gpu/drm/i915/display/intel_cursor_regs.h | 8 +- + drivers/gpu/drm/i915/display/intel_ddi.c | 36 +- + drivers/gpu/drm/i915/display/intel_display.c | 76 +- + drivers/gpu/drm/i915/display/intel_display.h | 3 +- + .../drm/i915/display/intel_display_clock_gating.c | 67 +- + .../drm/i915/display/intel_display_clock_gating.h | 16 +- + .../gpu/drm/i915/display/intel_display_debugfs.c | 21 +- + .../drm/i915/display/intel_display_power_well.c | 6 +- + drivers/gpu/drm/i915/display/intel_display_regs.h | 18 +- + drivers/gpu/drm/i915/display/intel_display_types.h | 17 +- + drivers/gpu/drm/i915/display/intel_display_wa.c | 4 +- + drivers/gpu/drm/i915/display/intel_display_wa.h | 9 - + drivers/gpu/drm/i915/display/intel_dp.c | 52 +- + drivers/gpu/drm/i915/display/intel_dp_mst.c | 4 + + drivers/gpu/drm/i915/display/intel_dpll.c | 22 +- + drivers/gpu/drm/i915/display/intel_dpll_mgr.c | 221 +- + drivers/gpu/drm/i915/display/intel_dpll_mgr.h | 22 + + drivers/gpu/drm/i915/display/intel_frontbuffer.c | 3 +- + drivers/gpu/drm/i915/display/intel_hdmi.c | 53 +- + drivers/gpu/drm/i915/display/intel_hdmi.h | 11 +- + drivers/gpu/drm/i915/display/intel_hti.c | 3 - + .../gpu/drm/i915/display/intel_modeset_verify.c | 1 - + drivers/gpu/drm/i915/display/intel_parent.c | 10 +- + drivers/gpu/drm/i915/display/intel_parent.h | 3 +- + drivers/gpu/drm/i915/display/intel_psr.c | 18 +- + drivers/gpu/drm/i915/display/intel_snps_phy.c | 60 +- + drivers/gpu/drm/i915/display/intel_snps_phy.h | 2 + + drivers/gpu/drm/i915/display/intel_tdf.h | 25 - + drivers/gpu/drm/i915/display/intel_vga.c | 7 +- + drivers/gpu/drm/i915/display/intel_vrr.c | 395 +- + drivers/gpu/drm/i915/display/intel_vrr.h | 2 + + drivers/gpu/drm/i915/display/skl_universal_plane.c | 4 + + drivers/gpu/drm/i915/gem/i915_gem_execbuffer.c | 2 +- + drivers/gpu/drm/i915/gem/i915_gem_shmem.c | 10 +- + drivers/gpu/drm/i915/gt/intel_rps.c | 3 +- + drivers/gpu/drm/i915/gt/selftest_hangcheck.c | 2 +- + drivers/gpu/drm/i915/gt/selftest_rc6.c | 4 +- + drivers/gpu/drm/i915/gvt/handlers.c | 2 +- + drivers/gpu/drm/i915/i915_dpt.c | 1 - + drivers/gpu/drm/i915/i915_reg.h | 2 +- + drivers/gpu/drm/i915/i915_switcheroo.c | 11 +- + drivers/gpu/drm/i915/intel_clock_gating.c | 30 +- + drivers/gpu/drm/i915/intel_gvt_mmio_table.c | 2 +- + drivers/gpu/drm/i915/intel_pcode.c | 14 +- + drivers/gpu/drm/i915/intel_pcode.h | 2 +- + drivers/gpu/drm/i915/selftests/i915_sw_fence.c | 2 +- + drivers/gpu/drm/imagination/pvr_ccb.c | 43 +- + drivers/gpu/drm/imagination/pvr_device.c | 25 +- + drivers/gpu/drm/imagination/pvr_fw.c | 23 +- + drivers/gpu/drm/imagination/pvr_fw.h | 2 +- + drivers/gpu/drm/imagination/pvr_fw_meta.c | 18 +- + drivers/gpu/drm/imagination/pvr_fw_riscv.c | 40 +- + drivers/gpu/drm/imagination/pvr_fw_startstop.c | 22 +- + drivers/gpu/drm/imagination/pvr_fw_trace.c | 1 - + drivers/gpu/drm/imagination/pvr_rogue_cr_defs.h | 920 +- + drivers/gpu/drm/imagination/pvr_rogue_defs.h | 22 +- + drivers/gpu/drm/imagination/pvr_rogue_riscv.h | 2 - + drivers/gpu/drm/imagination/pvr_trace.h | 14 +- + drivers/gpu/drm/imx/dc/dc-crtc.c | 2 +- + drivers/gpu/drm/imx/dc/dc-kms.c | 8 +- + drivers/gpu/drm/imx/dc/dc-plane.c | 2 +- + drivers/gpu/drm/imx/dcss/dcss-crtc.c | 2 +- + drivers/gpu/drm/imx/dcss/dcss-plane.c | 2 +- + drivers/gpu/drm/imx/ipuv3/ipuv3-crtc.c | 18 +- + drivers/gpu/drm/imx/ipuv3/ipuv3-plane.c | 21 +- + drivers/gpu/drm/ingenic/ingenic-drm-drv.c | 4 +- + drivers/gpu/drm/ingenic/ingenic-ipu.c | 2 +- + drivers/gpu/drm/kmb/kmb_crtc.c | 2 +- + drivers/gpu/drm/kmb/kmb_dsi.c | 9 +- + drivers/gpu/drm/kmb/kmb_plane.c | 2 +- + drivers/gpu/drm/logicvc/logicvc_crtc.c | 2 +- + drivers/gpu/drm/logicvc/logicvc_layer.c | 2 +- + drivers/gpu/drm/loongson/lsdc_crtc.c | 30 +- + drivers/gpu/drm/loongson/lsdc_drv.c | 4 +- + drivers/gpu/drm/loongson/lsdc_plane.c | 2 +- + drivers/gpu/drm/mcde/mcde_display.c | 272 +- + drivers/gpu/drm/mcde/mcde_drm.h | 12 +- + drivers/gpu/drm/mcde/mcde_drv.c | 3 +- + drivers/gpu/drm/mcde/mcde_dsi.c | 8 +- + drivers/gpu/drm/mediatek/mtk_crtc.c | 18 +- + drivers/gpu/drm/mediatek/mtk_dsi.c | 10 +- + drivers/gpu/drm/mediatek/mtk_plane.c | 21 +- + drivers/gpu/drm/meson/meson_crtc.c | 2 +- + drivers/gpu/drm/meson/meson_encoder_cvbs.c | 11 +- + drivers/gpu/drm/meson/meson_encoder_dsi.c | 11 +- + drivers/gpu/drm/meson/meson_encoder_hdmi.c | 11 +- + drivers/gpu/drm/meson/meson_overlay.c | 2 +- + drivers/gpu/drm/meson/meson_plane.c | 2 +- + drivers/gpu/drm/mgag200/mgag200_drv.h | 8 +- + drivers/gpu/drm/mgag200/mgag200_mode.c | 15 +- + drivers/gpu/drm/msm/disp/dpu1/dpu_crtc.c | 18 +- + drivers/gpu/drm/msm/disp/dpu1/dpu_plane.c | 18 +- + drivers/gpu/drm/msm/disp/mdp4/mdp4_crtc.c | 2 +- + drivers/gpu/drm/msm/disp/mdp4/mdp4_plane.c | 2 +- + drivers/gpu/drm/msm/disp/mdp5/mdp5_crtc.c | 20 +- + drivers/gpu/drm/msm/disp/mdp5/mdp5_plane.c | 16 +- + drivers/gpu/drm/msm/msm_gem_shrinker.c | 22 +- + drivers/gpu/drm/mxsfb/lcdif_kms.c | 19 +- + drivers/gpu/drm/mxsfb/mxsfb_kms.c | 6 +- + drivers/gpu/drm/nouveau/dispnv04/dfp.c | 5 +- + drivers/gpu/drm/nouveau/dispnv50/disp.c | 4 +- + drivers/gpu/drm/nouveau/dispnv50/head.c | 14 +- + drivers/gpu/drm/nouveau/dispnv50/headca7d.c | 21 +- + drivers/gpu/drm/nouveau/dispnv50/wndw.c | 17 +- + .../gpu/drm/nouveau/include/nvhw/class/clca7d.h | 4 + + drivers/gpu/drm/nouveau/include/nvkm/subdev/pci.h | 1 + + drivers/gpu/drm/nouveau/nouveau_abi16.c | 4 + + drivers/gpu/drm/nouveau/nouveau_acpi.c | 32 +- + drivers/gpu/drm/nouveau/nouveau_acpi.h | 10 +- + drivers/gpu/drm/nouveau/nouveau_bios.c | 2 +- + drivers/gpu/drm/nouveau/nouveau_connector.c | 152 +- + drivers/gpu/drm/nouveau/nouveau_connector.h | 12 +- + drivers/gpu/drm/nouveau/nouveau_dp.c | 6 +- + drivers/gpu/drm/nouveau/nouveau_drm.c | 114 +- + drivers/gpu/drm/nouveau/nouveau_exec.c | 2 +- + drivers/gpu/drm/nouveau/nouveau_fence.c | 2 +- + drivers/gpu/drm/nouveau/nouveau_led.c | 2 +- + drivers/gpu/drm/nouveau/nouveau_svm.c | 13 + + drivers/gpu/drm/nouveau/nouveau_vga.c | 34 +- + drivers/gpu/drm/nouveau/nvkm/engine/device/base.c | 2 +- + drivers/gpu/drm/nouveau/nvkm/subdev/bios/init.c | 2 +- + drivers/gpu/drm/nouveau/nvkm/subdev/clk/gk20a.h | 4 +- + drivers/gpu/drm/nouveau/nvkm/subdev/pci/Kbuild | 1 + + drivers/gpu/drm/nouveau/nvkm/subdev/pci/mcp79.c | 35 + + drivers/gpu/drm/omapdrm/dss/dsi.c | 361 +- + drivers/gpu/drm/omapdrm/dss/hdmi.h | 6 + + drivers/gpu/drm/omapdrm/dss/hdmi4.c | 29 +- + drivers/gpu/drm/omapdrm/dss/hdmi5.c | 37 +- + drivers/gpu/drm/omapdrm/dss/hdmi_common.c | 14 + + drivers/gpu/drm/omapdrm/omap_crtc.c | 17 +- + drivers/gpu/drm/omapdrm/omap_drv.c | 1 - + drivers/gpu/drm/omapdrm/omap_plane.c | 13 +- + drivers/gpu/drm/panel/Kconfig | 25 + + drivers/gpu/drm/panel/Makefile | 2 + + drivers/gpu/drm/panel/panel-edp.c | 5 + + drivers/gpu/drm/panel/panel-himax-hx83102.c | 24 +- + drivers/gpu/drm/panel/panel-himax-hx83112a.c | 26 +- + drivers/gpu/drm/panel/panel-himax-hx83112b.c | 23 +- + drivers/gpu/drm/panel/panel-himax-hx8394.c | 26 +- + drivers/gpu/drm/panel/panel-ilitek-ili7836a.c | 297 + + drivers/gpu/drm/panel/panel-ilitek-ili9805.c | 21 +- + drivers/gpu/drm/panel/panel-ilitek-ili9806e-core.c | 12 +- + drivers/gpu/drm/panel/panel-ilitek-ili9806e-core.h | 1 - + drivers/gpu/drm/panel/panel-ilitek-ili9806e-dsi.c | 16 +- + drivers/gpu/drm/panel/panel-ilitek-ili9806e-spi.c | 6 - + drivers/gpu/drm/panel/panel-ilitek-ili9881c.c | 15 +- + drivers/gpu/drm/panel/panel-ilitek-ili9882t.c | 24 +- + drivers/gpu/drm/panel/panel-jdi-fhd-r63452.c | 19 +- + drivers/gpu/drm/panel/panel-jdi-lt070me05000.c | 32 +- + drivers/gpu/drm/panel/panel-leadtek-ltk050h3146w.c | 20 +- + drivers/gpu/drm/panel/panel-leadtek-ltk500hd1829.c | 20 +- + drivers/gpu/drm/panel/panel-novatek-nt36532.c | 431 + + drivers/gpu/drm/panel/panel-samsung-dsi.h | 38 + + drivers/gpu/drm/panel/panel-samsung-s6d16d0.c | 74 +- + drivers/gpu/drm/panel/panel-samsung-s6d7aa0.c | 20 +- + drivers/gpu/drm/panel/panel-samsung-s6e3fa7.c | 20 +- + drivers/gpu/drm/panel/panel-samsung-s6e3fc2x01.c | 91 +- + drivers/gpu/drm/panel/panel-samsung-s6e3ha2.c | 30 +- + drivers/gpu/drm/panel/panel-samsung-s6e3ha8.c | 90 +- + drivers/gpu/drm/panel/panel-samsung-s6e63j0x03.c | 31 +- + drivers/gpu/drm/panel/panel-samsung-s6e63m0-dsi.c | 13 +- + drivers/gpu/drm/panel/panel-samsung-s6e63m0-spi.c | 6 - + drivers/gpu/drm/panel/panel-samsung-s6e63m0.c | 12 +- + drivers/gpu/drm/panel/panel-samsung-s6e63m0.h | 1 - + .../drm/panel/panel-samsung-s6e88a0-ams427ap24.c | 20 +- + .../drm/panel/panel-samsung-s6e88a0-ams452ef01.c | 26 +- + drivers/gpu/drm/panel/panel-samsung-s6e8aa0.c | 19 +- + .../gpu/drm/panel/panel-samsung-s6e8fc0-m1906f9.c | 46 +- + drivers/gpu/drm/panel/panel-samsung-sofef00.c | 35 +- + drivers/gpu/drm/panel/panel-sharp-ls043t1le01.c | 31 +- + drivers/gpu/drm/panel/panel-sharp-ls060t1sx01.c | 20 +- + drivers/gpu/drm/panel/panel-sony-td4353-jdi.c | 20 +- + .../gpu/drm/panel/panel-sony-tulip-truly-nt35521.c | 20 +- + drivers/gpu/drm/panel/panel-visionox-r66451.c | 23 +- + drivers/gpu/drm/panel/panel-visionox-rm69299.c | 21 +- + drivers/gpu/drm/panfrost/panfrost_devfreq.c | 2 +- + drivers/gpu/drm/panfrost/panfrost_device.c | 1 - + drivers/gpu/drm/panfrost/panfrost_drv.c | 2 +- + drivers/gpu/drm/panthor/panthor_device.h | 286 +- + drivers/gpu/drm/panthor/panthor_drv.c | 1 + + drivers/gpu/drm/panthor/panthor_fw.c | 22 +- + drivers/gpu/drm/panthor/panthor_gem.c | 14 +- + drivers/gpu/drm/panthor/panthor_gpu.c | 117 +- + drivers/gpu/drm/panthor/panthor_mmu.c | 44 +- + drivers/gpu/drm/panthor/panthor_mmu.h | 3 +- + drivers/gpu/drm/panthor/panthor_pwr.c | 24 +- + drivers/gpu/drm/panthor/panthor_sched.c | 591 +- + drivers/gpu/drm/panthor/panthor_sched.h | 5 + + drivers/gpu/drm/panthor/panthor_trace.h | 38 + + drivers/gpu/drm/pl111/pl111_display.c | 207 +- + drivers/gpu/drm/pl111/pl111_drm.h | 5 +- + drivers/gpu/drm/pl111/pl111_drv.c | 19 +- + drivers/gpu/drm/pl111/pl111_versatile.c | 18 - + drivers/gpu/drm/qxl/qxl_display.c | 6 +- + drivers/gpu/drm/radeon/atombios_encoders.c | 36 + + drivers/gpu/drm/radeon/radeon_device.c | 18 +- + drivers/gpu/drm/radeon/radeon_fence.c | 4 +- + drivers/gpu/drm/renesas/rcar-du/Kconfig | 12 + + drivers/gpu/drm/renesas/rcar-du/Makefile | 1 + + drivers/gpu/drm/renesas/rcar-du/rcar_dsc.c | 153 + + drivers/gpu/drm/renesas/rcar-du/rcar_du_crtc.c | 17 +- + drivers/gpu/drm/renesas/rcar-du/rcar_du_encoder.c | 17 +- + drivers/gpu/drm/renesas/rcar-du/rcar_du_plane.c | 19 +- + drivers/gpu/drm/renesas/rcar-du/rcar_du_vsp.c | 17 +- + drivers/gpu/drm/renesas/rcar-du/rcar_mipi_dsi.c | 51 +- + drivers/gpu/drm/renesas/rz-du/rzg2l_du_crtc.c | 15 +- + drivers/gpu/drm/renesas/rz-du/rzg2l_du_vsp.c | 15 +- + drivers/gpu/drm/renesas/shmobile/shmob_drm_crtc.c | 12 +- + drivers/gpu/drm/renesas/shmobile/shmob_drm_plane.c | 19 +- + drivers/gpu/drm/rockchip/dw-mipi-dsi-rockchip.c | 66 +- + drivers/gpu/drm/rockchip/rockchip_drm_vop.c | 20 +- + drivers/gpu/drm/rockchip/rockchip_drm_vop2.c | 20 +- + drivers/gpu/drm/scheduler/sched_entity.c | 10 +- + drivers/gpu/drm/scheduler/sched_fence.c | 3 +- + drivers/gpu/drm/scheduler/sched_internal.h | 3 +- + drivers/gpu/drm/scheduler/sched_main.c | 8 +- + drivers/gpu/drm/scheduler/tests/tests_basic.c | 2 +- + drivers/gpu/drm/sitronix/st7571.c | 2 +- + drivers/gpu/drm/sitronix/st7920.c | 24 +- + drivers/gpu/drm/solomon/ssd130x-spi.c | 25 +- + drivers/gpu/drm/solomon/ssd130x.c | 443 +- + drivers/gpu/drm/solomon/ssd130x.h | 10 +- + drivers/gpu/drm/sprd/sprd_dpu.c | 4 +- + drivers/gpu/drm/sti/sti_crtc.c | 2 +- + drivers/gpu/drm/sti/sti_cursor.c | 2 +- + drivers/gpu/drm/sti/sti_gdp.c | 2 +- + drivers/gpu/drm/sti/sti_hqvdp.c | 2 +- + drivers/gpu/drm/stm/ltdc.c | 6 +- + drivers/gpu/drm/sun4i/sun4i_crtc.c | 2 +- + drivers/gpu/drm/sun4i/sun4i_hdmi_enc.c | 3 +- + drivers/gpu/drm/sun4i/sun4i_layer.c | 21 +- + drivers/gpu/drm/sun4i/sun8i_ui_layer.c | 2 +- + drivers/gpu/drm/sun4i/sun8i_vi_layer.c | 2 +- + drivers/gpu/drm/sysfb/drm_sysfb_helper.h | 12 +- + drivers/gpu/drm/sysfb/drm_sysfb_modeset.c | 70 +- + drivers/gpu/drm/sysfb/efidrm.c | 17 +- + drivers/gpu/drm/sysfb/ofdrm.c | 18 +- + drivers/gpu/drm/sysfb/vesadrm.c | 18 +- + drivers/gpu/drm/tegra/dc.c | 18 +- + drivers/gpu/drm/tegra/dsi.c | 16 +- + drivers/gpu/drm/tegra/plane.c | 28 +- + drivers/gpu/drm/tegra/rgb.c | 15 +- + drivers/gpu/drm/tests/Makefile | 1 + + drivers/gpu/drm/tests/drm_connector_test.c | 42 +- + drivers/gpu/drm/tests/drm_hdmi_state_helper_test.c | 56 +- + drivers/gpu/drm/tests/drm_kunit_helpers.c | 4 +- + .../{drm_panic_test.c => drm_panic_helper_test.c} | 75 +- + drivers/gpu/drm/tidss/tidss_dispc.c | 38 +- + drivers/gpu/drm/tidss/tidss_dispc.h | 2 +- + drivers/gpu/drm/tidss/tidss_encoder.c | 10 +- + drivers/gpu/drm/tidss/tidss_plane.c | 2 + + drivers/gpu/drm/tilcdc/tilcdc_crtc.c | 7 +- + drivers/gpu/drm/tilcdc/tilcdc_plane.c | 2 +- + drivers/gpu/drm/tiny/appletbdrm.c | 14 +- + drivers/gpu/drm/tiny/arcpgu.c | 201 +- + drivers/gpu/drm/tiny/bochs.c | 6 +- + drivers/gpu/drm/tiny/cirrus-qemu.c | 2 +- + drivers/gpu/drm/tiny/gm12u320.c | 138 +- + drivers/gpu/drm/tiny/pixpaper.c | 2 +- + drivers/gpu/drm/tiny/repaper.c | 138 +- + drivers/gpu/drm/tiny/sharp-memory.c | 2 +- + drivers/gpu/drm/ttm/ttm_bo.c | 3 +- + drivers/gpu/drm/ttm/ttm_module.c | 1 - + drivers/gpu/drm/tve200/tve200_display.c | 220 +- + drivers/gpu/drm/tve200/tve200_drm.h | 6 +- + drivers/gpu/drm/tve200/tve200_drv.c | 12 +- + drivers/gpu/drm/udl/udl_modeset.c | 2 +- + drivers/gpu/drm/v3d/v3d_drv.h | 22 +- + drivers/gpu/drm/v3d/v3d_gem.c | 3 - + drivers/gpu/drm/v3d/v3d_perfmon.c | 13 +- + drivers/gpu/drm/v3d/v3d_sched.c | 4 - + drivers/gpu/drm/v3d/v3d_submit.c | 19 +- + drivers/gpu/drm/vboxvideo/vbox_mode.c | 4 +- + drivers/gpu/drm/vc4/tests/vc4_mock_crtc.c | 2 +- + drivers/gpu/drm/vc4/vc4_crtc.c | 14 +- + drivers/gpu/drm/vc4/vc4_drv.c | 1 - + drivers/gpu/drm/vc4/vc4_drv.h | 15 +- + drivers/gpu/drm/vc4/vc4_gem.c | 40 +- + drivers/gpu/drm/vc4/vc4_hdmi.c | 5 +- + drivers/gpu/drm/vc4/vc4_irq.c | 7 +- + drivers/gpu/drm/vc4/vc4_plane.c | 15 +- + drivers/gpu/drm/vc4/vc4_txp.c | 2 +- + drivers/gpu/drm/vc4/vc4_v3d.c | 37 +- + drivers/gpu/drm/verisilicon/vs_crtc.c | 2 +- + drivers/gpu/drm/verisilicon/vs_cursor_plane.c | 10 +- + drivers/gpu/drm/verisilicon/vs_plane.c | 34 +- + drivers/gpu/drm/verisilicon/vs_plane.h | 4 +- + drivers/gpu/drm/verisilicon/vs_primary_plane.c | 9 +- + drivers/gpu/drm/virtio/virtgpu_display.c | 14 +- + drivers/gpu/drm/virtio/virtgpu_plane.c | 14 +- + drivers/gpu/drm/vkms/tests/gen_yuv_conversion.py | 87 + + drivers/gpu/drm/vkms/tests/vkms_format_test.c | 40 +- + drivers/gpu/drm/vkms/vkms_colorop.c | 66 +- + drivers/gpu/drm/vkms/vkms_composer.c | 36 +- + drivers/gpu/drm/vkms/vkms_crtc.c | 18 +- + drivers/gpu/drm/vkms/vkms_drv.h | 2 +- + drivers/gpu/drm/vkms/vkms_formats.c | 112 +- + drivers/gpu/drm/vkms/vkms_formats.h | 2 +- + drivers/gpu/drm/vkms/vkms_plane.c | 91 +- + drivers/gpu/drm/vmwgfx/vmwgfx_kms.c | 39 +- + drivers/gpu/drm/vmwgfx/vmwgfx_kms.h | 4 +- + drivers/gpu/drm/vmwgfx/vmwgfx_ldu.c | 6 +- + drivers/gpu/drm/vmwgfx/vmwgfx_scrn.c | 6 +- + drivers/gpu/drm/vmwgfx/vmwgfx_stdu.c | 6 +- + drivers/gpu/drm/xe/Makefile | 5 +- + drivers/gpu/drm/xe/abi/guc_actions_slpc_abi.h | 1 + + drivers/gpu/drm/xe/abi/guc_klvs_abi.h | 1 + + drivers/gpu/drm/xe/abi/xe_log_abi.h | 200 + + drivers/gpu/drm/xe/abi/xe_sigid_abi.h | 172 + + drivers/gpu/drm/xe/display/xe_display.c | 22 +- + drivers/gpu/drm/xe/display/xe_display_pcode.c | 4 +- + drivers/gpu/drm/xe/display/xe_display_rpm.c | 2 - + drivers/gpu/drm/xe/display/xe_display_wa.c | 13 +- + drivers/gpu/drm/xe/display/xe_display_wa.h | 9 + + drivers/gpu/drm/xe/display/xe_dsb_buffer.c | 2 +- + drivers/gpu/drm/xe/display/xe_fb_pin.c | 2 +- + drivers/gpu/drm/xe/display/xe_panic.c | 3 +- + drivers/gpu/drm/xe/display/xe_tdf.c | 15 - + drivers/gpu/drm/xe/instructions/xe_mi_commands.h | 7 +- + drivers/gpu/drm/xe/regs/xe_gt_regs.h | 36 +- + drivers/gpu/drm/xe/regs/xe_lrc_layout.h | 2 + + drivers/gpu/drm/xe/regs/xe_regs.h | 2 + + drivers/gpu/drm/xe/tests/Makefile | 1 + + drivers/gpu/drm/xe/tests/xe_any_kunit.c | 213 + + .../gpu/drm/xe/tests/xe_guc_klv_helpers_kunit.c | 303 + + drivers/gpu/drm/xe/tests/xe_kunit_helpers.c | 4 + + drivers/gpu/drm/xe/tests/xe_log_kunit.c | 553 + + drivers/gpu/drm/xe/tests/xe_pci.c | 25 +- + drivers/gpu/drm/xe/xe_any.h | 137 + + drivers/gpu/drm/xe/xe_bo.c | 80 +- + drivers/gpu/drm/xe/xe_bo.h | 13 +- + drivers/gpu/drm/xe/xe_bo_types.h | 10 +- + drivers/gpu/drm/xe/xe_configfs.c | 139 +- + drivers/gpu/drm/xe/xe_configfs.h | 4 + + drivers/gpu/drm/xe/xe_cpu_bind.c | 296 + + drivers/gpu/drm/xe/xe_cpu_bind.h | 111 + + drivers/gpu/drm/xe/xe_debugfs.c | 66 + + drivers/gpu/drm/xe/xe_debugfs.h | 4 + + drivers/gpu/drm/xe/xe_defaults.h | 1 + + drivers/gpu/drm/xe/xe_devcoredump.c | 50 +- + drivers/gpu/drm/xe/xe_devcoredump.h | 15 +- + drivers/gpu/drm/xe/xe_device.c | 164 +- + drivers/gpu/drm/xe/xe_device.h | 2 +- + drivers/gpu/drm/xe/xe_device_sysfs.c | 36 + + drivers/gpu/drm/xe/xe_device_types.h | 57 +- + drivers/gpu/drm/xe/xe_dma_buf.c | 22 +- + drivers/gpu/drm/xe/xe_drm_client.c | 2 +- + drivers/gpu/drm/xe/xe_drm_ras.c | 65 + + drivers/gpu/drm/xe/xe_drm_ras.h | 3 + + drivers/gpu/drm/xe/xe_drm_ras_types.h | 3 + + drivers/gpu/drm/xe/xe_exec_queue.c | 239 +- + drivers/gpu/drm/xe/xe_exec_queue.h | 16 +- + drivers/gpu/drm/xe/xe_exec_queue_types.h | 36 +- + drivers/gpu/drm/xe/xe_execlist.c | 4 +- + drivers/gpu/drm/xe/xe_ggtt.c | 74 +- + drivers/gpu/drm/xe/xe_gsc.c | 3 +- + drivers/gpu/drm/xe/xe_gt.c | 24 +- + drivers/gpu/drm/xe/xe_gt.h | 13 + + drivers/gpu/drm/xe/xe_gt_ccs_mode.c | 6 - + drivers/gpu/drm/xe/xe_gt_debugfs.c | 252 +- + drivers/gpu/drm/xe/xe_gt_idle.c | 84 +- + drivers/gpu/drm/xe/xe_gt_idle.h | 1 + + drivers/gpu/drm/xe/xe_gt_printk.h | 3 + + drivers/gpu/drm/xe/xe_gt_sriov_pf_config.c | 299 +- + drivers/gpu/drm/xe/xe_gt_sriov_pf_config.h | 10 + + drivers/gpu/drm/xe/xe_gt_sriov_printk.h | 3 + + drivers/gpu/drm/xe/xe_gt_stats.c | 7 + + drivers/gpu/drm/xe/xe_gt_stats_types.h | 22 + + drivers/gpu/drm/xe/xe_gt_types.h | 29 + + drivers/gpu/drm/xe/xe_guc.c | 34 +- + drivers/gpu/drm/xe/xe_guc.h | 2 + + drivers/gpu/drm/xe/xe_guc_ads.c | 9 +- + drivers/gpu/drm/xe/xe_guc_capture.c | 2 - + drivers/gpu/drm/xe/xe_guc_ct.c | 114 +- + drivers/gpu/drm/xe/xe_guc_ct.h | 38 +- + drivers/gpu/drm/xe/xe_guc_engine_activity.c | 6 +- + drivers/gpu/drm/xe/xe_guc_exec_queue_types.h | 32 +- + drivers/gpu/drm/xe/xe_guc_hwconfig.c | 2 +- + drivers/gpu/drm/xe/xe_guc_klv_helpers.c | 78 + + drivers/gpu/drm/xe/xe_guc_klv_helpers.h | 5 + + drivers/gpu/drm/xe/xe_guc_log.c | 7 +- + drivers/gpu/drm/xe/xe_guc_pagefault.c | 80 +- + drivers/gpu/drm/xe/xe_guc_pc.c | 18 +- + drivers/gpu/drm/xe/xe_guc_submit.c | 472 +- + drivers/gpu/drm/xe/xe_guc_submit.h | 5 + + drivers/gpu/drm/xe/xe_guc_submit_types.h | 2 + + drivers/gpu/drm/xe/xe_guc_tlb_inval.c | 29 + + drivers/gpu/drm/xe/xe_guc_types.h | 6 + + drivers/gpu/drm/xe/xe_hwmon.c | 95 +- + drivers/gpu/drm/xe/xe_log.c | 247 + + drivers/gpu/drm/xe/xe_log.h | 194 + + drivers/gpu/drm/xe/xe_lrc.c | 112 +- + drivers/gpu/drm/xe/xe_lrc.h | 5 + + drivers/gpu/drm/xe/xe_lrc_types.h | 4 + + drivers/gpu/drm/xe/xe_mert.c | 2 +- + drivers/gpu/drm/xe/xe_migrate.c | 884 +- + drivers/gpu/drm/xe/xe_migrate.h | 95 +- + drivers/gpu/drm/xe/xe_mmio.c | 5 +- + drivers/gpu/drm/xe/xe_mmio_gem.c | 7 +- + drivers/gpu/drm/xe/xe_module.c | 4 + + drivers/gpu/drm/xe/xe_module.h | 1 + + drivers/gpu/drm/xe/xe_nvm.c | 1 + + drivers/gpu/drm/xe/xe_oa.c | 7 +- + drivers/gpu/drm/xe/xe_pagefault.c | 841 +- + drivers/gpu/drm/xe/xe_pagefault.h | 137 + + drivers/gpu/drm/xe/xe_pagefault_types.h | 211 +- + drivers/gpu/drm/xe/xe_pci.c | 79 +- + drivers/gpu/drm/xe/xe_pci_error.c | 13 +- + drivers/gpu/drm/xe/xe_pci_types.h | 4 +- + drivers/gpu/drm/xe/xe_pcode.c | 27 +- + drivers/gpu/drm/xe/xe_pcode.h | 6 +- + drivers/gpu/drm/xe/xe_pm.c | 23 + + drivers/gpu/drm/xe/xe_pm.h | 1 + + drivers/gpu/drm/xe/xe_printk.h | 3 + + drivers/gpu/drm/xe/xe_pt.c | 813 +- + drivers/gpu/drm/xe/xe_pt.h | 12 +- + drivers/gpu/drm/xe/xe_pt_types.h | 49 +- + drivers/gpu/drm/xe/xe_pxp_submit.c | 5 +- + drivers/gpu/drm/xe/xe_ras.c | 245 +- + drivers/gpu/drm/xe/xe_ras.h | 2 + + drivers/gpu/drm/xe/xe_ras_types.h | 50 + + drivers/gpu/drm/xe/xe_res_cursor.h | 5 +- + drivers/gpu/drm/xe/xe_ring_ops.c | 90 +- + drivers/gpu/drm/xe/xe_ring_ops_types.h | 24 + + drivers/gpu/drm/xe/xe_sched_job.c | 101 +- + drivers/gpu/drm/xe/xe_sched_job.h | 56 + + drivers/gpu/drm/xe/xe_sched_job_types.h | 49 +- + drivers/gpu/drm/xe/xe_sriov_packet.c | 31 +- + drivers/gpu/drm/xe/xe_sriov_pf_provision.c | 4 +- + drivers/gpu/drm/xe/xe_sriov_printk.h | 3 + + drivers/gpu/drm/xe/xe_sriov_vf_ccs.c | 1 - + drivers/gpu/drm/xe/xe_survivability_mode.c | 85 +- + drivers/gpu/drm/xe/xe_svm.c | 250 +- + drivers/gpu/drm/xe/xe_svm.h | 97 +- + drivers/gpu/drm/xe/xe_sync.c | 20 +- + drivers/gpu/drm/xe/xe_sysctrl.c | 92 + + drivers/gpu/drm/xe/xe_sysctrl.h | 2 + + drivers/gpu/drm/xe/xe_sysctrl_event.c | 28 +- + drivers/gpu/drm/xe/xe_sysctrl_mailbox.c | 7 +- + drivers/gpu/drm/xe/xe_sysctrl_mailbox_types.h | 52 + + drivers/gpu/drm/xe/xe_tile_printk.h | 3 + + drivers/gpu/drm/xe/xe_tile_sriov_printk.h | 3 + + drivers/gpu/drm/xe/xe_tile_types.h | 4 + + drivers/gpu/drm/xe/xe_tlb_inval.c | 96 +- + drivers/gpu/drm/xe/xe_tlb_inval.h | 1 + + drivers/gpu/drm/xe/xe_tlb_inval_job.c | 28 +- + drivers/gpu/drm/xe/xe_tlb_inval_job.h | 4 +- + drivers/gpu/drm/xe/xe_tlb_inval_types.h | 27 + + drivers/gpu/drm/xe/xe_trace.h | 2 +- + drivers/gpu/drm/xe/xe_trace_bo.h | 58 +- + drivers/gpu/drm/xe/xe_ttm_vram_mgr.c | 600 +- + drivers/gpu/drm/xe/xe_ttm_vram_mgr.h | 3 + + drivers/gpu/drm/xe/xe_ttm_vram_mgr_types.h | 40 + + drivers/gpu/drm/xe/xe_tuning.c | 4 + + drivers/gpu/drm/xe/xe_uc_fw.c | 1 + + drivers/gpu/drm/xe/xe_userptr.c | 21 +- + drivers/gpu/drm/xe/xe_vm.c | 570 +- + drivers/gpu/drm/xe/xe_vm.h | 3 + + drivers/gpu/drm/xe/xe_vm_madvise.c | 6 - + drivers/gpu/drm/xe/xe_vm_types.h | 56 +- + drivers/gpu/drm/xe/xe_vram.c | 255 +- + drivers/gpu/drm/xe/xe_vram.h | 10 + + drivers/gpu/drm/xe/xe_wa.c | 4 +- + drivers/gpu/drm/xe/xe_wa_oob.rules | 2 + + drivers/gpu/drm/xen/xen_drm_front.h | 6 +- + drivers/gpu/drm/xen/xen_drm_front_kms.c | 188 +- + drivers/gpu/drm/xlnx/zynqmp_kms.c | 18 +- + drivers/gpu/tests/gpu_buddy_test.c | 160 +- + drivers/gpu/vga/vga_switcheroo.c | 41 +- + drivers/pci/vgaarb.c | 13 +- + drivers/vfio/pci/vfio_pci_core.c | 9 +- + drivers/video/fbdev/core/fbcon.c | 8 - + include/drm/display/drm_dp.h | 1 + + include/drm/display/drm_dp_helper.h | 6 + + include/drm/display/drm_hdmi_helper.h | 15 + + include/drm/display/drm_hdmi_state_helper.h | 11 +- + include/drm/display/drm_scdc.h | 21 +- + include/drm/display/drm_scdc_helper.h | 105 +- + include/drm/drm_atomic.h | 89 +- + include/drm/drm_atomic_state_helper.h | 6 - + include/drm/drm_bridge.h | 49 + + include/drm/drm_client.h | 14 + + include/drm/drm_client_event.h | 3 + + include/drm/drm_colorop.h | 127 + + include/drm/drm_connector.h | 216 +- + include/drm/drm_crtc.h | 12 - + include/drm/drm_device.h | 1 + + include/drm/drm_edid.h | 2 + + include/drm/drm_gem.h | 14 +- + include/drm/drm_gem_atomic_helper.h | 13 +- + include/drm/drm_gpusvm.h | 77 +- + include/drm/drm_mipi_dbi.h | 2 +- + include/drm/drm_mode_config.h | 4 +- + include/drm/drm_modeset_helper_vtables.h | 53 +- + include/drm/drm_pagemap.h | 7 + + include/drm/drm_panic.h | 117 +- + include/drm/drm_panic_helper.h | 41 + + include/drm/drm_plane.h | 69 +- + include/drm/drm_print.h | 3 + + include/drm/drm_ras.h | 34 + + include/drm/drm_simple_kms_helper.h | 7 +- + include/drm/drm_sysfs.h | 4 - + include/drm/intel/display_parent_interface.h | 15 +- + include/drm/intel/pciids.h | 1 + + include/drm/ttm/ttm_resource.h | 2 +- + include/linux/dma-fence.h | 2 +- + include/linux/gpu_buddy.h | 101 +- + include/linux/hdmi.h | 12 + + include/linux/vga_switcheroo.h | 29 +- + include/linux/vgaarb.h | 8 +- + include/sound/omap-hdmi-audio.h | 1 + + include/uapi/drm/amdgpu_drm.h | 41 + + include/uapi/drm/drm_mode.h | 12 + + include/uapi/drm/drm_ras.h | 18 + + include/uapi/drm/ivpu_accel.h | 17 +- + include/uapi/drm/xe_drm.h | 22 +- + kernel/cgroup/dmem.c | 4 +- + rust/bindings/bindings_helper.h | 4 +- + rust/kernel/drm/gem/mod.rs | 13 +- + rust/kernel/drm/gem/shmem.rs | 5 +- + sound/soc/ti/omap-hdmi.c | 51 +- + 1386 files changed, 151793 insertions(+), 26503 deletions(-) + create mode 100644 Documentation/ABI/testing/sysfs-driver-intel-xe-gpu + create mode 100644 Documentation/devicetree/bindings/display/bridge/renesas,r8a779g0-dsc.yaml + create mode 100644 Documentation/devicetree/bindings/display/panel/ilitek,ili7836a.yaml + create mode 100644 Documentation/devicetree/bindings/display/panel/novatek,nt36532.yaml + create mode 100644 Documentation/devicetree/bindings/display/solomon,ssd1351.yaml + create mode 100644 Documentation/gpu/amdgpu/ualink.rst + create mode 100644 Documentation/gpu/xe/xe_sigid.rst + create mode 100644 drivers/gpu/drm/amd/amdgpu/amdgpu_sdma_types.h + create mode 100644 drivers/gpu/drm/amd/amdgpu/amdgpu_ualink.c + create mode 100644 drivers/gpu/drm/amd/amdgpu/amdgpu_ualink.h + create mode 100644 drivers/gpu/drm/amd/amdgpu/amdgpu_vm_internal.h + create mode 100644 drivers/gpu/drm/amd/amdgpu/nbio_v7_11_5.c + create mode 100644 drivers/gpu/drm/amd/amdgpu/nbio_v7_11_5.h + create mode 100644 drivers/gpu/drm/amd/amdgpu/ualink_v1_0.c + create mode 100644 drivers/gpu/drm/amd/amdgpu/ualink_v1_0.h + create mode 100644 drivers/gpu/drm/amd/display/dc/clk_mgr/dcn10/dcn10_clk_mgr.c + create mode 100644 drivers/gpu/drm/amd/display/dc/clk_mgr/dcn10/dcn10_clk_mgr.h + create mode 100644 drivers/gpu/drm/amd/display/dc/clk_mgr/dcn60/dcn60_smu_driver_if.h + delete mode 100644 drivers/gpu/drm/amd/display/dc/dc_edid_parser.c + create mode 100644 drivers/gpu/drm/amd/display/dc/dc_memory_pool.c + create mode 100644 drivers/gpu/drm/amd/display/dc/dc_memory_pool.h + delete mode 100644 drivers/gpu/drm/amd/display/dc/dml2_0/dml21/dml21_wrapper.h + create mode 100644 drivers/gpu/drm/amd/display/dc/dml2_wrapper/Makefile + rename drivers/gpu/drm/amd/display/dc/{dml2_0/dml21 => dml2_wrapper/dml21_wrapper}/dml21_translation_helper.c (96%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0/dml21 => dml2_wrapper/dml21_wrapper}/dml21_translation_helper.h (94%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0/dml21 => dml2_wrapper/dml21_wrapper}/dml21_utils.c (97%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0/dml21 => dml2_wrapper/dml21_wrapper}/dml21_utils.h (96%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0/dml21 => dml2_wrapper/dml21_wrapper}/dml21_wrapper.c (100%) + create mode 100644 drivers/gpu/drm/amd/display/dc/dml2_wrapper/dml21_wrapper/dml21_wrapper.h + rename drivers/gpu/drm/amd/display/dc/{dml2_0/dml21 => dml2_wrapper/dml21_wrapper}/dml21_wrapper_fpu.c (96%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0/dml21 => dml2_wrapper/dml21_wrapper}/dml21_wrapper_fpu.h (95%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_dc_resource_mgmt.c (96%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_dc_resource_mgmt.h (100%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_dc_types.h (100%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_internal_types.h (99%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_mall_phantom.c (97%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_mall_phantom.h (100%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_policy.c (99%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_policy.h (100%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_translation_helper.c (99%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_translation_helper.h (100%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_utils.c (98%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_utils.h (100%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_wrapper.c (97%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_wrapper.h (100%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_wrapper_fpu.c (93%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_wrapper_fpu.h (100%) + create mode 100644 drivers/gpu/drm/amd/display/dc/inc/hw/rmcm.h + create mode 100644 drivers/gpu/drm/amd/display/dc/rmcm/Makefile + create mode 100644 drivers/gpu/drm/amd/display/dc/rmcm/dcn42/dcn42_rmcm.c + create mode 100644 drivers/gpu/drm/amd/display/dc/rmcm/dcn42/dcn42_rmcm.h + create mode 100644 drivers/gpu/drm/amd/display/dc/rmcm/dcn60/dcn60_rmcm.c + rename drivers/gpu/drm/amd/display/dc/{dc_edid_parser.h => rmcm/dcn60/dcn60_rmcm.h} (71%) + delete mode 100644 drivers/gpu/drm/amd/display/dc/sspl/spl_custom_float.c + delete mode 100644 drivers/gpu/drm/amd/display/dc/sspl/spl_custom_float.h + delete mode 100644 drivers/gpu/drm/amd/display/dc/sspl/spl_fixpt31_32.c + delete mode 100644 drivers/gpu/drm/amd/display/dc/sspl/spl_fixpt31_32.h + create mode 100644 drivers/gpu/drm/amd/display/dc/sspl/spl_namespace.h + create mode 100644 drivers/gpu/drm/amd/include/asic_reg/nbio/nbio_7_11_5_offset.h + create mode 100644 drivers/gpu/drm/amd/include/asic_reg/nbio/nbio_7_11_5_sh_mask.h + create mode 100644 drivers/gpu/drm/amd/include/ivsrcid/mpnht/irqsrcs_mpnht_15_0.h + create mode 100644 drivers/gpu/drm/amd/ras/core/aca_v5_0.c + create mode 100644 drivers/gpu/drm/amd/ras/core/aca_v5_0.h + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_bert.c + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_bert.h + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_eeprom_mgr.c + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_eeprom_mgr.h + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_mce.c + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_mce.h + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_mp1_v15_0.c + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_mp1_v15_0.h + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_psp_v15_0.c + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_psp_v15_0.h + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_umc_v15_0.c + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_umc_v15_0.h + create mode 100644 drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_bert.c + create mode 100644 drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_bert.h + create mode 100644 drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mce.c + create mode 100644 drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mce.h + create mode 100644 drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mp1.c + rename drivers/gpu/drm/amd/ras/ras_mgr/{amdgpu_ras_mp1_v13_0.h => amdgpu_ras_mp1.h} (86%) + delete mode 100644 drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mp1_v13_0.c + create mode 100644 drivers/gpu/drm/drm_panic_helper.c + rename drivers/gpu/drm/{drm_panic_qr.rs => drm_panic_helper_qr.rs} (99%) + create mode 100644 drivers/gpu/drm/drm_panic_internal.h + delete mode 100644 drivers/gpu/drm/i915/display/intel_tdf.h + create mode 100644 drivers/gpu/drm/nouveau/nvkm/subdev/pci/mcp79.c + create mode 100644 drivers/gpu/drm/panel/panel-ilitek-ili7836a.c + create mode 100644 drivers/gpu/drm/panel/panel-novatek-nt36532.c + create mode 100644 drivers/gpu/drm/panel/panel-samsung-dsi.h + create mode 100644 drivers/gpu/drm/renesas/rcar-du/rcar_dsc.c + rename drivers/gpu/drm/tests/{drm_panic_test.c => drm_panic_helper_test.c} (77%) + create mode 100755 drivers/gpu/drm/vkms/tests/gen_yuv_conversion.py + create mode 100644 drivers/gpu/drm/xe/abi/xe_log_abi.h + create mode 100644 drivers/gpu/drm/xe/abi/xe_sigid_abi.h + create mode 100644 drivers/gpu/drm/xe/display/xe_display_wa.h + delete mode 100644 drivers/gpu/drm/xe/display/xe_tdf.c + create mode 100644 drivers/gpu/drm/xe/tests/xe_any_kunit.c + create mode 100644 drivers/gpu/drm/xe/tests/xe_log_kunit.c + create mode 100644 drivers/gpu/drm/xe/xe_any.h + create mode 100644 drivers/gpu/drm/xe/xe_cpu_bind.c + create mode 100644 drivers/gpu/drm/xe/xe_cpu_bind.h + create mode 100644 drivers/gpu/drm/xe/xe_log.c + create mode 100644 drivers/gpu/drm/xe/xe_log.h + create mode 100644 include/drm/drm_panic_helper.h +Merging drm-exynos/for-linux-next (3a8660878839f Linux 6.18-rc1) +$ git merge -m Merge branch 'for-linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/daeinki/drm-exynos.git drm-exynos/for-linux-next +Already up to date. +Merging drm-misc/for-linux-next (54e61ea6c492d drm/panel: Add driver for Raydium RM69220 DDIC) +$ git merge -m Merge branch 'for-linux-next' of https://gitlab.freedesktop.org/drm/misc/kernel.git drm-misc/for-linux-next +Auto-merging Documentation/devicetree/bindings/display/fsl,lcdif.yaml +Auto-merging MAINTAINERS +Auto-merging drivers/accel/ivpu/ivpu_job.c +Auto-merging drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c +Auto-merging drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_connector.c +Auto-merging drivers/gpu/drm/bridge/samsung-dsim.c +Auto-merging drivers/gpu/drm/clients/drm_fbdev_client.c +Auto-merging drivers/gpu/drm/clients/drm_fbdev_helper.c +Auto-merging drivers/gpu/drm/drm_bridge.c +Auto-merging drivers/gpu/drm/radeon/radeon_device.c +Merge made by the 'ort' strategy. + .../display/allwinner,sun6i-a31-mipi-dsi.yaml | 10 +- + .../display/bridge/fsl,imx8qxp-pxl2dpi.yaml | 4 +- + .../bindings/display/bridge/lontium,lt8912b.yaml | 8 +- + .../bridge/megachips,stdp2690-ge-b850v3-fw.yaml | 2 +- + .../bindings/display/bridge/nxp,tda998x.yaml | 2 +- + .../bindings/display/bridge/toshiba,tc358764.yaml | 2 +- + .../bindings/display/bridge/toshiba,tc358775.yaml | 12 +- + .../bindings/display/faraday,tve200.yaml | 2 +- + .../devicetree/bindings/display/fsl,lcdif.yaml | 12 +- + .../devicetree/bindings/display/himax,hx8357.yaml | 2 +- + .../bindings/display/imx/fsl,imx-lcdc.yaml | 2 +- + .../bindings/display/imx/fsl,imx6q-ldb.yaml | 54 +- + .../bindings/display/mediatek/mediatek,ethdr.yaml | 90 +-- + .../bindings/display/mediatek/mediatek,hdmi.yaml | 24 +- + .../bindings/display/panel/ilitek,il79900a.yaml | 13 +- + .../bindings/display/panel/novatek,nt51021.yaml | 87 ++ + .../panel/panel-simple-lvds-dual-ports.yaml | 2 + + .../bindings/display/panel/panel-simple.yaml | 2 + + .../bindings/display/panel/raydium,rm69220.yaml | 74 ++ + .../bindings/display/panel/samsung,ams495qa01.yaml | 2 +- + .../bindings/display/panel/visionox,vtdr6130.yaml | 8 +- + .../bindings/display/sitronix,st7567.yaml | 4 +- + .../devicetree/bindings/display/st,stm32-dsi.yaml | 46 +- + .../devicetree/bindings/display/st,stm32-ltdc.yaml | 6 +- + .../bindings/display/st,stm32mp25-lvds.yaml | 4 +- + .../display/tegra/nvidia,tegra20-host1x.yaml | 52 +- + Documentation/gpu/drm-kms-helpers.rst | 8 +- + Documentation/gpu/todo.rst | 33 +- + MAINTAINERS | 13 + + drivers/accel/amdxdna/aie2_ctx.c | 117 ++- + drivers/accel/amdxdna/aie2_msg_priv.h | 1 + + drivers/accel/amdxdna/aie2_pci.c | 15 +- + drivers/accel/amdxdna/aie2_pci.h | 6 +- + drivers/accel/amdxdna/amdxdna_mailbox.c | 2 +- + drivers/accel/amdxdna/npu4_regs.c | 1 + + drivers/accel/ethosu/ethosu_drv.c | 2 +- + drivers/accel/ivpu/ivpu_fw.c | 65 +- + drivers/accel/ivpu/ivpu_gem.c | 5 +- + drivers/accel/ivpu/ivpu_ipc.c | 11 +- + drivers/accel/ivpu/ivpu_job.c | 6 +- + drivers/accel/ivpu/ivpu_ms.c | 17 +- + drivers/accel/rocket/rocket_gem.c | 2 +- + drivers/gpu/drm/Kconfig | 7 +- + drivers/gpu/drm/Makefile | 5 +- + drivers/gpu/drm/adp/adp-mipi.c | 1 + + drivers/gpu/drm/amd/amdgpu/amdgpu_display.c | 2 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_gem.c | 20 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_kms.c | 2 +- + drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c | 1 - + .../amd/display/amdgpu_dm/amdgpu_dm_backlight.c | 2 +- + .../amd/display/amdgpu_dm/amdgpu_dm_connector.c | 1 - + drivers/gpu/drm/arm/display/komeda/komeda_crtc.c | 1 + + drivers/gpu/drm/armada/armada_fbdev.c | 2 +- + drivers/gpu/drm/bridge/Kconfig | 18 +- + drivers/gpu/drm/bridge/analogix/Kconfig | 1 + + drivers/gpu/drm/bridge/analogix/analogix_dp_core.c | 40 +- + drivers/gpu/drm/bridge/aux-bridge.c | 1 + + drivers/gpu/drm/bridge/cadence/cdns-dsi-core.c | 1 + + drivers/gpu/drm/bridge/fsl-ldb.c | 19 +- + drivers/gpu/drm/bridge/imx/imx93-pdfc.c | 1 + + drivers/gpu/drm/bridge/panel.c | 563 ------------- + drivers/gpu/drm/bridge/samsung-dsim.c | 24 +- + drivers/gpu/drm/bridge/ssd2825.c | 23 +- + drivers/gpu/drm/bridge/synopsys/dw-mipi-dsi.c | 11 +- + drivers/gpu/drm/bridge/synopsys/dw-mipi-dsi2.c | 11 +- + drivers/gpu/drm/bridge/tc358767.c | 64 +- + drivers/gpu/drm/bridge/tc358768.c | 25 +- + drivers/gpu/drm/bridge/ti-tdp158.c | 1 + + drivers/gpu/drm/bridge/waveshare-dsi.c | 17 +- + drivers/gpu/drm/clients/Makefile | 3 +- + drivers/gpu/drm/clients/drm_fbdev_client.c | 2 +- + .../drm_fbdev_helper.c} | 34 +- + drivers/gpu/drm/display/drm_bridge_connector.c | 1 + + drivers/gpu/drm/display/drm_dp_mst_topology.c | 6 + + drivers/gpu/drm/drm_bridge.c | 17 - + drivers/gpu/drm/drm_connector.c | 2 +- + drivers/gpu/drm/drm_crtc_helper_internal.h | 9 + + drivers/gpu/drm/drm_fbdev_dma.c | 2 +- + drivers/gpu/drm/drm_fbdev_shmem.c | 2 +- + drivers/gpu/drm/drm_fbdev_ttm.c | 2 +- + drivers/gpu/drm/drm_kms_helper_common.c | 39 + + drivers/gpu/drm/drm_of.c | 63 -- + drivers/gpu/drm/drm_panel.c | 664 +++++++++++++++- + drivers/gpu/drm/drm_panel_backlight_quirks.c | 2 +- + drivers/gpu/drm/drm_panel_orientation_quirks.c | 2 +- + drivers/gpu/drm/drm_syncobj.c | 35 +- + drivers/gpu/drm/drm_timeout.c | 95 +++ + drivers/gpu/drm/exynos/exynos_dp.c | 36 +- + drivers/gpu/drm/exynos/exynos_drm_fbdev.c | 2 +- + drivers/gpu/drm/gma500/fbdev.c | 2 +- + drivers/gpu/drm/i915/display/intel_fbdev.c | 2 +- + drivers/gpu/drm/i915/gem/i915_gem_wait.c | 17 +- + drivers/gpu/drm/imagination/pvr_fw.c | 130 +-- + drivers/gpu/drm/imagination/pvr_fw.h | 18 +- + drivers/gpu/drm/imx/dc/dc-kms.c | 1 + + drivers/gpu/drm/imx/dcss/Kconfig | 1 + + drivers/gpu/drm/imx/lcdc/imx-lcdc.c | 261 ++++-- + drivers/gpu/drm/ingenic/Kconfig | 1 + + drivers/gpu/drm/lima/lima_gem.c | 2 +- + drivers/gpu/drm/logicvc/Kconfig | 1 + + drivers/gpu/drm/mcde/Kconfig | 2 +- + drivers/gpu/drm/mcde/mcde_display.c | 1 + + drivers/gpu/drm/mcde/mcde_dsi.c | 45 +- + drivers/gpu/drm/msm/dp/dp_display.c | 1 + + drivers/gpu/drm/msm/dsi/dsi.c | 3 +- + drivers/gpu/drm/msm/msm_debugfs.c | 2 +- + drivers/gpu/drm/msm/msm_fbdev.c | 2 +- + drivers/gpu/drm/mxsfb/mxsfb_drv.c | 43 + + drivers/gpu/drm/mxsfb/mxsfb_drv.h | 2 + + drivers/gpu/drm/mxsfb/mxsfb_kms.c | 12 + + drivers/gpu/drm/nouveau/dispnv50/disp.c | 1 - + drivers/gpu/drm/omapdrm/dss/omapdss.h | 1 - + drivers/gpu/drm/omapdrm/dss/output.c | 42 +- + drivers/gpu/drm/omapdrm/omap_debugfs.c | 2 +- + drivers/gpu/drm/omapdrm/omap_fbdev.c | 2 +- + drivers/gpu/drm/panel/Kconfig | 29 +- + drivers/gpu/drm/panel/Makefile | 2 + + drivers/gpu/drm/panel/panel-ebbg-ft8719.c | 23 +- + drivers/gpu/drm/panel/panel-edp.c | 1 + + drivers/gpu/drm/panel/panel-ilitek-ili9882t.c | 234 +++++- + drivers/gpu/drm/panel/panel-novatek-nt51021.c | 875 +++++++++++++++++++++ + drivers/gpu/drm/panel/panel-raydium-rm69220.c | 417 ++++++++++ + drivers/gpu/drm/panel/panel-ronbo-rb070d30.c | 14 +- + drivers/gpu/drm/panel/panel-simple.c | 60 ++ + drivers/gpu/drm/panel/panel-visionox-vtdr6130.c | 221 +++++- + drivers/gpu/drm/panfrost/panfrost_drv.c | 2 +- + drivers/gpu/drm/panthor/panthor_drv.c | 1 - + drivers/gpu/drm/pl111/Kconfig | 1 + + drivers/gpu/drm/radeon/radeon_device.c | 2 +- + drivers/gpu/drm/radeon/radeon_fbdev.c | 2 +- + drivers/gpu/drm/renesas/rz-du/Kconfig | 13 + + drivers/gpu/drm/renesas/rz-du/Makefile | 1 + + drivers/gpu/drm/renesas/rz-du/rzg2l_du_drv.c | 24 +- + drivers/gpu/drm/renesas/rz-du/rzg2l_du_drv.h | 3 +- + drivers/gpu/drm/renesas/rz-du/rzg2l_du_encoder.c | 24 + + drivers/gpu/drm/renesas/rz-du/rzg2l_mipi_dsi.c | 162 +++- + drivers/gpu/drm/renesas/rz-du/rzg3l_lvds.c | 275 +++++++ + drivers/gpu/drm/renesas/rz-du/rzg3l_lvds_regs.h | 25 + + drivers/gpu/drm/renesas/shmobile/shmob_drm_crtc.c | 5 +- + drivers/gpu/drm/rockchip/Kconfig | 2 + + drivers/gpu/drm/rockchip/analogix_dp-rockchip.c | 11 - + drivers/gpu/drm/rockchip/rockchip_drm_gem.c | 2 +- + drivers/gpu/drm/scheduler/sched_entity.c | 51 +- + drivers/gpu/drm/scheduler/sched_main.c | 1 - + drivers/gpu/drm/sitronix/st7571.c | 1 - + drivers/gpu/drm/solomon/ssd130x.c | 30 +- + drivers/gpu/drm/stm/Kconfig | 1 + + drivers/gpu/drm/tegra/fbdev.c | 2 +- + drivers/gpu/drm/tegra/rgb.c | 1 + + drivers/gpu/drm/tegra/submit.c | 145 +++- + drivers/gpu/drm/tegra/uapi.c | 51 +- + drivers/gpu/drm/tegra/uapi.h | 3 +- + drivers/gpu/drm/tidss/Kconfig | 1 + + drivers/gpu/drm/tve200/Kconfig | 2 +- + drivers/gpu/drm/tve200/tve200_drm.h | 1 - + drivers/gpu/drm/tve200/tve200_drv.c | 31 +- + drivers/gpu/drm/v3d/v3d_bo.c | 3 +- + drivers/gpu/drm/v3d/v3d_drv.h | 10 - + drivers/gpu/drm/vboxvideo/vbox_mode.c | 2 +- + drivers/gpu/drm/vc4/vc4_gem.c | 3 +- + drivers/gpu/drm/verisilicon/vs_bridge.c | 11 +- + drivers/gpu/drm/verisilicon/vs_bridge.h | 1 - + drivers/gpu/drm/xe/xe_wait_user_fence.c | 2 +- + drivers/gpu/host1x/channel.c | 6 +- + drivers/gpu/host1x/context.c | 383 ++++++++- + drivers/gpu/host1x/context.h | 18 +- + drivers/gpu/host1x/dev.c | 52 +- + drivers/gpu/host1x/dev.h | 3 + + drivers/gpu/host1x/fence.c | 43 + + drivers/gpu/host1x/hw/cdma_hw.c | 3 + + drivers/gpu/host1x/hw/channel_hw.c | 3 +- + drivers/gpu/host1x/hw/debug_hw_1x06.c | 3 + + drivers/gpu/host1x/syncpt.c | 21 +- + drivers/video/fbdev/efifb.c | 2 +- + include/drm/bridge/analogix_dp.h | 1 - + .../drm_fbdev_helper.h} | 4 +- + include/drm/drm_bridge.h | 54 -- + include/drm/drm_of.h | 15 +- + include/drm/drm_panel.h | 99 ++- + include/drm/{drm_utils.h => drm_panel_quirks.h} | 10 +- + include/drm/drm_timeout.h | 17 + + include/drm/gpu_scheduler.h | 10 +- + include/linux/host1x.h | 53 +- + include/uapi/drm/tegra_drm.h | 4 +- + 184 files changed, 5054 insertions(+), 1843 deletions(-) + create mode 100644 Documentation/devicetree/bindings/display/panel/novatek,nt51021.yaml + create mode 100644 Documentation/devicetree/bindings/display/panel/raydium,rm69220.yaml + delete mode 100644 drivers/gpu/drm/bridge/panel.c + rename drivers/gpu/drm/{drm_fb_helper.c => clients/drm_fbdev_helper.c} (97%) + create mode 100644 drivers/gpu/drm/drm_timeout.c + create mode 100644 drivers/gpu/drm/panel/panel-novatek-nt51021.c + create mode 100644 drivers/gpu/drm/panel/panel-raydium-rm69220.c + create mode 100644 drivers/gpu/drm/renesas/rz-du/rzg3l_lvds.c + create mode 100644 drivers/gpu/drm/renesas/rz-du/rzg3l_lvds_regs.h + rename include/drm/{drm_fb_helper.h => clients/drm_fbdev_helper.h} (99%) + rename include/drm/{drm_utils.h => drm_panel_quirks.h} (57%) + create mode 100644 include/drm/drm_timeout.h +Merging amdgpu/drm-next (41505ac433cc4 Revert "drm/amd/display: Attach blend mode property for alpha planes") +$ git merge -m Merge branch 'drm-next' of https://gitlab.freedesktop.org/agd5f/linux.git amdgpu/drm-next +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_device.c +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_display.c +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_gem.c +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_kms.c +Auto-merging drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c +Auto-merging drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_backlight.c +Auto-merging drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_connector.c +Merge made by the 'ort' strategy. + drivers/gpu/drm/amd/amdgpu/Makefile | 13 +- + drivers/gpu/drm/amd/amdgpu/amdgpu.h | 2 + + drivers/gpu/drm/amd/amdgpu/amdgpu_acp.c | 13 +- + .../gpu/drm/amd/amdgpu/amdgpu_amdkfd_gc_9_4_3.c | 3 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v9.c | 2 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_atomfirmware.c | 2 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c | 10 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_debugfs.c | 8 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_debugfs.h | 2 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_device.c | 4 + + drivers/gpu/drm/amd/amdgpu/amdgpu_discovery.c | 87 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_display.c | 2 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_dma_buf.c | 20 + + drivers/gpu/drm/amd/amdgpu/amdgpu_drv.c | 43 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_fence.c | 41 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_gem.c | 89 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_gfx.c | 146 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_gfx.h | 21 + + drivers/gpu/drm/amd/amdgpu/amdgpu_gmc.c | 237 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_gmc.h | 8 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_irq.c | 35 + + drivers/gpu/drm/amd/amdgpu/amdgpu_irq.h | 1 + + drivers/gpu/drm/amd/amdgpu/amdgpu_jpeg.c | 4 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_kms.c | 12 + + drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c | 749 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h | 59 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_object.c | 3 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_psp.c | 57 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_psp.h | 2 + + drivers/gpu/drm/amd/amdgpu/amdgpu_ras.c | 3 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_rlc.h | 36 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_ttm.c | 8 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_ualink.c | 85 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_ucode.c | 38 + + drivers/gpu/drm/amd/amdgpu/amdgpu_ucode.h | 5 + + drivers/gpu/drm/amd/amdgpu/amdgpu_userq.c | 2 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_userq_fence.c | 9 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_vcn.c | 4 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_virt.c | 37 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_vm.c | 18 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_vm.h | 1 - + drivers/gpu/drm/amd/amdgpu/amdgpu_vm_internal.h | 43 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_vm_pt.c | 40 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_vpe.c | 32 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_vpe.h | 1 + + drivers/gpu/drm/amd/amdgpu/cik_ih.c | 31 - + drivers/gpu/drm/amd/amdgpu/cz_ih.c | 31 - + drivers/gpu/drm/amd/amdgpu/gfx_v10_0.c | 6 +- + drivers/gpu/drm/amd/amdgpu/gfx_v11_0.c | 71 +- + drivers/gpu/drm/amd/amdgpu/gfx_v12_0.c | 75 +- + drivers/gpu/drm/amd/amdgpu/gfx_v12_1.c | 38 +- + drivers/gpu/drm/amd/amdgpu/gfx_v6_0.c | 9 + + drivers/gpu/drm/amd/amdgpu/gfx_v9_0.c | 4 +- + drivers/gpu/drm/amd/amdgpu/gfx_v9_4_2.c | 11 +- + drivers/gpu/drm/amd/amdgpu/gfx_v9_4_2.h | 2 +- + drivers/gpu/drm/amd/amdgpu/gfx_v9_4_3.c | 4 +- + drivers/gpu/drm/amd/amdgpu/gfxhub_v12_1.c | 24 +- + drivers/gpu/drm/amd/amdgpu/gmc_v12_0.c | 2 +- + drivers/gpu/drm/amd/amdgpu/gmc_v12_1.c | 96 +- + drivers/gpu/drm/amd/amdgpu/gmc_v6_0.c | 85 - + drivers/gpu/drm/amd/amdgpu/gmc_v7_0.c | 83 - + drivers/gpu/drm/amd/amdgpu/gmc_v9_0.c | 7 - + drivers/gpu/drm/amd/amdgpu/hdp_v8_0.c | 132 + + drivers/gpu/drm/amd/amdgpu/hdp_v8_0.h | 31 + + drivers/gpu/drm/amd/amdgpu/iceland_ih.c | 31 - + drivers/gpu/drm/amd/amdgpu/ih_v6_0.c | 7 - + drivers/gpu/drm/amd/amdgpu/ih_v6_1.c | 7 - + drivers/gpu/drm/amd/amdgpu/ih_v7_0.c | 7 - + drivers/gpu/drm/amd/amdgpu/ih_v8_0.c | 781 + + drivers/gpu/drm/amd/amdgpu/ih_v8_0.h | 28 + + drivers/gpu/drm/amd/amdgpu/isp_v4_1_1.c | 2 +- + drivers/gpu/drm/amd/amdgpu/lsdma_v8_0.c | 138 + + drivers/gpu/drm/amd/amdgpu/lsdma_v8_0.h | 31 + + drivers/gpu/drm/amd/amdgpu/mes_userqueue.c | 52 +- + drivers/gpu/drm/amd/amdgpu/mes_userqueue.h | 4 + + drivers/gpu/drm/amd/amdgpu/mes_v11_0.c | 198 +- + drivers/gpu/drm/amd/amdgpu/mes_v12_0.c | 184 +- + drivers/gpu/drm/amd/amdgpu/mmhub_v5_0.c | 603 + + drivers/gpu/drm/amd/amdgpu/mmhub_v5_0.h | 28 + + drivers/gpu/drm/amd/amdgpu/navi10_ih.c | 7 - + drivers/gpu/drm/amd/amdgpu/nbif_v7_10.c | 350 + + drivers/gpu/drm/amd/amdgpu/nbif_v7_10.h | 32 + + drivers/gpu/drm/amd/amdgpu/psp_gfx_if.h | 5 + + drivers/gpu/drm/amd/amdgpu/psp_v15_0_3.c | 836 + + drivers/gpu/drm/amd/amdgpu/psp_v15_0_3.h | 32 + + drivers/gpu/drm/amd/amdgpu/sdma_v4_0.c | 2 +- + drivers/gpu/drm/amd/amdgpu/sdma_v4_4_2.c | 2 +- + drivers/gpu/drm/amd/amdgpu/sdma_v5_0.c | 2 +- + drivers/gpu/drm/amd/amdgpu/sdma_v5_2.c | 2 +- + drivers/gpu/drm/amd/amdgpu/sdma_v6_0.c | 2 +- + drivers/gpu/drm/amd/amdgpu/sdma_v7_0.c | 2 +- + drivers/gpu/drm/amd/amdgpu/sdma_v7_1.c | 9 +- + drivers/gpu/drm/amd/amdgpu/si_ih.c | 30 - + drivers/gpu/drm/amd/amdgpu/smuio_v15_0_3.c | 84 + + drivers/gpu/drm/amd/amdgpu/smuio_v15_0_3.h | 30 + + drivers/gpu/drm/amd/amdgpu/soc_v1_0.c | 17 - + drivers/gpu/drm/amd/amdgpu/soc_v1_0.h | 22 + + drivers/gpu/drm/amd/amdgpu/tonga_ih.c | 32 - + drivers/gpu/drm/amd/amdgpu/vcn_v1_0.c | 2 +- + drivers/gpu/drm/amd/amdgpu/vega10_ih.c | 8 - + drivers/gpu/drm/amd/amdgpu/vega20_ih.c | 8 - + drivers/gpu/drm/amd/amdgpu/vi.c | 1 + + drivers/gpu/drm/amd/amdgpu/vpe_v3_0.c | 351 + + drivers/gpu/drm/amd/amdgpu/vpe_v3_0.h | 29 + + drivers/gpu/drm/amd/amdkfd/kfd_device.c | 3 + + .../gpu/drm/amd/amdkfd/kfd_device_queue_manager.c | 36 +- + .../gpu/drm/amd/amdkfd/kfd_device_queue_manager.h | 1 + + drivers/gpu/drm/amd/amdkfd/kfd_int_process_v9.c | 12 +- + drivers/gpu/drm/amd/amdkfd/kfd_priv.h | 2 - + drivers/gpu/drm/amd/amdkfd/kfd_svm.c | 13 +- + drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c | 22 +- + .../amd/display/amdgpu_dm/amdgpu_dm_backlight.c | 92 +- + .../amd/display/amdgpu_dm/amdgpu_dm_connector.c | 59 +- + .../amd/display/amdgpu_dm/amdgpu_dm_mst_types.c | 6 +- + .../amdgpu_dm/tests/amdgpu_dm_connector_test.c | 155 + + .../amd/display/amdgpu_dm/tests/amdgpu_dm_test.c | 10 - + drivers/gpu/drm/amd/display/dc/Makefile | 3 + + .../amd/display/dc/clk_mgr/dcn60/dcn60_clk_mgr.c | 203 +- + drivers/gpu/drm/amd/display/dc/core/dc.c | 30 +- + .../gpu/drm/amd/display/dc/core/dc_hw_sequencer.c | 146 +- + drivers/gpu/drm/amd/display/dc/core/dc_resource.c | 2 +- + .../gpu/drm/amd/display/dc/core2/CMakeLists.txt | 10 + + drivers/gpu/drm/amd/display/dc/core2/dc_core2.c | 34 + + drivers/gpu/drm/amd/display/dc/core2/dc_core2.h | 19 + + .../gpu/drm/amd/display/dc/core2/dc_core2_init.c | 13 + + .../gpu/drm/amd/display/dc/core2/dc_core2_init.h | 10 + + drivers/gpu/drm/amd/display/dc/dc.h | 6 +- + .../amd/display/dc/dc_api_dispatch/CMakeLists.txt | 9 + + .../drm/amd/display/dc/dc_api_dispatch/api_shim.c | 32 + + .../drm/amd/display/dc/dc_api_dispatch/api_shim.h | 19 + + drivers/gpu/drm/amd/display/dc/dc_dmub_srv.c | 9 +- + drivers/gpu/drm/amd/display/dc/dc_hw_types.h | 9 + + drivers/gpu/drm/amd/display/dc/dc_memory_pool.c | 10 +- + drivers/gpu/drm/amd/display/dc/dml/Makefile | 2 +- + .../gpu/drm/amd/display/dc/dml/dcn314/dcn314_fpu.c | 4 +- + drivers/gpu/drm/amd/display/dc/dml2_0/Makefile | 2 +- + .../src/dml2_core/dml2_core_dcn6_calcs_dchub.c | 2 +- + .../gpu/drm/amd/display/dc/dml2_wrapper/Makefile | 5 +- + drivers/gpu/drm/amd/display/dc/dpp/Makefile | 2 +- + .../gpu/drm/amd/display/dc/dpp/dcn30/dcn30_dpp.h | 22 + + .../drm/amd/display/dc/dpp/dcn30/dcn30_dpp_cm.c | 10 +- + .../gpu/drm/amd/display/dc/dpp/dcn60/dcn60_dpp.c | 2 +- + .../gpu/drm/amd/display/dc/dpp/dcn60/dcn60_dpp.h | 3 + + .../drm/amd/display/dc/dpp/dcn60/dcn60_dpp_cm.c | 115 + + .../amd/display/dc/hubbub/dcn401/dcn401_hubbub.c | 1 + + .../drm/amd/display/dc/hubbub/dcn42/dcn42_hubbub.c | 42 +- + .../drm/amd/display/dc/hubbub/dcn42/dcn42_hubbub.h | 7 +- + .../drm/amd/display/dc/hubbub/dcn60/dcn60_hubbub.c | 9 + + .../gpu/drm/amd/display/dc/hubp/dcn60/dcn60_hubp.c | 5 +- + .../drm/amd/display/dc/hwss/dce110/dce110_hwseq.c | 10 +- + .../drm/amd/display/dc/hwss/dcn20/dcn20_hwseq.c | 10 +- + .../drm/amd/display/dc/hwss/dcn201/dcn201_hwseq.c | 3 +- + .../drm/amd/display/dc/hwss/dcn30/dcn30_hwseq.c | 4 +- + .../drm/amd/display/dc/hwss/dcn32/dcn32_hwseq.c | 6 +- + .../drm/amd/display/dc/hwss/dcn401/dcn401_hwseq.c | 50 +- + .../drm/amd/display/dc/hwss/dcn60/dcn60_hwseq.c | 5 + + drivers/gpu/drm/amd/display/dc/hwss/hw_sequencer.h | 51 +- + .../gpu/drm/amd/display/dc/inc/dc_core_interface.h | 31 + + .../drm/amd/display/dc/inc/hw/clk_mgr_internal.h | 9 + + drivers/gpu/drm/amd/display/dc/inc/hw/dccg.h | 9 - + drivers/gpu/drm/amd/display/dc/inc/hw/dchubbub.h | 4 + + drivers/gpu/drm/amd/display/dc/inc/hw/opp.h | 3 +- + drivers/gpu/drm/amd/display/dc/link/link_dpms.c | 2 + + .../drm/amd/display/dc/link/protocols/link_ddc.c | 8 +- + .../drm/amd/display/dc/link/protocols/link_ddc.h | 1 + + .../dc/link/protocols/link_edp_panel_control.c | 4 +- + .../amd/display/dc/link/protocols/link_hdmi_frl.c | 56 +- + .../gpu/drm/amd/display/dc/opp/dcn20/dcn20_opp.c | 5 +- + .../gpu/drm/amd/display/dc/opp/dcn20/dcn20_opp.h | 3 +- + .../gpu/drm/amd/display/dc/optc/dcn60/dcn60_optc.c | 83 +- + .../amd/display/dc/resource/dce60/dce60_resource.c | 13 +- + .../amd/display/dc/resource/dce80/dce80_resource.c | 2 + + .../display/dc/resource/dcn42b/dcn42b_resource.c | 2 + + .../amd/display/dc/resource/dcn60/dcn60_resource.c | 2 + + drivers/gpu/drm/amd/display/dc/sspl/spl_os_types.h | 1 - + drivers/gpu/drm/amd/display/modules/power/power.c | 2 + + .../amd/include/asic_reg/hdp/hdp_8_0_1_offset.h | 213 + + .../amd/include/asic_reg/hdp/hdp_8_0_1_sh_mask.h | 665 + + .../include/asic_reg/lsdma/lsdma_8_0_1_offset.h | 385 + + .../include/asic_reg/lsdma/lsdma_8_0_1_sh_mask.h | 1399 + + .../include/asic_reg/mmhub/mmhub_5_0_1_offset.h | 1494 +- + .../include/asic_reg/mmhub/mmhub_5_0_1_sh_mask.h | 7489 ++- + .../drm/amd/include/asic_reg/mp/mp_15_0_3_offset.h | 1105 + + .../amd/include/asic_reg/mp/mp_15_0_3_sh_mask.h | 1646 + + .../amd/include/asic_reg/nbif/nbif_7_10_0_offset.h | 15938 +++++ + .../include/asic_reg/nbif/nbif_7_10_0_sh_mask.h | 60316 +++++++++++++++++++ + .../amd/include/asic_reg/oss/osssys_8_0_1_offset.h | 275 + + .../include/asic_reg/oss/osssys_8_0_1_sh_mask.h | 984 + + .../amd/include/asic_reg/pcie/pcie_6_1_3_offset.h | 1755 + + .../amd/include/asic_reg/pcie/pcie_6_1_3_sh_mask.h | 10821 ++++ + .../include/asic_reg/smuio/smuio_15_0_3_offset.h | 79 + + .../include/asic_reg/smuio/smuio_15_0_3_sh_mask.h | 164 + + .../amd/include/asic_reg/vpe/vpe_3_0_0_offset.h | 877 + + .../amd/include/asic_reg/vpe/vpe_3_0_0_sh_mask.h | 2661 + + drivers/gpu/drm/amd/include/discovery.h | 42 + + drivers/gpu/drm/amd/include/mes_v11_api_def.h | 18 + + drivers/gpu/drm/amd/include/mes_v12_api_def.h | 17 + + drivers/gpu/drm/amd/include/soc_v2_0_ih_clientid.h | 51 + + drivers/gpu/drm/amd/pm/inc/amdgpu_dpm.h | 2 + + drivers/gpu/drm/amd/pm/legacy-dpm/si_dpm.c | 39 +- + .../gpu/drm/amd/pm/swsmu/smu15/smu_v15_0_8_ppt.c | 5 +- + drivers/gpu/drm/amd/ras/core/Makefile | 2 + + drivers/gpu/drm/amd/ras/core/eeprom.c | 6 +- + drivers/gpu/drm/amd/ras/core/ras_eeprom_mgr.c | 1 + + drivers/gpu/drm/amd/ras/core/ras_gfx.c | 3 + + drivers/gpu/drm/amd/ras/core/ras_mp1.c | 2 + + drivers/gpu/drm/amd/ras/core/ras_nbio.c | 3 + + drivers/gpu/drm/amd/ras/core/ras_psp.c | 4 + + drivers/gpu/drm/amd/ras/core/ras_psp_v15_0_3.c | 114 + + drivers/gpu/drm/amd/ras/core/ras_psp_v15_0_3.h | 31 + + drivers/gpu/drm/amd/ras/core/ras_umc.c | 4 + + drivers/gpu/drm/amd/ras/core/ras_umc_v13_0.c | 221 + + drivers/gpu/drm/amd/ras/core/ras_umc_v13_0.h | 70 + + .../drm/amd/ras/ras_mgr/amdgpu_ras_eeprom_i2c.c | 1 + + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mce.c | 15 +- + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mgr.c | 26 +- + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mp1.c | 6 +- + drivers/gpu/drm/radeon/radeon_bios.c | 6 +- + drivers/gpu/drm/radeon/radeon_combios.c | 51 + + drivers/gpu/drm/radeon/radeon_mode.h | 3 +- + include/uapi/drm/amdgpu_drm.h | 2 + + 221 files changed, 116083 insertions(+), 1292 deletions(-) + create mode 100644 drivers/gpu/drm/amd/amdgpu/hdp_v8_0.c + create mode 100644 drivers/gpu/drm/amd/amdgpu/hdp_v8_0.h + create mode 100644 drivers/gpu/drm/amd/amdgpu/ih_v8_0.c + create mode 100644 drivers/gpu/drm/amd/amdgpu/ih_v8_0.h + create mode 100644 drivers/gpu/drm/amd/amdgpu/lsdma_v8_0.c + create mode 100644 drivers/gpu/drm/amd/amdgpu/lsdma_v8_0.h + create mode 100644 drivers/gpu/drm/amd/amdgpu/mmhub_v5_0.c + create mode 100644 drivers/gpu/drm/amd/amdgpu/mmhub_v5_0.h + create mode 100644 drivers/gpu/drm/amd/amdgpu/nbif_v7_10.c + create mode 100644 drivers/gpu/drm/amd/amdgpu/nbif_v7_10.h + create mode 100644 drivers/gpu/drm/amd/amdgpu/psp_v15_0_3.c + create mode 100644 drivers/gpu/drm/amd/amdgpu/psp_v15_0_3.h + create mode 100644 drivers/gpu/drm/amd/amdgpu/smuio_v15_0_3.c + create mode 100644 drivers/gpu/drm/amd/amdgpu/smuio_v15_0_3.h + create mode 100644 drivers/gpu/drm/amd/amdgpu/vpe_v3_0.c + create mode 100644 drivers/gpu/drm/amd/amdgpu/vpe_v3_0.h + create mode 100644 drivers/gpu/drm/amd/display/dc/core2/CMakeLists.txt + create mode 100644 drivers/gpu/drm/amd/display/dc/core2/dc_core2.c + create mode 100644 drivers/gpu/drm/amd/display/dc/core2/dc_core2.h + create mode 100644 drivers/gpu/drm/amd/display/dc/core2/dc_core2_init.c + create mode 100644 drivers/gpu/drm/amd/display/dc/core2/dc_core2_init.h + create mode 100644 drivers/gpu/drm/amd/display/dc/dc_api_dispatch/CMakeLists.txt + create mode 100644 drivers/gpu/drm/amd/display/dc/dc_api_dispatch/api_shim.c + create mode 100644 drivers/gpu/drm/amd/display/dc/dc_api_dispatch/api_shim.h + create mode 100644 drivers/gpu/drm/amd/display/dc/dpp/dcn60/dcn60_dpp_cm.c + create mode 100644 drivers/gpu/drm/amd/display/dc/inc/dc_core_interface.h + create mode 100644 drivers/gpu/drm/amd/include/asic_reg/hdp/hdp_8_0_1_offset.h + create mode 100644 drivers/gpu/drm/amd/include/asic_reg/hdp/hdp_8_0_1_sh_mask.h + create mode 100644 drivers/gpu/drm/amd/include/asic_reg/lsdma/lsdma_8_0_1_offset.h + create mode 100644 drivers/gpu/drm/amd/include/asic_reg/lsdma/lsdma_8_0_1_sh_mask.h + create mode 100644 drivers/gpu/drm/amd/include/asic_reg/mp/mp_15_0_3_offset.h + create mode 100644 drivers/gpu/drm/amd/include/asic_reg/mp/mp_15_0_3_sh_mask.h + create mode 100644 drivers/gpu/drm/amd/include/asic_reg/nbif/nbif_7_10_0_offset.h + create mode 100644 drivers/gpu/drm/amd/include/asic_reg/nbif/nbif_7_10_0_sh_mask.h + create mode 100644 drivers/gpu/drm/amd/include/asic_reg/oss/osssys_8_0_1_offset.h + create mode 100644 drivers/gpu/drm/amd/include/asic_reg/oss/osssys_8_0_1_sh_mask.h + create mode 100644 drivers/gpu/drm/amd/include/asic_reg/pcie/pcie_6_1_3_offset.h + create mode 100644 drivers/gpu/drm/amd/include/asic_reg/pcie/pcie_6_1_3_sh_mask.h + create mode 100644 drivers/gpu/drm/amd/include/asic_reg/smuio/smuio_15_0_3_offset.h + create mode 100644 drivers/gpu/drm/amd/include/asic_reg/smuio/smuio_15_0_3_sh_mask.h + create mode 100644 drivers/gpu/drm/amd/include/asic_reg/vpe/vpe_3_0_0_offset.h + create mode 100644 drivers/gpu/drm/amd/include/asic_reg/vpe/vpe_3_0_0_sh_mask.h + create mode 100644 drivers/gpu/drm/amd/include/soc_v2_0_ih_clientid.h + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_psp_v15_0_3.c + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_psp_v15_0_3.h + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_umc_v13_0.c + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_umc_v13_0.h +Merging drm-intel/for-linux-next (2c1bb96681bea drm/intel: rename I915_GTT_VIEW_* enumerations to INTEL_GTT_VIEW_*) +$ git merge -m Merge branch 'for-linux-next' of https://gitlab.freedesktop.org/drm/i915/kernel.git drm-intel/for-linux-next +Auto-merging drivers/gpu/drm/i915/display/intel_display_params.h +Auto-merging drivers/gpu/drm/i915/display/intel_display_types.h +Auto-merging drivers/gpu/drm/i915/display/intel_dp_link_training.c +Auto-merging drivers/gpu/drm/i915/display/intel_dp_mst.c +Auto-merging drivers/gpu/drm/i915/display/intel_psr.c +Auto-merging drivers/gpu/drm/i915/display/intel_vrr.c +Auto-merging drivers/gpu/drm/i915/display/skl_universal_plane.c +Auto-merging drivers/gpu/drm/xe/Makefile +Auto-merging drivers/gpu/drm/xe/display/xe_fb_pin.c +Auto-merging drivers/gpu/drm/xe/xe_device_types.h +Auto-merging drivers/gpu/drm/xe/xe_pci.c +CONFLICT (content): Merge conflict in drivers/gpu/drm/xe/xe_pci.c +Auto-merging drivers/gpu/drm/xe/xe_pm.c +Auto-merging drivers/gpu/drm/xe/xe_pm.h +Resolved 'drivers/gpu/drm/xe/xe_pci.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 4ad3bf9484964] Merge branch 'for-linux-next' of https://gitlab.freedesktop.org/drm/i915/kernel.git +$ git diff -M --stat --summary HEAD^.. + drivers/gpu/drm/i915/Makefile | 1 + + drivers/gpu/drm/i915/display/intel_alpm.c | 166 ++++++++++-- + drivers/gpu/drm/i915/display/intel_alpm.h | 3 + + drivers/gpu/drm/i915/display/intel_bios.c | 262 ++++++++++++++++++- + drivers/gpu/drm/i915/display/intel_bios.h | 14 ++ + drivers/gpu/drm/i915/display/intel_color.c | 76 ++++-- + .../gpu/drm/i915/display/intel_color_pipeline.c | 34 ++- + .../gpu/drm/i915/display/intel_crtc_state_dump.c | 9 + + drivers/gpu/drm/i915/display/intel_cx0_phy.c | 5 +- + drivers/gpu/drm/i915/display/intel_ddi.c | 37 ++- + drivers/gpu/drm/i915/display/intel_ddi_buf_trans.c | 90 ++++++- + drivers/gpu/drm/i915/display/intel_dip.c | 215 ++++++++++++++++ + drivers/gpu/drm/i915/display/intel_dip.h | 53 ++++ + drivers/gpu/drm/i915/display/intel_dip_regs.h | 38 +++ + drivers/gpu/drm/i915/display/intel_display.c | 60 ++++- + drivers/gpu/drm/i915/display/intel_display_core.h | 7 + + drivers/gpu/drm/i915/display/intel_display_irq.c | 18 +- + .../gpu/drm/i915/display/intel_display_limits.h | 1 + + .../gpu/drm/i915/display/intel_display_params.h | 5 + + drivers/gpu/drm/i915/display/intel_display_power.c | 9 + + .../gpu/drm/i915/display/intel_display_power_map.c | 42 +++- + drivers/gpu/drm/i915/display/intel_display_rpm.c | 7 + + drivers/gpu/drm/i915/display/intel_display_rpm.h | 1 + + drivers/gpu/drm/i915/display/intel_display_types.h | 24 +- + drivers/gpu/drm/i915/display/intel_dp.c | 279 +++++++++++++++------ + .../gpu/drm/i915/display/intel_dp_aux_backlight.c | 78 +++--- + drivers/gpu/drm/i915/display/intel_dp_hdcp.c | 176 ++++++------- + .../gpu/drm/i915/display/intel_dp_link_training.c | 54 ++-- + drivers/gpu/drm/i915/display/intel_dp_mst.c | 2 +- + drivers/gpu/drm/i915/display/intel_dp_test.c | 77 +++--- + drivers/gpu/drm/i915/display/intel_fb.c | 20 +- + drivers/gpu/drm/i915/display/intel_fbc.c | 12 +- + drivers/gpu/drm/i915/display/intel_hotplug.c | 5 + + drivers/gpu/drm/i915/display/intel_lspcon.c | 82 +++--- + drivers/gpu/drm/i915/display/intel_parent.c | 4 +- + drivers/gpu/drm/i915/display/intel_parent.h | 6 +- + drivers/gpu/drm/i915/display/intel_plane.c | 53 +++- + drivers/gpu/drm/i915/display/intel_psr.c | 76 +++--- + drivers/gpu/drm/i915/display/intel_psr_regs.h | 2 + + drivers/gpu/drm/i915/display/intel_vrr.c | 14 +- + drivers/gpu/drm/i915/display/intel_vrr_regs.h | 6 - + drivers/gpu/drm/i915/display/skl_universal_plane.c | 57 +++-- + drivers/gpu/drm/i915/gem/i915_gem_domain.c | 4 +- + drivers/gpu/drm/i915/gem/i915_gem_mman.c | 14 +- + drivers/gpu/drm/i915/gem/i915_gem_object.h | 2 +- + drivers/gpu/drm/i915/gem/selftests/i915_gem_mman.c | 4 +- + drivers/gpu/drm/i915/gt/intel_engine_cs.c | 2 +- + .../gpu/drm/i915/gt/intel_execlists_submission.c | 17 +- + drivers/gpu/drm/i915/gt/intel_reset.c | 8 +- + drivers/gpu/drm/i915/gt/selftest_engine_pm.c | 8 +- + drivers/gpu/drm/i915/gt/uc/intel_guc.h | 2 +- + drivers/gpu/drm/i915/gvt/aperture_gm.c | 4 +- + drivers/gpu/drm/i915/i915_debugfs.c | 22 +- + drivers/gpu/drm/i915/i915_dpt.c | 2 - + drivers/gpu/drm/i915/i915_drv.h | 2 - + drivers/gpu/drm/i915/i915_fb_pin.c | 6 - + drivers/gpu/drm/i915/i915_gem.c | 6 +- + drivers/gpu/drm/i915/i915_gem.h | 6 +- + drivers/gpu/drm/i915/i915_gtt_view_types.h | 74 ------ + drivers/gpu/drm/i915/i915_overlay.c | 5 - + drivers/gpu/drm/i915/i915_request.c | 2 - + drivers/gpu/drm/i915/i915_vma.c | 44 ++-- + drivers/gpu/drm/i915/i915_vma.h | 12 +- + drivers/gpu/drm/i915/i915_vma_types.h | 29 ++- + drivers/gpu/drm/i915/selftests/i915_vma.c | 60 ++--- + drivers/gpu/drm/i915/selftests/igt_atomic.c | 7 + + drivers/gpu/drm/i915/vlv_iosf_sb.c | 6 +- + drivers/gpu/drm/xe/Makefile | 1 + + .../gpu/drm/xe/compat-i915-headers/i915_config.h | 16 -- + .../xe/compat-i915-headers/i915_gtt_view_types.h | 7 - + drivers/gpu/drm/xe/display/xe_display.c | 10 +- + drivers/gpu/drm/xe/display/xe_display_rpm.c | 6 + + drivers/gpu/drm/xe/display/xe_fb_pin.c | 28 +-- + drivers/gpu/drm/xe/xe_device_types.h | 13 + + drivers/gpu/drm/xe/xe_pci.c | 22 +- + drivers/gpu/drm/xe/xe_pm.c | 45 ++++ + drivers/gpu/drm/xe/xe_pm.h | 2 + + include/drm/intel/display_parent_interface.h | 9 +- + include/drm/intel/gtt_view_types.h | 79 ++++++ + 79 files changed, 1987 insertions(+), 779 deletions(-) + create mode 100644 drivers/gpu/drm/i915/display/intel_dip.c + create mode 100644 drivers/gpu/drm/i915/display/intel_dip.h + create mode 100644 drivers/gpu/drm/i915/display/intel_dip_regs.h + delete mode 100644 drivers/gpu/drm/i915/i915_gtt_view_types.h + delete mode 100644 drivers/gpu/drm/xe/compat-i915-headers/i915_config.h + delete mode 100644 drivers/gpu/drm/xe/compat-i915-headers/i915_gtt_view_types.h + create mode 100644 include/drm/intel/gtt_view_types.h +Merging drm-msm/msm-next (5e4a3f7b26206 drm/msm/dp: add stream-aware link register accessors) +$ git merge -m Merge branch 'msm-next' of https://gitlab.freedesktop.org/drm/msm.git drm-msm/msm-next +Auto-merging drivers/gpu/drm/msm/dp/dp_display.c +Auto-merging drivers/gpu/drm/msm/msm_gem_shrinker.c +Merge made by the 'ort' strategy. + .../devicetree/bindings/display/msm/gmu.yaml | 114 ++- + .../devicetree/bindings/display/msm/gpu.yaml | 7 + + .../devicetree/bindings/display/msm/hdmi.yaml | 4 +- + .../devicetree/bindings/display/msm/qcom,mdp5.yaml | 1 + + drivers/gpu/drm/msm/adreno/a3xx_gpu.c | 12 +- + drivers/gpu/drm/msm/adreno/a6xx_catalog.c | 994 +++++++++++++++++++++ + drivers/gpu/drm/msm/adreno/a6xx_gmu.c | 231 +++-- + drivers/gpu/drm/msm/adreno/a6xx_gmu.h | 1 + + drivers/gpu/drm/msm/adreno/a6xx_gpu.h | 17 + + drivers/gpu/drm/msm/adreno/a6xx_hfi.c | 96 ++ + drivers/gpu/drm/msm/adreno/a6xx_hfi.h | 1 + + drivers/gpu/drm/msm/adreno/a8xx_gpu.c | 18 +- + drivers/gpu/drm/msm/adreno/adreno_gpu.h | 33 +- + drivers/gpu/drm/msm/disp/dpu1/dpu_hw_catalog.c | 2 + + drivers/gpu/drm/msm/disp/mdp5/mdp5_cfg.c | 81 +- + drivers/gpu/drm/msm/dp/dp_ctrl.c | 161 +++- + drivers/gpu/drm/msm/dp/dp_ctrl.h | 8 +- + drivers/gpu/drm/msm/dp/dp_debug.c | 1 + + drivers/gpu/drm/msm/dp/dp_display.c | 115 ++- + drivers/gpu/drm/msm/dp/dp_panel.c | 191 +++- + drivers/gpu/drm/msm/dp/dp_panel.h | 17 +- + drivers/gpu/drm/msm/dp/dp_reg.h | 56 ++ + drivers/gpu/drm/msm/dsi/dsi_host.c | 2 + + drivers/gpu/drm/msm/dsi/dsi_manager.c | 2 +- + drivers/gpu/drm/msm/hdmi/hdmi_bridge.c | 34 +- + drivers/gpu/drm/msm/hdmi/hdmi_phy.c | 16 +- + drivers/gpu/drm/msm/msm_drv.c | 24 + + drivers/gpu/drm/msm/msm_gem_shrinker.c | 9 +- + drivers/gpu/drm/msm/msm_io_utils.c | 6 +- + drivers/gpu/drm/msm/registers/adreno/a6xx.xml | 8 + + 30 files changed, 1996 insertions(+), 266 deletions(-) +Merging drm-msm-lumag/msm-next-lumag (5e4a3f7b26206 drm/msm/dp: add stream-aware link register accessors) +$ git merge -m Merge branch 'msm-next-lumag' of https://gitlab.freedesktop.org/lumag/msm.git drm-msm-lumag/msm-next-lumag +Already up to date. +Merging drm-xe/drm-xe-next (cf4171a20d139 drm/xe/guc: Fix race around q->guc->suspend_pending access) +$ git merge -m Merge branch 'drm-xe-next' of https://gitlab.freedesktop.org/drm/xe/kernel.git drm-xe/drm-xe-next +Auto-merging drivers/gpu/drm/xe/xe_vm.c +Merge made by the 'ort' strategy. + drivers/gpu/drm/xe/xe_device.c | 29 ++++++++++++++ + drivers/gpu/drm/xe/xe_device.h | 2 + + drivers/gpu/drm/xe/xe_execlist.c | 6 ++- + drivers/gpu/drm/xe/xe_gpu_scheduler_types.h | 5 ++- + drivers/gpu/drm/xe/xe_guc_exec_queue_types.h | 5 ++- + drivers/gpu/drm/xe/xe_guc_submit.c | 58 ++++++++++++++++++++-------- + drivers/gpu/drm/xe/xe_lrc.c | 38 +++++++++++------- + drivers/gpu/drm/xe/xe_module.c | 24 ++++++++++-- + drivers/gpu/drm/xe/xe_vm.c | 18 +++++++-- + include/drm/intel/pciids.h | 3 +- + 10 files changed, 148 insertions(+), 40 deletions(-) +$ git am -3 ../patches/0001-drm-xe-Fix-up-merge-issue.patch +Applying: drm: xe: Fix up merge issue +Using index info to reconstruct a base tree... +M drivers/gpu/drm/xe/xe_ttm_vram_mgr.c +Falling back to patching base and 3-way merge... +Auto-merging drivers/gpu/drm/xe/xe_ttm_vram_mgr.c +No changes -- Patch already applied. +Merging drm-rust/for-linux-next (df718311a8418 gpu: nova-core: document the GIN interrupt controller and GSP events) +$ git merge -m Merge branch 'for-linux-next' of https://gitlab.freedesktop.org/drm/rust/kernel.git drm-rust/for-linux-next +Auto-merging MAINTAINERS +Auto-merging rust/bindings/bindings_helper.h +Auto-merging rust/helpers/helpers.c +Auto-merging rust/kernel/bitfield.rs +Auto-merging rust/kernel/lib.rs +Auto-merging rust/kernel/mem.rs +CONFLICT (add/add): Merge conflict in rust/kernel/mem.rs +Auto-merging rust/kernel/pci/irq.rs +Auto-merging rust/macros/lib.rs +Resolved 'rust/kernel/mem.rs' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master b7550a1b7f045] Merge branch 'for-linux-next' of https://gitlab.freedesktop.org/drm/rust/kernel.git +$ git diff -M --stat --summary HEAD^.. + Documentation/gpu/nova/core/fsp.rst | 2 + + Documentation/gpu/nova/core/interrupts.rst | 603 ++++++++++++ + Documentation/gpu/nova/core/pramin.rst | 128 +++ + Documentation/gpu/nova/core/todo.rst | 2 +- + Documentation/gpu/nova/index.rst | 2 + + MAINTAINERS | 7 + + drivers/gpu/drm/nova/driver.rs | 19 +- + drivers/gpu/drm/nova/file.rs | 119 ++- + drivers/gpu/drm/tyr/driver.rs | 1 + + drivers/gpu/drm/tyr/fw.rs | 7 +- + drivers/gpu/drm/tyr/gpu.rs | 6 +- + drivers/gpu/drm/tyr/regs.rs | 57 +- + drivers/gpu/nova-core/Kconfig | 15 + + drivers/gpu/nova-core/api.rs | 68 ++ + drivers/gpu/nova-core/driver.rs | 128 ++- + drivers/gpu/nova-core/falcon.rs | 164 ++-- + drivers/gpu/nova-core/falcon/fsp.rs | 63 +- + drivers/gpu/nova-core/falcon/gsp.rs | 98 +- + drivers/gpu/nova-core/falcon/hal.rs | 22 +- + drivers/gpu/nova-core/falcon/hal/ga102.rs | 76 +- + drivers/gpu/nova-core/falcon/hal/tu102.rs | 55 +- + drivers/gpu/nova-core/falcon/sec2.rs | 37 +- + drivers/gpu/nova-core/fb.rs | 22 +- + drivers/gpu/nova-core/fb/hal/gb100.rs | 59 +- + drivers/gpu/nova-core/fb/regs.rs | 34 +- + drivers/gpu/nova-core/firmware.rs | 3 +- + drivers/gpu/nova-core/firmware/booter.rs | 2 +- + drivers/gpu/nova-core/firmware/fwsec/bootloader.rs | 30 +- + drivers/gpu/nova-core/firmware/gsp.rs | 119 +-- + .../gpu/nova-core/firmware/{fsp.rs => gsp_fmc.rs} | 59 +- + drivers/gpu/nova-core/firmware/radix3.rs | 144 +++ + drivers/gpu/nova-core/firmware/riscv.rs | 8 +- + drivers/gpu/nova-core/fsp.rs | 45 +- + drivers/gpu/nova-core/gpu.rs | 354 +++++-- + drivers/gpu/nova-core/gpu/regs.rs | 86 ++ + drivers/gpu/nova-core/gsp.rs | 39 +- + drivers/gpu/nova-core/gsp/boot.rs | 26 +- + drivers/gpu/nova-core/gsp/cmdq.rs | 264 +++-- + drivers/gpu/nova-core/gsp/commands.rs | 48 +- + drivers/gpu/nova-core/gsp/fw.rs | 18 +- + drivers/gpu/nova-core/gsp/fw/commands.rs | 28 + + drivers/gpu/nova-core/gsp/hal.rs | 16 +- + drivers/gpu/nova-core/gsp/hal/gh100.rs | 16 +- + drivers/gpu/nova-core/gsp/hal/tu102.rs | 41 +- + drivers/gpu/nova-core/gsp/regs.rs | 9 +- + drivers/gpu/nova-core/gsp/sequencer.rs | 14 +- + drivers/gpu/nova-core/irq.rs | 29 + + drivers/gpu/nova-core/irq/doorbell_test.rs | 251 +++++ + drivers/gpu/nova-core/irq/gsp.rs | 195 ++++ + drivers/gpu/nova-core/irq/hal.rs | 67 ++ + drivers/gpu/nova-core/irq/hal/gh100.rs | 40 + + drivers/gpu/nova-core/irq/hal/tu102.rs | 43 + + drivers/gpu/nova-core/irq/interrupt_tree.rs | 682 +++++++++++++ + drivers/gpu/nova-core/irq/regs.rs | 88 ++ + drivers/gpu/nova-core/mctp.rs | 16 +- + drivers/gpu/nova-core/mm.rs | 335 +++++++ + drivers/gpu/nova-core/mm/bar_user.rs | 420 ++++++++ + drivers/gpu/nova-core/mm/hal.rs | 56 ++ + drivers/gpu/nova-core/mm/hal/gb100.rs | 35 + + drivers/gpu/nova-core/mm/hal/gh100.rs | 35 + + drivers/gpu/nova-core/mm/hal/tu102.rs | 37 + + drivers/gpu/nova-core/mm/pagetable.rs | 424 ++++++++ + drivers/gpu/nova-core/mm/pagetable/map.rs | 345 +++++++ + drivers/gpu/nova-core/mm/pagetable/ver2.rs | 275 ++++++ + drivers/gpu/nova-core/mm/pagetable/ver3.rs | 421 ++++++++ + drivers/gpu/nova-core/mm/pagetable/walk.rs | 244 +++++ + drivers/gpu/nova-core/mm/pramin.rs | 312 ++++++ + drivers/gpu/nova-core/mm/regs.rs | 70 ++ + drivers/gpu/nova-core/mm/tlb.rs | 120 +++ + drivers/gpu/nova-core/mm/vmm.rs | 346 +++++++ + drivers/gpu/nova-core/nova_core.rs | 5 + + drivers/gpu/nova-core/num.rs | 2 +- + drivers/gpu/nova-core/regs.rs | 326 ++++--- + drivers/gpu/nova-core/selftest.rs | 64 ++ + drivers/gpu/nova-core/vbios.rs | 11 +- + include/uapi/drm/nova_drm.h | 128 +++ + rust/bindings/bindings_helper.h | 1 + + rust/helpers/dma_fence.c | 49 + + rust/helpers/helpers.c | 1 + + rust/helpers/pci.c | 6 + + rust/kernel/auxiliary.rs | 17 +- + rust/kernel/bitfield.rs | 9 + + rust/kernel/debugfs.rs | 26 +- + rust/kernel/debugfs/entry.rs | 4 +- + rust/kernel/debugfs/file_ops.rs | 29 +- + rust/kernel/device_id.rs | 3 +- + rust/kernel/dma.rs | 141 ++- + rust/kernel/dma_buf/dma_fence.rs | 1022 ++++++++++++++++++++ + rust/kernel/dma_buf/mod.rs | 14 + + rust/kernel/io.rs | 220 +++-- + rust/kernel/io/register.rs | 803 ++++----------- + rust/kernel/io/resource.rs | 8 + + rust/kernel/lib.rs | 2 + + rust/kernel/maple_tree.rs | 30 +- + rust/kernel/mem.rs | 230 +++++ + rust/kernel/pci.rs | 14 + + rust/kernel/pci/irq.rs | 68 +- + rust/kernel/sync/atomic.rs | 4 +- + rust/kernel/uaccess.rs | 18 +- + rust/macros/io/mod.rs | 3 + + rust/macros/io/register.rs | 296 ++++++ + rust/macros/lib.rs | 9 + + samples/rust/rust_dma.rs | 30 +- + samples/rust/rust_driver_pci.rs | 4 + + 104 files changed, 9939 insertions(+), 1707 deletions(-) + create mode 100644 Documentation/gpu/nova/core/interrupts.rst + create mode 100644 Documentation/gpu/nova/core/pramin.rst + create mode 100644 drivers/gpu/nova-core/api.rs + rename drivers/gpu/nova-core/firmware/{fsp.rs => gsp_fmc.rs} (67%) + create mode 100644 drivers/gpu/nova-core/firmware/radix3.rs + create mode 100644 drivers/gpu/nova-core/gpu/regs.rs + create mode 100644 drivers/gpu/nova-core/irq.rs + create mode 100644 drivers/gpu/nova-core/irq/doorbell_test.rs + create mode 100644 drivers/gpu/nova-core/irq/gsp.rs + create mode 100644 drivers/gpu/nova-core/irq/hal.rs + create mode 100644 drivers/gpu/nova-core/irq/hal/gh100.rs + create mode 100644 drivers/gpu/nova-core/irq/hal/tu102.rs + create mode 100644 drivers/gpu/nova-core/irq/interrupt_tree.rs + create mode 100644 drivers/gpu/nova-core/irq/regs.rs + create mode 100644 drivers/gpu/nova-core/mm.rs + create mode 100644 drivers/gpu/nova-core/mm/bar_user.rs + create mode 100644 drivers/gpu/nova-core/mm/hal.rs + create mode 100644 drivers/gpu/nova-core/mm/hal/gb100.rs + create mode 100644 drivers/gpu/nova-core/mm/hal/gh100.rs + create mode 100644 drivers/gpu/nova-core/mm/hal/tu102.rs + create mode 100644 drivers/gpu/nova-core/mm/pagetable.rs + create mode 100644 drivers/gpu/nova-core/mm/pagetable/map.rs + create mode 100644 drivers/gpu/nova-core/mm/pagetable/ver2.rs + create mode 100644 drivers/gpu/nova-core/mm/pagetable/ver3.rs + create mode 100644 drivers/gpu/nova-core/mm/pagetable/walk.rs + create mode 100644 drivers/gpu/nova-core/mm/pramin.rs + create mode 100644 drivers/gpu/nova-core/mm/regs.rs + create mode 100644 drivers/gpu/nova-core/mm/tlb.rs + create mode 100644 drivers/gpu/nova-core/mm/vmm.rs + create mode 100644 drivers/gpu/nova-core/selftest.rs + create mode 100644 rust/helpers/dma_fence.c + create mode 100644 rust/kernel/dma_buf/dma_fence.rs + create mode 100644 rust/kernel/dma_buf/mod.rs + create mode 100644 rust/macros/io/mod.rs + create mode 100644 rust/macros/io/register.rs +Merging drm-nova/nova-next (93296e9d9528f gpu: nova-core: vbios: store reference to Device where relevant) +$ git merge -m Merge branch 'nova-next' of https://gitlab.freedesktop.org/drm/nova.git drm-nova/nova-next +Already up to date. +Merging etnaviv/etnaviv/next (6bde14ba5f7ef drm/etnaviv: add optional reset support) +$ git merge -m Merge branch 'etnaviv/next' of https://git.pengutronix.de/git/lst/linux etnaviv/etnaviv/next +Already up to date. +Merging fbdev/for-next (9f4c6043f33c9 fbdev: atafb: avoid cast when assigning buffer address pointer) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/deller/linux-fbdev.git fbdev/for-next +Auto-merging drivers/video/fbdev/efifb.c +Merge made by the 'ort' strategy. + drivers/video/fbdev/acornfb.c | 57 ++++++------------------- + drivers/video/fbdev/acornfb.h | 9 +--- + drivers/video/fbdev/atafb.c | 28 +----------- + drivers/video/fbdev/aty/aty128fb.c | 2 +- + drivers/video/fbdev/aty/atyfb_base.c | 4 +- + drivers/video/fbdev/aty/mach64_cursor.c | 2 +- + drivers/video/fbdev/aty/radeon_base.c | 4 +- + drivers/video/fbdev/aty/radeon_monitor.c | 4 +- + drivers/video/fbdev/aty/radeonfb.h | 2 +- + drivers/video/fbdev/au1100fb.c | 2 +- + drivers/video/fbdev/core/Kconfig | 6 --- + drivers/video/fbdev/efifb.c | 4 +- + drivers/video/fbdev/grvga.c | 2 +- + drivers/video/fbdev/gxt4500.c | 1 + + drivers/video/fbdev/kyro/STG4000OverlayDevice.c | 2 +- + drivers/video/fbdev/kyro/fbdev.c | 4 +- + drivers/video/fbdev/maxinefb.c | 2 +- + drivers/video/fbdev/mmp/core.c | 2 +- + drivers/video/fbdev/mmp/hw/mmp_ctrl.c | 4 +- + drivers/video/fbdev/mmp/hw/mmp_ctrl.h | 4 +- + drivers/video/fbdev/mmp/hw/mmp_spi.c | 2 +- + drivers/video/fbdev/nvidia/nvidia.c | 1 + + drivers/video/fbdev/ocfb.c | 2 +- + drivers/video/fbdev/omap2/omapfb/dss/dispc.c | 4 +- + drivers/video/fbdev/pm2fb.c | 2 +- + drivers/video/fbdev/s1d13xxxfb.c | 6 +-- + drivers/video/fbdev/sh_mobile_lcdcfb.c | 37 ++++++++++------ + drivers/video/fbdev/skeletonfb.c | 6 +-- + drivers/video/fbdev/sstfb.c | 5 ++- + drivers/video/fbdev/tgafb.c | 6 +-- + drivers/video/fbdev/tridentfb.c | 2 +- + drivers/video/fbdev/udlfb.c | 5 +++ + drivers/video/fbdev/via/dvi.c | 6 +-- + drivers/video/fbdev/via/hw.c | 2 +- + drivers/video/fbdev/via/viafbdev.c | 4 +- + drivers/video/sticore.c | 2 +- + include/video/mmp_disp.h | 6 +-- + 37 files changed, 100 insertions(+), 143 deletions(-) +$ git am -3 ../patches/0001-fix-up-for-drm-hyperv-Remove-reference-to-hyperv_fb-.patch +Applying: fix up for "drm/hyperv: Remove reference to hyperv_fb driver" +$ git reset HEAD^ +Unstaged changes after reset: +M drivers/gpu/drm/hyperv/Kconfig +$ git add -A . +$ git commit -v -a --amend +warning: notes ref refs/notes/commits is invalid +[master a879361275b0d] Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/deller/linux-fbdev.git + Date: Sat Oct 3 00:43:39 2026 +0200 +Merging regmap/for-next (117e6a5fd98fb Merge regmap/for-7.4 into regmap-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/broonie/regmap.git regmap/for-next +Merge made by the 'ort' strategy. + drivers/base/regmap/internal.h | 12 +++ + drivers/base/regmap/regcache-rbtree.c | 4 +- + drivers/base/regmap/regcache.c | 28 ++---- + drivers/base/regmap/regmap-debugfs.c | 24 ++--- + drivers/base/regmap/regmap-kunit.c | 98 +++++++++++++++++- + drivers/base/regmap/regmap-ram.c | 26 +++-- + drivers/base/regmap/regmap-raw-ram.c | 21 ++-- + drivers/base/regmap/regmap.c | 184 +++++++++++----------------------- + 8 files changed, 219 insertions(+), 178 deletions(-) +Merging sound/for-next (cb8e306fd99c7 ALSA: Drop unused snd_card_free_on_error()) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/tiwai/sound.git sound/for-next +Auto-merging MAINTAINERS +Auto-merging drivers/acpi/scan.c +Auto-merging sound/core/pcm_native.c +Merge made by the 'ort' strategy. + Documentation/sound/alsa-configuration.rst | 6 + + Documentation/sound/cards/hdspm.rst | 109 ++--- + MAINTAINERS | 7 + + drivers/acpi/scan.c | 1 + + drivers/platform/x86/serial-multi-instantiate.c | 1 + + include/sound/control.h | 2 +- + include/sound/core.h | 1 - + include/sound/emu10k1.h | 4 +- + include/sound/emux_legacy.h | 2 +- + include/sound/rawmidi.h | 1 + + include/uapi/sound/asound.h | 3 +- + sound/core/control.c | 2 + + sound/core/init.c | 48 +- + sound/core/pcm_native.c | 6 +- + sound/core/rawmidi.c | 83 +++- + sound/core/seq/Kconfig | 7 +- + sound/core/seq/oss/seq_oss.c | 3 + + sound/core/seq/oss/seq_oss_init.c | 2 +- + sound/core/timer.c | 1 + + sound/core/timer_compat.c | 1 + + sound/core/ump.c | 2 +- + sound/drivers/dummy.c | 10 +- + sound/drivers/mpu401/mpu401_uart.c | 8 +- + sound/drivers/serial-generic.c | 10 +- + sound/firewire/bebob/bebob_hwdep.c | 2 +- + sound/hda/codecs/conexant.c | 67 +++ + sound/hda/codecs/hdmi/hdmi.c | 2 +- + sound/hda/codecs/realtek/alc269.c | 147 +++++- + sound/hda/codecs/realtek/alc882.c | 2 +- + sound/hda/codecs/side-codecs/cs35l41_hda_i2c.c | 3 + + .../hda/codecs/side-codecs/cs35l41_hda_property.c | 28 ++ + sound/hda/codecs/sigmatel.c | 6 +- + sound/hda/common/codec.c | 2 +- + sound/hda/controllers/acpi.c | 41 ++ + sound/hda/core/bus.c | 2 +- + sound/hda/core/component.c | 5 +- + sound/hda/core/controller.c | 5 +- + sound/i2c/other/ak4113.c | 2 +- + sound/i2c/other/ak4114.c | 2 +- + sound/i2c/other/ak4xxx-adda.c | 2 +- + sound/isa/galaxy/galaxy.c | 10 +- + sound/isa/opti9xx/opti92x-ad1848.c | 2 +- + sound/isa/sc6000.c | 10 +- + sound/mips/hal2.c | 4 +- + sound/mips/sgio2audio.c | 13 +- + sound/oss/dmasound/dmasound_q40.c | 65 ++- + sound/pci/ac97/ac97_codec.c | 2 +- + sound/pci/ac97/ac97_patch.c | 2 +- + sound/pci/ad1889.c | 13 +- + sound/pci/ali5451/ali5451.c | 13 +- + sound/pci/als4000.c | 13 +- + sound/pci/atiixp.c | 13 +- + sound/pci/atiixp_modem.c | 13 +- + sound/pci/au88x0/au88x0.c | 11 +- + sound/pci/au88x0/au88x0.h | 2 +- + sound/pci/au88x0/au88x0_mpu401.c | 2 +- + sound/pci/aw2/aw2-saa7146.c | 4 +- + sound/pci/azt3328.c | 11 +- + sound/pci/bt87x.c | 13 +- + sound/pci/ca0106/ca0106_main.c | 11 +- + sound/pci/cs4281.c | 15 +- + sound/pci/cs46xx/cs46xx_lib.c | 6 +- + sound/pci/cs46xx/dsp_spos_scb_lib.c | 8 +- + sound/pci/cs5535audio/cs5535audio.c | 13 +- + sound/pci/echoaudio/echoaudio.c | 18 +- + sound/pci/echoaudio/echoaudio.h | 4 +- + sound/pci/emu10k1/emu10k1x.c | 13 +- + sound/pci/ens1370.c | 13 +- + sound/pci/es1938.c | 13 +- + sound/pci/es1968.c | 13 +- + sound/pci/fm801.c | 13 +- + sound/pci/ice1712/ice1724.c | 13 +- + sound/pci/intel8x0.c | 14 +- + sound/pci/intel8x0m.c | 13 +- + sound/pci/lola/lola.c | 13 +- + sound/pci/lx6464es/lx_core.c | 2 +- + sound/pci/lx6464es/lx_defs.h | 2 +- + sound/pci/maestro3.c | 11 +- + sound/pci/oxygen/oxygen_lib.c | 15 +- + sound/pci/oxygen/oxygen_pcm.c | 37 +- + sound/pci/pcxhr/pcxhr_core.c | 4 +- + sound/pci/riptide/riptide.c | 15 +- + sound/pci/rme32.c | 13 +- + sound/pci/rme96.c | 13 +- + sound/pci/rme9652/hdsp.c | 2 +- + sound/pci/sis7019.c | 13 +- + sound/pci/sonicvibes.c | 13 +- + sound/pci/trident/trident_memory.c | 2 +- + sound/pci/via82xx.c | 24 +- + sound/pci/via82xx_modem.c | 13 +- + sound/pci/vx222/vx222_ops.c | 2 +- + sound/pci/ymfpci/ymfpci.c | 13 +- + sound/pcmcia/vx/vxp_ops.c | 2 +- + sound/ppc/tumbler.c | 2 +- + sound/soc/codecs/tas675x.c | 10 +- + sound/usb/Makefile | 1 + + sound/usb/caiaq/Makefile | 2 +- + sound/usb/caiaq/device.c | 12 + + sound/usb/caiaq/device.h | 13 + + sound/usb/caiaq/input.c | 29 +- + sound/usb/caiaq/lcd.c | 258 +++++++++++ + sound/usb/caiaq/lcd.h | 7 + + sound/usb/clock.c | 27 ++ + sound/usb/midi.c | 69 +-- + sound/usb/mixer.c | 253 +++++++---- + sound/usb/mixer.h | 2 + + sound/usb/mixer_evo.c | 505 +++++++++++++++++++++ + sound/usb/mixer_evo.h | 12 + + sound/usb/mixer_maps.c | 13 + + sound/usb/mixer_quirks.c | 5 + + sound/usb/mixer_us16x08.h | 2 +- + sound/usb/quirks-table.h | 253 ++++++++++- + sound/usb/quirks.c | 235 +++++++++- + sound/usb/usbaudio.h | 6 + + sound/x86/intel_hdmi_audio.c | 10 +- + tools/testing/selftests/alsa/mixer-test.c | 255 +++++++++++ + tools/testing/selftests/alsa/utimer-test.c | 6 + + 117 files changed, 2509 insertions(+), 721 deletions(-) + create mode 100644 sound/usb/caiaq/lcd.c + create mode 100644 sound/usb/caiaq/lcd.h + create mode 100644 sound/usb/mixer_evo.c + create mode 100644 sound/usb/mixer_evo.h +Merging ieee1394/for-next (a6e7c3836b812 firewire: cdev: use kzalloc_flex() to allocate structure with byte array) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/ieee1394/linux1394.git ieee1394/for-next +Merge made by the 'ort' strategy. + drivers/firewire/.kunitconfig | 1 + + drivers/firewire/Kconfig | 16 ++ + drivers/firewire/config-rom-generator-test.c | 407 +++++++++++++++++++++++++++ + drivers/firewire/config-rom-parser-test.c | 355 +++++++++++++++++++++++ + drivers/firewire/core-card.c | 11 +- + drivers/firewire/core-cdev.c | 299 ++++++++++---------- + drivers/firewire/core-device.c | 4 + + drivers/firewire/core-transaction.c | 146 +++++----- + drivers/firewire/device-attribute-test.c | 4 +- + drivers/firewire/ohci-serdes-test.c | 4 +- + drivers/firewire/ohci.c | 267 ++++++++++++------ + drivers/firewire/ohci.h | 2 +- + drivers/firewire/uapi-test.c | 20 ++ + include/linux/firewire.h | 50 ++-- + include/uapi/linux/firewire-cdev.h | 4 +- + tools/firewire/nosy-dump.c | 10 +- + 16 files changed, 1252 insertions(+), 348 deletions(-) + create mode 100644 drivers/firewire/config-rom-generator-test.c + create mode 100644 drivers/firewire/config-rom-parser-test.c +Merging sound-asoc/for-next (0d6ef8b530db8 Merge asoc/for-7.4 into asoc-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/broonie/sound.git sound-asoc/for-next +Auto-merging Documentation/devicetree/bindings/vendor-prefixes.yaml +Auto-merging MAINTAINERS +Auto-merging sound/soc/amd/acp-es8336.c +Auto-merging sound/soc/amd/acp/acp3x-es83xx/acp3x-es83xx.c +Auto-merging sound/soc/intel/boards/sof_cirrus_common.c +Merge made by the 'ort' strategy. + .../devicetree/bindings/sound/adi,adau1372.yaml | 10 +- + .../devicetree/bindings/sound/adi,adau7118.yaml | 22 +- + .../bindings/sound/airoha,an7581-afe.yaml | 41 + + .../bindings/sound/airoha,an7581-wm8960.yaml | 71 + + .../bindings/sound/amlogic,gx-sound-card.yaml | 20 +- + .../bindings/sound/asahi-kasei,ak4619.yaml | 16 +- + .../bindings/sound/audio-graph-port.yaml | 10 +- + .../devicetree/bindings/sound/cirrus,cs35l45.yaml | 60 +- + .../devicetree/bindings/sound/cirrus,cs4270.yaml | 12 +- + .../devicetree/bindings/sound/cirrus,cs42l42.yaml | 42 +- + .../devicetree/bindings/sound/cirrus,cs42l84.yaml | 18 +- + .../devicetree/bindings/sound/cirrus,cs42xx8.yaml | 56 +- + .../bindings/sound/davinci-evm-audio.txt | 49 - + .../devicetree/bindings/sound/dialog,da7219.yaml | 62 +- + .../bindings/sound/esstech,es9039q2m.yaml | 94 + + .../devicetree/bindings/sound/everest,es8316.yaml | 5 + + .../bindings/sound/foursemi,fs2105s.yaml | 2 +- + .../devicetree/bindings/sound/fsl,easrc.yaml | 45 + + .../devicetree/bindings/sound/fsl,imx-asrc.yaml | 45 +- + .../devicetree/bindings/sound/fsl,sai.yaml | 14 +- + .../bindings/sound/hisilicon,hi6210-i2s.yaml | 4 + + .../devicetree/bindings/sound/imx-audio-card.yaml | 4 +- + .../devicetree/bindings/sound/imx-audmux.yaml | 8 +- + .../bindings/sound/invensense,ics43432.yaml | 8 +- + .../bindings/sound/loongson,ls-audio-card.yaml | 4 +- + .../bindings/sound/mediatek,mt2701-audio.yaml | 10 +- + .../bindings/sound/mediatek,mt8173-afe-pcm.yaml | 22 +- + .../bindings/sound/mediatek,mt8183-audio.yaml | 82 +- + .../sound/mt8186-mt6366-da7219-max98357.yaml | 36 +- + .../sound/mt8186-mt6366-rt1019-rt5682s.yaml | 36 +- + .../sound/mt8192-mt6359-rt1015-rt5682.yaml | 46 +- + .../devicetree/bindings/sound/mt8195-mt6359.yaml | 50 +- + .../devicetree/bindings/sound/nuvoton,nau8360.yaml | 115 + + .../bindings/sound/qcom,q6apm-lpass-dais.yaml | 12 +- + .../bindings/sound/qcom,q6dsp-lpass-ports.yaml | 16 +- + .../devicetree/bindings/sound/qcom,sm8250.yaml | 1 + + .../bindings/sound/qcom,wcd9378-sdw.yaml | 105 + + .../devicetree/bindings/sound/realtek,rt5677.yaml | 10 + + .../devicetree/bindings/sound/renesas,rsnd.yaml | 16 +- + .../devicetree/bindings/sound/renesas,rz-ssi.yaml | 32 +- + .../bindings/sound/stericsson,ux500-msp-i2s.yaml | 96 + + .../bindings/sound/ti,da830-evm-audio.yaml | 83 + + .../devicetree/bindings/sound/ti,tas2781.yaml | 4 +- + .../devicetree/bindings/sound/ti,tas5805m.yaml | 10 +- + .../bindings/sound/ti,tlv320dac3100.yaml | 2 +- + .../devicetree/bindings/sound/ti,tpa6130a2.yaml | 2 +- + .../devicetree/bindings/sound/ux500-mop500.txt | 39 - + .../devicetree/bindings/sound/ux500-msp.txt | 42 - + .../devicetree/bindings/vendor-prefixes.yaml | 2 + + MAINTAINERS | 18 + + drivers/firmware/cirrus/cs_dsp.c | 2 +- + drivers/gpib/fmh_gpib/fmh_gpib.c | 10 +- + include/dt-bindings/sound/qcom,q6dsp-lpass-ports.h | 52 + + include/linux/dmaengine.h | 9 - + include/linux/firmware/cirrus/cs_dsp.h | 2 +- + include/sound/sdca.h | 23 +- + include/sound/sdca_class.h | 70 + + include/sound/simple_card_utils.h | 2 + + include/sound/soc-component.h | 14 +- + include/sound/soc-dai.h | 179 +- + include/sound/soc.h | 16 +- + include/sound/soc_sdw_utils.h | 20 + + include/sound/tas2781-dsp.h | 13 + + include/sound/tas2781.h | 31 + + sound/hda/core/intel-nhlt.c | 4 +- + sound/soc/amd/acp-es8336.c | 3 +- + sound/soc/amd/acp/Kconfig | 4 +- + sound/soc/amd/acp/acp3x-es83xx/acp3x-es83xx.c | 12 + + sound/soc/amd/acp/soc_amd_sdw_common.h | 13 + + sound/soc/atmel/atmel-pcm-dma.c | 2 +- + sound/soc/bcm/cygnus-ssp.h | 2 +- + sound/soc/codecs/88pm860x-codec.c | 6 + + sound/soc/codecs/Kconfig | 49 + + sound/soc/codecs/Makefile | 8 + + sound/soc/codecs/ab8500-codec.c | 157 +- + sound/soc/codecs/adau1372.c | 12 + + sound/soc/codecs/adau1373.c | 12 + + sound/soc/codecs/adau1701.c | 15 +- + sound/soc/codecs/adau17x1.c | 17 +- + sound/soc/codecs/adau1977.c | 13 + + sound/soc/codecs/adau7118.c | 12 + + sound/soc/codecs/ak4104.c | 7 + + sound/soc/codecs/ak4118.c | 7 + + sound/soc/codecs/ak4458.c | 9 + + sound/soc/codecs/ak4535.c | 6 + + sound/soc/codecs/ak4619.c | 32 + + sound/soc/codecs/ak4642.c | 14 +- + sound/soc/codecs/ak4671.c | 7 + + sound/soc/codecs/ak5386.c | 6 + + sound/soc/codecs/ak5558.c | 7 + + sound/soc/codecs/alc5623.c | 11 + + sound/soc/codecs/alc5632.c | 13 + + sound/soc/codecs/arizona-jack.c | 1 + + sound/soc/codecs/arizona.c | 23 + + sound/soc/codecs/cpcap.c | 87 +- + sound/soc/codecs/cs35l33.c | 8 +- + sound/soc/codecs/cs35l34.c | 2 +- + sound/soc/codecs/cs35l35.c | 7 + + sound/soc/codecs/cs35l36.c | 56 + + sound/soc/codecs/cs35l41.c | 10 + + sound/soc/codecs/cs35l45-tables.c | 4 + + sound/soc/codecs/cs35l45.c | 97 +- + sound/soc/codecs/cs35l45.h | 23 + + sound/soc/codecs/cs35l56-shared-test.c | 56 + + sound/soc/codecs/cs35l56.c | 10 + + sound/soc/codecs/cs40l50-codec.c | 9 + + sound/soc/codecs/cs4234.c | 9 + + sound/soc/codecs/cs4265.c | 12 + + sound/soc/codecs/cs4270.c | 6 + + sound/soc/codecs/cs4271.c | 6 + + sound/soc/codecs/cs42l42.c | 11 +- + sound/soc/codecs/cs42l43.c | 12 + + sound/soc/codecs/cs42l51.c | 7 + + sound/soc/codecs/cs42l52.c | 22 +- + sound/soc/codecs/cs42l56.c | 17 +- + sound/soc/codecs/cs42l73.c | 11 +- + sound/soc/codecs/cs42l84.c | 6 + + sound/soc/codecs/cs42xx8.c | 8 + + sound/soc/codecs/cs43130.c | 14 + + sound/soc/codecs/cs4341.c | 8 + + sound/soc/codecs/cs4349.c | 12 + + sound/soc/codecs/cs48l32.c | 23 + + sound/soc/codecs/cs530x.c | 9 + + sound/soc/codecs/cs53l30.c | 8 + + sound/soc/codecs/cx20442.c | 2 +- + sound/soc/codecs/cx2072x.c | 12 + + sound/soc/codecs/da7210.c | 9 +- + sound/soc/codecs/da7218.c | 141 +- + sound/soc/codecs/da7219.c | 12 + + sound/soc/codecs/da732x.c | 17 + + sound/soc/codecs/da9055.c | 8 + + sound/soc/codecs/es7134.c | 6 + + sound/soc/codecs/es7241.c | 7 + + sound/soc/codecs/es8311.c | 27 +- + sound/soc/codecs/es8316.c | 27 +- + sound/soc/codecs/es8323.c | 32 +- + sound/soc/codecs/es8326.c | 181 +- + sound/soc/codecs/es8326.h | 4 + + sound/soc/codecs/es8328.c | 20 +- + sound/soc/codecs/es8375.c | 30 +- + sound/soc/codecs/es8389.c | 65 +- + sound/soc/codecs/es9039q2m.c | 2008 +++++++++++++++++ + sound/soc/codecs/fs210x.c | 5 +- + sound/soc/codecs/hda.c | 7 +- + sound/soc/codecs/hdac_hda.c | 2 +- + sound/soc/codecs/hdac_hdmi.c | 2 +- + sound/soc/codecs/inno_rk3036.c | 12 + + sound/soc/codecs/isabelle.c | 13 + + sound/soc/codecs/lm49453.c | 15 + + sound/soc/codecs/lpass-rx-macro.c | 6 +- + sound/soc/codecs/lpass-tx-macro.c | 4 +- + sound/soc/codecs/lpass-va-macro.c | 4 +- + sound/soc/codecs/lpass-wsa-macro.c | 48 +- + sound/soc/codecs/max98088.c | 12 + + sound/soc/codecs/max98090.c | 35 +- + sound/soc/codecs/max98095.c | 24 +- + sound/soc/codecs/max98371.c | 7 + + sound/soc/codecs/max98373-i2c.c | 10 + + sound/soc/codecs/max98388.c | 10 + + sound/soc/codecs/max98390.c | 10 + + sound/soc/codecs/max98396.c | 12 + + sound/soc/codecs/max9850.c | 11 + + sound/soc/codecs/max98520.c | 10 + + sound/soc/codecs/max9860.c | 23 + + sound/soc/codecs/max9867.c | 10 + + sound/soc/codecs/max98925.c | 8 + + sound/soc/codecs/max98926.c | 8 + + sound/soc/codecs/max98927.c | 14 +- + sound/soc/codecs/mc13783.c | 12 + + sound/soc/codecs/ml26124.c | 6 + + sound/soc/codecs/mt6359.c | 8 +- + sound/soc/codecs/nau8325.c | 11 + + sound/soc/codecs/nau8360-dsp.c | 634 ++++++ + sound/soc/codecs/nau8360-dsp.h | 122 ++ + sound/soc/codecs/nau8360.c | 2309 ++++++++++++++++++++ + sound/soc/codecs/nau8360.h | 917 ++++++++ + sound/soc/codecs/nau8540.c | 15 +- + sound/soc/codecs/nau8810.c | 12 + + sound/soc/codecs/nau8821.c | 11 + + sound/soc/codecs/nau8822.c | 27 +- + sound/soc/codecs/nau8824.c | 11 + + sound/soc/codecs/nau8825.c | 29 +- + sound/soc/codecs/ntp8835.c | 7 + + sound/soc/codecs/ntp8918.c | 7 + + sound/soc/codecs/pcm5102a.c | 30 +- + sound/soc/codecs/pcm6240.c | 8 +- + sound/soc/codecs/rk3308_codec.c | 11 + + sound/soc/codecs/rk3328_codec.c | 9 + + sound/soc/codecs/rt1011.c | 44 +- + sound/soc/codecs/rt1015.c | 91 +- + sound/soc/codecs/rt1016.c | 14 +- + sound/soc/codecs/rt1019.c | 10 + + sound/soc/codecs/rt1305.c | 14 +- + sound/soc/codecs/rt1308.c | 14 +- + sound/soc/codecs/rt1318.c | 13 +- + sound/soc/codecs/rt1320-sdw.c | 12 +- + sound/soc/codecs/rt274.c | 8 + + sound/soc/codecs/rt286.c | 8 + + sound/soc/codecs/rt298.c | 8 + + sound/soc/codecs/rt5514-spi.c | 37 +- + sound/soc/codecs/rt5514.c | 39 +- + sound/soc/codecs/rt5616.c | 23 +- + sound/soc/codecs/rt5631.c | 10 + + sound/soc/codecs/rt5631.h | 8 +- + sound/soc/codecs/rt5640.c | 22 +- + sound/soc/codecs/rt5645.c | 23 +- + sound/soc/codecs/rt5651.c | 12 +- + sound/soc/codecs/rt5659.c | 14 +- + sound/soc/codecs/rt5660.c | 14 +- + sound/soc/codecs/rt5663.c | 10 + + sound/soc/codecs/rt5665.c | 14 +- + sound/soc/codecs/rt5668.c | 22 + + sound/soc/codecs/rt5670.c | 12 +- + sound/soc/codecs/rt5677-spi.c | 21 +- + sound/soc/codecs/rt5677.c | 10 + + sound/soc/codecs/rt5682.c | 22 + + sound/soc/codecs/rt5682s.c | 34 +- + sound/soc/codecs/rt712-sdca-dmic.c | 4 +- + sound/soc/codecs/rt712-sdca-sdw.c | 1 + + sound/soc/codecs/rt712-sdca-sdw.h | 3 + + sound/soc/codecs/rt712-sdca.c | 141 +- + sound/soc/codecs/rt712-sdca.h | 6 + + sound/soc/codecs/rt721-sdca-sdw.c | 15 + + sound/soc/codecs/rt721-sdca.c | 498 +++-- + sound/soc/codecs/rt721-sdca.h | 17 + + sound/soc/codecs/rt766-sdca.c | 2 +- + sound/soc/codecs/rt9120.c | 9 + + sound/soc/codecs/rt9123.c | 13 + + sound/soc/codecs/rtq9124.c | 9 + + sound/soc/codecs/rtq9128.c | 9 + + sound/soc/codecs/sgtl5000.c | 11 + + sound/soc/codecs/si476x.c | 18 + + sound/soc/codecs/sma1303.c | 13 + + sound/soc/codecs/sma1307.c | 27 +- + sound/soc/codecs/sn624x-sdca-sdw.c | 1888 ++++++++++++++++ + sound/soc/codecs/sn624x-sdca.h | 152 ++ + sound/soc/codecs/src4xxx.c | 8 + + sound/soc/codecs/ssm2518.c | 13 + + sound/soc/codecs/ssm2602.c | 13 + + sound/soc/codecs/ssm3515.c | 10 + + sound/soc/codecs/ssm4567.c | 13 + + sound/soc/codecs/sta32x.c | 116 +- + sound/soc/codecs/sta350.c | 15 +- + sound/soc/codecs/sta529.c | 7 + + sound/soc/codecs/tas2552.c | 14 + + sound/soc/codecs/tas2562.c | 10 + + sound/soc/codecs/tas2764.c | 14 +- + sound/soc/codecs/tas2770.c | 14 +- + sound/soc/codecs/tas2780.c | 10 + + sound/soc/codecs/tas2781-comlib-i2c.c | 4 +- + sound/soc/codecs/tas2781-comlib.c | 10 +- + sound/soc/codecs/tas2781-fmwlib.c | 40 +- + sound/soc/codecs/tas2781-i2c.c | 320 ++- + sound/soc/codecs/tas2783-sdw.c | 218 +- + sound/soc/codecs/tas2783.h | 4 +- + sound/soc/codecs/tas5086.c | 7 + + sound/soc/codecs/tas571x.c | 7 + + sound/soc/codecs/tas5720.c | 9 + + sound/soc/codecs/tas6424.c | 9 + + sound/soc/codecs/tfa9879.c | 9 + + sound/soc/codecs/tlv320adc3xxx.c | 14 + + sound/soc/codecs/tlv320adcx140.c | 12 + + sound/soc/codecs/tlv320aic23.c | 9 + + sound/soc/codecs/tlv320aic26.c | 8 + + sound/soc/codecs/tlv320aic31xx.c | 41 +- + sound/soc/codecs/tlv320aic32x4-clk.c | 2 +- + sound/soc/codecs/tlv320aic32x4.c | 102 +- + sound/soc/codecs/tlv320aic3x.c | 42 +- + sound/soc/codecs/tlv320dac33.c | 10 +- + sound/soc/codecs/tscs454.c | 15 + + sound/soc/codecs/twl4030.c | 80 +- + sound/soc/codecs/uda1334.c | 6 + + sound/soc/codecs/uda1342.c | 7 + + sound/soc/codecs/uda1380.c | 11 + + sound/soc/codecs/wcd-mbhc-v2.c | 8 +- + sound/soc/codecs/wcd9335.c | 25 +- + sound/soc/codecs/wcd934x.c | 2 +- + sound/soc/codecs/wcd9378-sdca.c | 1080 +++++++++ + sound/soc/codecs/wcd9378-sdca.h | 20 + + sound/soc/codecs/wcd9378-sdw.c | 51 + + sound/soc/codecs/wm2200.c | 10 + + sound/soc/codecs/wm5100.c | 10 + + sound/soc/codecs/wm8350.c | 24 + + sound/soc/codecs/wm8400.c | 9 + + sound/soc/codecs/wm8510.c | 12 + + sound/soc/codecs/wm8523.c | 13 + + sound/soc/codecs/wm8524.c | 6 + + sound/soc/codecs/wm8580.c | 20 + + sound/soc/codecs/wm8711.c | 13 + + sound/soc/codecs/wm8728.c | 9 + + sound/soc/codecs/wm8731.c | 13 + + sound/soc/codecs/wm8737.c | 11 + + sound/soc/codecs/wm8741.c | 13 + + sound/soc/codecs/wm8750.c | 13 + + sound/soc/codecs/wm8753.c | 20 + + sound/soc/codecs/wm8770.c | 11 + + sound/soc/codecs/wm8776.c | 13 + + sound/soc/codecs/wm8804.c | 15 +- + sound/soc/codecs/wm8900.c | 18 + + sound/soc/codecs/wm8903.c | 18 + + sound/soc/codecs/wm8904.c | 20 +- + sound/soc/codecs/wm8940.c | 13 + + sound/soc/codecs/wm8955.c | 42 +- + sound/soc/codecs/wm8958-dsp2.c | 8 +- + sound/soc/codecs/wm8960.c | 13 + + sound/soc/codecs/wm8961.c | 18 + + sound/soc/codecs/wm8962.c | 74 +- + sound/soc/codecs/wm8971.c | 13 + + sound/soc/codecs/wm8974.c | 17 + + sound/soc/codecs/wm8978.c | 12 + + sound/soc/codecs/wm8983.c | 12 + + sound/soc/codecs/wm8985.c | 53 +- + sound/soc/codecs/wm8988.c | 13 + + sound/soc/codecs/wm8990.c | 9 + + sound/soc/codecs/wm8991.c | 9 + + sound/soc/codecs/wm8993.c | 18 + + sound/soc/codecs/wm8994.c | 23 + + sound/soc/codecs/wm8995.c | 61 +- + sound/soc/codecs/wm8996.c | 14 +- + sound/soc/codecs/wm9081.c | 18 + + sound/soc/codecs/wm9713.c | 13 + + sound/soc/codecs/wm_hubs.c | 2 +- + sound/soc/codecs/zl38060.c | 6 + + sound/soc/dwc/dwc-i2s.c | 9 + + sound/soc/fsl/Kconfig | 2 +- + sound/soc/fsl/fsl_asrc.c | 110 +- + sound/soc/fsl/fsl_asrc_common.h | 7 +- + sound/soc/fsl/fsl_asrc_dma.c | 38 +- + sound/soc/fsl/fsl_asrc_m2m.c | 17 +- + sound/soc/fsl/fsl_easrc.c | 125 +- + sound/soc/fsl/fsl_mqs.c | 14 + + sound/soc/fsl/fsl_xcvr.c | 28 +- + sound/soc/generic/audio-graph-card.c | 1 + + sound/soc/generic/audio-graph-card2.c | 4 +- + sound/soc/generic/simple-card-utils.c | 23 +- + sound/soc/generic/simple-card.c | 13 +- + sound/soc/hisilicon/hi6210-i2s.c | 7 + + sound/soc/img/img-i2s-in.c | 12 +- + sound/soc/img/img-i2s-out.c | 13 +- + sound/soc/img/img-parallel-out.c | 8 +- + sound/soc/intel/atom/sst-mfld-dsp.h | 2 +- + sound/soc/intel/atom/sst/sst.h | 2 +- + sound/soc/intel/atom/sst/sst_acpi.c | 14 +- + sound/soc/intel/avs/boards/hdaudio.c | 8 +- + sound/soc/intel/avs/pcm.c | 11 +- + sound/soc/intel/avs/topology.c | 25 - + sound/soc/intel/avs/topology.h | 3 - + sound/soc/intel/boards/Kconfig | 1 + + sound/soc/intel/boards/sof_cirrus_common.c | 2 +- + sound/soc/intel/common/soc-acpi-intel-arl-match.c | 71 + + sound/soc/intel/common/soc-acpi-intel-lnl-match.c | 71 + + sound/soc/intel/common/soc-acpi-intel-mtl-match.c | 71 + + sound/soc/intel/common/soc-acpi-intel-ptl-match.c | 85 + + sound/soc/intel/common/sof-function-topology-lib.c | 289 ++- + sound/soc/intel/common/sof-function-topology-lib.h | 3 + + sound/soc/jz4740/jz4740-i2s.c | 7 + + sound/soc/kirkwood/kirkwood-i2s.c | 7 + + sound/soc/loongson/loongson_i2s.c | 6 + + sound/soc/mediatek/Kconfig | 27 +- + sound/soc/mediatek/Makefile | 1 + + sound/soc/mediatek/an7581/Makefile | 9 + + sound/soc/mediatek/an7581/an7581-afe-common.h | 49 + + sound/soc/mediatek/an7581/an7581-afe-pcm.c | 515 +++++ + sound/soc/mediatek/an7581/an7581-dai-etdm.c | 433 ++++ + sound/soc/mediatek/an7581/an7581-reg.h | 114 + + sound/soc/mediatek/an7581/an7581-wm8960.c | 161 ++ + sound/soc/mediatek/common/mtk-afe-fe-dai.c | 14 +- + sound/soc/mediatek/common/mtk-base-afe.h | 2 + + sound/soc/mediatek/mt2701/mt2701-afe-clock-ctrl.c | 61 +- + sound/soc/mediatek/mt2701/mt2701-afe-pcm.c | 4 +- + sound/soc/mediatek/mt2701/mt2701-cs42448.c | 4 +- + sound/soc/mediatek/mt2701/mt2701-wm8960.c | 4 +- + sound/soc/mediatek/mt6797/mt6797-afe-clk.c | 23 +- + sound/soc/mediatek/mt6797/mt6797-afe-pcm.c | 8 +- + sound/soc/mediatek/mt7986/mt7986-afe-pcm.c | 8 +- + sound/soc/mediatek/mt7986/mt7986-dai-etdm.c | 2 +- + sound/soc/mediatek/mt7986/mt7986-wm8960.c | 4 +- + sound/soc/mediatek/mt8173/mt8173-afe-pcm.c | 26 +- + sound/soc/mediatek/mt8183/mt8183-afe-clk.c | 31 +- + sound/soc/mediatek/mt8183/mt8183-afe-pcm.c | 21 +- + sound/soc/mediatek/mt8186/mt8186-afe-clk.c | 95 +- + sound/soc/mediatek/mt8186/mt8186-afe-gpio.c | 6 +- + sound/soc/mediatek/mt8186/mt8186-afe-pcm.c | 23 +- + sound/soc/mediatek/mt8188/mt8188-afe-clk.c | 115 +- + sound/soc/mediatek/mt8188/mt8188-afe-pcm.c | 36 +- + sound/soc/mediatek/mt8188/mt8188-audsys-clk.c | 12 +- + sound/soc/mediatek/mt8188/mt8188-dai-adda.c | 5 +- + sound/soc/mediatek/mt8188/mt8188-dai-dmic.c | 10 +- + sound/soc/mediatek/mt8189/mt8189-afe-clk.c | 134 +- + sound/soc/mediatek/mt8189/mt8189-afe-pcm.c | 61 +- + sound/soc/mediatek/mt8189/mt8189-dai-i2s.c | 18 +- + sound/soc/mediatek/mt8189/mt8189-dai-tdm.c | 19 +- + sound/soc/mediatek/mt8192/mt8192-afe-clk.c | 163 +- + sound/soc/mediatek/mt8192/mt8192-afe-gpio.c | 94 +- + sound/soc/mediatek/mt8192/mt8192-afe-pcm.c | 23 +- + sound/soc/mediatek/mt8192/mt8192-dai-adda.c | 44 +- + sound/soc/mediatek/mt8192/mt8192-dai-i2s.c | 20 +- + sound/soc/mediatek/mt8192/mt8192-dai-tdm.c | 18 +- + .../mediatek/mt8192/mt8192-mt6359-rt1015-rt5682.c | 37 +- + sound/soc/mediatek/mt8195/mt8195-afe-pcm.c | 2 +- + sound/soc/mediatek/mt8196/mt8196-afe-pcm.c | 14 +- + sound/soc/mxs/mxs-saif.c | 10 + + sound/soc/pxa/mmp-sspa.c | 6 + + sound/soc/pxa/pxa-ssp.c | 11 + + sound/soc/pxa/pxa2xx-i2s.c | 6 + + sound/soc/pxa/pxa2xx-pcm-lib.c | 9 +- + sound/soc/qcom/common.c | 61 +- + sound/soc/qcom/common.h | 5 +- + sound/soc/qcom/qdsp6/q6apm-lpass-dais.c | 2 + + sound/soc/qcom/qdsp6/q6asm-dai.c | 2 +- + sound/soc/qcom/qdsp6/q6dsp-lpass-ports.c | 54 + + sound/soc/qcom/sc8280xp.c | 132 +- + sound/soc/qcom/sdm845.c | 49 +- + sound/soc/renesas/rz-ssi.c | 9 + + sound/soc/renesas/siu_dai.c | 6 + + sound/soc/renesas/ssi.c | 13 + + sound/soc/rockchip/rk3399_gru_sound.c | 4 +- + sound/soc/rockchip/rockchip_i2s.c | 25 +- + sound/soc/rockchip/rockchip_i2s_tdm.c | 27 +- + sound/soc/rockchip/rockchip_pdm.c | 14 +- + sound/soc/rockchip/rockchip_sai.c | 31 +- + sound/soc/samsung/aries_wm8994.c | 14 +- + sound/soc/samsung/pcm.c | 19 +- + sound/soc/samsung/spdif.c | 8 +- + sound/soc/samsung/tm2_wm5110.c | 12 +- + sound/soc/sdca/Kconfig | 6 +- + sound/soc/sdca/sdca_asoc.c | 4 +- + sound/soc/sdca/sdca_class.c | 148 +- + sound/soc/sdca/sdca_class.h | 35 - + sound/soc/sdca/sdca_class_function.c | 50 +- + sound/soc/sdca/sdca_device.c | 4 + + sound/soc/sdca/sdca_fdl.c | 18 +- + sound/soc/sdca/sdca_functions.c | 15 +- + sound/soc/sdw_utils/Makefile | 4 +- + sound/soc/sdw_utils/soc_sdw_rt711.c | 1 + + sound/soc/sdw_utils/soc_sdw_rt_mf_sdca.c | 6 + + sound/soc/sdw_utils/soc_sdw_rt_sdca_jack_common.c | 8 + + sound/soc/sdw_utils/soc_sdw_senary_amp.c | 80 + + sound/soc/sdw_utils/soc_sdw_senary_dmic.c | 49 + + sound/soc/sdw_utils/soc_sdw_senary_sdca.c | 63 + + .../sdw_utils/soc_sdw_senary_sdca_jack_common.c | 197 ++ + sound/soc/sdw_utils/soc_sdw_utils.c | 384 ++++ + sound/soc/soc-component.c | 8 +- + sound/soc/soc-compress.c | 5 +- + sound/soc/soc-core.c | 175 +- + sound/soc/soc-dai.c | 418 +++- + sound/soc/soc-dapm.c | 9 +- + sound/soc/soc-generic-dmaengine-pcm.c | 10 +- + sound/soc/soc-internal.h | 39 + + sound/soc/soc-ops.c | 4 +- + sound/soc/soc-pcm.c | 144 +- + sound/soc/sof/amd/Kconfig | 1 + + sound/soc/sof/amd/acp-common.c | 17 + + sound/soc/sof/amd/acp-dsp-offset.h | 17 + + sound/soc/sof/amd/acp.c | 391 +++- + sound/soc/sof/amd/acp.h | 16 + + sound/soc/sof/amd/acp7x.h | 37 + + sound/soc/sof/amd/pci-acp7x.c | 2 + + sound/soc/sof/imx/imx-common.c | 1 + + sound/soc/sof/imx/imx8.c | 26 +- + sound/soc/sof/intel/apl.c | 1 + + sound/soc/sof/intel/cnl.c | 1 + + sound/soc/sof/intel/hda-dai.c | 26 +- + sound/soc/sof/intel/icl.c | 1 + + sound/soc/sof/intel/mtl.c | 1 + + sound/soc/sof/intel/skl.c | 1 + + sound/soc/sof/intel/tgl.c | 1 + + sound/soc/sof/ipc4-priv.h | 10 +- + sound/soc/sof/ipc4-topology.c | 65 +- + sound/soc/sof/mediatek/mt8186/mt8186.c | 2 +- + sound/soc/sof/mediatek/mt8195/mt8195.c | 2 +- + sound/soc/sof/topology.c | 3 +- + sound/soc/sprd/sprd-pcm-compress.c | 7 +- + sound/soc/sprd/sprd-pcm-dma.c | 10 +- + sound/soc/sti/uniperif_player.c | 29 +- + sound/soc/sti/uniperif_reader.c | 27 +- + sound/soc/stm/stm32_i2s.c | 12 + + sound/soc/stm/stm32_sai_sub.c | 15 + + sound/soc/stm/stm32_spdifrx.c | 2 +- + sound/soc/sunxi/sun4i-i2s.c | 13 + + sound/soc/sunxi/sun8i-codec.c | 13 + + sound/soc/tegra/tegra20_i2s.c | 10 + + sound/soc/tegra/tegra210_adx.c | 12 +- + sound/soc/tegra/tegra210_adx.h | 2 +- + sound/soc/tegra/tegra210_i2s.c | 13 + + sound/soc/tegra/tegra30_i2s.c | 10 + + sound/soc/ti/davinci-i2s.c | 16 +- + sound/soc/ti/davinci-mcasp.c | 17 +- + sound/soc/ti/omap-mcbsp.c | 14 +- + sound/soc/uniphier/aio-cpu.c | 9 + + sound/soc/uniphier/aio-dma.c | 5 +- + sound/soc/uniphier/aio.h | 2 +- + sound/soc/ux500/Kconfig | 23 +- + sound/soc/ux500/Makefile | 3 - + sound/soc/ux500/mop500.c | 167 -- + sound/soc/ux500/mop500_ab8500.c | 437 ---- + sound/soc/ux500/mop500_ab8500.h | 17 - + sound/soc/ux500/ux500_msp_dai.c | 33 +- + sound/soc/ux500/ux500_msp_i2s.c | 2 +- + sound/soc/xilinx/xlnx_formatter_pcm.c | 79 +- + sound/soc/xtensa/xtfpga-i2s.c | 12 +- + 501 files changed, 20884 insertions(+), 3583 deletions(-) + create mode 100644 Documentation/devicetree/bindings/sound/airoha,an7581-afe.yaml + create mode 100644 Documentation/devicetree/bindings/sound/airoha,an7581-wm8960.yaml + delete mode 100644 Documentation/devicetree/bindings/sound/davinci-evm-audio.txt + create mode 100644 Documentation/devicetree/bindings/sound/esstech,es9039q2m.yaml + create mode 100644 Documentation/devicetree/bindings/sound/nuvoton,nau8360.yaml + create mode 100644 Documentation/devicetree/bindings/sound/qcom,wcd9378-sdw.yaml + create mode 100644 Documentation/devicetree/bindings/sound/stericsson,ux500-msp-i2s.yaml + create mode 100644 Documentation/devicetree/bindings/sound/ti,da830-evm-audio.yaml + delete mode 100644 Documentation/devicetree/bindings/sound/ux500-mop500.txt + delete mode 100644 Documentation/devicetree/bindings/sound/ux500-msp.txt + create mode 100644 include/sound/sdca_class.h + create mode 100644 sound/soc/codecs/es9039q2m.c + create mode 100644 sound/soc/codecs/nau8360-dsp.c + create mode 100644 sound/soc/codecs/nau8360-dsp.h + create mode 100644 sound/soc/codecs/nau8360.c + create mode 100644 sound/soc/codecs/nau8360.h + create mode 100644 sound/soc/codecs/sn624x-sdca-sdw.c + create mode 100644 sound/soc/codecs/sn624x-sdca.h + create mode 100644 sound/soc/codecs/wcd9378-sdca.c + create mode 100644 sound/soc/codecs/wcd9378-sdca.h + create mode 100644 sound/soc/codecs/wcd9378-sdw.c + create mode 100644 sound/soc/mediatek/an7581/Makefile + create mode 100644 sound/soc/mediatek/an7581/an7581-afe-common.h + create mode 100644 sound/soc/mediatek/an7581/an7581-afe-pcm.c + create mode 100644 sound/soc/mediatek/an7581/an7581-dai-etdm.c + create mode 100644 sound/soc/mediatek/an7581/an7581-reg.h + create mode 100644 sound/soc/mediatek/an7581/an7581-wm8960.c + delete mode 100644 sound/soc/sdca/sdca_class.h + create mode 100644 sound/soc/sdw_utils/soc_sdw_senary_amp.c + create mode 100644 sound/soc/sdw_utils/soc_sdw_senary_dmic.c + create mode 100644 sound/soc/sdw_utils/soc_sdw_senary_sdca.c + create mode 100644 sound/soc/sdw_utils/soc_sdw_senary_sdca_jack_common.c + create mode 100644 sound/soc/soc-internal.h + create mode 100644 sound/soc/sof/amd/acp7x.h + delete mode 100644 sound/soc/ux500/mop500.c + delete mode 100644 sound/soc/ux500/mop500_ab8500.c + delete mode 100644 sound/soc/ux500/mop500_ab8500.h +Merging modules/modules-next (ff7360c5c7313 module: Remove the error-injection.h include from linux/module.h) +$ git merge -m Merge branch 'modules-next' of https://git.kernel.org/pub/scm/linux/kernel/git/modules/linux.git modules/modules-next +Auto-merging drivers/gpu/drm/i915/gt/intel_engine_cs.c +Auto-merging drivers/gpu/drm/xe/xe_device.c +Auto-merging drivers/gpu/drm/xe/xe_exec_queue.c +Auto-merging drivers/gpu/drm/xe/xe_ggtt.c +Auto-merging drivers/gpu/drm/xe/xe_guc.c +Auto-merging drivers/gpu/drm/xe/xe_guc_ads.c +Auto-merging drivers/gpu/drm/xe/xe_guc_ct.c +Auto-merging drivers/gpu/drm/xe/xe_guc_log.c +Auto-merging drivers/gpu/drm/xe/xe_mmio.c +Auto-merging drivers/gpu/drm/xe/xe_oa.c +Auto-merging drivers/gpu/drm/xe/xe_pm.c +Auto-merging drivers/gpu/drm/xe/xe_pt.c +Auto-merging drivers/gpu/drm/xe/xe_sync.c +Auto-merging drivers/gpu/drm/xe/xe_tuning.c +Auto-merging drivers/gpu/drm/xe/xe_uc_fw.c +Auto-merging drivers/gpu/drm/xe/xe_vm.c +Auto-merging drivers/gpu/drm/xe/xe_wa.c +Auto-merging fs/coredump.c +Auto-merging fs/nfsd/nfs4layouts.c +Auto-merging fs/nfsd/nfs4recover.c +Auto-merging kernel/module/main.c +Auto-merging kernel/reboot.c +Auto-merging kernel/time/jiffies.c +Auto-merging net/bridge/br_stp_if.c +Merge made by the 'ort' strategy. + arch/x86/kernel/cpu/mce/dev-mcelog.c | 2 +- + block/blk-core.c | 1 + + drivers/block/drbd/drbd_nl.c | 1 + + drivers/gpu/drm/i915/display/intel_connector.c | 1 + + .../gpu/drm/i915/display/intel_display_driver.c | 1 + + drivers/gpu/drm/i915/gt/intel_engine_cs.c | 1 + + drivers/gpu/drm/i915/gt/intel_gt.c | 2 + + drivers/gpu/drm/i915/gt/uc/intel_guc_ct.c | 1 + + drivers/gpu/drm/i915/gt/uc/intel_uc.c | 1 + + drivers/gpu/drm/i915/gt/uc/intel_uc_fw.c | 1 + + drivers/gpu/drm/i915/i915_driver.c | 1 + + drivers/gpu/drm/i915/i915_pci.c | 2 + + drivers/gpu/drm/i915/intel_uncore.c | 1 + + drivers/gpu/drm/xe/xe_device.c | 2 +- + drivers/gpu/drm/xe/xe_exec_queue.c | 1 + + drivers/gpu/drm/xe/xe_ggtt.c | 2 +- + drivers/gpu/drm/xe/xe_guc.c | 1 + + drivers/gpu/drm/xe/xe_guc_ads.c | 2 +- + drivers/gpu/drm/xe/xe_guc_ct.c | 2 +- + drivers/gpu/drm/xe/xe_guc_log.c | 2 +- + drivers/gpu/drm/xe/xe_guc_relay.c | 2 +- + drivers/gpu/drm/xe/xe_hw_engine_class_sysfs.c | 1 + + drivers/gpu/drm/xe/xe_hw_engine_group.c | 2 + + drivers/gpu/drm/xe/xe_mmio.c | 1 + + drivers/gpu/drm/xe/xe_oa.c | 1 + + drivers/gpu/drm/xe/xe_pm.c | 2 +- + drivers/gpu/drm/xe/xe_pt.c | 2 + + drivers/gpu/drm/xe/xe_pxp.c | 2 + + drivers/gpu/drm/xe/xe_sriov.c | 2 +- + drivers/gpu/drm/xe/xe_sync.c | 1 + + drivers/gpu/drm/xe/xe_tile.c | 2 +- + drivers/gpu/drm/xe/xe_tuning.c | 2 + + drivers/gpu/drm/xe/xe_uc_fw.c | 2 +- + drivers/gpu/drm/xe/xe_vm.c | 1 + + drivers/gpu/drm/xe/xe_wa.c | 2 +- + drivers/gpu/drm/xe/xe_wopcm.c | 2 +- + drivers/greybus/svc_watchdog.c | 1 + + drivers/macintosh/windfarm_core.c | 1 + + drivers/net/netdevsim/tc.c | 1 + + drivers/pnp/pnpbios/core.c | 2 +- + drivers/video/fbdev/uvesafb.c | 1 + + fs/coredump.c | 2 +- + fs/nfs/cache_lib.c | 2 +- + fs/nfsd/nfs4layouts.c | 2 +- + fs/nfsd/nfs4recover.c | 1 + + fs/ocfs2/stackglue.c | 1 + + include/linux/compat.h | 1 + + include/linux/kmod.h | 12 +---- + include/linux/module.h | 1 - + include/linux/module_symbol.h | 7 ++- + include/linux/syscalls.h | 1 + + kernel/cgroup/cgroup-v1.c | 2 +- + kernel/module/kallsyms.c | 59 ++++++++++++---------- + kernel/module/kmod.c | 3 +- + kernel/module/main.c | 1 + + kernel/power/process.c | 2 +- + kernel/reboot.c | 2 +- + kernel/time/jiffies.c | 1 + + kernel/umh.c | 2 +- + lib/kobject_uevent.c | 2 +- + net/bridge/br_stp_if.c | 2 +- + net/core/skb_fault_injection.c | 1 + + scripts/faddr2line | 2 +- + scripts/mod/modpost.h | 2 +- + security/keys/request_key.c | 2 +- + security/tomoyo/common.h | 2 +- + tools/perf/util/symbol.h | 4 +- + 67 files changed, 108 insertions(+), 72 deletions(-) +Merging input/next (5b051ea055b31 dt-bindings: input: Add AD7147 CapTouch schema) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/dtor/input.git input/next +Auto-merging Documentation/devicetree/bindings/input/mediatek,mt6779-keypad.yaml +Auto-merging MAINTAINERS +Auto-merging drivers/input/keyboard/atkbd.c +Merge made by the 'ort' strategy. + .../devicetree/bindings/input/adc-joystick.yaml | 20 +- + .../bindings/input/adi,ad7147_captouch.yaml | 48 ++++ + .../devicetree/bindings/input/cypress-sf.yaml | 14 +- + .../bindings/input/mediatek,mt6779-keypad.yaml | 10 +- + .../bindings/input/microchip,cap11xx.yaml | 14 +- + .../bindings/input/touchscreen/eeti,exc3000.yaml | 16 +- + .../bindings/input/touchscreen/sis,9200-ts.yaml | 61 +++++ + .../bindings/input/touchscreen/sis_i2c.txt | 31 --- + .../bindings/input/touchscreen/ti,tsc2007.yaml | 12 +- + MAINTAINERS | 2 +- + drivers/input/input-mt.c | 5 +- + drivers/input/keyboard/atkbd.c | 5 +- + drivers/input/keyboard/snvs_pwrkey.c | 9 +- + drivers/input/keyboard/st-keyscan.c | 13 +- + drivers/input/matrix-keymap.c | 7 + + drivers/input/serio/gscps2.c | 293 +++++++++++---------- + 16 files changed, 319 insertions(+), 241 deletions(-) + create mode 100644 Documentation/devicetree/bindings/input/adi,ad7147_captouch.yaml + create mode 100644 Documentation/devicetree/bindings/input/touchscreen/sis,9200-ts.yaml + delete mode 100644 Documentation/devicetree/bindings/input/touchscreen/sis_i2c.txt +Merging block/for-next (67cd826ded183 Merge branch 'for-7.4/io_uring' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/axboe/linux.git block/for-next +Auto-merging block/blk-core.c +Auto-merging io_uring/io-wq.c +Merge made by the 'ort' strategy. + arch/m68k/mac/config.c | 31 +- + block/bdev.c | 2 +- + block/blk-core.c | 7 +- + block/blk-mq.c | 1 + + block/blk-settings.c | 18 +- + block/blk-zoned.c | 872 +++++++++++++++++++++++-------------- + block/blk.h | 6 + + block/fops.c | 40 +- + drivers/block/amiflop.c | 71 ++- + drivers/block/rnull/configfs.rs | 44 +- + drivers/block/rnull/rnull.rs | 15 +- + drivers/block/swim.c | 457 ++++++++++--------- + drivers/block/swim3.c | 8 +- + drivers/block/swim_asm.S | 340 ++++++++------- + drivers/block/ublk_drv.c | 10 +- + drivers/nvme/host/ioctl.c | 19 +- + include/linux/blkdev.h | 11 +- + include/linux/io_uring.h | 32 ++ + include/linux/io_uring/cmd.h | 34 +- + include/linux/io_uring_types.h | 2 + + io_uring/cancel.c | 89 +++- + io_uring/cancel.h | 13 +- + io_uring/fdinfo.c | 11 +- + io_uring/io-wq.c | 12 +- + io_uring/io_uring.c | 225 ++++++++-- + io_uring/io_uring.h | 31 -- + io_uring/memmap.c | 8 + + io_uring/notif.c | 2 + + io_uring/refs.h | 27 ++ + io_uring/rsrc.c | 42 +- + io_uring/rw.c | 4 + + io_uring/timeout.c | 18 + + io_uring/timeout.h | 2 + + io_uring/uring_cmd.c | 23 +- + io_uring/uring_cmd.h | 2 +- + rust/kernel/block/mq/gen_disk.rs | 72 +-- + rust/kernel/block/mq/operations.rs | 16 +- + rust/kernel/block/mq/request.rs | 16 +- + rust/kernel/block/mq/tag_set.rs | 35 +- + 39 files changed, 1703 insertions(+), 965 deletions(-) +$ git am -3 ../patches/0001-Revert-block-remove-bio_last_bvec_all.patch +Applying: Revert "block: remove bio_last_bvec_all" +$ git reset HEAD^ +Unstaged changes after reset: +M Documentation/block/biovecs.rst +M include/linux/bio.h +$ git add -A . +$ git commit -v -a --amend +warning: notes ref refs/notes/commits is invalid +[master 77a35520490e5] Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/axboe/linux.git + Date: Sat Oct 3 00:43:49 2026 +0200 +$ git am -3 ../patches/0001-Fixup-for-blk-zoned-mismerge.patch +Applying: Fixup for blk-zoned mismerge +Using index info to reconstruct a base tree... +M block/blk-zoned.c +Falling back to patching base and 3-way merge... +No changes -- Patch already applied. +Merging device-mapper/for-next (c0df022cb0fa2 dm-writecache: fix REQ_FUA processing) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/device-mapper/linux-dm.git device-mapper/for-next +Merge made by the 'ort' strategy. + Documentation/admin-guide/device-mapper/verity.rst | 5 ++ + drivers/md/Kconfig | 14 ++++ + drivers/md/dm-cache-metadata.c | 2 +- + drivers/md/dm-clone-metadata.c | 3 - + drivers/md/dm-crypt.c | 23 +++--- + drivers/md/dm-delay.c | 3 +- + drivers/md/dm-exception-store.h | 2 +- + drivers/md/dm-mpath.c | 5 +- + drivers/md/dm-pcache/backing_dev.c | 2 +- + drivers/md/dm-raid.c | 2 +- + drivers/md/dm-raid1.c | 2 +- + drivers/md/dm-thin.c | 3 - + drivers/md/dm-unstripe.c | 2 +- + drivers/md/dm-vdo/block-map.c | 2 +- + drivers/md/dm-vdo/dm-vdo-target.c | 2 +- + drivers/md/dm-vdo/indexer/chapter-index.c | 15 +++- + drivers/md/dm-vdo/indexer/delta-index.c | 17 +++- + drivers/md/dm-vdo/indexer/volume.c | 6 -- + drivers/md/dm-verity-verify-sig.c | 4 +- + drivers/md/dm-writecache.c | 93 ++++++++++++++-------- + drivers/md/dm-zone.c | 6 +- + drivers/md/dm.c | 2 +- + drivers/md/persistent-data/dm-bitset.c | 5 +- + drivers/md/persistent-data/dm-btree-remove.c | 2 +- + 24 files changed, 136 insertions(+), 86 deletions(-) +Merging libata/for-next (cfce1dc635041 ata: libata: Fix scsi_done() documentation) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/libata/linux libata/for-next +Auto-merging drivers/ata/Kconfig +Auto-merging drivers/ata/ahci.c +Auto-merging drivers/ata/libahci.c +Auto-merging drivers/ata/libahci_platform.c +Auto-merging drivers/ata/libata-core.c +Auto-merging drivers/ata/libata-scsi.c +Auto-merging include/linux/libata.h +Merge made by the 'ort' strategy. + Documentation/driver-api/libata.rst | 4 +- + drivers/ata/Kconfig | 17 +- + drivers/ata/Makefile | 2 +- + drivers/ata/ahci.c | 19 +- + drivers/ata/ahci_brcm.c | 1 + + drivers/ata/ahci_ceva.c | 1 + + drivers/ata/ahci_da850.c | 6 +- + drivers/ata/ahci_qoriq.c | 1 + + drivers/ata/ahci_st.c | 78 +-- + drivers/ata/ata_generic.c | 16 + + drivers/ata/libahci.c | 29 +- + drivers/ata/libahci_platform.c | 4 + + drivers/ata/libata-core.c | 81 ++- + drivers/ata/libata-scsi.c | 7 +- + drivers/ata/libata-sff.c | 13 +- + drivers/ata/pata_arasan_cf.c | 6 +- + drivers/ata/pata_cswarp.c | 183 +++++++ + drivers/ata/pata_parport/pata_parport.c | 9 +- + drivers/ata/sata_dwc_460ex.c | 5 +- + drivers/ata/sata_fsl.c | 26 +- + drivers/ata/sata_inic162x.c | 903 -------------------------------- + drivers/ata/sata_qstor.c | 8 +- + include/linux/libata.h | 1 + + 23 files changed, 416 insertions(+), 1004 deletions(-) + create mode 100644 drivers/ata/pata_cswarp.c + delete mode 100644 drivers/ata/sata_inic162x.c +Merging pcmcia/pcmcia-next (b3c26ea81ccc5 pcmcia: remove obsolete host controller drivers) +$ git merge -m Merge branch 'pcmcia-next' of https://git.kernel.org/pub/scm/linux/kernel/git/brodo/linux.git pcmcia/pcmcia-next +Already up to date. +Merging mmc/next (fd9af27f6c231 dt-bindings: mmc: sun4i-a10-mmc: add Allwinner B288) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/ulfh/mmc.git mmc/next +Auto-merging CREDITS +Auto-merging Documentation/devicetree/bindings/mmc/qcom,sdhci-msm.yaml +Auto-merging MAINTAINERS +Auto-merging drivers/mmc/host/Kconfig +Merge made by the 'ort' strategy. + CREDITS | 5 + + .../bindings/mmc/allwinner,sun4i-a10-mmc.yaml | 2 + + .../bindings/mmc/amlogic,meson-gx-mmc.yaml | 29 +- + .../devicetree/bindings/mmc/cdns,sd6hc.yaml | 105 + + .../bindings/mmc/mmc-controller-common.yaml | 10 + + .../devicetree/bindings/mmc/qcom,sdhci-msm.yaml | 2 + + Documentation/driver-api/mmc/mmc-async-req.rst | 64 +- + Documentation/driver-api/mmc/mmc-tools.rst | 1 + + MAINTAINERS | 12 +- + drivers/memstick/core/ms_block.c | 2 +- + drivers/memstick/host/r592.c | 2 +- + drivers/mmc/core/block.c | 2 +- + drivers/mmc/core/core.c | 8 +- + drivers/mmc/core/core.h | 10 +- + drivers/mmc/core/crypto.c | 2 +- + drivers/mmc/core/host.c | 39 +- + drivers/mmc/core/mmc.c | 2 +- + drivers/mmc/core/mmc_ops.c | 22 +- + drivers/mmc/core/sdio_cis.c | 5 +- + drivers/mmc/host/Kconfig | 31 - + drivers/mmc/host/Makefile | 2 +- + drivers/mmc/host/atmel-mci.c | 5 +- + drivers/mmc/host/bcm2835.c | 2 +- + drivers/mmc/host/cavium-thunderx.c | 5 +- + drivers/mmc/host/davinci_mmc.c | 6 +- + drivers/mmc/host/dw_mmc-rockchip.c | 2 +- + drivers/mmc/host/dw_mmc.c | 316 ++- + drivers/mmc/host/dw_mmc.h | 18 +- + drivers/mmc/host/meson-gx-mmc.c | 31 +- + drivers/mmc/host/mvsdio.c | 2 +- + drivers/mmc/host/rtsx_usb_sdmmc.c | 5 + + .../host/{sdhci-cadence.c => sdhci-cadence-core.c} | 303 ++- + drivers/mmc/host/sdhci-cadence-phy-v6.c | 975 ++++++++ + drivers/mmc/host/sdhci-cadence.h | 118 + + drivers/mmc/host/sdhci-msm.c | 34 +- + drivers/mmc/host/sdhci-of-arasan.c | 4 +- + drivers/mmc/host/sdhci-of-dwcmshc.c | 13 + + drivers/mmc/host/sdhci-of-esdhc.c | 2 +- + drivers/mmc/host/sdhci-omap.c | 2 +- + drivers/mmc/host/sdhci-pci-core.c | 8 +- + drivers/mmc/host/sdhci-pxav3.c | 13 +- + drivers/mmc/host/sdhci.h | 2 +- + drivers/mmc/host/sunplus-mmc.c | 5 + + drivers/mmc/host/ushc.c | 5 +- + drivers/mmc/host/vub300.c | 2493 -------------------- + drivers/staging/greybus/sdio.c | 11 +- + include/linux/mmc/host.h | 6 +- + 47 files changed, 1825 insertions(+), 2918 deletions(-) + create mode 100644 Documentation/devicetree/bindings/mmc/cdns,sd6hc.yaml + rename drivers/mmc/host/{sdhci-cadence.c => sdhci-cadence-core.c} (67%) + create mode 100644 drivers/mmc/host/sdhci-cadence-phy-v6.c + create mode 100644 drivers/mmc/host/sdhci-cadence.h + delete mode 100644 drivers/mmc/host/vub300.c +Merging mfd/for-mfd-next (319633b06ff2b dt-bindings: mfd: qcom,tcsr: Add compatible for MSM8952) +$ git merge -m Merge branch 'for-mfd-next' of https://git.kernel.org/pub/scm/linux/kernel/git/lee/mfd.git mfd/for-mfd-next +Auto-merging MAINTAINERS +Auto-merging drivers/gpio/Kconfig +Auto-merging drivers/gpio/Makefile +Merge made by the 'ort' strategy. + .../devicetree/bindings/input/cpcap-pwrbutton.txt | 20 - + .../bindings/input/motorola,cpcap-pwrbutton.yaml | 32 ++ + .../leds/backlight/ti,lm3533-backlight.yaml | 69 +++ + .../devicetree/bindings/leds/ti,lm3533-leds.yaml | 67 +++ + .../devicetree/bindings/leds/ti,lm3533.yaml | 169 +++++++ + .../devicetree/bindings/mfd/motorola,cpcap.yaml | 408 +++++++++++++++ + .../devicetree/bindings/mfd/motorola-cpcap.txt | 78 --- + .../devicetree/bindings/mfd/qcom,spmi-pmic.yaml | 1 + + .../devicetree/bindings/mfd/qcom,tcsr.yaml | 1 + + .../devicetree/bindings/mfd/rohm,bd71815-pmic.yaml | 9 +- + .../devicetree/bindings/mfd/rohm,bd71828-pmic.yaml | 9 +- + .../devicetree/bindings/mfd/rohm,bd72720-pmic.yaml | 29 +- + .../devicetree/bindings/mfd/rohm,bd73800-pmic.yaml | 218 ++++++++ + .../devicetree/bindings/mfd/rohm,pmic-pins.yaml | 72 +++ + .../devicetree/bindings/mfd/stericsson,ab8500.yaml | 150 +++++- + .../bindings/mfd/ti,keystone-devctrl.yaml | 90 ++++ + .../devicetree/bindings/mfd/ti,tps61050.yaml | 92 ++++ + .../devicetree/bindings/mfd/ti,tps65910.yaml | 8 + + .../devicetree/bindings/mfd/ti,twl6040.yaml | 156 ++++++ + .../bindings/mfd/ti-keystone-devctrl.txt | 19 - + Documentation/devicetree/bindings/mfd/tps6105x.txt | 62 --- + Documentation/devicetree/bindings/mfd/twl6040.txt | 67 --- + .../bindings/regulator/rohm,bd73800-regulator.yaml | 99 ++++ + MAINTAINERS | 2 + + drivers/clk/clk-bd718x7.c | 8 + + drivers/gpio/Kconfig | 12 + + drivers/gpio/Makefile | 1 + + drivers/gpio/gpio-bd73800.c | 209 ++++++++ + drivers/iio/light/lm3533-als.c | 185 +++---- + drivers/leds/leds-lm3533.c | 121 ++--- + drivers/mfd/Kconfig | 17 +- + drivers/mfd/ab8500-core.c | 2 +- + drivers/mfd/cs42l43-i2c.c | 10 + + drivers/mfd/cs42l43-sdw.c | 9 + + drivers/mfd/cs42l43.c | 30 +- + drivers/mfd/cs42l43.h | 1 + + drivers/mfd/da903x.c | 34 +- + drivers/mfd/da9062-core.c | 24 +- + drivers/mfd/da9150-core.c | 8 +- + drivers/mfd/intel-lpss.c | 24 +- + drivers/mfd/intel_quark_i2c_gpio.c | 52 +- + drivers/mfd/iqs62x.c | 8 +- + drivers/mfd/khadas-mcu.c | 104 +++- + drivers/mfd/lm3533-core.c | 386 ++++++-------- + drivers/mfd/lm3533-ctrlbank.c | 33 +- + drivers/mfd/max77843.c | 8 +- + drivers/mfd/motorola-cpcap.c | 143 +++--- + drivers/mfd/mt6360-core.c | 8 +- + drivers/mfd/rk8xx-core.c | 2 +- + drivers/mfd/rk8xx-i2c.c | 4 +- + drivers/mfd/rn5t618.c | 8 +- + drivers/mfd/rohm-bd71828.c | 147 +++++- + drivers/mfd/ti_am335x_tscadc.c | 8 +- + drivers/mfd/tps6586x.c | 8 +- + drivers/mfd/twl-core.c | 31 +- + drivers/mfd/wcd934x.c | 8 +- + drivers/mfd/wm831x-auxadc.c | 2 +- + drivers/regulator/Kconfig | 4 +- + drivers/regulator/bd71828-regulator.c | 558 ++++++++++++++++++++- + drivers/rtc/rtc-bd70528.c | 8 + + drivers/thermal/khadas_mcu_fan.c | 108 +++- + drivers/video/backlight/lm3533_bl.c | 188 +++---- + include/linux/mfd/cs42l43-regs.h | 1 + + include/linux/mfd/da9150/core.h | 1 + + include/linux/mfd/khadas-mcu.h | 19 +- + include/linux/mfd/lm3533.h | 75 +-- + include/linux/mfd/motorola-cpcap.h | 7 + + include/linux/mfd/rohm-bd73800.h | 306 +++++++++++ + include/linux/mfd/rohm-generic.h | 1 + + 69 files changed, 3747 insertions(+), 1111 deletions(-) + delete mode 100644 Documentation/devicetree/bindings/input/cpcap-pwrbutton.txt + create mode 100644 Documentation/devicetree/bindings/input/motorola,cpcap-pwrbutton.yaml + create mode 100644 Documentation/devicetree/bindings/leds/backlight/ti,lm3533-backlight.yaml + create mode 100644 Documentation/devicetree/bindings/leds/ti,lm3533-leds.yaml + create mode 100644 Documentation/devicetree/bindings/leds/ti,lm3533.yaml + create mode 100644 Documentation/devicetree/bindings/mfd/motorola,cpcap.yaml + delete mode 100644 Documentation/devicetree/bindings/mfd/motorola-cpcap.txt + create mode 100644 Documentation/devicetree/bindings/mfd/rohm,bd73800-pmic.yaml + create mode 100644 Documentation/devicetree/bindings/mfd/rohm,pmic-pins.yaml + create mode 100644 Documentation/devicetree/bindings/mfd/ti,keystone-devctrl.yaml + create mode 100644 Documentation/devicetree/bindings/mfd/ti,tps61050.yaml + create mode 100644 Documentation/devicetree/bindings/mfd/ti,twl6040.yaml + delete mode 100644 Documentation/devicetree/bindings/mfd/ti-keystone-devctrl.txt + delete mode 100644 Documentation/devicetree/bindings/mfd/tps6105x.txt + delete mode 100644 Documentation/devicetree/bindings/mfd/twl6040.txt + create mode 100644 Documentation/devicetree/bindings/regulator/rohm,bd73800-regulator.yaml + create mode 100644 drivers/gpio/gpio-bd73800.c + create mode 100644 include/linux/mfd/rohm-bd73800.h +Merging backlight/for-backlight-next (5e1631df673f3 backlight: ili9320/ktd253: Fix typos in comments) +$ git merge -m Merge branch 'for-backlight-next' of https://git.kernel.org/pub/scm/linux/kernel/git/lee/backlight.git backlight/for-backlight-next +Merge made by the 'ort' strategy. + drivers/video/backlight/ili9320.c | 2 +- + drivers/video/backlight/ktd253-backlight.c | 2 +- + 2 files changed, 2 insertions(+), 2 deletions(-) +Merging battery/for-next (4fc88ba435dad power: supply: cros_charge-control: adopt EC charge state on probe) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/sre/linux-power-supply.git battery/for-next +Auto-merging drivers/power/supply/max17042_battery.c +Auto-merging include/linux/power_supply.h +Merge made by the 'ort' strategy. + Documentation/ABI/testing/sysfs-class-power | 39 ++++- + .../ABI/testing/sysfs-class-power-ltc4162l | 2 + + Documentation/ABI/testing/sysfs-class-power-mp2629 | 2 +- + Documentation/ABI/testing/sysfs-class-power-rt9467 | 2 + + Documentation/ABI/testing/sysfs-class-power-rt9471 | 2 + + .../devicetree/bindings/power/supply/bq27xxx.yaml | 2 +- + drivers/power/reset/keystone-reset.c | 3 +- + drivers/power/supply/bq24190_charger.c | 71 +++++++++- + drivers/power/supply/bq24257_charger.c | 43 +++++- + drivers/power/supply/bq2515x_charger.c | 15 +- + drivers/power/supply/bq25630_charger.c | 84 +++++++++++ + drivers/power/supply/bq25890_charger.c | 2 +- + drivers/power/supply/bq27xxx_battery.c | 18 ++- + drivers/power/supply/bq27xxx_battery_i2c.c | 2 + + drivers/power/supply/charger-manager.c | 2 +- + drivers/power/supply/cros_charge-control.c | 157 ++++++++++++++++++--- + drivers/power/supply/da9030_battery.c | 2 +- + drivers/power/supply/ltc4162-l-charger.c | 56 +++++++- + drivers/power/supply/max17042_battery.c | 2 +- + drivers/power/supply/mm8013.c | 1 + + drivers/power/supply/power_supply_sysfs.c | 17 +++ + drivers/power/supply/qcom_smbx.c | 98 ++++++++++--- + drivers/power/supply/rt9467-charger.c | 54 ++++++- + drivers/power/supply/rt9471.c | 54 ++++++- + drivers/power/supply/s2mu005-battery.c | 4 +- + drivers/power/supply/sbs-charger.c | 4 +- + drivers/power/supply/smb347-charger.c | 4 +- + include/linux/power/bq27xxx_battery.h | 1 + + include/linux/power_supply.h | 18 ++- + 29 files changed, 671 insertions(+), 90 deletions(-) +Merging regulator/for-next (6722016006986 Merge regulator/for-7.4 into regulator-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/broonie/regulator.git regulator/for-next +Auto-merging Documentation/devicetree/bindings/vendor-prefixes.yaml +Auto-merging MAINTAINERS +Auto-merging drivers/regulator/Kconfig +Merge made by the 'ort' strategy. + .../regulator/maxim,max77620-regulator.yaml | 12 +- + .../bindings/regulator/maxim,max77826.yaml | 4 +- + .../bindings/regulator/maxim,max77838.yaml | 4 +- + .../regulator/mediatek,mt6380-regulator.yaml | 117 ++++ + .../devicetree/bindings/regulator/mps,mp5416.yaml | 38 +- + .../devicetree/bindings/regulator/mps,mp886x.yaml | 30 +- + .../devicetree/bindings/regulator/mps,mpq4210.yaml | 68 ++ + .../devicetree/bindings/regulator/mps,mpq7920.yaml | 46 +- + .../bindings/regulator/mt6380-regulator.txt | 89 --- + .../bindings/regulator/nexperia,nex10000ub.yaml | 69 ++ + .../devicetree/bindings/regulator/nxp,pf0900.yaml | 2 +- + .../bindings/regulator/onnn,fan53880.yaml | 4 +- + .../bindings/regulator/pwm-regulator.yaml | 2 +- + .../bindings/regulator/qcom,rpmh-regulator.yaml | 16 + + .../regulator/richtek,rtmv20-regulator.yaml | 6 +- + .../bindings/regulator/silergy,sy8824x.yaml | 8 +- + .../bindings/regulator/silergy,sy8827n.yaml | 4 +- + .../devicetree/bindings/regulator/ti,tps65023.yaml | 90 +++ + .../devicetree/bindings/regulator/ti,tps6586x.yaml | 196 ++++++ + .../devicetree/bindings/regulator/tps65023.txt | 60 -- + .../devicetree/bindings/regulator/tps6586x.txt | 135 ---- + .../bindings/soc/mediatek/mediatek,pwrap.yaml | 3 + + .../devicetree/bindings/vendor-prefixes.yaml | 2 + + MAINTAINERS | 6 + + drivers/regulator/88pm886-regulator.c | 26 + + drivers/regulator/Kconfig | 21 +- + drivers/regulator/Makefile | 2 + + drivers/regulator/ab8500-ext.c | 2 +- + drivers/regulator/ab8500.c | 748 +++++++++++++++++++-- + drivers/regulator/axp20x-regulator.c | 6 +- + drivers/regulator/core.c | 10 +- + drivers/regulator/fixed.c | 4 + + drivers/regulator/mp886x.c | 34 +- + drivers/regulator/mpq4210.c | 243 +++++++ + drivers/regulator/nex10000ub-regulator.c | 147 ++++ + drivers/regulator/pca9450-regulator.c | 81 ++- + drivers/regulator/pf1550-regulator.c | 19 +- + drivers/regulator/pv88080-regulator.c | 4 +- + drivers/regulator/qcom-rpmh-regulator.c | 55 +- + drivers/regulator/rt6190-regulator.c | 18 +- + drivers/regulator/tps6105x-regulator.c | 2 +- + drivers/regulator/tps65185.c | 6 +- + include/linux/mfd/88pm886.h | 7 + + include/linux/regulator/driver.h | 2 +- + include/linux/regulator/pca9450.h | 4 +- + kernel/irq/irqdesc.c | 10 +- + kernel/softirq.c | 4 +- + 47 files changed, 1979 insertions(+), 487 deletions(-) + create mode 100644 Documentation/devicetree/bindings/regulator/mediatek,mt6380-regulator.yaml + create mode 100644 Documentation/devicetree/bindings/regulator/mps,mpq4210.yaml + delete mode 100644 Documentation/devicetree/bindings/regulator/mt6380-regulator.txt + create mode 100644 Documentation/devicetree/bindings/regulator/nexperia,nex10000ub.yaml + create mode 100644 Documentation/devicetree/bindings/regulator/ti,tps65023.yaml + create mode 100644 Documentation/devicetree/bindings/regulator/ti,tps6586x.yaml + delete mode 100644 Documentation/devicetree/bindings/regulator/tps65023.txt + delete mode 100644 Documentation/devicetree/bindings/regulator/tps6586x.txt + create mode 100644 drivers/regulator/mpq4210.c + create mode 100644 drivers/regulator/nex10000ub-regulator.c +Merging security/next (881f19c2ffbc7 lsm: remove redundant NULL check in security_init()) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/pcmoore/lsm.git security/next +Auto-merging fs/namei.c +Auto-merging fs/namespace.c +Auto-merging include/linux/lsm_hook_defs.h +CONFLICT (content): Merge conflict in include/linux/lsm_hook_defs.h +Auto-merging include/linux/ns/ns_common_types.h +CONFLICT (content): Merge conflict in include/linux/ns/ns_common_types.h +Auto-merging include/linux/sched.h +Auto-merging include/linux/security.h +CONFLICT (content): Merge conflict in include/linux/security.h +Auto-merging security/security.c +Auto-merging security/selinux/hooks.c +Auto-merging security/smack/smack_lsm.c +Auto-merging tools/testing/selftests/bpf/progs/lsm.c +Auto-merging tools/testing/selftests/landlock/fs_test.c +Resolved 'include/linux/lsm_hook_defs.h' using previous resolution. +Resolved 'include/linux/ns/ns_common_types.h' using previous resolution. +Resolved 'include/linux/security.h' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 050f8d78a9392] Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/pcmoore/lsm.git +$ git diff -M --stat --summary HEAD^.. + Documentation/admin-guide/LSM/index.rst | 20 +- + fs/cachefiles/security.c | 4 +- + fs/namei.c | 18 +- + fs/namespace.c | 3 +- + include/linux/cred.h | 17 +- + include/linux/lsm_audit.h | 7 +- + include/linux/lsm_hook_defs.h | 28 +-- + include/linux/lsm_hooks.h | 1 + + include/linux/ns/ns_common_types.h | 3 + + include/linux/sched.h | 8 +- + include/linux/security.h | 78 +++++--- + include/uapi/linux/nsfs.h | 1 + + init/init_task.c | 2 +- + kernel/auditsc.c | 5 +- + kernel/cred.c | 5 +- + kernel/nscommon.c | 17 +- + kernel/nsproxy.c | 6 + + rust/kernel/security.rs | 3 +- + security/lsm_audit.c | 8 +- + security/lsm_init.c | 9 +- + security/security.c | 115 +++++++++-- + security/selinux/hooks.c | 25 ++- + security/smack/smack_lsm.c | 9 +- + tools/testing/selftests/bpf/prog_tests/test_lsm.c | 231 ++++++++++++++++++++++ + tools/testing/selftests/bpf/progs/lsm.c | 79 ++++++++ + tools/testing/selftests/landlock/fs_test.c | 10 +- + 26 files changed, 600 insertions(+), 112 deletions(-) +$ git am -3 ../patches/0001-security-Fix-up-mismerge-and-additional-semantic-iss.patch +Applying: security: Fix up mismerge and additional semantic issues with vfs-brauner +Using index info to reconstruct a base tree... +M include/linux/lsm_hook_defs.h +M security/security.c +Falling back to patching base and 3-way merge... +Auto-merging security/security.c +$ git reset HEAD^ +Unstaged changes after reset: +M include/linux/security.h +M security/apparmor/af_unix.c +M security/apparmor/file.c +M security/apparmor/include/af_unix.h +M security/apparmor/include/file.h +M security/apparmor/include/net.h +M security/apparmor/lsm.c +M security/apparmor/net.c +M security/security.c +M security/selinux/hooks.c +M security/smack/smack_lsm.c +$ git add -A . +$ git commit -v -a --amend +warning: notes ref refs/notes/commits is invalid +[master a8fb6db3b8a82] Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/pcmoore/lsm.git + Date: Sat Oct 3 00:44:00 2026 +0200 +$ git am -3 ../patches/0001-security-More-merge-fixup-from-the-constification-of.patch +Applying: security: More merge fixup from the constification of idmap +$ git reset HEAD^ +Unstaged changes after reset: +M security/security.c +$ git add -A . +$ git commit -v -a --amend +warning: notes ref refs/notes/commits is invalid +[master ce2c10438aa83] Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/pcmoore/lsm.git + Date: Sat Oct 3 00:44:00 2026 +0200 +Merging apparmor/apparmor-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'apparmor-next' of https://git.kernel.org/pub/scm/linux/kernel/git/jj/linux-apparmor apparmor/apparmor-next +Already up to date. +Merging integrity/next-integrity (1dc317438308c integrity: Replace integrity_audit_message() with integrity_audit_msg()) +$ git merge -m Merge branch 'next-integrity' of https://git.kernel.org/pub/scm/linux/kernel/git/zohar/linux-integrity integrity/next-integrity +Auto-merging security/integrity/evm/evm_main.c +Auto-merging security/integrity/ima/ima_api.c +Auto-merging security/integrity/ima/ima_appraise.c +Auto-merging security/integrity/ima/ima_main.c +Auto-merging security/integrity/ima/ima_policy.c +Merge made by the 'ort' strategy. + security/integrity/evm/evm_main.c | 9 +++++---- + security/integrity/ima/ima_api.c | 8 ++++---- + security/integrity/ima/ima_appraise.c | 6 +++--- + security/integrity/ima/ima_fs.c | 7 ++++--- + security/integrity/ima/ima_init.c | 2 +- + security/integrity/ima/ima_main.c | 13 +++++++------ + security/integrity/ima/ima_policy.c | 11 ++++++----- + security/integrity/ima/ima_queue.c | 2 +- + security/integrity/ima/ima_queue_keys.c | 8 ++++---- + security/integrity/ima/ima_template_lib.c | 2 +- + security/integrity/integrity.h | 18 +++--------------- + security/integrity/integrity_audit.c | 12 ++---------- + 12 files changed, 41 insertions(+), 57 deletions(-) +Merging selinux/next (7ab68f08381f2 Automated merge of 'dev' into 'next') +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/pcmoore/selinux.git selinux/next +Auto-merging security/selinux/hooks.c +Auto-merging security/selinux/selinuxfs.c +Merge made by the 'ort' strategy. + security/selinux/avc.c | 42 ++++++------ + security/selinux/hooks.c | 9 ++- + security/selinux/include/avc.h | 7 +- + security/selinux/selinuxfs.c | 59 +++++++++-------- + security/selinux/ss/mls.c | 38 ----------- + security/selinux/ss/mls.h | 3 - + security/selinux/ss/mls_types.h | 3 - + security/selinux/ss/policydb.c | 137 ++++++++++++++++++++++++++++++++++++---- + 8 files changed, 194 insertions(+), 104 deletions(-) +Merging smack/next (fedc88e38ce97 smack: fix cred UAF in smack_file_send_sigiotask()) +$ git merge -m Merge branch 'next' of https://github.com/cschaufler/smack-next smack/next +Already up to date. +Merging tomoyo/master (72d3fcf802c45 Linux 7.3-rc5) +$ git merge -m Merge branch 'master' of git://git.code.sf.net/p/tomoyo/tomoyo.git tomoyo/master +Already up to date. +Merging tpmdd-tpm/for-next-tpm (015fb29a74834 tpm: Disable TPM on null key name mismatch) +$ git merge -m Merge branch 'for-next-tpm' of https://git.kernel.org/pub/scm/linux/kernel/git/jarkko/linux-tpmdd.git tpmdd-tpm/for-next-tpm +Already up to date. +Merging tpmdd-keys/for-next-keys (cf60aac3d7aca KEYS: Fix add_key() race with keyring restriction) +$ git merge -m Merge branch 'for-next-keys' of https://git.kernel.org/pub/scm/linux/kernel/git/jarkko/linux-tpmdd.git tpmdd-keys/for-next-keys +Auto-merging security/keys/trusted-keys/trusted_tpm1.c +Merge made by the 'ort' strategy. + security/keys/key.c | 6 ++-- + security/keys/persistent.c | 55 ++++++++++++++++++------------- + security/keys/trusted-keys/trusted_tpm1.c | 4 ++- + security/keys/trusted-keys/trusted_tpm2.c | 4 ++- + 4 files changed, 41 insertions(+), 28 deletions(-) +Merging watchdog/watchdog-next (8b5a9f09037e3 dt-bindings: watchdog: apple,wdt: Add t8140 compatible) +$ git merge -m Merge branch 'watchdog-next' of https://git.kernel.org/pub/scm/linux/kernel/git/groeck/linux-staging.git watchdog/watchdog-next +Auto-merging drivers/watchdog/hpwdt.c +Auto-merging drivers/watchdog/mtk_wdt.c +Merge made by the 'ort' strategy. + .../devicetree/bindings/watchdog/apple,wdt.yaml | 1 + + .../bindings/watchdog/mediatek,mtk-wdt.yaml | 2 + + .../bindings/watchdog/renesas,r9a09g057-wdt.yaml | 29 ++++- + .../devicetree/bindings/watchdog/samsung-wdt.yaml | 23 +++- + Documentation/watchdog/watchdog-parameters.rst | 2 + + drivers/watchdog/acquirewdt.c | 3 +- + drivers/watchdog/advantech_ec_wdt.c | 3 +- + drivers/watchdog/advantechwdt.c | 5 +- + drivers/watchdog/airoha_wdt.c | 5 +- + drivers/watchdog/alim1535_wdt.c | 5 +- + drivers/watchdog/alim7101_wdt.c | 5 +- + drivers/watchdog/arm_smc_wdt.c | 3 +- + drivers/watchdog/armada_37xx_wdt.c | 3 +- + drivers/watchdog/aspeed_wdt.c | 3 +- + drivers/watchdog/at91rm9200_wdt.c | 5 +- + drivers/watchdog/at91sam9_wdt.c | 5 +- + drivers/watchdog/atcwdt200_wdt.c | 5 +- + drivers/watchdog/ath79_wdt.c | 5 +- + drivers/watchdog/bcm2835_wdt.c | 3 +- + drivers/watchdog/bcm47xx_wdt.c | 5 +- + drivers/watchdog/bcm7038_wdt.c | 3 +- + drivers/watchdog/booke_wdt.c | 3 +- + drivers/watchdog/cadence_wdt.c | 13 +-- + drivers/watchdog/cgbc_wdt.c | 7 +- + drivers/watchdog/da9052_wdt.c | 5 +- + drivers/watchdog/da9055_wdt.c | 3 +- + drivers/watchdog/da9062_wdt.c | 8 +- + drivers/watchdog/davinci_wdt.c | 7 +- + drivers/watchdog/db8500_wdt.c | 5 +- + drivers/watchdog/dw_wdt.c | 3 +- + drivers/watchdog/ebc-c384_wdt.c | 5 +- + drivers/watchdog/eurotechwdt.c | 3 +- + drivers/watchdog/exar_wdt.c | 5 +- + drivers/watchdog/f71808e_wdt.c | 9 +- + drivers/watchdog/gef_wdt.c | 3 +- + drivers/watchdog/geodewdt.c | 5 +- + drivers/watchdog/gpio_wdt.c | 3 +- + drivers/watchdog/hpwdt.c | 3 +- + drivers/watchdog/i6300esb.c | 28 +++-- + drivers/watchdog/iTCO_wdt.c | 5 +- + drivers/watchdog/ib700wdt.c | 5 +- + drivers/watchdog/ibmasr.c | 3 +- + drivers/watchdog/ie6xx_wdt.c | 5 +- + drivers/watchdog/imgpdc_wdt.c | 5 +- + drivers/watchdog/imx2_wdt.c | 5 +- + drivers/watchdog/imx7ulp_wdt.c | 3 +- + drivers/watchdog/imx_sc_wdt.c | 3 +- + drivers/watchdog/indydog.c | 3 +- + drivers/watchdog/intel_oc_wdt.c | 5 +- + drivers/watchdog/it87_wdt.c | 7 +- + drivers/watchdog/jz4740_wdt.c | 7 +- + drivers/watchdog/keembay_wdt.c | 13 +-- + drivers/watchdog/kempld_wdt.c | 7 +- + drivers/watchdog/lenovo_se10_wdt.c | 5 +- + drivers/watchdog/lenovo_se30_wdt.c | 5 +- + drivers/watchdog/lenovo_se30g2_se60_wdt.c | 5 +- + drivers/watchdog/loongson1_wdt.c | 5 +- + drivers/watchdog/lpc18xx_wdt.c | 5 +- + drivers/watchdog/ma35d1_wdt.c | 3 +- + drivers/watchdog/max63xx_wdt.c | 9 +- + drivers/watchdog/max77620_wdt.c | 3 +- + drivers/watchdog/mena21_wdt.c | 3 +- + drivers/watchdog/menf21bmc_wdt.c | 3 +- + drivers/watchdog/menz69_wdt.c | 3 +- + drivers/watchdog/meson_gxbb_wdt.c | 5 +- + drivers/watchdog/meson_wdt.c | 3 +- + drivers/watchdog/mixcomwd.c | 3 +- + drivers/watchdog/mpc8xxx_wdt.c | 5 +- + drivers/watchdog/msc313e_wdt.c | 12 +-- + drivers/watchdog/mt7621_wdt.c | 3 +- + drivers/watchdog/mtk_wdt.c | 119 ++++++++++++++++++--- + drivers/watchdog/nct6694_wdt.c | 3 +- + drivers/watchdog/ni903x_wdt.c | 5 +- + drivers/watchdog/nic7018_wdt.c | 5 +- + drivers/watchdog/nv_tco.c | 5 +- + drivers/watchdog/octeon-wdt-main.c | 5 +- + drivers/watchdog/of_xilinx_wdt.c | 8 +- + drivers/watchdog/omap_wdt.c | 3 +- + drivers/watchdog/orion_wdt.c | 3 +- + drivers/watchdog/pc87413_wdt.c | 9 +- + drivers/watchdog/pcwd.c | 5 +- + drivers/watchdog/pcwd_pci.c | 7 +- + drivers/watchdog/pcwd_usb.c | 7 +- + drivers/watchdog/pika_wdt.c | 5 +- + drivers/watchdog/pm8916_wdt.c | 8 +- + drivers/watchdog/pnx4008_wdt.c | 5 +- + drivers/watchdog/pseries-wdt.c | 7 +- + drivers/watchdog/rc32434_wdt.c | 5 +- + drivers/watchdog/renesas_wdt.c | 3 +- + drivers/watchdog/rn5t618_wdt.c | 3 +- + drivers/watchdog/rt2880_wdt.c | 3 +- + drivers/watchdog/rti_wdt.c | 5 +- + drivers/watchdog/rzg2l_wdt.c | 3 +- + drivers/watchdog/rzv2h_wdt.c | 3 +- + drivers/watchdog/s32g_wdt.c | 5 +- + drivers/watchdog/s3c2410_wdt.c | 23 +++- + drivers/watchdog/sama5d4_wdt.c | 5 +- + drivers/watchdog/sbc60xxwdt.c | 5 +- + drivers/watchdog/sbc7240_wdt.c | 5 +- + drivers/watchdog/sbc8360.c | 3 +- + drivers/watchdog/sbc_epx_c3.c | 3 +- + drivers/watchdog/sbsa_gwdt.c | 24 ++++- + drivers/watchdog/sc1200wdt.c | 3 +- + drivers/watchdog/sch311x_wdt.c | 5 +- + drivers/watchdog/shwdt.c | 7 +- + drivers/watchdog/simatic-ipc-wdt.c | 3 +- + drivers/watchdog/sl28cpld_wdt.c | 3 +- + drivers/watchdog/smsc37b787_wdt.c | 3 +- + drivers/watchdog/softdog.c | 5 +- + drivers/watchdog/sp5100_tco.c | 68 ++++++++++-- + drivers/watchdog/sp805_wdt.c | 14 +-- + drivers/watchdog/starfive-wdt.c | 7 +- + drivers/watchdog/stmp3xxx_rtc_wdt.c | 13 +-- + drivers/watchdog/stpmic1_wdt.c | 3 +- + drivers/watchdog/sun4v_wdt.c | 5 +- + drivers/watchdog/sunplus_wdt.c | 3 +- + drivers/watchdog/sunxi_wdt.c | 3 +- + drivers/watchdog/tegra_wdt.c | 5 +- + drivers/watchdog/tqmx86_wdt.c | 3 +- + drivers/watchdog/ts4800_wdt.c | 3 +- + drivers/watchdog/twl4030_wdt.c | 3 +- + drivers/watchdog/txx9wdt.c | 7 +- + drivers/watchdog/uniphier_wdt.c | 5 +- + drivers/watchdog/via_wdt.c | 5 +- + drivers/watchdog/visconti_wdt.c | 3 +- + drivers/watchdog/w83627hf_wdt.c | 5 +- + drivers/watchdog/w83877f_wdt.c | 5 +- + drivers/watchdog/w83977f_wdt.c | 5 +- + drivers/watchdog/wafer5823wdt.c | 5 +- + drivers/watchdog/watchdog_core.c | 5 +- + drivers/watchdog/watchdog_dev.c | 9 +- + drivers/watchdog/wdat_wdt.c | 5 +- + drivers/watchdog/wdt.c | 9 +- + drivers/watchdog/wdt977.c | 7 +- + drivers/watchdog/wdt_pci.c | 9 +- + drivers/watchdog/wm831x_wdt.c | 3 +- + drivers/watchdog/wm8350_wdt.c | 3 +- + drivers/watchdog/xen_wdt.c | 5 +- + drivers/watchdog/xilinx_wwdt.c | 5 +- + drivers/watchdog/ziirave_wdt.c | 3 +- + include/dt-bindings/reset/mediatek,mt8167-wdt.h | 21 ++++ + 141 files changed, 689 insertions(+), 298 deletions(-) + create mode 100644 include/dt-bindings/reset/mediatek,mt8167-wdt.h +Merging iommu/next (cec2dc7663e71 Merge branches 'apple/dart', 'samsung/exynos', 'riscv', 'broadcom', 'intel/vt-d', 'amd/amd-vi' and 'core' into next) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/iommu/linux.git iommu/next +Auto-merging MAINTAINERS +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu.h +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_device.c +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_drv.c +Merge made by the 'ort' strategy. + .../bindings/display/brcm,bcm2835-hvs.yaml | 3 + + .../devicetree/bindings/iommu/apple,dart.yaml | 12 +- + .../bindings/iommu/brcm,bcm2712-iommu.yaml | 54 ++ + .../bindings/iommu/brcm,bcm2712-iommuc.yaml | 40 ++ + MAINTAINERS | 12 + + drivers/gpu/drm/amd/amdgpu/amdgpu.h | 1 + + drivers/gpu/drm/amd/amdgpu/amdgpu_device.c | 50 ++ + drivers/gpu/drm/amd/amdgpu/amdgpu_drv.c | 12 + + drivers/iommu/Kconfig | 17 +- + drivers/iommu/Makefile | 1 + + drivers/iommu/amd/amd_iommu.h | 3 + + drivers/iommu/amd/amd_iommu_types.h | 7 + + drivers/iommu/amd/init.c | 169 ++++--- + drivers/iommu/amd/iommu.c | 460 ++++++++++++----- + drivers/iommu/apple-dart.c | 74 ++- + drivers/iommu/bcm2712-iommu-cache.c | 84 ++++ + drivers/iommu/bcm2712-iommu-cache.h | 9 + + drivers/iommu/bcm2712-iommu.c | 550 +++++++++++++++++++++ + drivers/iommu/exynos-iommu.c | 37 +- + drivers/iommu/generic_pt/.kunitconfig | 1 + + drivers/iommu/generic_pt/Kconfig | 10 + + drivers/iommu/generic_pt/fmt/Makefile | 2 + + drivers/iommu/generic_pt/fmt/bcm2712.h | 288 +++++++++++ + drivers/iommu/generic_pt/fmt/defs_bcm2712.h | 18 + + drivers/iommu/generic_pt/fmt/iommu_bcm2712.c | 6 + + drivers/iommu/generic_pt/kunit_iommu_pt.h | 2 +- + drivers/iommu/intel/dmar.c | 72 ++- + drivers/iommu/intel/iommu.c | 156 +++++- + drivers/iommu/intel/iommu.h | 17 +- + drivers/iommu/intel/nested.c | 2 + + drivers/iommu/iommu-sva.c | 14 +- + drivers/iommu/iommu.c | 4 + + drivers/iommu/riscv/iommu.c | 12 + + include/linux/amd-iommu.h | 13 +- + include/linux/generic_pt/common.h | 6 + + include/linux/generic_pt/iommu.h | 12 + + 36 files changed, 1995 insertions(+), 235 deletions(-) + create mode 100644 Documentation/devicetree/bindings/iommu/brcm,bcm2712-iommu.yaml + create mode 100644 Documentation/devicetree/bindings/iommu/brcm,bcm2712-iommuc.yaml + create mode 100644 drivers/iommu/bcm2712-iommu-cache.c + create mode 100644 drivers/iommu/bcm2712-iommu-cache.h + create mode 100644 drivers/iommu/bcm2712-iommu.c + create mode 100644 drivers/iommu/generic_pt/fmt/bcm2712.h + create mode 100644 drivers/iommu/generic_pt/fmt/defs_bcm2712.h + create mode 100644 drivers/iommu/generic_pt/fmt/iommu_bcm2712.c +Merging audit/next (88bd5e852addf Automated merge of 'dev' into 'next') +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/pcmoore/audit.git audit/next +Auto-merging kernel/auditsc.c +Merge made by the 'ort' strategy. + include/linux/audit.h | 6 ++-- + kernel/audit.c | 2 +- + kernel/audit.h | 20 ++++++++---- + kernel/audit_tree.c | 1 + + kernel/audit_watch.c | 2 ++ + kernel/auditfilter.c | 86 +++++++++++++++++++++++++-------------------------- + kernel/auditsc.c | 4 ++- + 7 files changed, 66 insertions(+), 55 deletions(-) +Merging devicetree/for-next (bb91f8668c247 dt-bindings: ASoC: Convert MediaTek RT5650 codecs bindings to DT schema) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/robh/linux.git devicetree/for-next +Auto-merging Documentation/devicetree/bindings/i2c/xlnx,xps-iic-2.00.a.yaml +Auto-merging Documentation/devicetree/bindings/media/qcom,x1e80100-camss.yaml +Auto-merging Documentation/devicetree/bindings/mfd/ti,tps65910.yaml +Auto-merging Documentation/devicetree/bindings/trivial-devices.yaml +Auto-merging MAINTAINERS +Auto-merging drivers/of/irq.c +Merge made by the 'ort' strategy. + .../devicetree/bindings/arm/arm,cci-400.yaml | 6 +- + .../devicetree/bindings/arm/arm,coresight-cti.yaml | 2 +- + .../bindings/arm/mediatek/mediatek,audsys.yaml | 2 +- + .../bindings/arm/mediatek/mediatek,g3dsys.txt | 30 - + Documentation/devicetree/bindings/arm/omap/mpu.txt | 54 -- + .../devicetree/bindings/arm/ti/ti,omap-mpu.yaml | 58 ++ + .../devicetree/bindings/arm/vexpress-config.yaml | 4 + + .../devicetree/bindings/bus/omap-ocp2scp.txt | 29 - + .../devicetree/bindings/bus/ti,omap-ocp2scp.yaml | 74 ++ + .../bindings/display/bridge/sil,sii9022.yaml | 2 +- + .../display/mediatek/mediatek,hdmi-ddc.yaml | 11 +- + .../devicetree/bindings/dts-coding-style.rst | 5 +- + .../bindings/firmware/nvidia,tegra186-bpmp.yaml | 6 +- + .../devicetree/bindings/gpio/delta,tn48m-gpio.yaml | 2 +- + .../devicetree/bindings/gpu/arm,mali-utgard.yaml | 25 +- + .../devicetree/bindings/hwmon/adi,ltc2991.yaml | 2 +- + .../devicetree/bindings/hwmon/vexpress.txt | 23 - + .../bindings/i2c/xlnx,xps-iic-2.00.a.yaml | 4 +- + .../bindings/iio/light/upisemi,us5182.yaml | 2 +- + .../devicetree/bindings/input/matrix-keymap.yaml | 3 +- + .../devicetree/bindings/input/ti,tca8418.yaml | 21 +- + .../interrupt-controller/chrp,open-pic.yaml | 7 +- + .../bindings/interrupt-controller/qcom,pdc.yaml | 1 + + .../mailbox/allwinner,sun6i-a31-msgbox.yaml | 3 - + .../devicetree/bindings/mailbox/altera-mailbox.txt | 12 +- + .../bindings/mailbox/hisilicon,hi3660-mailbox.txt | 2 +- + .../bindings/mailbox/hisilicon,hi6220-mailbox.txt | 2 +- + .../devicetree/bindings/mailbox/mailbox.txt | 60 -- + .../bindings/mailbox/ti,omap-mailbox.yaml | 3 +- + .../bindings/media/amlogic,c3-mipi-csi2.yaml | 4 +- + .../devicetree/bindings/media/arm,mali-c55.yaml | 2 +- + .../devicetree/bindings/media/atmel,isc.yaml | 14 +- + .../bindings/media/brcm,bcm2835-unicam.yaml | 10 +- + .../devicetree/bindings/media/cdns,csi2rx.yaml | 58 +- + .../bindings/media/i2c/galaxycore,gc05a2.yaml | 2 +- + .../devicetree/bindings/media/i2c/ovti,ov2680.yaml | 26 +- + .../devicetree/bindings/media/i2c/ovti,ov2732.yaml | 6 +- + .../bindings/media/img,e5010-jpeg-enc.yaml | 22 +- + .../bindings/media/mediatek,vcodec-encoder.yaml | 2 +- + .../bindings/media/mediatek-jpeg-decoder.yaml | 4 +- + .../bindings/media/mediatek-jpeg-encoder.yaml | 2 +- + .../bindings/media/microchip,csi2dc.yaml | 52 +- + .../devicetree/bindings/media/microchip,xisc.yaml | 14 +- + .../devicetree/bindings/media/nxp,imx8-jpeg.yaml | 4 +- + .../bindings/media/qcom,msm8996-venus.yaml | 40 +- + .../bindings/media/qcom,x1e80100-camss.yaml | 26 +- + .../bindings/media/raspberrypi,pispbe.yaml | 12 +- + .../devicetree/bindings/media/samsung,fimc.yaml | 14 +- + .../devicetree/bindings/media/st,stm32-dcmi.yaml | 16 +- + .../devicetree/bindings/media/st,stm32-dcmipp.yaml | 14 +- + .../devicetree/bindings/media/ti,cal.yaml | 48 +- + .../devicetree/bindings/mfd/ti,tps65910.yaml | 2 +- + Documentation/devicetree/bindings/mux/reg-mux.yaml | 2 +- + .../bindings/nvmem/zii,rave-sp-eeprom.yaml | 2 +- + .../bindings/pci/hisilicon,kirin-pcie.yaml | 4 +- + .../bindings/pinctrl/sunplus,sp7021-pinctrl.yaml | 2 +- + .../bindings/power/reset/xlnx,zynqmp-power.yaml | 9 +- + .../devicetree/bindings/regulator/vexpress.txt | 32 - + .../bindings/scsi/hisilicon,hip05-sas-v1.yaml | 158 ++++ + .../devicetree/bindings/scsi/hisilicon-sas.txt | 98 --- + .../soc/hisilicon/hisilicon,hip05-cpld.yaml | 35 + + .../soc/mediatek/mediatek,mt2701-g3dsys.yaml | 58 ++ + .../bindings/soc/nuvoton/nuvoton,npcm-gcr.yaml | 18 +- + .../bindings/soc/qcom/qcom,aoss-qmp.yaml | 2 +- + .../bindings/sound/mediatek,mt8173-rt5650.yaml | 73 ++ + .../devicetree/bindings/sound/mt8173-rt5650.txt | 31 - + .../bindings/spi/aspeed,ast2600-fmc.yaml | 6 +- + .../devicetree/bindings/thermal/thermal-idle.yaml | 2 +- + .../devicetree/bindings/trivial-devices.yaml | 4 +- + Documentation/devicetree/of_unittest.rst | 2 +- + MAINTAINERS | 4 +- + drivers/of/fdt_address.c | 2 +- + drivers/of/irq.c | 39 +- + drivers/of/property.c | 23 +- + include/dt-bindings/clock/agilex-clock.h | 1 + + scripts/dtc/dt-check-style | 972 ++++++++++++--------- + .../dtc/dt-style-selftest/bad/dts-blank-lines.dts | 37 + + .../bad/dts-child-name-order.dtso | 33 + + .../dtc/dt-style-selftest/bad/dts-cont-align.dts | 27 + + .../dt-style-selftest/bad/dts-digit-node-order.dts | 40 + + .../bad/dts-digit-node-order.dtso | 41 + + .../bad/dts-extend-node-child-name-order.dtso | 26 + + .../bad/dts-extend-node-digit-node-order.dtso | 34 + + .../dtc/dt-style-selftest/bad/dts-line-length.dts | 22 + + .../dtc/dt-style-selftest/bad/dts-node-name.dts | 60 ++ + .../dt-style-selftest/bad/dts-property-name.dts | 28 + + .../dt-style-selftest/bad/dts-property-order.dts | 18 + + .../dt-style-selftest/bad/dts-property-order.dtso | 62 ++ + .../bad/dts-redundant-ws-strict.dts | 27 + + .../dtc/dt-style-selftest/bad/dts-redundant-ws.dts | 28 + + .../dt-style-selftest/bad/dts-redundant-ws.dtso | 9 + + .../dtc/dt-style-selftest/bad/dts-trailing-ws.dts | 16 + + .../dtc/dt-style-selftest/bad/dts-unused-label.dts | 21 + + .../dt-style-selftest/bad/yaml-blank-lines.yaml | 54 ++ + .../bad/yaml-child-addr-order.yaml | 2 +- + .../bad/yaml-child-name-order.yaml | 2 +- + .../dtc/dt-style-selftest/bad/yaml-cont-align.yaml | 9 +- + .../bad/yaml-digit-node-order.yaml | 2 +- + .../dtc/dt-style-selftest/bad/yaml-hex-case.yaml | 7 +- + .../dt-style-selftest/bad/yaml-indent-strict.yaml | 2 +- + .../bad/yaml-label-in-string.yaml | 2 +- + .../dt-style-selftest/bad/yaml-line-length.yaml | 5 +- + .../dt-style-selftest/bad/yaml-mixed-indent.yaml | 4 +- + .../dt-style-selftest/bad/yaml-multi-close.yaml | 2 +- + .../dtc/dt-style-selftest/bad/yaml-node-close.yaml | 2 +- + .../dtc/dt-style-selftest/bad/yaml-node-name.yaml | 54 ++ + .../bad/yaml-prop-order-device-type.yaml | 2 +- + .../dtc/dt-style-selftest/bad/yaml-prop-order.yaml | 7 +- + .../dt-style-selftest/bad/yaml-prop-pairing.yaml | 2 +- + .../dt-style-selftest/bad/yaml-property-name.yaml | 46 + + .../bad/yaml-redundant-ws-strict.yaml | 31 + + .../dt-style-selftest/bad/yaml-redundant-ws.yaml | 35 + + .../dt-style-selftest/bad/yaml-required-blank.yaml | 2 +- + scripts/dtc/dt-style-selftest/bad/yaml-tab.yaml | 2 +- + .../bad/yaml-trailing-comment.yaml | 2 +- + .../dt-style-selftest/bad/yaml-trailing-ws.yaml | 7 +- + .../bad/yaml-unclosed-comment.yaml | 2 +- + .../bad/yaml-unit-addr-prefix.yaml | 2 +- + .../dtc/dt-style-selftest/bad/yaml-unit-addr.yaml | 2 +- + .../dt-style-selftest/bad/yaml-unused-label.yaml | 2 +- + .../bad/yaml-value-ws-multiline.yaml | 2 +- + .../dtc/dt-style-selftest/bad/yaml-value-ws.yaml | 2 +- + .../expected/dts-blank-lines.dts.txt | 6 + + .../expected/dts-child-name-order.dts.txt | 1 + + .../expected/dts-child-name-order.dtso.txt | 3 + + .../expected/dts-cont-align.dts.txt | 11 + + .../expected/dts-digit-node-order.dts.txt | 2 + + .../expected/dts-digit-node-order.dtso.txt | 2 + + .../dts-extend-node-child-name-order.dtso.txt | 2 + + .../dts-extend-node-digit-node-order.dtso.txt | 2 + + .../expected/dts-line-length.dts.txt | 3 + + .../expected/dts-mixed-indent.dts.txt | 1 + + .../expected/dts-node-name.dts.txt | 13 + + .../expected/dts-property-name.dts.txt | 14 + + .../expected/dts-property-order.dts.txt | 16 +- + .../expected/dts-property-order.dtso.txt | 12 + + .../expected/dts-redundant-ws-strict.dts.txt | 13 + + .../expected/dts-redundant-ws.dts.txt | 10 + + .../expected/dts-redundant-ws.dtso.txt | 2 + + .../expected/dts-trailing-ws.dts.txt | 3 + + .../expected/dts-unused-label.dts.txt | 2 + + .../expected/yaml-blank-lines.yaml.txt | 6 + + .../expected/yaml-cont-align.yaml.txt | 4 +- + .../expected/yaml-hex-case.yaml.txt | 2 + + .../expected/yaml-line-length.yaml.txt | 1 + + .../expected/yaml-mixed-indent.yaml.txt | 1 + + .../expected/yaml-node-name.yaml.txt | 7 + + .../expected/yaml-prop-order.yaml.txt | 1 + + .../expected/yaml-prop-pairing.yaml.txt | 4 +- + .../expected/yaml-property-name.yaml.txt | 14 + + .../expected/yaml-redundant-ws-strict.yaml.txt | 5 + + .../expected/yaml-redundant-ws.yaml.txt | 4 + + .../expected/yaml-trailing-ws.yaml.txt | 2 + + .../expected/yaml-value-ws-multiline.yaml.txt | 1 + + .../good/dts-child-name-order.dtso | 44 + + .../dtc/dt-style-selftest/good/dts-cont-align.dts | 14 +- + .../good/dts-digit-node-order.dts | 3 - + .../good/dts-digit-node-order.dtso | 59 ++ + .../good/dts-extend-node-child-name-order.dtso | 26 + + .../good/dts-extend-node-digit-node-order.dtso | 34 + + .../dt-style-selftest/good/dts-property-order.dts | 8 + + .../dt-style-selftest/good/dts-property-order.dtso | 50 ++ + .../dtc/dt-style-selftest/good/yaml-4space.yaml | 2 +- + .../dt-style-selftest/good/yaml-cont-align.yaml | 33 + + .../good/yaml-tricky-parsing.yaml | 2 +- + scripts/dtc/dt-style-selftest/run.sh | 2 +- + 166 files changed, 2589 insertions(+), 1110 deletions(-) + delete mode 100644 Documentation/devicetree/bindings/arm/mediatek/mediatek,g3dsys.txt + delete mode 100644 Documentation/devicetree/bindings/arm/omap/mpu.txt + create mode 100644 Documentation/devicetree/bindings/arm/ti/ti,omap-mpu.yaml + delete mode 100644 Documentation/devicetree/bindings/bus/omap-ocp2scp.txt + create mode 100644 Documentation/devicetree/bindings/bus/ti,omap-ocp2scp.yaml + delete mode 100644 Documentation/devicetree/bindings/hwmon/vexpress.txt + delete mode 100644 Documentation/devicetree/bindings/mailbox/mailbox.txt + delete mode 100644 Documentation/devicetree/bindings/regulator/vexpress.txt + create mode 100644 Documentation/devicetree/bindings/scsi/hisilicon,hip05-sas-v1.yaml + delete mode 100644 Documentation/devicetree/bindings/scsi/hisilicon-sas.txt + create mode 100644 Documentation/devicetree/bindings/soc/hisilicon/hisilicon,hip05-cpld.yaml + create mode 100644 Documentation/devicetree/bindings/soc/mediatek/mediatek,mt2701-g3dsys.yaml + create mode 100644 Documentation/devicetree/bindings/sound/mediatek,mt8173-rt5650.yaml + delete mode 100644 Documentation/devicetree/bindings/sound/mt8173-rt5650.txt + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-blank-lines.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-child-name-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-cont-align.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-digit-node-order.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-digit-node-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-extend-node-child-name-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-extend-node-digit-node-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-line-length.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-node-name.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-property-name.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-property-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-redundant-ws-strict.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-redundant-ws.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-redundant-ws.dtso + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-trailing-ws.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-unused-label.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/yaml-blank-lines.yaml + create mode 100644 scripts/dtc/dt-style-selftest/bad/yaml-node-name.yaml + create mode 100644 scripts/dtc/dt-style-selftest/bad/yaml-property-name.yaml + create mode 100644 scripts/dtc/dt-style-selftest/bad/yaml-redundant-ws-strict.yaml + create mode 100644 scripts/dtc/dt-style-selftest/bad/yaml-redundant-ws.yaml + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-blank-lines.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-child-name-order.dtso.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-cont-align.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-digit-node-order.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-digit-node-order.dtso.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-extend-node-child-name-order.dtso.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-extend-node-digit-node-order.dtso.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-line-length.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-node-name.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-property-name.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-property-order.dtso.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-redundant-ws-strict.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-redundant-ws.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-redundant-ws.dtso.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-trailing-ws.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-unused-label.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/yaml-blank-lines.yaml.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/yaml-node-name.yaml.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/yaml-property-name.yaml.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/yaml-redundant-ws-strict.yaml.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/yaml-redundant-ws.yaml.txt + create mode 100644 scripts/dtc/dt-style-selftest/good/dts-child-name-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/good/dts-digit-node-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/good/dts-extend-node-child-name-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/good/dts-extend-node-digit-node-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/good/dts-property-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/good/yaml-cont-align.yaml +Merging dt-krzk/for-next (dfcea1641506d Merge branches 'next/dt' and 'next/dt64' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-dt.git dt-krzk/for-next +Merge made by the 'ort' strategy. + arch/arm/boot/dts/actions/owl-s500.dtsi | 12 ++++++------ + arch/arm/boot/dts/arm/vexpress-v2p-ca5s.dts | 2 +- + arch/arm/boot/dts/aspeed/aspeed-bmc-opp-witherspoon.dts | 2 +- + arch/arm/boot/dts/aspeed/aspeed-bmc-supermicro-x11spi.dts | 2 +- + arch/arm64/boot/dts/actions/s700.dtsi | 6 +++--- + arch/arm64/boot/dts/actions/s900.dtsi | 6 +++--- + arch/arm64/boot/dts/cavium/thunder-88xx.dtsi | 6 +++--- + 7 files changed, 18 insertions(+), 18 deletions(-) +Merging mailbox/for-next (14af7a96afa39 mailbox: add Axiado AX3005 mailbox driver) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/jassibrar/mailbox.git mailbox/for-next +Already up to date. +Merging spi/for-next (ed5f8f8b115db Merge spi/for-7.4 into spi-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/broonie/spi.git spi/for-next +Auto-merging MAINTAINERS +Auto-merging drivers/spi/spi-atmel.c +Merge made by the 'ort' strategy. + Documentation/ABI/testing/sysfs-class-spi-master | 6 +- + .../devicetree/bindings/spi/amlogic,a4-spisg.yaml | 30 +- + .../bindings/spi/amlogic,meson-gx-spicc.yaml | 13 +- + .../bindings/spi/microchip,pic32mzda-spi.yaml | 82 +++ + .../bindings/spi/microchip,spi-pic32.txt | 34 -- + .../bindings/spi/realtek,rtd1625-nor.yaml | 65 +++ + .../devicetree/bindings/spi/renesas,sh-msiof.yaml | 9 +- + .../bindings/spi/spi-peripheral-props.yaml | 8 + + Documentation/spi/multiple-data-lanes.rst | 10 +- + MAINTAINERS | 6 + + drivers/spi/Kconfig | 374 ++++++------ + drivers/spi/Makefile | 1 + + drivers/spi/spi-amlogic-spisg.c | 116 +++- + drivers/spi/spi-ar934x.c | 13 +- + drivers/spi/spi-at91-usart.c | 4 +- + drivers/spi/spi-atmel.c | 4 +- + drivers/spi/spi-bcm2835.c | 29 +- + drivers/spi/spi-bcm63xx.c | 2 +- + drivers/spi/spi-cadence-quadspi.c | 2 +- + drivers/spi/spi-dw-core.c | 126 +++-- + drivers/spi/spi-dw-dma.c | 290 +++++++++- + drivers/spi/spi-dw-mmio.c | 2 + + drivers/spi/spi-dw.h | 8 + + drivers/spi/spi-ep93xx.c | 8 +- + drivers/spi/spi-fsl-dspi.c | 35 +- + drivers/spi/spi-geni-qcom.c | 93 ++- + drivers/spi/spi-imx.c | 12 +- + drivers/spi/spi-ingenic.c | 10 +- + drivers/spi/spi-ma35d1-qspi.c | 27 +- + drivers/spi/spi-mem.c | 14 +- + drivers/spi/spi-mtk-nor.c | 1 + + drivers/spi/spi-mxic.c | 8 +- + drivers/spi/spi-nxp-xspi.c | 16 +- + drivers/spi/spi-omap2-mcspi.c | 4 +- + drivers/spi/spi-orion.c | 9 + + drivers/spi/spi-pic32.c | 2 +- + drivers/spi/spi-pl022.c | 19 +- + drivers/spi/spi-qpic-snand.c | 105 ++-- + drivers/spi/spi-realtek-rtl.c | 36 +- + drivers/spi/spi-rockchip-sfc.c | 21 +- + drivers/spi/spi-rspi.c | 49 +- + drivers/spi/spi-rtk-nor.c | 624 +++++++++++++++++++++ + drivers/spi/spi-s3c64xx.c | 2 +- + drivers/spi/spi-sg2044-nor.c | 60 +- + drivers/spi/spi-sh-msiof.c | 18 +- + drivers/spi/spi-stm32.c | 2 +- + drivers/spi/spi-sun6i.c | 2 +- + drivers/spi/spi-sunplus-sp7021.c | 13 +- + drivers/spi/spi-tegra210-quad.c | 517 ++++++++++++++--- + drivers/spi/spi-virtio.c | 22 +- + drivers/spi/spi-xilinx.c | 2 +- + drivers/spi/spi.c | 15 +- + include/linux/spi/spi.h | 9 + + tools/spi/Makefile | 2 +- + tools/spi/spidev_test.c | 330 ++++++++--- + 55 files changed, 2567 insertions(+), 754 deletions(-) + create mode 100644 Documentation/devicetree/bindings/spi/microchip,pic32mzda-spi.yaml + delete mode 100644 Documentation/devicetree/bindings/spi/microchip,spi-pic32.txt + create mode 100644 Documentation/devicetree/bindings/spi/realtek,rtd1625-nor.yaml + create mode 100644 drivers/spi/spi-rtk-nor.c +Merging tip/master (8f511d67b4fb0 Merge branch into tip/master: 'x86/tdx') +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/tip/tip.git tip/master +Auto-merging Documentation/ABI/testing/sysfs-devices-system-cpu +Auto-merging Documentation/arch/arm64/silicon-errata.rst +Auto-merging Documentation/devicetree/bindings/interrupt-controller/qcom,pdc.yaml +Auto-merging Documentation/scheduler/index.rst +CONFLICT (content): Merge conflict in Documentation/scheduler/index.rst +Auto-merging MAINTAINERS +Auto-merging arch/Kconfig +Auto-merging arch/arm64/Kconfig +Auto-merging arch/arm64/Kconfig.platforms +Auto-merging arch/arm64/configs/defconfig +CONFLICT (content): Merge conflict in arch/arm64/configs/defconfig +Auto-merging arch/arm64/include/asm/preempt.h +Auto-merging arch/loongarch/Kconfig +Auto-merging arch/powerpc/Kconfig +Auto-merging arch/riscv/Kconfig +Auto-merging arch/s390/Kconfig +Auto-merging arch/s390/kernel/hiperdispatch.c +Auto-merging arch/um/kernel/um_arch.c +Auto-merging arch/x86/Kconfig +Auto-merging arch/x86/crypto/aesni-intel_glue.c +Auto-merging arch/x86/include/asm/string_64.h +Auto-merging drivers/irqchip/irq-aclint-sswi.c +Auto-merging drivers/resctrl/mpam_resctrl.c +Auto-merging fs/aio.c +Auto-merging fs/exec.c +Auto-merging include/linux/sched.h +Auto-merging io_uring/rw.c +Auto-merging kernel/events/core.c +Auto-merging kernel/exit.c +Auto-merging kernel/sched/fair.c +Auto-merging net/core/pktgen.c +Resolved 'Documentation/scheduler/index.rst' using previous resolution. +Resolved 'arch/arm64/configs/defconfig' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 1e66b89ba7f4f] Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/tip/tip.git +$ git diff -M --stat --summary HEAD^.. + Documentation/ABI/testing/sysfs-devices-system-cpu | 24 +- + Documentation/ABI/testing/sysfs-platform-ts5500 | 54 -- + Documentation/arch/arm64/silicon-errata.rst | 3 + + Documentation/arch/x86/tdx.rst | 21 + + Documentation/arch/x86/xstate.rst | 54 ++ + .../bindings/interrupt-controller/qcom,pdc.yaml | 1 + + Documentation/driver-api/index.rst | 1 + + Documentation/driver-api/steal-governor.rst | 151 ++++ + Documentation/filesystems/resctrl.rst | 19 +- + Documentation/scheduler/index.rst | 1 + + Documentation/scheduler/sched-paravirt.rst | 67 ++ + MAINTAINERS | 14 +- + arch/Kconfig | 38 - + arch/arm/kernel/perf_regs.c | 8 +- + arch/arm64/Kconfig | 10 +- + arch/arm64/Kconfig.platforms | 1 + + arch/arm64/boot/dts/qcom/purwa.dtsi | 5 + + arch/arm64/configs/defconfig | 2 - + arch/arm64/include/asm/preempt.h | 10 - + arch/arm64/kernel/paravirt.c | 4 +- + arch/arm64/kernel/perf_regs.c | 8 +- + arch/csky/kernel/perf_regs.c | 8 +- + arch/loongarch/Kconfig | 1 - + arch/loongarch/kernel/paravirt.c | 4 +- + arch/loongarch/kernel/perf_regs.c | 8 +- + arch/mips/kernel/perf_regs.c | 8 +- + arch/parisc/kernel/perf_regs.c | 8 +- + arch/powerpc/Kconfig | 1 - + arch/powerpc/perf/perf_regs.c | 2 +- + arch/powerpc/platforms/pseries/setup.c | 4 +- + arch/riscv/Kconfig | 1 - + arch/riscv/include/asm/smp.h | 12 + + arch/riscv/kernel/paravirt.c | 4 +- + arch/riscv/kernel/perf_regs.c | 8 +- + arch/riscv/kernel/sbi-ipi.c | 4 +- + arch/riscv/kernel/smp.c | 12 - + arch/s390/Kconfig | 1 - + arch/s390/include/asm/preempt.h | 11 - + arch/s390/kernel/hiperdispatch.c | 10 +- + arch/s390/kernel/perf_regs.c | 2 +- + arch/um/kernel/um_arch.c | 78 +- + arch/x86/Kconfig | 10 - + arch/x86/boot/compressed/error.c | 19 - + arch/x86/boot/compressed/error.h | 1 - + arch/x86/boot/compressed/mem.c | 42 - + arch/x86/boot/compressed/sev.h | 2 - + arch/x86/boot/compressed/tdx-shared.c | 2 + + arch/x86/boot/early_serial_console.c | 13 +- + arch/x86/boot/string.c | 13 +- + arch/x86/coco/sev/core.c | 32 +- + arch/x86/coco/tdx/tdx-shared.c | 31 + + arch/x86/coco/tdx/tdx.c | 41 +- + arch/x86/crypto/aegis128-aesni-glue.c | 3 +- + arch/x86/crypto/aesni-intel_glue.c | 7 +- + arch/x86/crypto/aria_aesni_avx2_glue.c | 11 +- + arch/x86/crypto/aria_aesni_avx_glue.c | 11 +- + arch/x86/crypto/aria_gfni_avx512_glue.c | 11 +- + arch/x86/crypto/camellia_aesni_avx2_glue.c | 11 +- + arch/x86/crypto/camellia_aesni_avx_glue.c | 11 +- + arch/x86/crypto/cast5_avx_glue.c | 7 +- + arch/x86/crypto/cast6_avx_glue.c | 7 +- + arch/x86/crypto/serpent_avx2_glue.c | 9 +- + arch/x86/crypto/serpent_avx_glue.c | 7 +- + arch/x86/crypto/sm4_aesni_avx2_glue.c | 11 +- + arch/x86/crypto/sm4_aesni_avx_glue.c | 11 +- + arch/x86/crypto/twofish_avx_glue.c | 6 +- + arch/x86/events/amd/uncore.c | 39 +- + arch/x86/events/core.c | 488 ++++++++++- + arch/x86/events/intel/core.c | 154 ++-- + arch/x86/events/intel/ds.c | 232 ++++-- + arch/x86/events/intel/lbr.c | 2 +- + arch/x86/events/perf_event.h | 217 ++++- + arch/x86/include/asm/cpufeatures.h | 5 +- + arch/x86/include/asm/cpuid/api.h | 2 +- + arch/x86/include/asm/fpu/regset.h | 6 +- + arch/x86/include/asm/fpu/sched.h | 6 +- + arch/x86/include/asm/fpu/types.h | 25 - + arch/x86/include/asm/fpu/xstate.h | 3 + + arch/x86/include/asm/kvm-x86-ops.h | 1 + + arch/x86/include/asm/kvm_host.h | 10 +- + arch/x86/include/asm/local.h | 4 +- + arch/x86/include/asm/math_emu.h | 15 - + arch/x86/include/asm/msr-index.h | 10 + + arch/x86/include/asm/perf_event.h | 46 +- + arch/x86/include/asm/preempt.h | 28 - + arch/x86/include/asm/processor.h | 2 +- + arch/x86/include/asm/sev.h | 4 + + arch/x86/include/asm/shared/string.h | 52 ++ + arch/x86/include/asm/shared/tdx.h | 6 + + arch/x86/include/asm/special_insns.h | 13 - + arch/x86/include/asm/string.h | 21 +- + arch/x86/include/asm/string_64.h | 1 - + arch/x86/include/asm/tdx.h | 23 + + arch/x86/include/asm/tdx_global_metadata.h | 9 +- + arch/x86/include/asm/traps.h | 2 - + arch/x86/include/uapi/asm/perf_regs.h | 53 ++ + arch/x86/include/uapi/asm/sigcontext.h | 15 + + arch/x86/kernel/asm-offsets.c | 1 + + arch/x86/kernel/cpu/amd.c | 21 +- + arch/x86/kernel/cpu/bugs.c | 18 +- + arch/x86/kernel/cpu/common.c | 98 ++- + arch/x86/kernel/cpu/microcode/intel-ucode-defs.h | 91 +- + arch/x86/kernel/cpu/mtrr/amd.c | 9 +- + arch/x86/kernel/cpu/resctrl/ctrlmondata.c | 6 + + arch/x86/kernel/cpu/resctrl/intel_aet.c | 34 + + arch/x86/kernel/cpu/scattered.c | 2 + + arch/x86/kernel/cpu/sgx/main.c | 8 +- + arch/x86/kernel/cpu/vmware.c | 4 +- + arch/x86/kernel/crash.c | 6 +- + arch/x86/kernel/fpu/bugs.c | 4 - + arch/x86/kernel/fpu/core.c | 46 +- + arch/x86/kernel/fpu/init.c | 6 +- + arch/x86/kernel/fpu/regset.c | 6 - + arch/x86/kernel/fpu/signal.c | 136 +-- + arch/x86/kernel/fpu/xstate.c | 62 +- + arch/x86/kernel/kvm.c | 4 +- + arch/x86/kernel/perf_regs.c | 172 +++- + arch/x86/kernel/shstk.c | 2 +- + arch/x86/kvm/mmu/mmu.c | 4 + + arch/x86/kvm/svm/sev.c | 2 + + arch/x86/kvm/vmx/pmu_intel.c | 28 +- + arch/x86/kvm/vmx/tdx.c | 99 ++- + arch/x86/kvm/vmx/tdx.h | 2 + + arch/x86/kvm/vmx/vmx.c | 10 +- + arch/x86/kvm/vmx/vmx.h | 15 +- + arch/x86/platform/Makefile | 1 - + arch/x86/platform/olpc/olpc-xo15-sci.c | 4 +- + arch/x86/platform/pvh/enlighten.c | 3 +- + arch/x86/platform/ts5500/Makefile | 2 - + arch/x86/platform/ts5500/ts5500.c | 341 -------- + arch/x86/virt/svm/sev.c | 172 +++- + arch/x86/virt/vmx/tdx/seamcall_internal.h | 19 + + arch/x86/virt/vmx/tdx/tdx.c | 439 ++++++++-- + arch/x86/virt/vmx/tdx/tdx.h | 10 +- + arch/x86/virt/vmx/tdx/tdx_global_metadata.c | 23 +- + arch/x86/virt/vmx/tdx/tdxcall.S | 10 +- + arch/x86/xen/pmu.c | 5 +- + drivers/base/cpu.c | 12 + + drivers/clocksource/timer-clint.c | 4 +- + drivers/crypto/ccp/sev-dev.c | 2 + + drivers/firmware/efi/libstub/x86-stub.c | 39 + + drivers/irqchip/Kconfig | 6 +- + drivers/irqchip/irq-aclint-sswi.c | 4 +- + drivers/irqchip/irq-al-fic.c | 2 +- + drivers/irqchip/irq-gic-v3-its.c | 6 +- + drivers/irqchip/irq-gic-v3.c | 12 +- + drivers/irqchip/irq-gic-v5.c | 9 +- + drivers/irqchip/irq-gic.c | 8 +- + drivers/irqchip/irq-lan966x-oic.c | 1 - + drivers/irqchip/irq-mtk-cirq.c | 2 +- + drivers/irqchip/irq-pruss-intc.c | 38 +- + drivers/irqchip/irq-riscv-imsic-early.c | 4 +- + drivers/irqchip/irq-riscv-imsic-state.c | 13 +- + drivers/irqchip/irq-riscv-imsic-state.h | 1 - + drivers/irqchip/irq-sifive-plic.c | 2 +- + drivers/irqchip/irq-vic.c | 2 +- + drivers/irqchip/qcom-pdc.c | 3 + + drivers/resctrl/mpam_resctrl.c | 5 + + drivers/soc/fsl/qe/qe_ports_ic.c | 1 - + drivers/virt/Kconfig | 17 + + drivers/virt/Makefile | 1 + + drivers/virt/coco/tdx-guest/tdx-guest.c | 6 +- + drivers/virt/steal_governor.c | 296 +++++++ + drivers/xen/time.c | 4 +- + fs/aio.c | 2 +- + fs/exec.c | 13 +- + fs/proc/uptime.c | 6 +- + fs/resctrl/ctrlmondata.c | 6 +- + fs/resctrl/monitor.c | 2 + + fs/resctrl/pseudo_lock.c | 9 +- + fs/resctrl/rdtgroup.c | 8 +- + include/asm-generic/preempt.h | 10 - + include/linux/bitmap.h | 14 + + include/linux/cpumask.h | 42 + + include/linux/hrtimer.h | 10 +- + include/linux/interrupt.h | 6 +- + include/linux/irq-entry-common.h | 17 +- + include/linux/kernel.h | 20 - + include/linux/kernel_stat.h | 11 + + include/linux/list.h | 6 +- + include/linux/perf_event.h | 23 + + include/linux/perf_regs.h | 36 +- + include/linux/posix-timers.h | 39 +- + include/linux/preempt.h | 20 +- + include/linux/resctrl.h | 19 + + include/linux/resctrl_types.h | 2 + + include/linux/sched.h | 32 +- + include/linux/sched/cputime.h | 6 +- + include/linux/sched/task.h | 1 - + include/linux/wait.h | 2 +- + include/uapi/linux/perf_event.h | 49 +- + include/uapi/linux/sched.h | 2 +- + io_uring/rw.c | 2 +- + kernel/Kconfig.kexec | 2 +- + kernel/Kconfig.preempt | 13 +- + kernel/cpu.c | 6 + + kernel/crash_core.c | 2 +- + kernel/entry/common.c | 17 +- + kernel/events/core.c | 177 +++- + kernel/exit.c | 13 +- + kernel/futex/requeue.c | 2 +- + kernel/futex/waitwake.c | 8 +- + kernel/irq/irqdomain.c | 1 + + kernel/locking/rtmutex.c | 2 +- + kernel/sched/core.c | 444 +++++----- + kernel/sched/cputime.c | 4 +- + kernel/sched/deadline.c | 9 +- + kernel/sched/debug.c | 21 +- + kernel/sched/ext/ext.c | 18 +- + kernel/sched/ext/ext.h | 7 + + kernel/sched/fair.c | 131 ++- + kernel/sched/idle.c | 5 +- + kernel/sched/rt.c | 7 +- + kernel/sched/sched.h | 80 +- + kernel/sched/stop_task.c | 5 +- + kernel/sched/wait.c | 22 +- + kernel/time/hrtimer.c | 23 +- + kernel/time/posix-cpu-timers.c | 103 ++- + kernel/time/posix-timers.c | 26 +- + kernel/time/posix-timers.h | 3 + + kernel/time/sleep_timeout.c | 4 +- + kernel/time/tick-sched.c | 30 +- + kernel/time/time_test.c | 16 + + kernel/time/timeconv.c | 6 +- + kernel/time/timekeeping.c | 4 +- + kernel/time/timer.c | 2 +- + kernel/time/timer_migration.c | 6 +- + lib/bitmap.c | 17 + + lib/crc/x86/crc-pclmul-template.h | 6 +- + lib/crypto/x86/blake2s.h | 4 +- + lib/crypto/x86/chacha.h | 3 +- + lib/crypto/x86/nh.h | 4 +- + lib/crypto/x86/poly1305.h | 7 +- + lib/crypto/x86/sha1.h | 4 +- + lib/crypto/x86/sha256.h | 4 +- + lib/crypto/x86/sha512.h | 3 +- + lib/crypto/x86/sm3.h | 3 +- + lib/raid/xor/Makefile | 2 +- + lib/raid/xor/x86/xor-avx512.c | 122 +++ + lib/raid/xor/x86/xor_arch.h | 31 +- + net/core/pktgen.c | 4 +- + tools/objtool/Documentation/klp-test-design.txt | 286 +++++++ + tools/objtool/Documentation/klp-write-tests.txt | 266 ++++++ + tools/objtool/Makefile | 15 +- + .../tests/generic/fixtures/abs_and_addressable.c | 44 + + tools/objtool/tests/generic/fixtures/basic.c | 20 + + .../objtool/tests/generic/fixtures/changed_data.c | 16 + + .../objtool/tests/generic/fixtures/checksum_data.c | 116 +++ + .../objtool/tests/generic/fixtures/checksum_insn.c | 78 ++ + .../tests/generic/fixtures/checksum_position.c | 42 + + .../objtool/tests/generic/fixtures/checksum_skip.c | 47 ++ + .../objtool/tests/generic/fixtures/cold_function.c | 21 + + .../objtool/tests/generic/fixtures/cross_module.c | 25 + + .../tests/generic/fixtures/data_alignment.c | 29 + + .../tests/generic/fixtures/function_removal.c | 25 + + .../tests/generic/fixtures/init_reference.c | 17 + + tools/objtool/tests/generic/fixtures/jump_label.c | 76 ++ + tools/objtool/tests/generic/fixtures/klp_funcs.c | 31 + + .../tests/generic/fixtures/local_to_global.c | 34 + + tools/objtool/tests/generic/fixtures/new_data.c | 23 + + .../tests/generic/fixtures/new_export_ref.c | 35 + + .../objtool/tests/generic/fixtures/new_function.c | 21 + + tools/objtool/tests/generic/fixtures/no_modinfo.c | 11 + + .../tests/generic/fixtures/special_section.c | 24 + + .../generic/fixtures/special_section_shared.c | 31 + + tools/objtool/tests/generic/fixtures/static_call.c | 59 ++ + .../objtool/tests/generic/fixtures/static_local.c | 17 + + .../generic/fixtures/static_local_uncorrelated.c | 41 + + .../objtool/tests/generic/fixtures/switch_rodata.c | 31 + + .../tests/generic/fixtures/symid_discarded.c | 25 + + tools/objtool/tests/generic/fixtures/sympos_dup.c | 32 + + .../tests/generic/fixtures/sympos_vmlinux.c | 40 + + .../tests/generic/fixtures/thinlto_ambiguity.c | 57 ++ + .../objtool/tests/generic/fixtures/thinlto_local.c | 39 + + tools/objtool/tests/generic/fixtures/ubsan_noise.c | 49 ++ + .../tests/generic/test-abs-and-addressable.sh | 50 ++ + tools/objtool/tests/generic/test-basic.sh | 17 + + tools/objtool/tests/generic/test-changed-data.sh | 18 + + tools/objtool/tests/generic/test-checksum-data.sh | 61 ++ + tools/objtool/tests/generic/test-checksum-debug.sh | 49 ++ + tools/objtool/tests/generic/test-checksum-insn.sh | 49 ++ + .../tests/generic/test-checksum-position.sh | 51 ++ + tools/objtool/tests/generic/test-checksum-skip.sh | 81 ++ + tools/objtool/tests/generic/test-checksum-value.sh | 37 + + tools/objtool/tests/generic/test-cold-function.sh | 39 + + tools/objtool/tests/generic/test-data-alignment.sh | 40 + + .../generic/test-export-symbol-for-modules.sh | 39 + + .../objtool/tests/generic/test-function-removal.sh | 34 + + tools/objtool/tests/generic/test-init-reference.sh | 29 + + .../tests/generic/test-jump-label-exempt-keys.sh | 51 ++ + tools/objtool/tests/generic/test-jump-label-key.sh | 44 + + .../tests/generic/test-jump-label-module-key.sh | 22 + + .../generic/test-jump-label-module-static-key.sh | 45 + + .../tests/generic/test-jump-label-new-key.sh | 51 ++ + .../tests/generic/test-klp-funcs-content.sh | 45 + + .../tests/generic/test-local-to-global-flip.sh | 63 ++ + .../objtool/tests/generic/test-local-vs-export.sh | 32 + + .../objtool/tests/generic/test-missing-checksum.sh | 18 + + .../objtool/tests/generic/test-missing-modinfo.sh | 16 + + .../tests/generic/test-modname-normalize.sh | 26 + + tools/objtool/tests/generic/test-module-object.sh | 31 + + .../tests/generic/test-module-vmlinux-reloc.sh | 40 + + tools/objtool/tests/generic/test-new-data.sh | 27 + + tools/objtool/tests/generic/test-new-export-ref.sh | 46 ++ + tools/objtool/tests/generic/test-new-function.sh | 16 + + tools/objtool/tests/generic/test-post-link.sh | 42 + + .../tests/generic/test-special-section-shared.sh | 26 + + .../objtool/tests/generic/test-special-section.sh | 20 + + .../generic/test-static-call-annotate-stripped.sh | 42 + + .../tests/generic/test-static-call-module-key.sh | 36 + + .../objtool/tests/generic/test-static-call-new.sh | 45 + + .../generic/test-static-local-uncorrelated.sh | 41 + + tools/objtool/tests/generic/test-static-local.sh | 24 + + tools/objtool/tests/generic/test-switch-rodata.sh | 53 ++ + .../objtool/tests/generic/test-symid-discarded.sh | 44 + + tools/objtool/tests/generic/test-sympos-vmlinux.sh | 57 ++ + tools/objtool/tests/generic/test-sympos.sh | 51 ++ + .../tests/generic/test-symvers-parse-error.sh | 23 + + .../tests/generic/test-thinlto-ambiguity.sh | 77 ++ + tools/objtool/tests/generic/test-thinlto-local.sh | 48 ++ + tools/objtool/tests/generic/test-ubsan-noise.sh | 48 ++ + tools/objtool/tests/lib.sh | 911 +++++++++++++++++++++ + tools/objtool/tests/run-tests.sh | 239 ++++++ + tools/objtool/tests/x86/fixtures/alt_annotate.c | 57 ++ + tools/objtool/tests/x86/fixtures/checksum_alt.c | 66 ++ + .../objtool/tests/x86/fixtures/empty_alternative.c | 77 ++ + tools/objtool/tests/x86/fixtures/kcfi.c | 39 + + .../objtool/tests/x86/fixtures/special_sections.c | 77 ++ + .../tests/x86/fixtures/static_call_no_key.c | 32 + + tools/objtool/tests/x86/test-alt-annotation.sh | 38 + + tools/objtool/tests/x86/test-checksum-alt.sh | 45 + + tools/objtool/tests/x86/test-empty-alternative.sh | 31 + + tools/objtool/tests/x86/test-kcfi.sh | 39 + + .../tests/x86/test-manual-klp-static-call.sh | 40 + + tools/objtool/tests/x86/test-special-sections.sh | 42 + + tools/perf/trace/beauty/include/uapi/linux/sched.h | 2 +- + .../testing/selftests/timers/clocksource-switch.c | 23 +- + tools/testing/selftests/timers/posix_timers.c | 7 +- + tools/testing/selftests/timers/raw_skew.c | 12 + + tools/testing/selftests/x86/Makefile | 5 +- + .../selftests/x86/sigframe_fpu_portability.c | 235 ++++++ + tools/testing/selftests/x86/xstate.c | 12 - + tools/testing/selftests/x86/xstate.h | 20 + + 343 files changed, 10162 insertions(+), 2233 deletions(-) + delete mode 100644 Documentation/ABI/testing/sysfs-platform-ts5500 + create mode 100644 Documentation/driver-api/steal-governor.rst + create mode 100644 Documentation/scheduler/sched-paravirt.rst + delete mode 100644 arch/x86/include/asm/math_emu.h + create mode 100644 arch/x86/include/asm/shared/string.h + delete mode 100644 arch/x86/platform/ts5500/Makefile + delete mode 100644 arch/x86/platform/ts5500/ts5500.c + create mode 100644 drivers/virt/steal_governor.c + create mode 100644 lib/raid/xor/x86/xor-avx512.c + create mode 100644 tools/objtool/Documentation/klp-test-design.txt + create mode 100644 tools/objtool/Documentation/klp-write-tests.txt + create mode 100644 tools/objtool/tests/generic/fixtures/abs_and_addressable.c + create mode 100644 tools/objtool/tests/generic/fixtures/basic.c + create mode 100644 tools/objtool/tests/generic/fixtures/changed_data.c + create mode 100644 tools/objtool/tests/generic/fixtures/checksum_data.c + create mode 100644 tools/objtool/tests/generic/fixtures/checksum_insn.c + create mode 100644 tools/objtool/tests/generic/fixtures/checksum_position.c + create mode 100644 tools/objtool/tests/generic/fixtures/checksum_skip.c + create mode 100644 tools/objtool/tests/generic/fixtures/cold_function.c + create mode 100644 tools/objtool/tests/generic/fixtures/cross_module.c + create mode 100644 tools/objtool/tests/generic/fixtures/data_alignment.c + create mode 100644 tools/objtool/tests/generic/fixtures/function_removal.c + create mode 100644 tools/objtool/tests/generic/fixtures/init_reference.c + create mode 100644 tools/objtool/tests/generic/fixtures/jump_label.c + create mode 100644 tools/objtool/tests/generic/fixtures/klp_funcs.c + create mode 100644 tools/objtool/tests/generic/fixtures/local_to_global.c + create mode 100644 tools/objtool/tests/generic/fixtures/new_data.c + create mode 100644 tools/objtool/tests/generic/fixtures/new_export_ref.c + create mode 100644 tools/objtool/tests/generic/fixtures/new_function.c + create mode 100644 tools/objtool/tests/generic/fixtures/no_modinfo.c + create mode 100644 tools/objtool/tests/generic/fixtures/special_section.c + create mode 100644 tools/objtool/tests/generic/fixtures/special_section_shared.c + create mode 100644 tools/objtool/tests/generic/fixtures/static_call.c + create mode 100644 tools/objtool/tests/generic/fixtures/static_local.c + create mode 100644 tools/objtool/tests/generic/fixtures/static_local_uncorrelated.c + create mode 100644 tools/objtool/tests/generic/fixtures/switch_rodata.c + create mode 100644 tools/objtool/tests/generic/fixtures/symid_discarded.c + create mode 100644 tools/objtool/tests/generic/fixtures/sympos_dup.c + create mode 100644 tools/objtool/tests/generic/fixtures/sympos_vmlinux.c + create mode 100644 tools/objtool/tests/generic/fixtures/thinlto_ambiguity.c + create mode 100644 tools/objtool/tests/generic/fixtures/thinlto_local.c + create mode 100644 tools/objtool/tests/generic/fixtures/ubsan_noise.c + create mode 100755 tools/objtool/tests/generic/test-abs-and-addressable.sh + create mode 100755 tools/objtool/tests/generic/test-basic.sh + create mode 100755 tools/objtool/tests/generic/test-changed-data.sh + create mode 100755 tools/objtool/tests/generic/test-checksum-data.sh + create mode 100755 tools/objtool/tests/generic/test-checksum-debug.sh + create mode 100755 tools/objtool/tests/generic/test-checksum-insn.sh + create mode 100755 tools/objtool/tests/generic/test-checksum-position.sh + create mode 100755 tools/objtool/tests/generic/test-checksum-skip.sh + create mode 100755 tools/objtool/tests/generic/test-checksum-value.sh + create mode 100755 tools/objtool/tests/generic/test-cold-function.sh + create mode 100755 tools/objtool/tests/generic/test-data-alignment.sh + create mode 100755 tools/objtool/tests/generic/test-export-symbol-for-modules.sh + create mode 100755 tools/objtool/tests/generic/test-function-removal.sh + create mode 100755 tools/objtool/tests/generic/test-init-reference.sh + create mode 100755 tools/objtool/tests/generic/test-jump-label-exempt-keys.sh + create mode 100755 tools/objtool/tests/generic/test-jump-label-key.sh + create mode 100755 tools/objtool/tests/generic/test-jump-label-module-key.sh + create mode 100755 tools/objtool/tests/generic/test-jump-label-module-static-key.sh + create mode 100755 tools/objtool/tests/generic/test-jump-label-new-key.sh + create mode 100755 tools/objtool/tests/generic/test-klp-funcs-content.sh + create mode 100755 tools/objtool/tests/generic/test-local-to-global-flip.sh + create mode 100755 tools/objtool/tests/generic/test-local-vs-export.sh + create mode 100755 tools/objtool/tests/generic/test-missing-checksum.sh + create mode 100755 tools/objtool/tests/generic/test-missing-modinfo.sh + create mode 100755 tools/objtool/tests/generic/test-modname-normalize.sh + create mode 100755 tools/objtool/tests/generic/test-module-object.sh + create mode 100755 tools/objtool/tests/generic/test-module-vmlinux-reloc.sh + create mode 100755 tools/objtool/tests/generic/test-new-data.sh + create mode 100755 tools/objtool/tests/generic/test-new-export-ref.sh + create mode 100755 tools/objtool/tests/generic/test-new-function.sh + create mode 100755 tools/objtool/tests/generic/test-post-link.sh + create mode 100755 tools/objtool/tests/generic/test-special-section-shared.sh + create mode 100755 tools/objtool/tests/generic/test-special-section.sh + create mode 100755 tools/objtool/tests/generic/test-static-call-annotate-stripped.sh + create mode 100755 tools/objtool/tests/generic/test-static-call-module-key.sh + create mode 100755 tools/objtool/tests/generic/test-static-call-new.sh + create mode 100755 tools/objtool/tests/generic/test-static-local-uncorrelated.sh + create mode 100755 tools/objtool/tests/generic/test-static-local.sh + create mode 100755 tools/objtool/tests/generic/test-switch-rodata.sh + create mode 100755 tools/objtool/tests/generic/test-symid-discarded.sh + create mode 100755 tools/objtool/tests/generic/test-sympos-vmlinux.sh + create mode 100755 tools/objtool/tests/generic/test-sympos.sh + create mode 100755 tools/objtool/tests/generic/test-symvers-parse-error.sh + create mode 100755 tools/objtool/tests/generic/test-thinlto-ambiguity.sh + create mode 100755 tools/objtool/tests/generic/test-thinlto-local.sh + create mode 100755 tools/objtool/tests/generic/test-ubsan-noise.sh + create mode 100644 tools/objtool/tests/lib.sh + create mode 100755 tools/objtool/tests/run-tests.sh + create mode 100644 tools/objtool/tests/x86/fixtures/alt_annotate.c + create mode 100644 tools/objtool/tests/x86/fixtures/checksum_alt.c + create mode 100644 tools/objtool/tests/x86/fixtures/empty_alternative.c + create mode 100644 tools/objtool/tests/x86/fixtures/kcfi.c + create mode 100644 tools/objtool/tests/x86/fixtures/special_sections.c + create mode 100644 tools/objtool/tests/x86/fixtures/static_call_no_key.c + create mode 100755 tools/objtool/tests/x86/test-alt-annotation.sh + create mode 100755 tools/objtool/tests/x86/test-checksum-alt.sh + create mode 100755 tools/objtool/tests/x86/test-empty-alternative.sh + create mode 100755 tools/objtool/tests/x86/test-kcfi.sh + create mode 100755 tools/objtool/tests/x86/test-manual-klp-static-call.sh + create mode 100755 tools/objtool/tests/x86/test-special-sections.sh + create mode 100644 tools/testing/selftests/x86/sigframe_fpu_portability.c +Merging kexec/kexec-next (9d0b028715b00 Merge branch 'kexec-7.4' into kexec-next) +$ git merge -m Merge branch 'kexec-next' of https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git kexec/kexec-next +Auto-merging include/linux/mm.h +Auto-merging mm/memory-failure.c +Merge made by the 'ort' strategy. + arch/riscv/kernel/kexec_elf.c | 2 +- + include/linux/kexec.h | 15 --------------- + include/linux/mm.h | 14 ++++++++++++++ + kernel/kexec_core.c | 10 ++++++++++ + kernel/kexec_file.c | 22 ++++++++++++++++++++-- + mm/memory-failure.c | 40 ++++++++++++++++++++++++++++++++++++++++ + 6 files changed, 85 insertions(+), 18 deletions(-) +Merging liveupdate/next (5221d141653a8 Merge branch 'kho-bootmem' into next) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git liveupdate/next +Auto-merging Documentation/admin-guide/kernel-parameters.txt +Auto-merging MAINTAINERS +Auto-merging drivers/pci/pci.c +Auto-merging drivers/pci/pci.h +Auto-merging drivers/pci/probe.c +Auto-merging drivers/pci/quirks.c +Auto-merging include/linux/memblock.h +Auto-merging include/linux/pci.h +Auto-merging mm/Kconfig +Auto-merging mm/memblock.c +CONFLICT (content): Merge conflict in mm/memblock.c +Auto-merging mm/mm_init.c +CONFLICT (content): Merge conflict in mm/mm_init.c +Resolved 'mm/memblock.c' using previous resolution. +Resolved 'mm/mm_init.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master f9809e5c8da6e] Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git +$ git diff -M --stat --summary HEAD^.. + Documentation/PCI/index.rst | 1 + + Documentation/PCI/liveupdate.rst | 35 + + Documentation/admin-guide/kernel-parameters.txt | 22 +- + Documentation/admin-guide/mm/kho.rst | 18 +- + Documentation/core-api/kho/index.rst | 36 +- + Documentation/core-api/liveupdate.rst | 5 + + MAINTAINERS | 15 + + arch/x86/boot/compressed/kaslr.c | 12 +- + arch/x86/include/uapi/asm/setup_data.h | 4 +- + arch/x86/kernel/e820.c | 11 +- + arch/x86/kernel/kexec-bzimage64.c | 6 +- + arch/x86/kernel/setup.c | 2 +- + arch/x86/realmode/init.c | 2 +- + drivers/firmware/efi/efi-init.c | 6 +- + drivers/of/fdt.c | 8 +- + drivers/of/kexec.c | 16 +- + drivers/pci/Kconfig | 15 + + drivers/pci/Makefile | 1 + + drivers/pci/liveupdate.c | 1015 +++++++++++++++++++++++ + drivers/pci/liveupdate.h | 62 ++ + drivers/pci/pci-driver.c | 9 +- + drivers/pci/pci.c | 79 +- + drivers/pci/pci.h | 5 + + drivers/pci/probe.c | 16 +- + drivers/pci/quirks.c | 58 +- + include/asm-generic/kexec_handover.h | 2 +- + include/linux/kexec.h | 2 +- + include/linux/kexec_handover.h | 16 +- + include/linux/kho/abi/pci.h | 66 ++ + include/linux/liveupdate.h | 22 + + include/linux/memblock.h | 30 +- + include/linux/pci.h | 7 + + include/linux/pci_liveupdate.h | 72 ++ + kernel/kexec_file.c | 4 +- + kernel/liveupdate/Kconfig | 1 - + kernel/liveupdate/kexec_handover.c | 347 ++++---- + kernel/liveupdate/kexec_handover_debugfs.c | 24 +- + kernel/liveupdate/kexec_handover_internal.h | 4 +- + kernel/liveupdate/luo_file.c | 84 ++ + kernel/liveupdate/luo_internal.h | 17 + + mm/Kconfig | 4 - + mm/memblock.c | 56 +- + mm/memfd_luo.c | 3 +- + mm/mm_init.c | 4 +- + tools/testing/memblock/internal.h | 2 +- + 45 files changed, 1871 insertions(+), 355 deletions(-) + create mode 100644 Documentation/PCI/liveupdate.rst + create mode 100644 drivers/pci/liveupdate.c + create mode 100644 drivers/pci/liveupdate.h + create mode 100644 include/linux/kho/abi/pci.h + create mode 100644 include/linux/pci_liveupdate.h +Merging clockevents/timers/drivers/next (1b8b356b4b06e clocksource/drivers/armada: Unwind timer clock on init failure) +$ git merge -m Merge branch 'timers/drivers/next' of https://git.kernel.org/pub/scm/linux/kernel/git/daniel.lezcano/linux.git clockevents/timers/drivers/next +Auto-merging drivers/clocksource/timer-ti-dm.c +Auto-merging drivers/pwm/pwm-samsung.c +Merge made by the 'ort' strategy. +Merging edac/edac-for-next (898e6a2c5ce0e Merge ras/edac-drivers into for-next) +$ git merge -m Merge branch 'edac-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/ras/ras.git edac/edac-for-next +Auto-merging MAINTAINERS +Auto-merging arch/x86/include/asm/msr-index.h +Auto-merging drivers/base/cacheinfo.c +Merge made by the 'ort' strategy. + MAINTAINERS | 26 +++- + arch/x86/include/asm/mce.h | 5 + + arch/x86/include/asm/msr-index.h | 2 + + drivers/base/cacheinfo.c | 17 +++ + drivers/edac/Kconfig | 22 +++ + drivers/edac/Makefile | 5 +- + drivers/edac/altera_edac.c | 67 ++++----- + drivers/edac/bluefield_edac.c | 4 +- + drivers/edac/dummy_edac.c | 191 +++++++++++++++++++++++++ + drivers/edac/i10nm_base.c | 2 + + drivers/edac/ie31200_edac.c | 4 +- + drivers/edac/intel-bff.c | 295 +++++++++++++++++++++++++++++++++++++++ + drivers/edac/loongson_edac.c | 4 +- + drivers/edac/skx_base.c | 2 +- + drivers/edac/versalnet_edac.c | 1 + + include/linux/cacheinfo.h | 12 +- + 16 files changed, 603 insertions(+), 56 deletions(-) + create mode 100644 drivers/edac/dummy_edac.c + create mode 100644 drivers/edac/intel-bff.c +Merging ftrace/for-next (ca3c36c6c4ded Merge probes/for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git ftrace/for-next +Auto-merging lib/bootconfig.c +Auto-merging tools/bootconfig/main.c +Merge made by the 'ort' strategy. + include/linux/bootconfig.h | 7 +- + kernel/kprobes.c | 43 +-- + kernel/trace/trace_probe.c | 2 +- + lib/bootconfig.c | 22 +- + tools/bootconfig/main.c | 2 - + tools/testing/selftests/ftrace/Makefile | 4 +- + tools/testing/selftests/ftrace/boottime-ktap | 6 + + tools/testing/selftests/ftrace/boottime/Makefile | 10 + + tools/testing/selftests/ftrace/boottime/README | 74 ++++ + .../ftrace/boottime/bootconfigs/01-kprobe.bconf | 4 + + .../ftrace/boottime/bootconfigs/02-synth.bconf | 4 + + .../ftrace/boottime/bootconfigs/03-eprobe.bconf | 4 + + .../ftrace/boottime/bootconfigs/04-fprobe.bconf | 4 + + .../ftrace/boottime/bootconfigs/05-tprobe.bconf | 4 + + .../ftrace/boottime/bootconfigs/06-instance.bconf | 5 + + .../boottime/cmdlines/cmdline-01-ftrace.cmdline | 1 + + .../cmdlines/cmdline-02-trace-event.cmdline | 1 + + .../cmdlines/cmdline-03-trace-buf-size.cmdline | 1 + + .../cmdlines/cmdline-04-trace-options.cmdline | 1 + + .../cmdlines/cmdline-05-trace-clock.cmdline | 1 + + .../cmdlines/cmdline-06-trace-instance.cmdline | 1 + + .../cmdlines/persistent-01-reserve-mem.cmdline | 1 + + .../cmdlines/persistent-02-backup-instance.cmdline | 1 + + .../selftests/ftrace/boottime/run_boottime_test.sh | 403 +++++++++++++++++++++ + .../selftests/ftrace/boottime/tests/01-kprobe.sh | 25 ++ + .../selftests/ftrace/boottime/tests/02-synth.sh | 25 ++ + .../selftests/ftrace/boottime/tests/03-eprobe.sh | 25 ++ + .../selftests/ftrace/boottime/tests/04-fprobe.sh | 25 ++ + .../selftests/ftrace/boottime/tests/05-tprobe.sh | 25 ++ + .../selftests/ftrace/boottime/tests/06-instance.sh | 30 ++ + .../ftrace/boottime/tests/cmdline-01-ftrace.sh | 19 + + .../boottime/tests/cmdline-02-trace-event.sh | 26 ++ + .../boottime/tests/cmdline-03-trace-buf-size.sh | 29 ++ + .../boottime/tests/cmdline-04-trace-options.sh | 23 ++ + .../boottime/tests/cmdline-05-trace-clock.sh | 19 + + .../boottime/tests/cmdline-06-trace-instance.sh | 24 ++ + .../boottime/tests/persistent-01-reserve-mem.sh | 28 ++ + .../tests/persistent-02-backup-instance.sh | 40 ++ + tools/testing/selftests/ftrace/config | 6 + + 39 files changed, 917 insertions(+), 58 deletions(-) + create mode 100755 tools/testing/selftests/ftrace/boottime-ktap + create mode 100644 tools/testing/selftests/ftrace/boottime/Makefile + create mode 100644 tools/testing/selftests/ftrace/boottime/README + create mode 100644 tools/testing/selftests/ftrace/boottime/bootconfigs/01-kprobe.bconf + create mode 100644 tools/testing/selftests/ftrace/boottime/bootconfigs/02-synth.bconf + create mode 100644 tools/testing/selftests/ftrace/boottime/bootconfigs/03-eprobe.bconf + create mode 100644 tools/testing/selftests/ftrace/boottime/bootconfigs/04-fprobe.bconf + create mode 100644 tools/testing/selftests/ftrace/boottime/bootconfigs/05-tprobe.bconf + create mode 100644 tools/testing/selftests/ftrace/boottime/bootconfigs/06-instance.bconf + create mode 100644 tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-01-ftrace.cmdline + create mode 100644 tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-02-trace-event.cmdline + create mode 100644 tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-03-trace-buf-size.cmdline + create mode 100644 tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-04-trace-options.cmdline + create mode 100644 tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-05-trace-clock.cmdline + create mode 100644 tools/testing/selftests/ftrace/boottime/cmdlines/cmdline-06-trace-instance.cmdline + create mode 100644 tools/testing/selftests/ftrace/boottime/cmdlines/persistent-01-reserve-mem.cmdline + create mode 100644 tools/testing/selftests/ftrace/boottime/cmdlines/persistent-02-backup-instance.cmdline + create mode 100755 tools/testing/selftests/ftrace/boottime/run_boottime_test.sh + create mode 100644 tools/testing/selftests/ftrace/boottime/tests/01-kprobe.sh + create mode 100644 tools/testing/selftests/ftrace/boottime/tests/02-synth.sh + create mode 100644 tools/testing/selftests/ftrace/boottime/tests/03-eprobe.sh + create mode 100644 tools/testing/selftests/ftrace/boottime/tests/04-fprobe.sh + create mode 100644 tools/testing/selftests/ftrace/boottime/tests/05-tprobe.sh + create mode 100644 tools/testing/selftests/ftrace/boottime/tests/06-instance.sh + create mode 100644 tools/testing/selftests/ftrace/boottime/tests/cmdline-01-ftrace.sh + create mode 100644 tools/testing/selftests/ftrace/boottime/tests/cmdline-02-trace-event.sh + create mode 100644 tools/testing/selftests/ftrace/boottime/tests/cmdline-03-trace-buf-size.sh + create mode 100644 tools/testing/selftests/ftrace/boottime/tests/cmdline-04-trace-options.sh + create mode 100644 tools/testing/selftests/ftrace/boottime/tests/cmdline-05-trace-clock.sh + create mode 100644 tools/testing/selftests/ftrace/boottime/tests/cmdline-06-trace-instance.sh + create mode 100644 tools/testing/selftests/ftrace/boottime/tests/persistent-01-reserve-mem.sh + create mode 100644 tools/testing/selftests/ftrace/boottime/tests/persistent-02-backup-instance.sh +$ git am -3 ../patches/0001-ftrace-Fix-semantic-conflict-with-mm-tree.patch +Applying: ftrace: Fix semantic conflict with mm tree +Using index info to reconstruct a base tree... +M kernel/trace/trace_printk.c +Falling back to patching base and 3-way merge... +Auto-merging kernel/trace/trace_printk.c +No changes -- Patch already applied. +Merging rcu/next (5826bf4c62811 Merge branches 'misc.2026.10.01a' and 'torture.2026.10.01a' into HEAD) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/rcu/linux rcu/next +Auto-merging Documentation/admin-guide/kernel-parameters.txt +Auto-merging include/linux/compiler.h +Merge made by the 'ort' strategy. + Documentation/RCU/stallwarn.rst | 22 +- + Documentation/admin-guide/kernel-parameters.txt | 23 +- + include/linux/compiler.h | 17 + + include/linux/rcupdate.h | 10 +- + include/linux/rcuref.h | 2 +- + include/linux/srcu.h | 83 +++- + include/linux/srcutiny.h | 18 +- + include/linux/srcutree.h | 19 +- + kernel/rcu/Kconfig | 6 + + kernel/rcu/rcu.h | 14 + + kernel/rcu/rcutorture.c | 317 ++++++++++++- + kernel/rcu/srcutiny.c | 189 +++++++- + kernel/rcu/srcutree.c | 504 ++++++++++++++++++--- + kernel/rcu/tasks.h | 15 +- + kernel/rcu/tiny.c | 127 +++++- + kernel/rcu/tree.c | 164 ++++++- + kernel/rcu/tree.h | 6 + + kernel/rcu/tree_exp.h | 2 +- + kernel/rcu/tree_nocb.h | 3 +- + kernel/rcu/tree_plugin.h | 2 - + kernel/rcu/tree_stall.h | 36 +- + .../testing/selftests/bpf/prog_tests/rcu_reentry.c | 93 ++++ + tools/testing/selftests/bpf/progs/rcu_reentry.c | 51 +++ + .../testing/selftests/rcutorture/bin/kvm-again.sh | 11 +- + .../testing/selftests/rcutorture/bin/kvm-remote.sh | 22 +- + .../selftests/rcutorture/bin/srcu_lockdep.sh | 77 +++- + tools/testing/selftests/rcutorture/bin/torture.sh | 24 + + 27 files changed, 1677 insertions(+), 180 deletions(-) + create mode 100644 tools/testing/selftests/bpf/prog_tests/rcu_reentry.c + create mode 100644 tools/testing/selftests/bpf/progs/rcu_reentry.c +Merging paulmck/non-rcu/next (afb36d3025c84 Merge branches 'csd-lock.2026.09.03a', 'hazptr.2026.09.30a', 'nmi.2026.09.17a' and 'usb-mtu3.2026.09.29a' into HEAD) +$ git merge -m Merge branch 'non-rcu/next' of https://git.kernel.org/pub/scm/linux/kernel/git/paulmck/linux-rcu.git paulmck/non-rcu/next +Auto-merging Documentation/admin-guide/kernel-parameters.txt +Auto-merging init/main.c +Auto-merging kernel/Makefile +Auto-merging kernel/sched/core.c +Auto-merging lib/Kconfig.debug +Auto-merging tools/testing/selftests/rcutorture/bin/torture.sh +Merge made by the 'ort' strategy. + Documentation/admin-guide/kernel-parameters.txt | 96 +++ + arch/x86/kernel/nmi.c | 21 +- + drivers/usb/mtu3/mtu3_trace.h | 4 +- + include/linux/hazptr.h | 325 +++++++ + include/linux/torture.h | 6 +- + init/main.c | 2 + + kernel/Makefile | 2 +- + kernel/hazptr.c | 284 +++++++ + kernel/rcu/Kconfig.debug | 26 + + kernel/rcu/Makefile | 1 + + kernel/rcu/hazptrtorture.c | 946 +++++++++++++++++++++ + kernel/rcu/refscale.c | 43 + + kernel/rcu/update.c | 3 +- + kernel/sched/core.c | 2 + + kernel/smp.c | 77 +- + kernel/torture.c | 38 +- + lib/Kconfig.debug | 12 + + lib/Makefile | 1 + + lib/test_csd_lock.c | 173 ++++ + tools/testing/selftests/rcutorture/bin/kvm.sh | 9 +- + tools/testing/selftests/rcutorture/bin/torture.sh | 26 +- + .../selftests/rcutorture/configs/hazptr/CFLIST | 2 + + .../selftests/rcutorture/configs/hazptr/CFcommon | 2 + + .../selftests/rcutorture/configs/hazptr/NOPREEMPT | 19 + + .../rcutorture/configs/hazptr/NOPREEMPT.boot | 1 + + .../selftests/rcutorture/configs/hazptr/PREEMPT | 16 + + .../rcutorture/configs/hazptr/ver_functions.sh | 40 + + 27 files changed, 2139 insertions(+), 38 deletions(-) + create mode 100644 include/linux/hazptr.h + create mode 100644 kernel/hazptr.c + create mode 100644 kernel/rcu/hazptrtorture.c + create mode 100644 lib/test_csd_lock.c + create mode 100644 tools/testing/selftests/rcutorture/configs/hazptr/CFLIST + create mode 100644 tools/testing/selftests/rcutorture/configs/hazptr/CFcommon + create mode 100644 tools/testing/selftests/rcutorture/configs/hazptr/NOPREEMPT + create mode 100644 tools/testing/selftests/rcutorture/configs/hazptr/NOPREEMPT.boot + create mode 100644 tools/testing/selftests/rcutorture/configs/hazptr/PREEMPT + create mode 100644 tools/testing/selftests/rcutorture/configs/hazptr/ver_functions.sh +Merging kvm/next (d4b7fb647204f Merge branch 'kvm-xen-longmode' into HEAD) +$ git merge -m Merge branch 'next' of git://git.kernel.org/pub/scm/virt/kvm/kvm.git kvm/next +Auto-merging arch/x86/include/asm/kvm_host.h +Auto-merging include/linux/kvm_host.h +Auto-merging virt/kvm/kvm_main.c +Merge made by the 'ort' strategy. + arch/x86/include/asm/kvm_host.h | 3 +- + arch/x86/kvm/xen.c | 212 ++++++++++++++++++++++++---------------- + arch/x86/kvm/xen.h | 5 + + include/linux/kvm_host.h | 2 + + virt/kvm/kvm_main.c | 10 ++ + virt/kvm/pfncache.c | 18 ++-- + 6 files changed, 157 insertions(+), 93 deletions(-) +$ git am -3 ../patches/kvm-x86-static-cpu-has +Applying: Signed-off-by: Mark Brown +Using index info to reconstruct a base tree... +M arch/x86/kvm/msrs.c +Falling back to patching base and 3-way merge... +Auto-merging arch/x86/kvm/msrs.c +No changes -- Patch already applied. +Merging kvm-arm/next (73ed60746b21c Merge branch kvm-arm64/pkvm-allocator-7.4 into kvmarm-master/next) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/kvmarm/kvmarm.git kvm-arm/next +Auto-merging arch/arm64/Makefile +Auto-merging arch/arm64/include/asm/ptrace.h +CONFLICT (content): Merge conflict in arch/arm64/include/asm/ptrace.h +Auto-merging arch/arm64/include/asm/sysreg.h +CONFLICT (content): Merge conflict in arch/arm64/include/asm/sysreg.h +Auto-merging arch/arm64/kernel/pi/Makefile +Auto-merging arch/arm64/kvm/arm.c +Auto-merging arch/arm64/kvm/hyp/include/nvhe/pkvm.h +Auto-merging arch/arm64/kvm/hyp/include/nvhe/spinlock.h +Auto-merging arch/arm64/kvm/hyp/nvhe/hyp-main.c +CONFLICT (content): Merge conflict in arch/arm64/kvm/hyp/nvhe/hyp-main.c +Auto-merging arch/arm64/kvm/hyp/nvhe/pkvm.c +Auto-merging arch/arm64/kvm/mmu.c +Auto-merging arch/arm64/kvm/sys_regs.c +Auto-merging arch/arm64/kvm/vgic/vgic-init.c +Auto-merging arch/arm64/kvm/vgic/vgic-its.c +Auto-merging arch/arm64/kvm/vgic/vgic.c +Auto-merging arch/s390/kvm/s390/s390.c +Auto-merging arch/x86/kvm/mmu/mmu.c +Auto-merging arch/x86/kvm/vmx/tdx.c +Auto-merging drivers/irqchip/irq-gic-v5.c +Auto-merging include/linux/kvm_host.h +Auto-merging tools/testing/selftests/kvm/Makefile.kvm +Auto-merging virt/kvm/kvm_main.c +Resolved 'arch/arm64/include/asm/ptrace.h' using previous resolution. +Resolved 'arch/arm64/include/asm/sysreg.h' using previous resolution. +Recorded preimage for 'arch/arm64/kvm/hyp/nvhe/hyp-main.c' +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +Recorded resolution for 'arch/arm64/kvm/hyp/nvhe/hyp-main.c'. +[master 9aa9d5eaf78bc] Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/kvmarm/kvmarm.git +$ git diff -M --stat --summary HEAD^.. + Documentation/virt/kvm/api.rst | 67 +- + Documentation/virt/kvm/arm/fw-pseudo-registers.rst | 2 + + Documentation/virt/kvm/arm/pkvm.rst | 164 +- + Documentation/virt/kvm/devices/arm-vgic-v5.rst | 271 ++- + Documentation/virt/kvm/devices/vcpu.rst | 5 +- + arch/arm64/Makefile | 3 + + arch/arm64/include/asm/brk-imm.h | 1 + + arch/arm64/include/asm/esr.h | 34 +- + arch/arm64/include/asm/kvm_arm.h | 6 +- + arch/arm64/include/asm/kvm_asm.h | 7 + + arch/arm64/include/asm/kvm_emulate.h | 124 +- + arch/arm64/include/asm/kvm_hcall.h | 258 +++ + arch/arm64/include/asm/kvm_host.h | 238 +-- + arch/arm64/include/asm/kvm_hyp.h | 7 + + arch/arm64/include/asm/kvm_mmu.h | 19 +- + arch/arm64/include/asm/kvm_nested.h | 7 + + arch/arm64/include/asm/kvm_pgtable.h | 7 +- + arch/arm64/include/asm/kvm_pkvm.h | 146 +- + arch/arm64/include/asm/ptrace.h | 64 +- + arch/arm64/include/asm/stacktrace/nvhe.h | 10 +- + arch/arm64/include/asm/sysreg.h | 33 +- + arch/arm64/include/uapi/asm/kvm.h | 15 + + arch/arm64/kernel/pi/Makefile | 1 + + arch/arm64/kernel/vmlinux.lds.S | 2 +- + arch/arm64/kvm/Kconfig | 1 + + arch/arm64/kvm/Makefile | 3 +- + arch/arm64/kvm/arm.c | 96 +- + arch/arm64/kvm/guest.c | 33 +- + arch/arm64/kvm/hyp/entry.S | 2 +- + arch/arm64/kvm/hyp/exception.c | 30 +- + arch/arm64/kvm/hyp/hyp-constants.c | 2 - + arch/arm64/kvm/hyp/include/hyp/adjust_pc.h | 52 +- + arch/arm64/kvm/hyp/include/hyp/switch.h | 4 +- + arch/arm64/kvm/hyp/include/nvhe/alloc.h | 28 + + arch/arm64/kvm/hyp/include/nvhe/mm.h | 3 + + arch/arm64/kvm/hyp/include/nvhe/pkvm.h | 40 +- + arch/arm64/kvm/hyp/include/nvhe/spinlock.h | 4 + + arch/arm64/kvm/hyp/include/nvhe/trace.h | 5 +- + arch/arm64/kvm/hyp/nvhe/Makefile | 3 +- + arch/arm64/kvm/hyp/nvhe/alloc.c | 1211 +++++++++++++ + arch/arm64/kvm/hyp/nvhe/events.c | 2 + + arch/arm64/kvm/hyp/nvhe/ffa.c | 220 ++- + arch/arm64/kvm/hyp/nvhe/hyp-main.c | 1171 +++++++++--- + arch/arm64/kvm/hyp/nvhe/hyp.lds.S | 1 - + arch/arm64/kvm/hyp/nvhe/mem_protect.c | 20 +- + arch/arm64/kvm/hyp/nvhe/mm.c | 79 +- + arch/arm64/kvm/hyp/nvhe/pkvm.c | 487 ++++- + arch/arm64/kvm/hyp/nvhe/setup.c | 6 + + arch/arm64/kvm/hyp/nvhe/stacktrace.c | 3 +- + arch/arm64/kvm/hyp/nvhe/switch.c | 29 +- + arch/arm64/kvm/hyp/nvhe/sys_regs.c | 63 + + arch/arm64/kvm/hyp/nvhe/trace.c | 76 +- + arch/arm64/kvm/hyp/pgtable.c | 5 +- + arch/arm64/kvm/hyp/vgic-v5-sr.c | 45 + + arch/arm64/kvm/hyp_trace.c | 15 +- + arch/arm64/kvm/hypercalls.c | 11 + + arch/arm64/kvm/mmio.c | 2 + + arch/arm64/kvm/mmu.c | 555 ++++-- + arch/arm64/kvm/nested.c | 151 +- + arch/arm64/kvm/pauth.c | 2 +- + arch/arm64/kvm/pkvm.c | 220 ++- + arch/arm64/kvm/psci.c | 11 +- + arch/arm64/kvm/ptdump.c | 79 +- + arch/arm64/kvm/pvtime.c | 10 +- + arch/arm64/kvm/stacktrace.c | 3 - + arch/arm64/kvm/sys_regs.c | 63 +- + arch/arm64/kvm/sys_regs.h | 2 +- + arch/arm64/kvm/trace_pkvm.h | 45 + + arch/arm64/kvm/vgic-sys-reg-v5.c | 519 ++++++ + arch/arm64/kvm/vgic/vgic-init.c | 160 +- + arch/arm64/kvm/vgic/vgic-irqfd.c | 18 +- + arch/arm64/kvm/vgic/vgic-irs-v5.c | 1217 +++++++++++++ + arch/arm64/kvm/vgic/vgic-its.c | 6 +- + arch/arm64/kvm/vgic/vgic-kvm-device.c | 272 ++- + arch/arm64/kvm/vgic/vgic-mmio.c | 6 + + arch/arm64/kvm/vgic/vgic-mmio.h | 2 + + arch/arm64/kvm/vgic/vgic-v5-tables.c | 1872 ++++++++++++++++++++ + arch/arm64/kvm/vgic/vgic-v5-tables.h | 140 ++ + arch/arm64/kvm/vgic/vgic-v5.c | 1152 +++++++++++- + arch/arm64/kvm/vgic/vgic.c | 39 +- + arch/arm64/kvm/vgic/vgic.h | 21 + + arch/arm64/tools/sysreg | 51 + + arch/s390/kvm/s390/s390.c | 11 +- + arch/x86/kvm/mmu/mmu.c | 11 +- + arch/x86/kvm/vmx/tdx.c | 2 +- + drivers/irqchip/irq-gic-v5-irs.c | 19 +- + drivers/irqchip/irq-gic-v5.c | 111 +- + include/kvm/arm_vgic.h | 213 ++- + include/linux/irqchip/arm-gic-v5.h | 254 ++- + include/linux/irqchip/arm-vgic-info.h | 5 + + include/linux/kvm_host.h | 1 + + include/linux/trace_remote_event.h | 2 + + tools/arch/arm64/include/uapi/asm/kvm.h | 15 + + tools/testing/selftests/kvm/Makefile.kvm | 2 + + .../testing/selftests/kvm/arm64/debug-exceptions.c | 2 +- + .../testing/selftests/kvm/arm64/external_aborts.c | 3 +- + tools/testing/selftests/kvm/arm64/hypercalls.c | 31 +- + .../selftests/kvm/arm64/nv_pre_fault_memory_test.c | 158 ++ + tools/testing/selftests/kvm/arm64/sea_to_user.c | 75 +- + tools/testing/selftests/kvm/arm64/set_id_regs.c | 106 +- + tools/testing/selftests/kvm/arm64/vgic_v5.c | 1849 ++++++++++++++++++- + tools/testing/selftests/kvm/include/arm64/gic_v5.h | 105 ++ + tools/testing/selftests/kvm/include/test_util.h | 1 + + tools/testing/selftests/kvm/lib/test_util.c | 15 + + .../testing/selftests/kvm/pre_fault_memory_test.c | 150 +- + virt/kvm/kvm_main.c | 10 +- + 106 files changed, 13917 insertions(+), 1093 deletions(-) + create mode 100644 arch/arm64/include/asm/kvm_hcall.h + create mode 100644 arch/arm64/kvm/hyp/include/nvhe/alloc.h + create mode 100644 arch/arm64/kvm/hyp/nvhe/alloc.c + create mode 100644 arch/arm64/kvm/trace_pkvm.h + create mode 100644 arch/arm64/kvm/vgic-sys-reg-v5.c + create mode 100644 arch/arm64/kvm/vgic/vgic-irs-v5.c + create mode 100644 arch/arm64/kvm/vgic/vgic-v5-tables.c + create mode 100644 arch/arm64/kvm/vgic/vgic-v5-tables.h + create mode 100644 tools/testing/selftests/kvm/arm64/nv_pre_fault_memory_test.c +Merging kvms390/next (044ae0767d8cc KVM: s390: Kick PV cpus at the right time for service irqs) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/kvms390/linux.git kvms390/next +Auto-merging arch/s390/kernel/uv.c +Auto-merging arch/s390/kvm/s390/interrupt.c +Auto-merging arch/s390/kvm/s390/s390.c +Auto-merging arch/s390/kvm/s390/s390.h +Auto-merging tools/testing/selftests/kvm/Makefile.kvm +Merge made by the 'ort' strategy. + Documentation/virt/kvm/devices/vm.rst | 18 ++++ + arch/s390/kernel/uv.c | 41 +++---- + arch/s390/kvm/s390/intercept.c | 32 +++--- + arch/s390/kvm/s390/interrupt.c | 130 ++++++++++++++++++----- + arch/s390/kvm/s390/priv.c | 19 ++-- + arch/s390/kvm/s390/s390.c | 12 +-- + arch/s390/kvm/s390/s390.h | 1 + + tools/testing/selftests/kvm/Makefile.kvm | 1 + + tools/testing/selftests/kvm/s390/irq_injection.c | 74 +++++++++++++ + 9 files changed, 257 insertions(+), 71 deletions(-) + create mode 100644 tools/testing/selftests/kvm/s390/irq_injection.c +Merging kvm-ppc/topic/ppc-kvm (93f51579e7df2 Linux 7.3-rc4) +$ git merge -m Merge branch 'topic/ppc-kvm' of https://git.kernel.org/pub/scm/linux/kernel/git/powerpc/linux.git kvm-ppc/topic/ppc-kvm +Already up to date. +Merging kvm-riscv/riscv_kvm_next (41e81f7e3ef96 RISC-V: KVM: Fix HSM hart status error propagation) +$ git merge -m Merge branch 'riscv_kvm_next' of https://github.com/kvm-riscv/linux.git kvm-riscv/riscv_kvm_next +Already up to date. +Merging kvm-x86/next (6bd2905303c58 Merge branch 'vmx') +$ git merge -m Merge branch 'next' of https://github.com/kvm-x86/linux.git kvm-x86/next +Auto-merging Documentation/admin-guide/kernel-parameters.txt +Auto-merging Documentation/virt/kvm/api.rst +Auto-merging arch/arm64/kvm/mmu.c +CONFLICT (content): Merge conflict in arch/arm64/kvm/mmu.c +Auto-merging arch/arm64/kvm/nested.c +Auto-merging arch/x86/include/asm/cpufeatures.h +Auto-merging arch/x86/include/asm/kvm-x86-ops.h +Auto-merging arch/x86/include/asm/kvm_host.h +Auto-merging arch/x86/kernel/cpu/scattered.c +Auto-merging arch/x86/kvm/mmu/mmu.c +Auto-merging arch/x86/kvm/svm/sev.c +Auto-merging arch/x86/kvm/vmx/tdx.c +Auto-merging arch/x86/kvm/vmx/vmx.c +Auto-merging arch/x86/kvm/vmx/vmx.h +Auto-merging include/linux/kvm_host.h +Auto-merging tools/testing/selftests/kvm/Makefile.kvm +Auto-merging tools/testing/selftests/kvm/arm64/hypercalls.c +Auto-merging tools/testing/selftests/kvm/include/test_util.h +Auto-merging tools/testing/selftests/kvm/lib/test_util.c +Auto-merging virt/kvm/guest_memfd.c +Auto-merging virt/kvm/kvm_main.c +Resolved 'arch/arm64/kvm/mmu.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 76c8ecb752ea5] Merge branch 'next' of https://github.com/kvm-x86/linux.git +$ git diff -M --stat --summary HEAD^.. + Documentation/admin-guide/kernel-parameters.txt | 26 + + Documentation/virt/kvm/api.rst | 141 ++++- + .../virt/kvm/x86/amd-memory-encryption.rst | 22 +- + Documentation/virt/kvm/x86/hypercalls.rst | 2 +- + Documentation/virt/kvm/x86/intel-tdx.rst | 4 + + arch/arm64/kvm/mmu.c | 3 +- + arch/arm64/kvm/nested.c | 4 +- + arch/x86/include/asm/cpufeatures.h | 1 + + arch/x86/include/asm/kvm-x86-ops.h | 3 +- + arch/x86/include/asm/kvm_host.h | 32 +- + arch/x86/include/asm/svm.h | 18 +- + arch/x86/include/asm/vmx.h | 8 + + arch/x86/include/uapi/asm/svm.h | 2 + + arch/x86/kernel/cpu/scattered.c | 1 + + arch/x86/kvm/Kconfig | 15 +- + arch/x86/kvm/cpuid.c | 37 +- + arch/x86/kvm/cpuid.h | 1 - + arch/x86/kvm/hyperv.c | 9 +- + arch/x86/kvm/kvm_emulate.h | 4 +- + arch/x86/kvm/lapic.c | 4 +- + arch/x86/kvm/lapic.h | 2 +- + arch/x86/kvm/mmu/mmu.c | 183 ++++-- + arch/x86/kvm/mmu/paging_tmpl.h | 21 +- + arch/x86/kvm/msrs.c | 15 +- + arch/x86/kvm/pmu.c | 14 +- + arch/x86/kvm/regs.c | 16 +- + arch/x86/kvm/regs.h | 11 +- + arch/x86/kvm/reverse_cpuid.h | 2 +- + arch/x86/kvm/svm/avic.c | 22 +- + arch/x86/kvm/svm/nested.c | 47 +- + arch/x86/kvm/svm/sev.c | 112 ++-- + arch/x86/kvm/svm/svm.c | 199 +++++- + arch/x86/kvm/svm/svm.h | 14 +- + arch/x86/kvm/vmx/common.h | 90 ++- + arch/x86/kvm/vmx/main.c | 153 ++++- + arch/x86/kvm/vmx/nested.c | 60 +- + arch/x86/kvm/vmx/posted_intr.c | 24 +- + arch/x86/kvm/vmx/posted_intr.h | 8 +- + arch/x86/kvm/vmx/tdx.c | 141 +++-- + arch/x86/kvm/vmx/vmx.c | 309 +++------ + arch/x86/kvm/vmx/vmx.h | 46 +- + arch/x86/kvm/vmx/x86_ops.h | 6 +- + arch/x86/kvm/x86.c | 703 ++++++++++++--------- + arch/x86/kvm/x86.h | 17 +- + include/linux/kvm_host.h | 89 +-- + include/trace/events/kvm.h | 6 +- + include/uapi/linux/kvm.h | 16 + + tools/testing/selftests/kvm/Makefile.kvm | 4 +- + tools/testing/selftests/kvm/arm64/hypercalls.c | 2 +- + .../testing/selftests/kvm/arm64/page_fault_test.c | 18 +- + .../testing/selftests/kvm/arm64/vgic_lpi_stress.c | 21 +- + tools/testing/selftests/kvm/demand_paging_test.c | 3 +- + tools/testing/selftests/kvm/guest_memfd_test.c | 2 +- + tools/testing/selftests/kvm/include/kvm_util.h | 245 ++++++- + tools/testing/selftests/kvm/include/numaif.h | 43 ++ + tools/testing/selftests/kvm/include/test_util.h | 37 +- + tools/testing/selftests/kvm/include/x86/evmcs.h | 78 +-- + .../testing/selftests/kvm/include/x86/processor.h | 27 +- + tools/testing/selftests/kvm/include/x86/smm.h | 2 +- + tools/testing/selftests/kvm/include/x86/vmx.h | 186 +++--- + tools/testing/selftests/kvm/lib/arm64/processor.c | 4 +- + tools/testing/selftests/kvm/lib/elf.c | 48 +- + tools/testing/selftests/kvm/lib/guest_sprintf.c | 12 - + tools/testing/selftests/kvm/lib/io.c | 157 ----- + tools/testing/selftests/kvm/lib/kvm_util.c | 350 ++++++---- + .../selftests/kvm/lib/loongarch/processor.c | 5 +- + tools/testing/selftests/kvm/lib/riscv/processor.c | 4 +- + tools/testing/selftests/kvm/lib/s390/processor.c | 7 +- + tools/testing/selftests/kvm/lib/test_util.c | 7 - + tools/testing/selftests/kvm/lib/userfaultfd_util.c | 6 +- + tools/testing/selftests/kvm/lib/x86/memstress.c | 8 +- + tools/testing/selftests/kvm/lib/x86/processor.c | 23 +- + tools/testing/selftests/kvm/lib/x86/vmx.c | 159 +++-- + tools/testing/selftests/kvm/memslot_perf_test.c | 5 +- + tools/testing/selftests/kvm/s390/cmma_test.c | 19 +- + tools/testing/selftests/kvm/s390/irq_routing.c | 2 +- + .../testing/selftests/kvm/set_memory_region_test.c | 8 +- + tools/testing/selftests/kvm/steal_time.c | 20 +- + tools/testing/selftests/kvm/x86/amx_test.c | 2 +- + tools/testing/selftests/kvm/x86/aperfmperf_test.c | 10 +- + tools/testing/selftests/kvm/x86/cpuid_test.c | 2 +- + .../selftests/kvm/x86/evmcs_smm_controls_test.c | 6 +- + .../testing/selftests/kvm/x86/feature_msrs_test.c | 2 +- + .../testing/selftests/kvm/x86/fix_hypercall_test.c | 2 +- + .../kvm/x86/guest_memfd_conversions_test.c | 512 +++++++++++++++ + tools/testing/selftests/kvm/x86/hyperv_clock.c | 2 +- + tools/testing/selftests/kvm/x86/hyperv_evmcs.c | 87 +-- + tools/testing/selftests/kvm/x86/hyperv_svm_test.c | 2 +- + tools/testing/selftests/kvm/x86/kvm_buslock_test.c | 8 +- + .../selftests/kvm/x86/nested_close_kvm_test.c | 6 +- + .../selftests/kvm/x86/nested_dirty_log_test.c | 8 +- + .../selftests/kvm/x86/nested_emulation_test.c | 23 +- + .../selftests/kvm/x86/nested_exceptions_test.c | 33 +- + .../selftests/kvm/x86/nested_invalid_cr3_test.c | 14 +- + .../selftests/kvm/x86/nested_tdp_fault_test.c | 16 +- + .../selftests/kvm/x86/nested_tsc_adjust_test.c | 10 +- + .../selftests/kvm/x86/nested_tsc_scaling_test.c | 244 ++++--- + .../testing/selftests/kvm/x86/nested_x2apic_test.c | 40 +- + .../testing/selftests/kvm/x86/nx_huge_pages_test.c | 3 +- + .../kvm/x86/private_mem_conversions_test.c | 66 +- + .../selftests/kvm/x86/private_mem_kvm_exits_test.c | 36 +- + .../kvm/x86/save_restore_pf_stress_test.c | 14 +- + tools/testing/selftests/kvm/x86/set_boot_cpu_id.c | 2 +- + tools/testing/selftests/kvm/x86/set_sregs_test.c | 11 + + tools/testing/selftests/kvm/x86/sev_dbg_test.c | 6 +- + tools/testing/selftests/kvm/x86/sev_smoke_test.c | 3 +- + .../kvm/x86/smaller_maxphyaddr_emulation_test.c | 9 +- + tools/testing/selftests/kvm/x86/smm_test.c | 4 +- + tools/testing/selftests/kvm/x86/state_test.c | 74 +-- + .../selftests/kvm/x86/svm_nested_clear_efer_svme.c | 50 -- + .../selftests/kvm/x86/svm_nested_efer_test.c | 266 ++++++++ + .../selftests/kvm/x86/triple_fault_event_test.c | 8 +- + tools/testing/selftests/kvm/x86/tsc_msrs_test.c | 2 +- + .../selftests/kvm/x86/vmx_apic_access_test.c | 22 +- + .../selftests/kvm/x86/vmx_apicv_updates_test.c | 16 +- + .../x86/vmx_exception_with_invalid_guest_state.c | 2 +- + .../kvm/x86/vmx_invalid_nested_guest_state.c | 12 +- + .../selftests/kvm/x86/vmx_nested_la57_state_test.c | 12 +- + .../selftests/kvm/x86/vmx_preemption_timer_test.c | 28 +- + tools/testing/selftests/kvm/x86/xapic_ipi_test.c | 78 +-- + virt/kvm/Kconfig | 3 - + virt/kvm/guest_memfd.c | 649 ++++++++++++++++--- + virt/kvm/guest_memfd.h | 19 +- + virt/kvm/kvm_main.c | 159 +++-- + 124 files changed, 4515 insertions(+), 2243 deletions(-) + delete mode 100644 tools/testing/selftests/kvm/lib/io.c + create mode 100644 tools/testing/selftests/kvm/x86/guest_memfd_conversions_test.c + delete mode 100644 tools/testing/selftests/kvm/x86/svm_nested_clear_efer_svme.c + create mode 100644 tools/testing/selftests/kvm/x86/svm_nested_efer_test.c +$ git am -3 ../patches/0001-KVM-selftests-Fix-up-semantic-changes.patch +Applying: KVM: selftests: Fix up semantic changes +Using index info to reconstruct a base tree... +M tools/testing/selftests/kvm/lib/kvm_util.c +Falling back to patching base and 3-way merge... +Auto-merging tools/testing/selftests/kvm/lib/kvm_util.c +No changes -- Patch already applied. +$ git am -3 ../patches/0001-KVM-selftests-Fix-up-phy_pages_alloc-API-rework-inte.patch +Applying: KVM: selftests: Fix up phy_pages_alloc() API rework interaction +$ git reset HEAD^ +Unstaged changes after reset: +M tools/testing/selftests/kvm/arm64/vgic_its_save.c +$ git add -A . +$ git commit -v -a --amend +warning: notes ref refs/notes/commits is invalid +[master 410b2f873a9df] Merge branch 'next' of https://github.com/kvm-x86/linux.git + Date: Sat Oct 3 00:53:23 2026 +0200 +Merging xen-tip/linux-next (93f51579e7df2 Linux 7.3-rc4) +$ git merge -m Merge branch 'linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/xen/tip.git xen-tip/linux-next +Already up to date. +Merging percpu/for-next (8f0b4cce4481f Linux 6.19-rc1) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/dennis/percpu.git percpu/for-next +Already up to date. +Merging workqueues/for-next (fee1265a72572 Merge branch 'for-7.3-fixes' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/tj/wq.git workqueues/for-next +Auto-merging kernel/workqueue.c +Merge made by the 'ort' strategy. + include/trace/events/workqueue.h | 146 ++++++++++ + kernel/workqueue.c | 603 ++++++++++++++++++++++++++------------- + 2 files changed, 547 insertions(+), 202 deletions(-) +Merging sched-ext/for-next (6a119bf5e64d3 Merge branch 'for-7.4' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/tj/sched_ext.git sched-ext/for-next +Auto-merging init/Kconfig +Auto-merging kernel/sched/ext/cid.c +Auto-merging kernel/sched/ext/ext.c +Auto-merging kernel/sched/ext/sub.c +Auto-merging kernel/sched/sched.h +Auto-merging tools/sched_ext/include/scx/common.bpf.h +Merge made by the 'ort' strategy. + Documentation/scheduler/sched-ext.rst | 13 + + include/linux/sched/ext.h | 38 +- + init/Kconfig | 2 - + kernel/sched/ext/cid.c | 67 +- + kernel/sched/ext/cid.h | 11 +- + kernel/sched/ext/ext.c | 924 +++++++++++++++++---- + kernel/sched/ext/ext.h | 32 +- + kernel/sched/ext/idle.c | 14 +- + kernel/sched/ext/inlines.h | 8 +- + kernel/sched/ext/internal.h | 105 ++- + kernel/sched/ext/sub.c | 110 +-- + kernel/sched/ext/sub.h | 38 +- + kernel/sched/ext/types.h | 28 +- + kernel/sched/sched.h | 12 +- + kernel/sched/syscalls.c | 2 + + tools/sched_ext/Makefile | 2 +- + tools/sched_ext/include/scx/cid.bpf.h | 697 +++++++++++++--- + tools/sched_ext/include/scx/common.bpf.h | 94 ++- + tools/sched_ext/include/scx/compat.bpf.h | 84 ++ + tools/sched_ext/include/scx/compat.h | 28 +- + tools/sched_ext/include/scx/const-defs.h | 20 + + tools/sched_ext/include/scx/enum_defs.autogen.h | 10 + + tools/sched_ext/include/scx/enums.autogen.bpf.h | 21 +- + tools/sched_ext/include/scx/enums.autogen.h | 22 +- + tools/sched_ext/include/scx/enums_abi.autogen.h | 12 +- + tools/sched_ext/include/scx/features.bpf.h | 29 + + tools/sched_ext/scx_flatcg.bpf.c | 2 +- + tools/sched_ext/scx_qmap.bpf.c | 93 ++- + tools/sched_ext/scx_qmap.c | 13 +- + tools/sched_ext/scx_qmap.h | 1 + + tools/testing/selftests/sched_ext/.gitignore | 4 + + tools/testing/selftests/sched_ext/Makefile | 14 +- + .../testing/selftests/sched_ext/allowed_cpus.bpf.c | 24 +- + tools/testing/selftests/sched_ext/allowed_cpus.c | 123 ++- + .../selftests/sched_ext/cgroup_nr_cpus.bpf.c | 60 ++ + tools/testing/selftests/sched_ext/cgroup_nr_cpus.c | 419 ++++++++++ + tools/testing/selftests/sched_ext/config | 3 + + tools/testing/selftests/sched_ext/create_dsq.bpf.c | 35 + + tools/testing/selftests/sched_ext/create_dsq.c | 4 +- + .../selftests/sched_ext/dequeue_remote.bpf.c | 270 ++++++ + tools/testing/selftests/sched_ext/dequeue_remote.c | 204 +++++ + .../testing/selftests/sched_ext/enq_blocked.bpf.c | 116 +++ + tools/testing/selftests/sched_ext/enq_blocked.c | 917 ++++++++++++++++++++ + tools/testing/selftests/sched_ext/enq_blocked.h | 28 + + tools/testing/selftests/sched_ext/hotplug.c | 34 +- + tools/testing/selftests/sched_ext/kick.bpf.c | 164 ++++ + tools/testing/selftests/sched_ext/kick.c | 519 ++++++++++++ + tools/testing/selftests/sched_ext/kick_test.h | 29 + + tools/testing/selftests/sched_ext/nohz_tick.bpf.c | 57 +- + tools/testing/selftests/sched_ext/nohz_tick.c | 207 ++++- + tools/testing/selftests/sched_ext/nohz_tick_test.h | 13 + + tools/testing/selftests/sched_ext/rt_stall.c | 80 +- + tools/testing/selftests/sched_ext/runner.c | 2 +- + .../selftests/sched_ext/test_modules/Makefile | 13 + + .../sched_ext/test_modules/scx_enq_blocked_test.c | 195 +++++ + tools/testing/selftests/sched_ext/util.c | 143 ++++ + tools/testing/selftests/sched_ext/util.h | 12 + + 57 files changed, 5663 insertions(+), 558 deletions(-) + create mode 100644 tools/sched_ext/include/scx/const-defs.h + create mode 100644 tools/sched_ext/include/scx/features.bpf.h + create mode 100644 tools/testing/selftests/sched_ext/cgroup_nr_cpus.bpf.c + create mode 100644 tools/testing/selftests/sched_ext/cgroup_nr_cpus.c + create mode 100644 tools/testing/selftests/sched_ext/dequeue_remote.bpf.c + create mode 100644 tools/testing/selftests/sched_ext/dequeue_remote.c + create mode 100644 tools/testing/selftests/sched_ext/enq_blocked.bpf.c + create mode 100644 tools/testing/selftests/sched_ext/enq_blocked.c + create mode 100644 tools/testing/selftests/sched_ext/enq_blocked.h + create mode 100644 tools/testing/selftests/sched_ext/kick.bpf.c + create mode 100644 tools/testing/selftests/sched_ext/kick.c + create mode 100644 tools/testing/selftests/sched_ext/kick_test.h + create mode 100644 tools/testing/selftests/sched_ext/nohz_tick_test.h + create mode 100644 tools/testing/selftests/sched_ext/test_modules/Makefile + create mode 100644 tools/testing/selftests/sched_ext/test_modules/scx_enq_blocked_test.c +Merging drivers-x86/for-next (fe5030c8cc715 Merge branch 'platform-drivers-x86-intel-pmt' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/pdx86/platform-drivers-x86.git drivers-x86/for-next +Auto-merging MAINTAINERS +Auto-merging drivers/platform/x86/asus-laptop.c +Auto-merging drivers/platform/x86/hp/hp-wmi.c +Merge made by the 'ort' strategy. + Documentation/ABI/testing/debugfs-tpmi | 8 +- + .../admin-guide/laptops/thinkpad-acpi.rst | 26 +- + Documentation/wmi/devices/acer-wmi-battery.rst | 70 ++++ + Documentation/wmi/devices/bitland-mifs-wmi.rst | 64 ++- + MAINTAINERS | 3 +- + drivers/platform/arm64/acer-aspire1-ec.c | 2 +- + drivers/platform/arm64/huawei-gaokun-ec.c | 2 +- + drivers/platform/arm64/lenovo-thinkpad-t14s.c | 10 +- + drivers/platform/arm64/lenovo-yoga-c630.c | 2 +- + drivers/platform/mellanox/mlxbf-tmfifo.c | 1 - + drivers/platform/mellanox/mlxreg-hotplug.c | 4 +- + .../platform/surface/surface_aggregator_registry.c | 5 +- + drivers/platform/x86/Kconfig | 12 + + drivers/platform/x86/Makefile | 1 + + drivers/platform/x86/acer-wmi-battery.c | 450 +++++++++++++++++++++ + drivers/platform/x86/acer-wmi.c | 56 ++- + drivers/platform/x86/amd/pmc/pmc-quirks.c | 9 + + drivers/platform/x86/amd/pmc/pmc.c | 15 +- + drivers/platform/x86/amd/pmf/core.c | 1 + + drivers/platform/x86/amd/pmf/sps.c | 10 +- + drivers/platform/x86/amd/pmf/util.c | 1 + + drivers/platform/x86/asus-armoury.c | 29 +- + drivers/platform/x86/asus-armoury.h | 41 ++ + drivers/platform/x86/asus-laptop.c | 2 +- + drivers/platform/x86/asus-nb-wmi.c | 15 +- + drivers/platform/x86/asus-tf103c-dock.c | 2 +- + drivers/platform/x86/asus-wmi.c | 97 +++-- + drivers/platform/x86/asus-wmi.h | 1 - + drivers/platform/x86/bitland-mifs-wmi.c | 146 ++++--- + drivers/platform/x86/dell/alienware-wmi-wmax.c | 8 + + drivers/platform/x86/dell/dell-dw5826e-reset.c | 30 +- + drivers/platform/x86/dell/dell-wmi-aio.c | 200 ++++----- + drivers/platform/x86/hp/hp-bioscfg/bioscfg.c | 2 +- + drivers/platform/x86/hp/hp-wmi.c | 27 +- + drivers/platform/x86/hp/hp_accel.c | 2 +- + drivers/platform/x86/huawei-wmi.c | 1 + + drivers/platform/x86/intel/bxtwc_tmu.c | 5 +- + drivers/platform/x86/intel/bytcrc_pwrsrc.c | 2 +- + drivers/platform/x86/intel/crystal_cove_charger.c | 2 +- + drivers/platform/x86/intel/int0002_vgpio.c | 4 +- + drivers/platform/x86/intel/pmc/core.c | 1 + + drivers/platform/x86/intel/pmt/class.c | 35 +- + drivers/platform/x86/intel/pmt/class.h | 5 +- + drivers/platform/x86/intel/pmt/crashlog.c | 235 +++++++---- + drivers/platform/x86/intel/pmt/discovery.c | 2 +- + drivers/platform/x86/intel/pmt/telemetry.c | 3 + + drivers/platform/x86/intel/punit_ipc.c | 4 +- + .../x86/intel/speed_select_if/isst_if_common.c | 57 ++- + .../x86/intel/uncore-frequency/uncore-frequency.c | 1 + + drivers/platform/x86/intel/vsec.c | 19 +- + drivers/platform/x86/intel/vsec_tpmi.c | 2 +- + drivers/platform/x86/lenovo/ideapad-laptop.c | 4 + + drivers/platform/x86/lenovo/think-lmi.c | 2 +- + drivers/platform/x86/lenovo/thinkpad_acpi.c | 273 ++++++++----- + drivers/platform/x86/msi-laptop.c | 2 +- + drivers/platform/x86/panasonic-laptop.c | 14 +- + drivers/platform/x86/samsung-laptop.c | 2 +- + drivers/platform/x86/topstar-laptop.c | 6 +- + drivers/platform/x86/uniwill/uniwill-acpi.c | 2 +- + include/linux/intel_vsec.h | 14 +- + include/linux/platform_data/x86/asus-wmi.h | 4 +- + include/linux/string_choices.h | 6 + + include/uapi/linux/amd-pmf.h | 3 + + 63 files changed, 1525 insertions(+), 539 deletions(-) + create mode 100644 Documentation/wmi/devices/acer-wmi-battery.rst + create mode 100644 drivers/platform/x86/acer-wmi-battery.c +Merging chrome-platform/for-next (317e7abcaf60b platform/chrome: cros_ec_typec: Validate SVID and mode counts in discovery data) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/chrome-platform/linux.git chrome-platform/for-next +Merge made by the 'ort' strategy. + drivers/platform/chrome/cros_ec_ishtp.c | 2 +- + drivers/platform/chrome/cros_ec_proto.c | 43 +++++-- + drivers/platform/chrome/cros_ec_proto_test.c | 158 ++++++++++++++++++++++++- + drivers/platform/chrome/cros_ec_rpmsg.c | 2 +- + drivers/platform/chrome/cros_ec_typec.c | 10 ++ + drivers/platform/chrome/cros_usbpd_notify.c | 4 +- + include/linux/platform_data/cros_ec_commands.h | 72 +++++++++++ + 7 files changed, 272 insertions(+), 19 deletions(-) +Merging chrome-platform-firmware/for-firmware-next (f2e05ce9763fe firmware: coreboot: Add CFR firmware attributes driver) +$ git merge -m Merge branch 'for-firmware-next' of https://git.kernel.org/pub/scm/linux/kernel/git/chrome-platform/linux.git chrome-platform-firmware/for-firmware-next +Auto-merging MAINTAINERS +Auto-merging drivers/firmware/Kconfig +Auto-merging drivers/firmware/Makefile +Auto-merging drivers/platform/x86/Kconfig +Auto-merging drivers/platform/x86/Makefile +Auto-merging drivers/platform/x86/asus-armoury.c +Auto-merging drivers/platform/x86/hp/hp-bioscfg/bioscfg.c +Auto-merging drivers/platform/x86/lenovo/think-lmi.c +Merge made by the 'ort' strategy. + MAINTAINERS | 34 +- + drivers/firmware/Kconfig | 5 +- + drivers/firmware/Makefile | 3 +- + drivers/firmware/{google => coreboot}/Kconfig | 89 +- + drivers/firmware/coreboot/Makefile | 15 + + drivers/firmware/{google => coreboot}/cbmem.c | 0 + drivers/firmware/coreboot/coreboot-cfr.c | 1204 ++++++++++++++++++++ + .../firmware/{google => coreboot}/coreboot_table.c | 4 +- + .../firmware/{google => coreboot}/coreboot_table.h | 0 + .../{google => coreboot}/framebuffer-coreboot.c | 0 + drivers/firmware/{google => coreboot}/gsmi.c | 0 + .../{google => coreboot}/memconsole-coreboot.c | 0 + .../{google => coreboot}/memconsole-x86-legacy.c | 0 + drivers/firmware/{google => coreboot}/memconsole.c | 0 + drivers/firmware/{google => coreboot}/memconsole.h | 6 +- + drivers/firmware/{google => coreboot}/vpd.c | 0 + drivers/firmware/{google => coreboot}/vpd_decode.c | 0 + drivers/firmware/{google => coreboot}/vpd_decode.h | 0 + .../x86 => firmware}/firmware_attributes_class.c | 2 +- + drivers/firmware/google/Makefile | 14 - + drivers/platform/x86/Kconfig | 3 - + drivers/platform/x86/Makefile | 2 - + drivers/platform/x86/asus-armoury.c | 2 +- + drivers/platform/x86/dell/dell-wmi-sysman/sysman.c | 2 +- + drivers/platform/x86/hp/hp-bioscfg/bioscfg.c | 2 +- + drivers/platform/x86/lenovo/think-lmi.c | 2 +- + drivers/platform/x86/lenovo/wmi-other.c | 2 +- + drivers/platform/x86/samsung-galaxybook.c | 2 +- + .../linux/firmware_attributes.h | 6 +- + 29 files changed, 1335 insertions(+), 64 deletions(-) + rename drivers/firmware/{google => coreboot}/Kconfig (57%) + create mode 100644 drivers/firmware/coreboot/Makefile + rename drivers/firmware/{google => coreboot}/cbmem.c (100%) + create mode 100644 drivers/firmware/coreboot/coreboot-cfr.c + rename drivers/firmware/{google => coreboot}/coreboot_table.c (99%) + rename drivers/firmware/{google => coreboot}/coreboot_table.h (100%) + rename drivers/firmware/{google => coreboot}/framebuffer-coreboot.c (100%) + rename drivers/firmware/{google => coreboot}/gsmi.c (100%) + rename drivers/firmware/{google => coreboot}/memconsole-coreboot.c (100%) + rename drivers/firmware/{google => coreboot}/memconsole-x86-legacy.c (100%) + rename drivers/firmware/{google => coreboot}/memconsole.c (100%) + rename drivers/firmware/{google => coreboot}/memconsole.h (82%) + rename drivers/firmware/{google => coreboot}/vpd.c (100%) + rename drivers/firmware/{google => coreboot}/vpd_decode.c (100%) + rename drivers/firmware/{google => coreboot}/vpd_decode.h (100%) + rename drivers/{platform/x86 => firmware}/firmware_attributes_class.c (94%) + delete mode 100644 drivers/firmware/google/Makefile + rename drivers/platform/x86/firmware_attributes_class.h => include/linux/firmware_attributes.h (60%) +Merging hsi/for-next (e81250ec6b692 hsi: omap_ssi_core: fix missing DMA mask setup for SSI controller device) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/sre/linux-hsi.git hsi/for-next +Already up to date. +Merging leds-lj/for-leds-next (05b4738b0078f leds: trigger: Add led_trigger_notify_hw_control_changed() interface) +$ git merge -m Merge branch 'for-leds-next' of https://git.kernel.org/pub/scm/linux/kernel/git/lee/leds.git leds-lj/for-leds-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + Documentation/ABI/testing/sysfs-class-led | 28 ++- + .../ABI/testing/sysfs-class-led-trigger-netdev | 3 + + .../devicetree/bindings/leds/adi,ltc3208.yaml | 183 ++++++++++++++++ + .../devicetree/bindings/leds/leds-cpcap.txt | 29 --- + .../bindings/leds/motorola,cpcap-leds.yaml | 42 ++++ + Documentation/leds/leds-class.rst | 72 +++++++ + Documentation/leds/leds-st1202.rst | 5 + + MAINTAINERS | 8 + + drivers/leds/Kconfig | 12 ++ + drivers/leds/Makefile | 1 + + drivers/leds/flash/leds-aat1290.c | 5 +- + drivers/leds/flash/leds-ktd2692.c | 10 +- + drivers/leds/led-class-flash.c | 21 +- + drivers/leds/led-class.c | 46 ++-- + drivers/leds/led-core.c | 5 +- + drivers/leds/led-triggers.c | 177 ++++++++++++++- + drivers/leds/leds-cros_ec.c | 6 + + drivers/leds/leds-is31fl32xx.c | 20 +- + drivers/leds/leds-lm3692x.c | 4 +- + drivers/leds/leds-lp8860.c | 5 +- + drivers/leds/leds-ltc3208.c | 234 ++++++++++++++++++++ + drivers/leds/leds-max77705.c | 3 +- + drivers/leds/leds-pca9532.c | 9 +- + drivers/leds/leds-pca963x.c | 5 +- + drivers/leds/leds-ss4200.c | 2 +- + drivers/leds/leds-st1202.c | 240 ++++++++++++++++----- + drivers/leds/leds-syscon.c | 9 +- + drivers/leds/leds-turris-omnia.c | 7 + + drivers/leds/leds.h | 8 +- + drivers/leds/rgb/leds-qcom-lpg.c | 38 ++-- + drivers/leds/trigger/Kconfig | 10 + + drivers/leds/trigger/ledtrig-netdev.c | 13 +- + include/linux/leds.h | 24 +++ + 33 files changed, 1089 insertions(+), 195 deletions(-) + create mode 100644 Documentation/devicetree/bindings/leds/adi,ltc3208.yaml + delete mode 100644 Documentation/devicetree/bindings/leds/leds-cpcap.txt + create mode 100644 Documentation/devicetree/bindings/leds/motorola,cpcap-leds.yaml + create mode 100644 drivers/leds/leds-ltc3208.c +Merging ipmi/for-next (921fcdb737891 char: ipmi: Replace deprecated strcpy with strscpy) +$ git merge -m Merge branch 'for-next' of https://github.com/cminyard/linux-ipmi.git ipmi/for-next +Merge made by the 'ort' strategy. + drivers/char/ipmi/ipmi_watchdog.c | 12 ++++++------ + 1 file changed, 6 insertions(+), 6 deletions(-) +Merging driver-core/driver-core-next (f1850e443b0e4 docs: admin-guide: Handle TAINT_FORCED_BIND when parsing /proc/sys/kernel/tainted) +$ git merge -m Merge branch 'driver-core-next' of https://git.kernel.org/pub/scm/linux/kernel/git/driver-core/driver-core.git driver-core/driver-core-next +Auto-merging drivers/base/bus.c +Auto-merging include/linux/module.h +Auto-merging include/linux/panic.h +Auto-merging kernel/module/main.c +Auto-merging kernel/panic.c +Auto-merging lib/kobject_uevent.c +Auto-merging rust/kernel/auxiliary.rs +Merge made by the 'ort' strategy. + Documentation/admin-guide/tainted-kernels.rst | 54 ++++++++++++++------------- + arch/powerpc/perf/hv-24x7.c | 2 +- + arch/x86/kernel/cpu/mce/core.c | 6 +-- + drivers/base/bus.c | 3 ++ + drivers/base/core.c | 30 +++++++-------- + drivers/base/node.c | 17 ++++----- + drivers/base/platform.c | 19 +++++----- + drivers/base/property.c | 2 +- + drivers/base/transport_class.c | 2 +- + drivers/perf/alibaba_uncore_drw_pmu.c | 2 +- + drivers/perf/arm-cci.c | 2 +- + drivers/perf/arm-ccn.c | 2 +- + drivers/perf/arm_cspmu/arm_cspmu.h | 2 +- + drivers/perf/arm_dsu_pmu.c | 2 +- + drivers/perf/arm_spe_pmu.c | 6 +-- + drivers/perf/cxl_pmu.c | 10 ++--- + drivers/perf/fsl_imx8_ddr_perf.c | 2 +- + drivers/perf/fujitsu_uncore_pmu.c | 2 +- + drivers/perf/hisilicon/hisi_pcie_pmu.c | 2 +- + drivers/perf/hisilicon/hisi_uncore_pmu.h | 6 +-- + drivers/perf/hisilicon/hns3_pmu.c | 6 +-- + drivers/perf/nvidia_t410_c2c_pmu.c | 2 +- + drivers/perf/nvidia_t410_cmem_latency_pmu.c | 12 +++--- + drivers/perf/qcom_l3_pmu.c | 8 ++-- + drivers/perf/starfive_starlink_pmu.c | 2 +- + drivers/perf/xgene_pmu.c | 2 +- + include/linux/device.h | 14 +++---- + include/linux/kernfs.h | 2 +- + include/linux/module.h | 10 +++++ + include/linux/panic.h | 3 +- + include/trace/events/module.h | 3 +- + kernel/module/main.c | 17 +++++++-- + kernel/panic.c | 10 +++-- + lib/kobject_uevent.c | 2 +- + rust/helpers/dma.c | 5 +++ + rust/kernel/auxiliary.rs | 8 ++++ + rust/kernel/scatterlist.rs | 12 +++++- + tools/debugging/kernel-chktaint | 8 ++++ + 38 files changed, 180 insertions(+), 119 deletions(-) +Merging usb/usb-next (639df5d23876a dt-bindings: usb: qcom,snps-dwc3: Document the Nord DWC3 controller) +$ git merge -m Merge branch 'usb-next' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/usb.git usb/usb-next +Auto-merging Documentation/devicetree/bindings/usb/qcom,snps-dwc3.yaml +Auto-merging arch/arm/boot/dts/st/stm32mp131.dtsi +Auto-merging arch/arm/boot/dts/st/stm32mp151.dtsi +Auto-merging arch/arm64/boot/dts/st/stm32mp231.dtsi +Auto-merging arch/arm64/boot/dts/st/stm32mp251.dtsi +Auto-merging drivers/net/usb/r8152.c +Auto-merging drivers/usb/core/hub.c +Auto-merging drivers/usb/dwc2/hcd.c +Auto-merging drivers/usb/gadget/legacy/inode.c +Auto-merging drivers/usb/gadget/udc/aspeed-vhub/core.c +Auto-merging drivers/usb/gadget/udc/aspeed-vhub/dev.c +Auto-merging drivers/usb/gadget/udc/lpc32xx_udc.c +Auto-merging drivers/usb/typec/anx7411.c +Auto-merging drivers/usb/typec/tcpm/tcpm.c +Auto-merging drivers/usb/typec/tipd/core.c +Auto-merging drivers/usb/typec/ucsi/ucsi.c +Merge made by the 'ort' strategy. + .../devicetree/bindings/dma/ti/ti,cppi41.yaml | 109 + + .../devicetree/bindings/phy/ti,am335x-usb-phy.yaml | 49 + + .../devicetree/bindings/usb/am33xx-usb.txt | 200 - + .../devicetree/bindings/usb/da8xx-usb.txt | 81 - + .../devicetree/bindings/usb/generic-ehci.yaml | 24 + + .../devicetree/bindings/usb/generic-ohci.yaml | 24 + + .../devicetree/bindings/usb/gpio-sbu-mux.yaml | 1 + + .../devicetree/bindings/usb/ohci-da8xx.txt | 23 - + .../devicetree/bindings/usb/qcom,snps-dwc3.yaml | 22 + + .../bindings/usb/ti,am335x-usb-ctrl-module.yaml | 38 + + .../devicetree/bindings/usb/ti,am33xx-usb.yaml | 103 + + .../devicetree/bindings/usb/ti,da830-musb.yaml | 113 + + .../devicetree/bindings/usb/ti,da830-ohci.yaml | 61 + + .../devicetree/bindings/usb/ti,hd3ss3220.yaml | 11 + + .../devicetree/bindings/usb/ti,musb-am33xx.yaml | 113 + + Documentation/usb/usbmon.rst | 2 +- + arch/arm/boot/dts/st/stm32mp131.dtsi | 4 +- + arch/arm/boot/dts/st/stm32mp151.dtsi | 4 +- + arch/arm64/boot/dts/st/stm32mp231.dtsi | 64 + + arch/arm64/boot/dts/st/stm32mp251.dtsi | 63 + + drivers/net/usb/r8152.c | 2 +- + drivers/usb/cdns3/core.c | 4 +- + drivers/usb/cdns3/drd.c | 4 +- + drivers/usb/chipidea/ci_hdrc_imx.c | 8 +- + drivers/usb/chipidea/otg_fsm.c | 2 +- + drivers/usb/chipidea/udc.c | 2 +- + drivers/usb/common/ulpi.c | 4 +- + drivers/usb/common/usb-conn-gpio.c | 8 +- + drivers/usb/core/driver.c | 10 +- + drivers/usb/core/hub.c | 2 +- + drivers/usb/core/urb.c | 4 +- + drivers/usb/core/usb.c | 2 +- + drivers/usb/dwc2/core.c | 2 +- + drivers/usb/dwc2/core.h | 2 +- + drivers/usb/dwc2/debugfs.c | 6 +- + drivers/usb/dwc2/gadget.c | 98 +- + drivers/usb/dwc2/hcd.c | 4 +- + drivers/usb/dwc2/hcd_ddma.c | 2 +- + drivers/usb/dwc3/debugfs.c | 6 +- + drivers/usb/dwc3/dwc3-google.c | 4 +- + drivers/usb/dwc3/dwc3-imx.c | 2 +- + drivers/usb/dwc3/dwc3-imx8mp.c | 4 +- + drivers/usb/dwc3/dwc3-keystone.c | 5 +- + drivers/usb/dwc3/dwc3-omap.c | 5 +- + drivers/usb/fotg210/Kconfig | 5 + + drivers/usb/fotg210/fotg210-core.c | 273 +- + drivers/usb/fotg210/fotg210-hcd.c | 5666 +------------------- + drivers/usb/fotg210/fotg210-hcd.h | 689 --- + drivers/usb/fotg210/fotg210-udc.c | 114 +- + drivers/usb/fotg210/fotg210-udc.h | 2 +- + drivers/usb/fotg210/fotg210.h | 26 +- + drivers/usb/gadget/composite.c | 2 +- + drivers/usb/gadget/function/f_mass_storage.c | 7 +- + drivers/usb/gadget/legacy/inode.c | 205 +- + drivers/usb/gadget/udc/aspeed-vhub/core.c | 4 +- + drivers/usb/gadget/udc/aspeed-vhub/dev.c | 2 - + drivers/usb/gadget/udc/aspeed_udc.c | 5 +- + drivers/usb/gadget/udc/at91_udc.c | 1 - + drivers/usb/gadget/udc/atmel_usba_udc.c | 5 +- + drivers/usb/gadget/udc/bcm63xx_udc.c | 1 - + drivers/usb/gadget/udc/core.c | 9 +- + drivers/usb/gadget/udc/fsl_udc_core.c | 3 - + drivers/usb/gadget/udc/lpc32xx_udc.c | 1 - + drivers/usb/gadget/udc/r8a66597-udc.c | 4 +- + drivers/usb/gadget/udc/renesas_usbf.c | 8 +- + drivers/usb/gadget/udc/snps_udc_plat.c | 4 +- + drivers/usb/gadget/udc/tegra-xudc.c | 5 +- + drivers/usb/host/Kconfig | 13 + + drivers/usb/host/ehci-dbg.c | 3 +- + drivers/usb/host/ehci-hcd.c | 31 +- + drivers/usb/host/ehci-hub.c | 104 +- + drivers/usb/host/ehci-ppc-of.c | 47 +- + drivers/usb/host/ehci-timer.c | 3 +- + drivers/usb/host/ehci.h | 55 +- + drivers/usb/host/ohci-dbg.c | 2 +- + drivers/usb/host/ohci-hub.c | 2 +- + drivers/usb/host/ohci-ppc-of.c | 44 +- + drivers/usb/host/pci-quirks.c | 1 + + drivers/usb/host/xhci-tegra.c | 8 +- + drivers/usb/image/microtek.c | 1 + + drivers/usb/misc/apple-mfi-fastcharge.c | 2 +- + drivers/usb/misc/brcmstb-usb-pinmap.c | 9 +- + drivers/usb/misc/onboard_usb_dev.c | 2 +- + drivers/usb/misc/qcom_eud.c | 2 +- + drivers/usb/mon/mon_main.c | 3 + + drivers/usb/mtu3/mtu3_core.c | 4 +- + drivers/usb/musb/da8xx.c | 8 +- + drivers/usb/musb/musb_dsps.c | 25 + + drivers/usb/phy/phy-ab8500-usb.c | 12 +- + drivers/usb/phy/phy-generic.c | 3 +- + drivers/usb/phy/phy-gpio-vbus-usb.c | 5 +- + drivers/usb/renesas_usbhs/mod.c | 4 +- + drivers/usb/storage/alauda.c | 95 +- + drivers/usb/storage/ene_ub6250.c | 2 + + drivers/usb/storage/sddr09.c | 9 + + drivers/usb/storage/usb.c | 6 +- + drivers/usb/typec/anx7411.c | 4 +- + drivers/usb/typec/bus.c | 4 +- + drivers/usb/typec/hd3ss3220.c | 36 +- + drivers/usb/typec/mux/gpio-sbu-mux.c | 13 +- + drivers/usb/typec/mux/it5205.c | 2 +- + drivers/usb/typec/mux/ps883x.c | 48 + + drivers/usb/typec/tcpm/tcpci_maxim_core.c | 3 +- + drivers/usb/typec/tcpm/tcpci_mt6360.c | 1 - + drivers/usb/typec/tcpm/tcpci_mt6370.c | 2 +- + drivers/usb/typec/tcpm/tcpm.c | 57 +- + drivers/usb/typec/tipd/core.c | 3 + + drivers/usb/typec/ucsi/ucsi.c | 52 +- + drivers/usb/typec/ucsi/ucsi.h | 3 +- + drivers/usb/usbip/stub_main.c | 2 +- + include/linux/ulpi/driver.h | 4 +- + include/linux/usb.h | 6 +- + include/linux/usb/typec_altmode.h | 4 +- + 113 files changed, 2014 insertions(+), 7077 deletions(-) + create mode 100644 Documentation/devicetree/bindings/dma/ti/ti,cppi41.yaml + create mode 100644 Documentation/devicetree/bindings/phy/ti,am335x-usb-phy.yaml + delete mode 100644 Documentation/devicetree/bindings/usb/am33xx-usb.txt + delete mode 100644 Documentation/devicetree/bindings/usb/da8xx-usb.txt + delete mode 100644 Documentation/devicetree/bindings/usb/ohci-da8xx.txt + create mode 100644 Documentation/devicetree/bindings/usb/ti,am335x-usb-ctrl-module.yaml + create mode 100644 Documentation/devicetree/bindings/usb/ti,am33xx-usb.yaml + create mode 100644 Documentation/devicetree/bindings/usb/ti,da830-musb.yaml + create mode 100644 Documentation/devicetree/bindings/usb/ti,da830-ohci.yaml + create mode 100644 Documentation/devicetree/bindings/usb/ti,musb-am33xx.yaml + delete mode 100644 drivers/usb/fotg210/fotg210-hcd.h +Merging thunderbolt/next (a93a8e3200002 thunderbolt: stream: Make read return framing error to the userspace) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/westeri/thunderbolt.git thunderbolt/next +Auto-merging drivers/thunderbolt/nhi.c +Auto-merging drivers/thunderbolt/nhi.h +Auto-merging drivers/thunderbolt/pci.c +Auto-merging drivers/thunderbolt/quirks.c +Auto-merging drivers/thunderbolt/stream.c +Auto-merging drivers/thunderbolt/switch.c +Auto-merging drivers/thunderbolt/tb.c +Auto-merging include/linux/thunderbolt.h +Merge made by the 'ort' strategy. + drivers/net/thunderbolt/main.c | 6 +- + drivers/thunderbolt/debugfs.c | 126 ++++++++++++++---- + drivers/thunderbolt/dma_test.c | 10 +- + drivers/thunderbolt/eeprom.c | 23 +++- + drivers/thunderbolt/nhi.c | 222 ++++++++++++++++++++++--------- + drivers/thunderbolt/nhi.h | 2 + + drivers/thunderbolt/path.c | 2 +- + drivers/thunderbolt/pci.c | 113 ++++++++++++++++ + drivers/thunderbolt/quirks.c | 18 +++ + drivers/thunderbolt/sb_regs.h | 2 + + drivers/thunderbolt/stream.c | 287 +++++++++++++++++++++++++++++++---------- + drivers/thunderbolt/switch.c | 4 + + drivers/thunderbolt/tb.c | 78 ----------- + drivers/thunderbolt/tb.h | 5 +- + drivers/thunderbolt/usb4.c | 10 +- + include/linux/thunderbolt.h | 47 ++++++- + 16 files changed, 703 insertions(+), 252 deletions(-) +Merging usb-serial/usb-next (6583f9741341b USB: serial: wwan: replace __get_free_page() with kmalloc()) +$ git merge -m Merge branch 'usb-next' of https://git.kernel.org/pub/scm/linux/kernel/git/johan/usb-serial.git usb-serial/usb-next +Merge made by the 'ort' strategy. + drivers/usb/serial/usb_wwan.c | 6 +++--- + 1 file changed, 3 insertions(+), 3 deletions(-) +Merging tty/tty-next (c44a4925cdac0 serial: 8250: Fix section mismatch with CONFIG_MODULES=n) +$ git merge -m Merge branch 'tty-next' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/tty.git tty/tty-next +Auto-merging MAINTAINERS +Auto-merging drivers/tty/serial/8250/8250_omap.c +Auto-merging drivers/tty/serial/8250/8250_port.c +Auto-merging drivers/tty/serial/amba-pl011.c +Auto-merging drivers/tty/serial/kgdboc.c +Auto-merging drivers/tty/serial/qcom_geni_serial.c +CONFLICT (content): Merge conflict in drivers/tty/serial/qcom_geni_serial.c +Auto-merging drivers/tty/serial/serial_core.c +CONFLICT (content): Merge conflict in drivers/tty/serial/serial_core.c +Auto-merging drivers/tty/serial/sifive.c +Auto-merging drivers/tty/tty_port.c +Auto-merging include/linux/soc/qcom/geni-se.h +CONFLICT (content): Merge conflict in include/linux/soc/qcom/geni-se.h +Resolved 'drivers/tty/serial/qcom_geni_serial.c' using previous resolution. +Resolved 'drivers/tty/serial/serial_core.c' using previous resolution. +Resolved 'include/linux/soc/qcom/geni-se.h' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master a7cfa26c4fc0e] Merge branch 'tty-next' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/tty.git +$ git diff -M --stat --summary HEAD^.. + Documentation/admin-guide/devices.txt | 13 +- + Documentation/devicetree/bindings/serial/8250.yaml | 32 +- + .../devicetree/bindings/serial/cdns,uart.yaml | 1 + + .../devicetree/bindings/serial/renesas,hscif.yaml | 4 +- + .../devicetree/bindings/serial/renesas,scif.yaml | 4 +- + .../bindings/serial/snps-dw-apb-uart.yaml | 1 + + Documentation/driver-api/serial/driver.rst | 2 +- + MAINTAINERS | 2 +- + drivers/acpi/acpi_apd.c | 1 + + drivers/tty/Kconfig | 10 - + drivers/tty/Makefile | 1 - + drivers/tty/hvc/hvcs.c | 6 +- + drivers/tty/moxa.c | 2136 -------------------- + drivers/tty/pty.c | 5 +- + drivers/tty/serdev/serdev-ttyport.c | 2 + + drivers/tty/serial/8250/8250.h | 17 +- + drivers/tty/serial/8250/8250_airoha.c | 187 ++ + drivers/tty/serial/8250/8250_core.c | 25 +- + drivers/tty/serial/8250/8250_dma.c | 29 +- + drivers/tty/serial/8250/8250_dw.c | 19 +- + drivers/tty/serial/8250/8250_dwlib.c | 5 + + drivers/tty/serial/8250/8250_dwlib.h | 5 + + drivers/tty/serial/8250/8250_early.c | 1 + + drivers/tty/serial/8250/8250_hub6.c | 11 +- + drivers/tty/serial/8250/8250_mid.c | 16 +- + drivers/tty/serial/8250/8250_mxpcie.c | 30 +- + drivers/tty/serial/8250/8250_omap.c | 4 +- + drivers/tty/serial/8250/8250_pci.c | 2 +- + drivers/tty/serial/8250/8250_platform.c | 9 +- + drivers/tty/serial/8250/8250_pnp.c | 2 +- + drivers/tty/serial/8250/8250_port.c | 46 +- + drivers/tty/serial/8250/8250_uniphier.c | 28 +- + drivers/tty/serial/8250/Kconfig | 15 +- + drivers/tty/serial/8250/Makefile | 3 +- + drivers/tty/serial/Kconfig | 1 + + drivers/tty/serial/amba-pl011.c | 21 +- + drivers/tty/serial/arc_uart.c | 2 +- + drivers/tty/serial/atmel_serial.c | 8 +- + drivers/tty/serial/fsl_lpuart.c | 18 +- + drivers/tty/serial/kgdboc.c | 2 +- + drivers/tty/serial/qcom_geni_serial.c | 43 +- + drivers/tty/serial/samsung_tty.c | 27 +- + drivers/tty/serial/serial_core.c | 51 +- + drivers/tty/serial/serial_txx9.c | 62 +- + drivers/tty/serial/sh-sci.c | 22 +- + drivers/tty/serial/sifive.c | 2 +- + drivers/tty/serial/tegra-tcu.c | 22 +- + drivers/tty/sysrq.c | 5 +- + drivers/tty/tty_port.c | 8 +- + drivers/tty/vt/keyboard.c | 4 +- + include/linux/serial_core.h | 2 - + include/linux/soc/qcom/geni-se.h | 12 +- + 52 files changed, 534 insertions(+), 2452 deletions(-) + delete mode 100644 drivers/tty/moxa.c + create mode 100644 drivers/tty/serial/8250/8250_airoha.c +Merging char-misc/char-misc-next (04434a1d0f311 rust_binder: add transaction_log and failed_transaction_log) +$ git merge -m Merge branch 'char-misc-next' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/char-misc.git char-misc/char-misc-next +Auto-merging Documentation/devicetree/bindings/spi/spi-peripheral-props.yaml +Auto-merging Documentation/devicetree/bindings/trivial-devices.yaml +Auto-merging MAINTAINERS +CONFLICT (modify/delete): drivers/android/binder.c deleted in char-misc/char-misc-next and modified in HEAD. Version HEAD of drivers/android/binder.c left in tree. +Auto-merging drivers/android/binder/node.rs +Auto-merging drivers/android/binder/node/wrapper.rs +Auto-merging drivers/android/binder/page_range.rs +Auto-merging drivers/android/binder/rust_binderfs.c +Auto-merging drivers/android/binder/thread.rs +CONFLICT (modify/delete): drivers/android/binder_alloc.c deleted in char-misc/char-misc-next and modified in HEAD. Version HEAD of drivers/android/binder_alloc.c left in tree. +CONFLICT (modify/delete): drivers/android/binderfs.c deleted in char-misc/char-misc-next and modified in HEAD. Version HEAD of drivers/android/binderfs.c left in tree. +Auto-merging drivers/iio/accel/kionix-kx022a.c +Auto-merging drivers/iio/accel/kxcjk-1013.c +Auto-merging drivers/iio/accel/sca3000.c +Auto-merging drivers/iio/adc/ad7173.c +Auto-merging drivers/iio/adc/ade9000.c +CONFLICT (content): Merge conflict in drivers/iio/adc/ade9000.c +Auto-merging drivers/iio/adc/adi-axi-adc.c +Auto-merging drivers/iio/adc/pac1934.c +Auto-merging drivers/iio/adc/stm32-adc.c +Auto-merging drivers/iio/buffer/industrialio-buffer-dmaengine.c +Auto-merging drivers/iio/frequency/adf4377.c +Auto-merging drivers/iio/imu/adis16400.c +Auto-merging drivers/iio/light/lm3533-als.c +Auto-merging drivers/iio/pressure/rohm-bm1390.c +Auto-merging drivers/iio/proximity/aw96103.c +Auto-merging drivers/pci/quirks.c +Resolved 'drivers/iio/adc/ade9000.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git rm -f drivers/android/binder.c +drivers/android/binderfs.c +drivers/android/binder_alloc.c +fatal: pathspec 'drivers/android/binder.c +drivers/android/binderfs.c +drivers/android/binder_alloc.c' did not match any files +$ git commit --no-edit -v -a +[master 570bb5e88a398] Merge branch 'char-misc-next' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/char-misc.git +$ git diff -M --stat --summary HEAD^.. + Documentation/ABI/testing/debugfs-iio-ad9910 | 23 + + Documentation/ABI/testing/debugfs-iio-backend | 16 +- + Documentation/ABI/testing/sysfs-bus-iio | 115 + + Documentation/ABI/testing/sysfs-bus-iio-adc | 9 + + .../ABI/testing/sysfs-bus-iio-frequency-ad9910 | 31 + + .../bindings/iio/accel/adi,adis16201.yaml | 6 +- + .../devicetree/bindings/iio/accel/adi,adxl367.yaml | 13 +- + .../devicetree/bindings/iio/adc/adi,ad4080.yaml | 38 +- + .../devicetree/bindings/iio/adc/adi,ad4130.yaml | 3 +- + .../devicetree/bindings/iio/adc/adi,ad4851.yaml | 8 +- + .../devicetree/bindings/iio/adc/adi,ad7173.yaml | 2 +- + .../devicetree/bindings/iio/adc/adi,ad7768-1.yaml | 6 +- + .../devicetree/bindings/iio/adc/adi,ad7768.yaml | 270 + + .../devicetree/bindings/iio/adc/adi,ad7779.yaml | 30 +- + .../devicetree/bindings/iio/adc/adi,ade9000.yaml | 58 +- + .../devicetree/bindings/iio/adc/adi,max40080.yaml | 65 + + .../bindings/iio/adc/axiado,ax3000-saradc.yaml | 63 + + .../devicetree/bindings/iio/adc/lltc,ltc2497.yaml | 2 +- + .../bindings/iio/adc/maxim,max34408.yaml | 28 +- + .../bindings/iio/adc/mediatek,mt2701-auxadc.yaml | 1 + + .../bindings/iio/adc/nxp,lpc1850-adc.yaml | 2 +- + .../bindings/iio/adc/renesas,r9a09g077-adc.yaml | 22 +- + .../bindings/iio/adc/renesas,rzg2l-adc.yaml | 17 +- + .../devicetree/bindings/iio/adc/ti,ads1015.yaml | 6 + + .../devicetree/bindings/iio/adc/ti,ads1100.yaml | 10 +- + .../devicetree/bindings/iio/adc/ti,ads112c04.yaml | 149 + + .../devicetree/bindings/iio/adc/ti,ads1298.yaml | 18 +- + .../devicetree/bindings/iio/addac/adi,ad74115.yaml | 3 +- + .../bindings/iio/addac/adi,ad74413r.yaml | 3 +- + .../devicetree/bindings/iio/dac/adi,ad3530r.yaml | 15 +- + .../devicetree/bindings/iio/dac/adi,ad5529r.yaml | 254 + + .../devicetree/bindings/iio/dac/adi,ad5710r.yaml | 145 + + .../bindings/iio/dac/microchip,mcp47feb02.yaml | 219 +- + .../bindings/iio/frequency/adi,ad9910.yaml | 209 + + .../bindings/iio/imu/invensense,icm42600.yaml | 24 +- + .../devicetree/bindings/iio/light/adux1020.yaml | 8 +- + .../bindings/iio/light/amstaos,tsl2591.yaml | 2 +- + .../bindings/iio/light/capella,cm32181.yaml | 44 + + .../devicetree/bindings/iio/light/isl29018.yaml | 6 +- + .../bindings/iio/light/liteon,ltr501.yaml | 32 +- + .../devicetree/bindings/iio/light/noa1305.yaml | 4 +- + .../devicetree/bindings/iio/light/stk33xx.yaml | 19 +- + .../devicetree/bindings/iio/light/tsl2583.yaml | 4 +- + .../devicetree/bindings/iio/light/tsl2772.yaml | 14 +- + .../bindings/iio/light/vishay,veml6030.yaml | 21 +- + .../bindings/iio/pressure/infineon,dps310.yaml | 10 +- + .../iio/proximity/pulsedlight,lidar-lite-v2.yaml | 71 + + .../bindings/iio/proximity/vishay,vcnl3020.yaml | 6 +- + .../bindings/iio/temperature/adi,ltc2983.yaml | 1 + + .../devicetree/bindings/misc/pci1179,0220.yaml | 146 + + .../bindings/spi/spi-peripheral-props.yaml | 7 + + .../devicetree/bindings/trivial-devices.yaml | 6 +- + Documentation/iio/ad7768.rst | 275 + + Documentation/iio/ad9910.rst | 792 +++ + Documentation/iio/ade9000.rst | 26 +- + Documentation/iio/adis16475.rst | 2 +- + Documentation/iio/adis16480.rst | 2 +- + Documentation/iio/adis16550.rst | 2 +- + Documentation/iio/adxl313.rst | 2 +- + Documentation/iio/adxl345.rst | 2 +- + Documentation/iio/adxl380.rst | 10 +- + Documentation/iio/index.rst | 2 + + MAINTAINERS | 81 +- + drivers/android/Kconfig | 41 +- + drivers/android/Makefile | 3 - + drivers/android/binder.c | 7187 -------------------- + drivers/android/binder/Makefile | 4 +- + drivers/android/binder/allocation.rs | 17 +- + drivers/android/binder/context.rs | 8 +- + drivers/android/binder/defs.rs | 3 +- + drivers/android/binder/error.rs | 39 + + drivers/android/binder/freeze.rs | 17 +- + drivers/android/binder/node.rs | 10 +- + drivers/android/binder/node/wrapper.rs | 8 +- + drivers/android/binder/page_range.rs | 6 +- + drivers/android/binder/process.rs | 43 +- + drivers/android/binder/range_alloc/array.rs | 2 +- + drivers/android/binder/range_alloc/mod.rs | 2 +- + drivers/android/binder/range_alloc/tree.rs | 2 +- + drivers/android/binder/rust_binder_events.h | 42 + + drivers/android/binder/rust_binder_internal.h | 1 + + drivers/android/binder/rust_binder_main.rs | 45 +- + drivers/android/binder/rust_binderfs.c | 21 +- + drivers/android/binder/thread.rs | 74 +- + drivers/android/binder/trace.rs | 37 + + drivers/android/binder/transaction.rs | 202 +- + drivers/android/binder_alloc.c | 1398 ---- + drivers/android/binder_alloc.h | 189 - + drivers/android/binder_internal.h | 597 -- + drivers/android/binder_netlink.c | 32 - + drivers/android/binder_netlink.h | 21 - + drivers/android/binder_trace.h | 448 -- + drivers/android/binderfs.c | 785 --- + drivers/android/dbitmap.h | 169 - + drivers/android/tests/.kunitconfig | 7 - + drivers/android/tests/Makefile | 6 - + drivers/android/tests/binder_alloc_kunit.c | 572 -- + drivers/char/xilinx_hwicap/xilinx_hwicap.c | 18 +- + drivers/char/xillybus/xillyusb.c | 25 +- + drivers/greybus/greybus_trace.h | 2 +- + drivers/iio/accel/Kconfig | 7 +- + drivers/iio/accel/adis16201.c | 133 +- + drivers/iio/accel/adxl313_spi.c | 2 + + drivers/iio/accel/adxl367.c | 59 +- + drivers/iio/accel/adxl372_spi.c | 2 + + drivers/iio/accel/adxl380.c | 7 + + drivers/iio/accel/adxl380_spi.c | 2 + + drivers/iio/accel/bma220_spi.c | 3 +- + drivers/iio/accel/bma400_core.c | 2 - + drivers/iio/accel/kionix-kx022a.c | 22 +- + drivers/iio/accel/kionix-kx022a.h | 2 +- + drivers/iio/accel/kxcjk-1013.c | 1 + + drivers/iio/accel/mma8452.c | 145 +- + drivers/iio/accel/mma9551.c | 1 + + drivers/iio/accel/mma9553.c | 1 + + drivers/iio/accel/sca3000.c | 2 + + drivers/iio/adc/Kconfig | 86 +- + drivers/iio/adc/Makefile | 4 + + drivers/iio/adc/ad4080.c | 34 +- + drivers/iio/adc/ad4134.c | 1 - + drivers/iio/adc/ad7173.c | 3 +- + drivers/iio/adc/ad7192.c | 3 + + drivers/iio/adc/ad7606_spi.c | 6 +- + drivers/iio/adc/ad7768-1.c | 3 + + drivers/iio/adc/ad7768.c | 1921 ++++++ + drivers/iio/adc/ade9000.c | 180 +- + drivers/iio/adc/adi-axi-adc.c | 21 + + drivers/iio/adc/at91-sama5d2_adc.c | 10 +- + drivers/iio/adc/axiado_saradc.c | 277 + + drivers/iio/adc/bcm_iproc_adc.c | 101 +- + drivers/iio/adc/ltc2497-core.c | 297 +- + drivers/iio/adc/ltc2497.c | 42 + + drivers/iio/adc/ltc2497.h | 31 +- + drivers/iio/adc/max11205.c | 2 + + drivers/iio/adc/max14001.c | 1 + + drivers/iio/adc/max40080.c | 574 ++ + drivers/iio/adc/mcp3911.c | 2 + + drivers/iio/adc/meson_saradc.c | 2 +- + drivers/iio/adc/nxp-sar-adc.c | 2 +- + drivers/iio/adc/pac1934.c | 9 +- + drivers/iio/adc/rockchip_saradc.c | 16 + + drivers/iio/adc/rzg2l_adc.c | 29 +- + drivers/iio/adc/rzt2h_adc.c | 499 +- + drivers/iio/adc/sophgo-cv1800b-adc.c | 2 + + drivers/iio/adc/stm32-adc.c | 14 +- + drivers/iio/adc/stm32-dfsdm-adc.c | 4 +- + drivers/iio/adc/ti-adc128s052.c | 8 +- + drivers/iio/adc/ti-ads1015.c | 30 + + drivers/iio/adc/ti-ads1018.c | 6 +- + drivers/iio/adc/ti-ads1100.c | 188 +- + drivers/iio/adc/ti-ads112c04.c | 524 ++ + drivers/iio/adc/ti-ads112c14.c | 1274 +++- + drivers/iio/adc/ti-ads131m02.c | 2 + + drivers/iio/adc/ti-ads7950.c | 4 +- + drivers/iio/adc/ti_am335x_adc.c | 4 +- + drivers/iio/amplifiers/ad8366.c | 2 + + drivers/iio/buffer/industrialio-buffer-dmaengine.c | 8 +- + drivers/iio/chemical/ens160_core.c | 7 +- + drivers/iio/chemical/sps30.c | 11 +- + .../iio/common/cros_ec_sensors/cros_ec_sensors.c | 2 +- + .../common/cros_ec_sensors/cros_ec_sensors_core.c | 5 +- + drivers/iio/dac/Kconfig | 50 +- + drivers/iio/dac/Makefile | 5 +- + drivers/iio/dac/ad3530r.c | 321 +- + drivers/iio/dac/ad3552r.c | 2 +- + drivers/iio/dac/ad5529r.c | 546 ++ + drivers/iio/dac/ad5755.c | 3 + + drivers/iio/dac/ad5758.c | 8 +- + drivers/iio/dac/ad8460.c | 2 +- + .../iio/dac/{mcp47feb02.c => mcp47feb02-core.c} | 393 +- + drivers/iio/dac/mcp47feb02-i2c.c | 145 + + drivers/iio/dac/mcp47feb02-spi.c | 145 + + drivers/iio/dac/mcp47feb02.h | 41 + + drivers/iio/dac/mcp4821.c | 7 +- + drivers/iio/dummy/iio_simple_dummy.c | 2 +- + drivers/iio/frequency/Kconfig | 21 + + drivers/iio/frequency/Makefile | 1 + + drivers/iio/frequency/ad9910.c | 2347 +++++++ + drivers/iio/frequency/adf4350.c | 2 +- + drivers/iio/frequency/adf4377.c | 3 + + drivers/iio/gyro/adxrs290.c | 4 +- + drivers/iio/gyro/bmg160_core.c | 4 +- + drivers/iio/gyro/itg3200_buffer.c | 3 +- + drivers/iio/humidity/Kconfig | 8 +- + drivers/iio/humidity/am2315.c | 34 +- + drivers/iio/humidity/ens210.c | 4 +- + drivers/iio/humidity/hts221_core.c | 168 +- + drivers/iio/humidity/hts221_i2c.c | 11 +- + drivers/iio/humidity/hts221_spi.c | 11 +- + drivers/iio/iio_core.h | 7 + + drivers/iio/imu/adis16400.c | 8 +- + drivers/iio/imu/inv_icm42600/inv_icm42600.h | 5 +- + drivers/iio/imu/inv_icm42600/inv_icm42600_accel.c | 33 +- + drivers/iio/imu/inv_icm42600/inv_icm42600_buffer.c | 114 +- + drivers/iio/imu/inv_icm42600/inv_icm42600_core.c | 26 +- + drivers/iio/imu/inv_icm42600/inv_icm42600_gyro.c | 24 +- + drivers/iio/imu/inv_icm42600/inv_icm42600_i2c.c | 16 +- + drivers/iio/imu/inv_icm42600/inv_icm42600_spi.c | 16 +- + drivers/iio/imu/inv_mpu6050/inv_mpu_core.c | 6 +- + drivers/iio/imu/st_lsm6dsx/st_lsm6dsx_shub.c | 4 +- + drivers/iio/industrialio-backend.c | 33 + + drivers/iio/industrialio-core.c | 255 +- + drivers/iio/industrialio-gts-helper.c | 2 +- + drivers/iio/light/Kconfig | 14 + + drivers/iio/light/Makefile | 1 + + drivers/iio/light/apds9306.c | 2 +- + drivers/iio/light/apds9999.c | 12 +- + drivers/iio/light/cros_ec_light_prox.c | 2 +- + drivers/iio/light/iqs621-als.c | 24 +- + drivers/iio/light/isl29018.c | 7 +- + drivers/iio/light/isl29028.c | 33 +- + drivers/iio/light/lm3533-als.c | 11 +- + drivers/iio/light/ltr390.c | 6 - + drivers/iio/light/ltr501.c | 113 +- + drivers/iio/light/stk3310.c | 203 +- + drivers/iio/light/tsl2583.c | 15 +- + drivers/iio/light/tsl2772.c | 16 +- + drivers/iio/light/vcnl4000.c | 3 +- + drivers/iio/light/veml3328.c | 53 +- + drivers/iio/light/veml6031x00.c | 1387 ++++ + drivers/iio/magnetometer/ak8975.c | 234 +- + drivers/iio/position/iqs624-pos.c | 24 +- + drivers/iio/potentiometer/x9250.c | 7 +- + drivers/iio/pressure/Kconfig | 2 + + drivers/iio/pressure/Makefile | 2 + + drivers/iio/pressure/bmp280-spi.c | 2 + + drivers/iio/pressure/cros_ec_baro.c | 2 +- + drivers/iio/pressure/dps310.c | 313 +- + drivers/iio/pressure/rohm-bm1390.c | 6 +- + drivers/iio/proximity/aw96103.c | 2 +- + drivers/iio/temperature/iqs620at-temp.c | 9 +- + drivers/iio/temperature/ltc2983.c | 19 +- + drivers/iio/temperature/mlx90632.c | 10 +- + drivers/iio/temperature/mlx90635.c | 11 +- + drivers/iio/temperature/tmp117.c | 18 +- + drivers/iio/temperature/tsys01.c | 17 +- + drivers/iio/temperature/tsys02d.c | 12 +- + drivers/iio/test/Kconfig | 13 + + drivers/iio/test/Makefile | 1 + + drivers/iio/test/iio-test-channel-prefix.c | 307 + + drivers/iio/test/iio-test-rescale.c | 4 +- + drivers/misc/Kconfig | 15 + + drivers/misc/Makefile | 1 + + drivers/misc/ad525x_dpot.c | 9 +- + drivers/misc/cardreader/alcor_pci.c | 6 +- + drivers/misc/cb710/core.c | 8 +- + drivers/misc/enclosure.c | 3 +- + drivers/misc/genwqe/card_ddcb.c | 3 + + drivers/misc/ibmvmc.c | 7 +- + drivers/misc/isl29003.c | 10 +- + drivers/misc/lis3lv02d/lis3lv02d_spi.c | 6 +- + drivers/misc/mei/gsc-me.c | 2 +- + drivers/misc/mei/hw-me.c | 99 +- + drivers/misc/mei/init.c | 33 +- + drivers/misc/mei/mei_dev.h | 1 + + drivers/misc/mei/pci-csc.c | 141 +- + drivers/misc/mei/vsc-tp.c | 17 +- + drivers/misc/nsm.c | 7 +- + drivers/misc/pch_phub.c | 8 +- + drivers/misc/phantom.c | 8 +- + drivers/misc/tc9564-pci.c | 84 + + drivers/misc/tifm_7xx1.c | 8 +- + drivers/misc/tsl2550.c | 13 +- + drivers/pci/quirks.c | 1 + + drivers/peci/controller/peci-aspeed.c | 8 +- + drivers/peci/controller/peci-npcm.c | 6 +- + drivers/platform/goldfish/goldfish_pipe.c | 22 +- + drivers/pps/generators/pps_gen.c | 111 +- + .../iio/Documentation/sysfs-bus-iio-adc-ad7280a | 2 +- + drivers/staging/iio/Kconfig | 1 - + drivers/staging/iio/Makefile | 1 - + drivers/staging/iio/accel/Kconfig | 19 - + drivers/staging/iio/accel/Makefile | 6 - + drivers/staging/iio/accel/adis16203.c | 315 - + drivers/staging/iio/adc/ad7816.c | 33 +- + drivers/staging/iio/frequency/ad9832.c | 6 + + drivers/staging/iio/frequency/ad9834.c | 6 + + include/linux/iio/adc/qcom-adc5-gen3-common.h | 2 +- + include/linux/iio/backend.h | 6 + + include/linux/iio/iio-gts-helper.h | 11 +- + include/linux/iio/iio-opaque.h | 2 +- + include/linux/iio/iio.h | 25 +- + include/linux/notifier.h | 7 + + include/linux/pps_gen_kernel.h | 4 +- + include/uapi/linux/android/binder.h | 1 + + include/uapi/linux/iio/types.h | 1 + + kernel/notifier.c | 105 + + tools/iio/iio_event_monitor.c | 2 + + .../selftests/filesystems/binderfs/binderfs_test.c | 2 +- + .../testing/selftests/filesystems/binderfs/config | 3 +- + 290 files changed, 17178 insertions(+), 13981 deletions(-) + create mode 100644 Documentation/ABI/testing/debugfs-iio-ad9910 + create mode 100644 Documentation/ABI/testing/sysfs-bus-iio-adc + create mode 100644 Documentation/ABI/testing/sysfs-bus-iio-frequency-ad9910 + create mode 100644 Documentation/devicetree/bindings/iio/adc/adi,ad7768.yaml + create mode 100644 Documentation/devicetree/bindings/iio/adc/adi,max40080.yaml + create mode 100644 Documentation/devicetree/bindings/iio/adc/axiado,ax3000-saradc.yaml + create mode 100644 Documentation/devicetree/bindings/iio/adc/ti,ads112c04.yaml + create mode 100644 Documentation/devicetree/bindings/iio/dac/adi,ad5529r.yaml + create mode 100644 Documentation/devicetree/bindings/iio/dac/adi,ad5710r.yaml + create mode 100644 Documentation/devicetree/bindings/iio/frequency/adi,ad9910.yaml + create mode 100644 Documentation/devicetree/bindings/iio/light/capella,cm32181.yaml + create mode 100644 Documentation/devicetree/bindings/iio/proximity/pulsedlight,lidar-lite-v2.yaml + create mode 100644 Documentation/devicetree/bindings/misc/pci1179,0220.yaml + create mode 100644 Documentation/iio/ad7768.rst + create mode 100644 Documentation/iio/ad9910.rst + delete mode 100644 drivers/android/binder.c + delete mode 100644 drivers/android/binder_alloc.c + delete mode 100644 drivers/android/binder_alloc.h + delete mode 100644 drivers/android/binder_internal.h + delete mode 100644 drivers/android/binder_netlink.c + delete mode 100644 drivers/android/binder_netlink.h + delete mode 100644 drivers/android/binder_trace.h + delete mode 100644 drivers/android/binderfs.c + delete mode 100644 drivers/android/dbitmap.h + delete mode 100644 drivers/android/tests/.kunitconfig + delete mode 100644 drivers/android/tests/Makefile + delete mode 100644 drivers/android/tests/binder_alloc_kunit.c + create mode 100644 drivers/iio/adc/ad7768.c + create mode 100644 drivers/iio/adc/axiado_saradc.c + create mode 100644 drivers/iio/adc/max40080.c + create mode 100644 drivers/iio/adc/ti-ads112c04.c + create mode 100644 drivers/iio/dac/ad5529r.c + rename drivers/iio/dac/{mcp47feb02.c => mcp47feb02-core.c} (69%) + create mode 100644 drivers/iio/dac/mcp47feb02-i2c.c + create mode 100644 drivers/iio/dac/mcp47feb02-spi.c + create mode 100644 drivers/iio/dac/mcp47feb02.h + create mode 100644 drivers/iio/frequency/ad9910.c + create mode 100644 drivers/iio/light/veml6031x00.c + create mode 100644 drivers/iio/test/iio-test-channel-prefix.c + create mode 100644 drivers/misc/tc9564-pci.c + delete mode 100644 drivers/staging/iio/accel/Kconfig + delete mode 100644 drivers/staging/iio/accel/Makefile + delete mode 100644 drivers/staging/iio/accel/adis16203.c +Merging coresight/next (a328193461260 coresight: trbe: Hide enable_sink sysfs file) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/coresight/linux.git coresight/next +Merge made by the 'ort' strategy. + drivers/hwtracing/coresight/coresight-core.c | 1 + + drivers/hwtracing/coresight/coresight-sysfs.c | 15 +++++++++------ + drivers/hwtracing/coresight/coresight-trbe.c | 7 +++++++ + include/linux/coresight.h | 4 ++++ + 4 files changed, 21 insertions(+), 6 deletions(-) +Merging fastrpc/for-next (ef071c4906eb4 Merge branches 'fastrpc-fixes' and 'fastrpc-for-7.4' into fastrpc-for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/srini/fastrpc.git fastrpc/for-next +Auto-merging drivers/misc/fastrpc.c +Merge made by the 'ort' strategy. + drivers/misc/fastrpc.c | 200 ++++++++++++++++++++++++++----------------------- + 1 file changed, 107 insertions(+), 93 deletions(-) +Merging fpga/for-next (093da48782df3 fpga: altera-cvp: Retry teardown and reset CVP state on failure) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/fpga/linux-fpga.git fpga/for-next +Merge made by the 'ort' strategy. + drivers/fpga/altera-cvp.c | 43 ++++++++++++++++++++++++++++++++++------- + drivers/fpga/dfl.h | 1 - + drivers/fpga/fpga-mgr.c | 2 +- + drivers/fpga/stratix10-soc.c | 2 +- + drivers/fpga/xilinx-selectmap.c | 39 ++++++++++++++++++++++++++++--------- + 5 files changed, 68 insertions(+), 19 deletions(-) +Merging icc/icc-next (3fd56a2833c22 Merge branch 'icc-misc' into icc-next) +$ git merge -m Merge branch 'icc-next' of https://git.kernel.org/pub/scm/linux/kernel/git/djakov/icc.git icc/icc-next +Merge made by the 'ort' strategy. + .../bindings/interconnect/qcom,kuno-rpmh.yaml | 118 +++ + .../bindings/interconnect/qcom,osm-l3.yaml | 1 + + drivers/interconnect/core.c | 32 + + drivers/interconnect/mediatek/Makefile | 2 +- + drivers/interconnect/qcom/Kconfig | 57 ++ + drivers/interconnect/qcom/Makefile | 2 + + drivers/interconnect/qcom/bcm-voter.c | 59 +- + drivers/interconnect/qcom/bcm-voter.h | 1 + + drivers/interconnect/qcom/eliza.c | 18 +- + drivers/interconnect/qcom/icc-common.h | 9 + + drivers/interconnect/qcom/icc-rpm.c | 105 ++- + drivers/interconnect/qcom/icc-rpm.h | 6 +- + drivers/interconnect/qcom/icc-rpmh.c | 71 +- + drivers/interconnect/qcom/kuno.c | 988 +++++++++++++++++++++ + drivers/interconnect/qcom/milos.c | 66 +- + drivers/interconnect/qcom/msm8976.c | 3 +- + drivers/interconnect/qcom/msm8996.c | 1 + + drivers/interconnect/qcom/sm6350.c | 56 +- + drivers/interconnect/qcom/sm8650.c | 70 +- + drivers/interconnect/qcom/smd-rpm.c | 8 +- + drivers/interconnect/samsung/exynos.c | 1 + + include/dt-bindings/interconnect/qcom,kuno.h | 89 ++ + 22 files changed, 1608 insertions(+), 155 deletions(-) + create mode 100644 Documentation/devicetree/bindings/interconnect/qcom,kuno-rpmh.yaml + create mode 100644 drivers/interconnect/qcom/kuno.c + create mode 100644 include/dt-bindings/interconnect/qcom,kuno.h +Merging iio/togreg (a3b3580713f3a iio: dac: ad5758: Fix the offset calculation) +$ git merge -m Merge branch 'togreg' of https://git.kernel.org/pub/scm/linux/kernel/git/jic23/iio.git iio/togreg +Already up to date. +Merging nfc/for-next (fd73f4a665989 Linux 7.3-rc3) +$ git merge -m Merge branch 'for-next' of https://codeberg.org/linux-nfc/linux.git nfc/for-next +Already up to date. +Merging phy-next/next (c7f2322431cb6 dt-bindings: phy: ti,tcan104x-can: Fix property constrains) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/phy/linux-phy.git phy-next/next +Auto-merging Documentation/devicetree/bindings/phy/qcom,x1e80100-csi2-phy.yaml +CONFLICT (add/add): Merge conflict in Documentation/devicetree/bindings/phy/qcom,x1e80100-csi2-phy.yaml +Auto-merging MAINTAINERS +Auto-merging include/linux/phy/phy.h +Resolved 'Documentation/devicetree/bindings/phy/qcom,x1e80100-csi2-phy.yaml' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 2d48989500e24] Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/phy/linux-phy.git +$ git diff -M --stat --summary HEAD^.. + .../bindings/phy/mediatek,mt8195-dp-phy.yaml | 77 ++ + .../devicetree/bindings/phy/mediatek,tphy.yaml | 1 + + .../bindings/phy/qcom,m31-eusb2-phy.yaml | 1 + + .../bindings/phy/qcom,sc8280xp-qmp-pcie-phy.yaml | 2 + + .../phy/qcom,sc8280xp-qmp-usb43dp-phy.yaml | 59 +- + .../devicetree/bindings/phy/ti,tcan104x-can.yaml | 9 +- + drivers/phy/apple/atc.c | 2 - + drivers/phy/broadcom/phy-brcm-usb.c | 29 +- + drivers/phy/cadence/phy-cadence-sierra.c | 11 +- + drivers/phy/hisilicon/phy-hi3670-pcie.c | 2 + + drivers/phy/mediatek/phy-mtk-dp.c | 826 ++++++++++++++++++--- + drivers/phy/qualcomm/phy-qcom-qmp-combo.c | 402 ++++++++-- + drivers/phy/qualcomm/phy-qcom-qmp-pcie.c | 73 ++ + drivers/phy/qualcomm/phy-qcom-qmp-pcs-v8_50.h | 4 +- + drivers/phy/samsung/phy-exynos5-usbdrd.c | 78 +- + drivers/phy/socionext/phy-uniphier-usb3hs.c | 3 +- + drivers/phy/starfive/phy-jh7110-dphy-tx.c | 16 +- + include/dt-bindings/phy/phy-qcom-qmp.h | 1 + + include/linux/phy/phy-thunderbolt.h | 14 + + include/linux/phy/phy.h | 2 + + 20 files changed, 1387 insertions(+), 225 deletions(-) + create mode 100644 Documentation/devicetree/bindings/phy/mediatek,mt8195-dp-phy.yaml + create mode 100644 include/linux/phy/phy-thunderbolt.h +Merging soundwire/next (90b63b309fd6c soundwire: Intel: stop sdw clock in system suspend) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/soundwire.git soundwire/next +Auto-merging drivers/soundwire/dmi-quirks.c +Merge made by the 'ort' strategy. + drivers/soundwire/bus.c | 54 +++++++++++++++++++++++------------- + drivers/soundwire/bus_type.c | 16 +++++++---- + drivers/soundwire/dmi-quirks.c | 14 ++++++++++ + drivers/soundwire/intel.h | 6 ++-- + drivers/soundwire/intel_auxdevice.c | 11 +++++--- + drivers/soundwire/intel_bus_common.c | 8 +++--- + drivers/soundwire/qcom.c | 38 ++++++++++++++++++------- + drivers/soundwire/slave.c | 28 +++++++++++++------ + drivers/soundwire/stream.c | 8 +++--- + include/linux/soundwire/sdw.h | 4 +++ + include/linux/soundwire/sdw_intel.h | 2 +- + 11 files changed, 129 insertions(+), 60 deletions(-) +Merging extcon/extcon-next (8d3ae59288f1e Linux 7.2) +$ git merge -m Merge branch 'extcon-next' of https://git.kernel.org/pub/scm/linux/kernel/git/chanwoo/extcon.git extcon/extcon-next +Already up to date. +Merging gnss/gnss-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'gnss-next' of https://git.kernel.org/pub/scm/linux/kernel/git/johan/gnss.git gnss/gnss-next +Already up to date. +Merging vfio/next (b30b52c2fb82e vfio/pci: Restore 8-byte ioeventfd support) +$ git merge -m Merge branch 'next' of https://github.com/awilliam/linux-vfio.git vfio/next +Auto-merging Documentation/driver-api/index.rst +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + Documentation/driver-api/index.rst | 1 + + Documentation/driver-api/vfio-selftests.rst | 252 +++ + MAINTAINERS | 1 + + drivers/vfio/cdx/main.c | 17 +- + drivers/vfio/device_cdev.c | 12 + + drivers/vfio/fsl-mc/vfio_fsl_mc.c | 15 +- + drivers/vfio/mdev/mdev_core.c | 2 +- + drivers/vfio/pci/vfio_pci_rdwr.c | 3 - + drivers/vfio/platform/vfio_platform_common.c | 15 +- + drivers/vfio/vfio_iommu_type1.c | 12 +- + drivers/vfio/vfio_main.c | 7 - + include/linux/mlx5/cq.h | 10 - + include/linux/mlx5/device.h | 231 +-- + include/linux/mlx5/mlx5_ifc.h | 178 ++ + include/linux/mlx5/mlx5_ifc_macros.h | 185 ++ + tools/arch/arm64/include/asm/barrier.h | 4 + + tools/arch/x86/include/asm/barrier.h | 5 + + tools/include/asm-generic/io.h | 28 + + tools/include/asm/barrier.h | 8 + + tools/include/linux/stddef.h | 10 + + .../selftests/kvm/include/arm64/processor.h | 4 +- + tools/testing/selftests/kvm/irq_test.c | 2 - + tools/testing/selftests/vfio/.gitignore | 1 + + tools/testing/selftests/vfio/lib/drivers/dsa/dsa.c | 1 + + tools/testing/selftests/vfio/lib/drivers/igb/igb.c | 1 + + .../testing/selftests/vfio/lib/drivers/ioat/ioat.c | 1 + + .../testing/selftests/vfio/lib/drivers/mlx5/mlx5.c | 1928 ++++++++++++++++++++ + .../selftests/vfio/lib/drivers/mlx5/mlx5_hw.h | 114 ++ + .../selftests/vfio/lib/drivers/mlx5/mlx5_ifc.h | 1 + + .../vfio/lib/drivers/mlx5/mlx5_ifc_fpga.h | 1 + + .../vfio/lib/drivers/mlx5/mlx5_ifc_macros.h | 1 + + .../vfio/lib/drivers/nv_falcon/nv_falcon.c | 1 + + .../vfio/lib/include/libvfio/vfio_pci_device.h | 11 + + .../vfio/lib/include/libvfio/vfio_pci_driver.h | 6 + + tools/testing/selftests/vfio/lib/iova_allocator.c | 7 +- + tools/testing/selftests/vfio/lib/libvfio.mk | 1 + + tools/testing/selftests/vfio/lib/sysfs.c | 1 + + tools/testing/selftests/vfio/lib/vfio_pci_driver.c | 6 + + tools/testing/selftests/vfio/settings | 5 + + .../testing/selftests/vfio/vfio_pci_driver_test.c | 5 +- + .../selftests/vfio/vfio_pci_sriov_uapi_test.c | 35 + + 41 files changed, 2850 insertions(+), 279 deletions(-) + create mode 100644 Documentation/driver-api/vfio-selftests.rst + create mode 100644 include/linux/mlx5/mlx5_ifc_macros.h + create mode 100644 tools/include/linux/stddef.h + create mode 100644 tools/testing/selftests/vfio/lib/drivers/mlx5/mlx5.c + create mode 100644 tools/testing/selftests/vfio/lib/drivers/mlx5/mlx5_hw.h + create mode 120000 tools/testing/selftests/vfio/lib/drivers/mlx5/mlx5_ifc.h + create mode 120000 tools/testing/selftests/vfio/lib/drivers/mlx5/mlx5_ifc_fpga.h + create mode 120000 tools/testing/selftests/vfio/lib/drivers/mlx5/mlx5_ifc_macros.h + create mode 100644 tools/testing/selftests/vfio/settings +Merging w1/for-next (813a5b9a3c9f6 w1: fix spelling mistakes in comments across the subsystem) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-w1.git w1/for-next +Merge made by the 'ort' strategy. + drivers/w1/masters/ds2482.c | 2 +- + drivers/w1/masters/ds2490.c | 4 ++-- + drivers/w1/masters/omap_hdq.c | 2 +- + drivers/w1/slaves/w1_ds28e17.c | 2 +- + drivers/w1/slaves/w1_therm.c | 2 +- + drivers/w1/w1.c | 2 +- + drivers/w1/w1_family.c | 2 +- + drivers/w1/w1_netlink.h | 2 +- + 8 files changed, 9 insertions(+), 9 deletions(-) +Merging spmi/spmi-next (8cdeaa50eae8d Linux 7.2-rc2) +$ git merge -m Merge branch 'spmi-next' of https://git.kernel.org/pub/scm/linux/kernel/git/sboyd/spmi.git spmi/spmi-next +Already up to date. +Merging staging/staging-next (dbc2fa996f446 staging: rtl8723bs: Remove manual ifname allocation) +$ git merge -m Merge branch 'staging-next' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/staging.git staging/staging-next +Auto-merging CREDITS +Auto-merging MAINTAINERS +Auto-merging drivers/staging/greybus/uart.c +Auto-merging drivers/staging/rtl8723bs/os_dep/ioctl_cfg80211.c +Merge made by the 'ort' strategy. + CREDITS | 4 + + MAINTAINERS | 7 - + drivers/staging/axis-fifo/axis-fifo.c | 6 +- + drivers/staging/fbtft/fb_bd663474.c | 4 +- + drivers/staging/greybus/Kconfig | 19 - + drivers/staging/greybus/Makefile | 8 - + drivers/staging/greybus/TODO | 0 + drivers/staging/greybus/arche-apb-ctrl.c | 490 ------- + drivers/staging/greybus/arche-platform.c | 653 ---------- + drivers/staging/greybus/arche_platform.h | 28 - + drivers/staging/greybus/camera.c | 1367 -------------------- + drivers/staging/greybus/gb-camera.h | 127 -- + drivers/staging/greybus/gbphy.c | 2 +- + drivers/staging/greybus/loopback.c | 5 +- + drivers/staging/greybus/uart.c | 32 +- + drivers/staging/most/video/video.c | 1 - + drivers/staging/octeon/ethernet-tx.c | 2 +- + drivers/staging/rtl8723bs/core/rtw_ap.c | 6 +- + drivers/staging/rtl8723bs/core/rtw_cmd.c | 46 +- + drivers/staging/rtl8723bs/core/rtw_efuse.c | 150 +-- + drivers/staging/rtl8723bs/core/rtw_ieee80211.c | 10 - + drivers/staging/rtl8723bs/core/rtw_ioctl_set.c | 3 - + drivers/staging/rtl8723bs/core/rtw_mlme.c | 1018 ++++++++------- + drivers/staging/rtl8723bs/core/rtw_mlme_ext.c | 218 +--- + drivers/staging/rtl8723bs/core/rtw_pwrctrl.c | 3 - + drivers/staging/rtl8723bs/core/rtw_recv.c | 146 +-- + drivers/staging/rtl8723bs/core/rtw_security.c | 99 +- + drivers/staging/rtl8723bs/core/rtw_sta_mgt.c | 32 - + drivers/staging/rtl8723bs/core/rtw_wlan_util.c | 23 +- + drivers/staging/rtl8723bs/core/rtw_xmit.c | 56 +- + drivers/staging/rtl8723bs/hal/Hal8723BReg.h | 17 +- + drivers/staging/rtl8723bs/hal/HalBtc8723b1Ant.c | 16 - + drivers/staging/rtl8723bs/hal/HalBtc8723b1Ant.h | 2 - + drivers/staging/rtl8723bs/hal/HalBtc8723b2Ant.c | 8 - + drivers/staging/rtl8723bs/hal/HalBtc8723b2Ant.h | 2 - + drivers/staging/rtl8723bs/hal/HalBtcOutSrc.h | 3 +- + drivers/staging/rtl8723bs/hal/HalPhyRf.c | 11 +- + drivers/staging/rtl8723bs/hal/HalPhyRf_8723B.c | 77 -- + drivers/staging/rtl8723bs/hal/HalPhyRf_8723B.h | 6 +- + drivers/staging/rtl8723bs/hal/HalPwrSeqCmd.c | 6 - + drivers/staging/rtl8723bs/hal/hal_btcoex.c | 52 - + drivers/staging/rtl8723bs/hal/hal_com.c | 6 +- + drivers/staging/rtl8723bs/hal/hal_com_phycfg.c | 10 +- + drivers/staging/rtl8723bs/hal/hal_intf.c | 9 +- + drivers/staging/rtl8723bs/hal/odm.c | 34 +- + drivers/staging/rtl8723bs/hal/odm.h | 51 +- + drivers/staging/rtl8723bs/hal/odm_DIG.h | 1 - + drivers/staging/rtl8723bs/hal/odm_DynamicTxPower.h | 1 - + drivers/staging/rtl8723bs/hal/odm_EdcaTurboCheck.c | 4 +- + drivers/staging/rtl8723bs/hal/odm_HWConfig.c | 7 - + drivers/staging/rtl8723bs/hal/odm_HWConfig.h | 2 - + drivers/staging/rtl8723bs/hal/odm_precomp.h | 1 - + drivers/staging/rtl8723bs/hal/rtl8723b_cmd.c | 2 - + drivers/staging/rtl8723bs/hal/rtl8723b_dm.c | 14 - + drivers/staging/rtl8723bs/hal/rtl8723b_hal_init.c | 76 +- + drivers/staging/rtl8723bs/hal/rtl8723b_phycfg.c | 25 +- + drivers/staging/rtl8723bs/hal/rtl8723b_rf6052.c | 30 +- + drivers/staging/rtl8723bs/hal/rtl8723b_rxdesc.c | 13 +- + drivers/staging/rtl8723bs/hal/rtl8723bs_recv.c | 9 +- + drivers/staging/rtl8723bs/hal/rtl8723bs_xmit.c | 15 +- + drivers/staging/rtl8723bs/hal/sdio_halinit.c | 58 - + drivers/staging/rtl8723bs/hal/sdio_ops.c | 24 +- + drivers/staging/rtl8723bs/include/HalPwrSeqCmd.h | 2 - + drivers/staging/rtl8723bs/include/basic_types.h | 8 - + drivers/staging/rtl8723bs/include/cmd_osdep.h | 6 +- + drivers/staging/rtl8723bs/include/drv_types.h | 6 - + drivers/staging/rtl8723bs/include/hal_com.h | 7 +- + drivers/staging/rtl8723bs/include/hal_com_h2c.h | 2 - + drivers/staging/rtl8723bs/include/hal_com_phycfg.h | 2 +- + drivers/staging/rtl8723bs/include/hal_com_reg.h | 92 +- + drivers/staging/rtl8723bs/include/hal_data.h | 23 - + drivers/staging/rtl8723bs/include/hal_intf.h | 2 +- + drivers/staging/rtl8723bs/include/hal_pg.h | 2 +- + drivers/staging/rtl8723bs/include/hal_phy.h | 19 +- + drivers/staging/rtl8723bs/include/ieee80211.h | 10 - + drivers/staging/rtl8723bs/include/ioctl_cfg80211.h | 1 - + drivers/staging/rtl8723bs/include/osdep_intf.h | 1 - + drivers/staging/rtl8723bs/include/osdep_service.h | 4 +- + .../rtl8723bs/include/osdep_service_linux.h | 2 +- + drivers/staging/rtl8723bs/include/rtl8723b_cmd.h | 9 +- + drivers/staging/rtl8723bs/include/rtl8723b_dm.h | 10 +- + drivers/staging/rtl8723bs/include/rtl8723b_hal.h | 6 +- + drivers/staging/rtl8723bs/include/rtl8723b_spec.h | 72 +- + drivers/staging/rtl8723bs/include/rtl8723b_xmit.h | 10 +- + drivers/staging/rtl8723bs/include/rtw_ap.h | 1 - + drivers/staging/rtl8723bs/include/rtw_cmd.h | 463 ++++--- + drivers/staging/rtl8723bs/include/rtw_eeprom.h | 2 - + drivers/staging/rtl8723bs/include/rtw_efuse.h | 9 +- + drivers/staging/rtl8723bs/include/rtw_ht.h | 2 - + drivers/staging/rtl8723bs/include/rtw_io.h | 14 +- + drivers/staging/rtl8723bs/include/rtw_ioctl_set.h | 2 - + drivers/staging/rtl8723bs/include/rtw_mlme.h | 89 +- + drivers/staging/rtl8723bs/include/rtw_mlme_ext.h | 46 +- + drivers/staging/rtl8723bs/include/rtw_pwrctrl.h | 21 +- + drivers/staging/rtl8723bs/include/rtw_recv.h | 85 +- + drivers/staging/rtl8723bs/include/rtw_security.h | 3 +- + drivers/staging/rtl8723bs/include/rtw_xmit.h | 58 +- + drivers/staging/rtl8723bs/include/sdio_ops.h | 23 +- + drivers/staging/rtl8723bs/include/sta_info.h | 92 +- + drivers/staging/rtl8723bs/include/wlan_bssdef.h | 23 +- + drivers/staging/rtl8723bs/include/xmit_osdep.h | 14 +- + drivers/staging/rtl8723bs/os_dep/ioctl_cfg80211.c | 130 +- + drivers/staging/rtl8723bs/os_dep/os_intfs.c | 61 +- + drivers/staging/rtl8723bs/os_dep/sdio_intf.c | 4 +- + drivers/staging/rtl8723bs/os_dep/sdio_ops_linux.c | 1 - + drivers/staging/rtl8723bs/os_dep/xmit_linux.c | 3 +- + drivers/staging/sm750fb/sm750.c | 22 +- + drivers/staging/sm750fb/sm750_accel.h | 4 +- + drivers/staging/vme_user/vme_tsi148.c | 3 +- + 109 files changed, 1336 insertions(+), 5292 deletions(-) + delete mode 100644 drivers/staging/greybus/TODO + delete mode 100644 drivers/staging/greybus/arche-apb-ctrl.c + delete mode 100644 drivers/staging/greybus/arche-platform.c + delete mode 100644 drivers/staging/greybus/arche_platform.h + delete mode 100644 drivers/staging/greybus/camera.c + delete mode 100644 drivers/staging/greybus/gb-camera.h +Merging counter-next/counter-next (edac5cf356994 MAINTAINERS: Mark ftm-quaddec driver as orphaned) +$ git merge -m Merge branch 'counter-next' of https://git.kernel.org/pub/scm/linux/kernel/git/wbg/counter.git counter-next/counter-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 6 ++---- + 1 file changed, 2 insertions(+), 4 deletions(-) +Merging mux/for-next (ac7bde3c53166 mux: Add driver for Renesas RZ/V2H VBENCTL VBUS_SEL mux) +$ git merge -m Merge branch 'for-next' of https://gitlab.com/peda-linux/mux.git mux/for-next +Merge made by the 'ort' strategy. + drivers/mux/Kconfig | 13 ++++++++ + drivers/mux/Makefile | 2 ++ + drivers/mux/rzv2h-vbenctl.c | 81 +++++++++++++++++++++++++++++++++++++++++++++ + 3 files changed, 96 insertions(+) + create mode 100644 drivers/mux/rzv2h-vbenctl.c +Merging dmaengine/next (0a8dda0a15d39 dmaengine: bestcomm: use devm_platform_get_and_ioremap_resource() to simplify code) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/dmaengine.git dmaengine/next +CONFLICT (modify/delete): Documentation/devicetree/bindings/usb/am33xx-usb.txt deleted in HEAD and modified in dmaengine/next. Version dmaengine/next of Documentation/devicetree/bindings/usb/am33xx-usb.txt left in tree. +CONFLICT (modify/delete): Documentation/devicetree/bindings/usb/da8xx-usb.txt deleted in HEAD and modified in dmaengine/next. Version dmaengine/next of Documentation/devicetree/bindings/usb/da8xx-usb.txt left in tree. +Auto-merging MAINTAINERS +Auto-merging drivers/dma/dmaengine.c +Auto-merging drivers/dma/mmp_pdma.c +Auto-merging drivers/dma/pxa_dma.c +Auto-merging drivers/dma/sprd-dma.c +Auto-merging drivers/dma/sun6i-dma.c +Auto-merging drivers/dma/switchtec_dma.c +Auto-merging drivers/dma/xilinx/xilinx_dma.c +Auto-merging drivers/pci/controller/dwc/pcie-designware.c +Auto-merging include/linux/dmaengine.h +Automatic merge failed; fix conflicts and then commit the result. +$ git rm -f Documentation/devicetree/bindings/usb/am33xx-usb.txt +rm 'Documentation/devicetree/bindings/usb/am33xx-usb.txt' +$ git commit --no-edit -v -a +[master 2ff4c6346f73d] Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/dmaengine.git +$ git diff -M --stat --summary HEAD^.. + .../devicetree/bindings/dma/apple,admac.yaml | 10 +- + .../devicetree/bindings/dma/snps,dw-axi-dmac.yaml | 22 ++ + .../bindings/dma/ti,dra7-dma-crossbar.yaml | 94 ++++++ + .../devicetree/bindings/dma/ti-dma-crossbar.txt | 68 ---- + Documentation/driver-api/dmaengine/provider.rst | 2 +- + MAINTAINERS | 2 +- + drivers/dma/altera-msgdma.c | 10 +- + drivers/dma/amba-pl08x.c | 4 +- + drivers/dma/apple-admac.c | 24 ++ + drivers/dma/arm-dma350.c | 2 +- + drivers/dma/at_hdmac.c | 139 ++++---- + drivers/dma/at_xdmac.c | 159 +++++----- + drivers/dma/bestcomm/ata.c | 2 - + drivers/dma/bestcomm/bestcomm.c | 46 +-- + drivers/dma/bestcomm/gen_bd.c | 23 +- + drivers/dma/dma-jz4780.c | 10 +- + drivers/dma/dmaengine.c | 34 +- + drivers/dma/dw-axi-dmac/dw-axi-dmac-platform.c | 151 +++++---- + drivers/dma/dw-axi-dmac/dw-axi-dmac.h | 57 ++-- + drivers/dma/dw-edma/dw-edma-core.c | 349 ++++++++++++++------- + drivers/dma/dw-edma/dw-edma-core.h | 67 +++- + drivers/dma/dw-edma/dw-edma-pcie.c | 18 ++ + drivers/dma/dw-edma/dw-edma-v0-core.c | 59 ++-- + drivers/dma/dw-edma/dw-hdma-v0-core.c | 82 +++-- + drivers/dma/dw-edma/dw-hdma-v0-debugfs.c | 17 +- + drivers/dma/dw-edma/dw-hdma-v0-regs.h | 10 - + drivers/dma/dw/core.c | 47 ++- + drivers/dma/ep93xx_dma.c | 37 +-- + drivers/dma/fsl-edma-main.c | 5 +- + drivers/dma/fsl_raid.c | 38 ++- + drivers/dma/fsldma.c | 22 +- + drivers/dma/idma64.c | 10 +- + drivers/dma/img-mdc-dma.c | 2 +- + drivers/dma/loongson/loongson1-apb-dma.c | 23 +- + drivers/dma/loongson/loongson2-apb-cmc-dma.c | 19 +- + drivers/dma/loongson/loongson2-apb-dma.c | 9 +- + drivers/dma/mediatek/mtk-hsdma.c | 26 +- + drivers/dma/mmp_pdma.c | 3 +- + drivers/dma/moxart-dma.c | 23 +- + drivers/dma/mv_xor.c | 8 +- + drivers/dma/nbpfaxi.c | 2 +- + drivers/dma/owl-dma.c | 21 +- + drivers/dma/pch_dma.c | 41 ++- + drivers/dma/pl330.c | 4 +- + drivers/dma/pxa_dma.c | 52 +-- + drivers/dma/qcom/gpi.c | 9 +- + drivers/dma/sprd-dma.c | 45 +-- + drivers/dma/st_fdma.c | 2 +- + drivers/dma/ste_dma40.c | 15 +- + drivers/dma/stm32/stm32-dma.c | 69 ++-- + drivers/dma/stm32/stm32-dma3.c | 99 +++--- + drivers/dma/stm32/stm32-mdma.c | 96 +++--- + drivers/dma/sun4i-dma.c | 17 +- + drivers/dma/sun6i-dma.c | 62 ++-- + drivers/dma/switchtec_dma.c | 14 +- + drivers/dma/tegra186-gpc-dma.c | 4 +- + drivers/dma/tegra20-apb-dma.c | 2 +- + drivers/dma/ti/k3-udma.c | 6 +- + drivers/dma/timb_dma.c | 63 ++-- + drivers/dma/txx9dmac.c | 89 +++--- + drivers/dma/virt-dma.h | 16 + + drivers/dma/xilinx/xilinx_dma.c | 85 ++++- + drivers/dma/xilinx/zynqmp_dma.c | 74 +++-- + drivers/pci/controller/dwc/pcie-designware.c | 1 + + include/linux/dma/edma.h | 3 +- + include/linux/dma/imx-dma.h | 5 - + include/linux/dmaengine.h | 23 +- + include/linux/fsl/bestcomm/gen_bd.h | 8 - + include/trace/events/tegra_apb_dma.h | 6 +- + 69 files changed, 1481 insertions(+), 1185 deletions(-) + create mode 100644 Documentation/devicetree/bindings/dma/ti,dra7-dma-crossbar.yaml + delete mode 100644 Documentation/devicetree/bindings/dma/ti-dma-crossbar.txt +Merging cgroup/for-next (7965da4a7fc0a Merge branch 'for-7.3-fixes' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/tj/cgroup.git cgroup/for-next +Auto-merging Documentation/admin-guide/cgroup-v2.rst +Auto-merging block/blk-cgroup.c +Auto-merging kernel/cgroup/cgroup-internal.h +Auto-merging kernel/cgroup/cgroup-v1.c +Auto-merging kernel/cgroup/dmem.c +Merge made by the 'ort' strategy. + Documentation/admin-guide/cgroup-v2.rst | 62 ++++++-- + block/blk-cgroup.c | 11 +- + include/linux/cgroup-defs.h | 8 + + kernel/cgroup/cgroup-internal.h | 2 +- + kernel/cgroup/cgroup-v1.c | 26 ++-- + kernel/cgroup/cgroup.c | 181 ++++++++++++++--------- + kernel/cgroup/cpuset-internal.h | 8 +- + kernel/cgroup/cpuset-v1.c | 59 +++++++- + kernel/cgroup/cpuset.c | 176 ++++++---------------- + kernel/cgroup/debug.c | 18 +-- + kernel/cgroup/dmem.c | 4 +- + kernel/cgroup/freezer.c | 5 +- + kernel/cgroup/namespace.c | 2 - + tools/testing/selftests/cgroup/lib/cgroup_util.c | 45 +++++- + tools/testing/selftests/cgroup/settings | 1 + + tools/testing/selftests/cgroup/test_cpu.c | 5 +- + tools/testing/selftests/cgroup/with_stress.sh | 2 +- + 17 files changed, 354 insertions(+), 261 deletions(-) + create mode 100644 tools/testing/selftests/cgroup/settings +Merging scsi/for-next (6147f16c23efb Merge branch 'misc' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/jejb/scsi.git scsi/for-next +Auto-merging drivers/ata/libata-scsi.c +CONFLICT (content): Merge conflict in drivers/ata/libata-scsi.c +Auto-merging drivers/scsi/scsi_scan.c +Auto-merging drivers/ufs/host/ufs-qcom.c +Auto-merging drivers/usb/gadget/function/f_mass_storage.c +Resolved 'drivers/ata/libata-scsi.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master d6e36adb55a03] Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/jejb/scsi.git +$ git diff -M --stat --summary HEAD^.. + Documentation/scsi/scsi_mid_low_api.rst | 8 +- + drivers/ata/libata-eh.c | 17 +- + drivers/ata/libata-sata.c | 20 +- + drivers/ata/libata-scsi.c | 344 ++-- + drivers/ata/libata.h | 5 +- + drivers/cdrom/cdrom.c | 10 +- + drivers/message/fusion/mptfc.c | 8 +- + drivers/s390/scsi/zfcp_erp.c | 30 +- + drivers/s390/scsi/zfcp_ext.h | 2 +- + drivers/s390/scsi/zfcp_fsf.c | 9 +- + drivers/s390/scsi/zfcp_scsi.c | 6 +- + drivers/s390/scsi/zfcp_sysfs.c | 4 +- + drivers/scsi/3w-9xxx.c | 20 +- + drivers/scsi/3w-sas.c | 28 +- + drivers/scsi/3w-xxxx.c | 27 +- + drivers/scsi/53c700.c | 16 +- + drivers/scsi/BusLogic.c | 20 +- + drivers/scsi/a100u2w.c | 4 +- + drivers/scsi/a2091.c | 4 +- + drivers/scsi/a3000.c | 4 +- + drivers/scsi/aacraid/commsup.c | 8 +- + drivers/scsi/advansys.c | 8 +- + drivers/scsi/aha1542.c | 20 +- + drivers/scsi/aha1740.c | 12 +- + drivers/scsi/arm/eesox.c | 8 +- + drivers/scsi/arm/fas216.c | 20 +- + drivers/scsi/atp870u.c | 12 +- + drivers/scsi/bfa/bfad_bsg.c | 4 +- + drivers/scsi/ch.c | 55 +- + drivers/scsi/constants.c | 61 +- + drivers/scsi/csiostor/csio_attr.c | 4 +- + drivers/scsi/csiostor/csio_scsi.c | 4 +- + drivers/scsi/dc395x.c | 8 +- + drivers/scsi/device_handler/scsi_dh_alua.c | 55 +- + drivers/scsi/device_handler/scsi_dh_emc.c | 28 +- + drivers/scsi/device_handler/scsi_dh_hp_sw.c | 29 +- + drivers/scsi/device_handler/scsi_dh_rdac.c | 55 +- + drivers/scsi/esp_scsi.c | 32 +- + drivers/scsi/fcoe/fcoe.c | 4 +- + drivers/scsi/fdomain.c | 16 +- + drivers/scsi/gvp11.c | 4 +- + drivers/scsi/hosts.c | 17 +- + drivers/scsi/hpsa.c | 69 +- + drivers/scsi/hpsa_cmd.h | 23 - + drivers/scsi/hptiop.c | 8 +- + drivers/scsi/ibmvscsi/ibmvfc-core.c | 262 +-- + drivers/scsi/ibmvscsi/ibmvfc-nvme.c | 8 +- + drivers/scsi/ibmvscsi/ibmvscsi.c | 86 +- + drivers/scsi/ibmvscsi_tgt/ibmvscsi_tgt.c | 5 +- + drivers/scsi/imm.c | 4 +- + drivers/scsi/initio.c | 8 +- + drivers/scsi/ipr.c | 324 ++-- + drivers/scsi/ips.c | 26 +- + drivers/scsi/leapraid/leapraid_func.h | 7 - + drivers/scsi/leapraid/leapraid_os.c | 15 +- + drivers/scsi/libfc/fc_fcp.c | 8 +- + drivers/scsi/libiscsi.c | 3 +- + drivers/scsi/libsas/sas_scsi_host.c | 10 +- + drivers/scsi/lpfc/lpfc_els.c | 30 +- + drivers/scsi/lpfc/lpfc_hbadisc.c | 44 +- + drivers/scsi/lpfc/lpfc_init.c | 16 +- + drivers/scsi/lpfc/lpfc_scsi.c | 26 +- + drivers/scsi/lpfc/lpfc_sli.c | 4 +- + drivers/scsi/mac53c94.c | 8 +- + drivers/scsi/megaraid.c | 8 +- + drivers/scsi/megaraid/mega_common.h | 2 - + drivers/scsi/megaraid/megaraid_mbox.c | 12 +- + drivers/scsi/megaraid/megaraid_sas_base.c | 18 +- + drivers/scsi/mesh.c | 24 +- + drivers/scsi/mpi3mr/mpi/mpi30_cnfg.h | 77 +- + drivers/scsi/mpi3mr/mpi/mpi30_image.h | 7 +- + drivers/scsi/mpi3mr/mpi/mpi30_ioc.h | 15 +- + drivers/scsi/mpi3mr/mpi/mpi30_transport.h | 2 +- + drivers/scsi/mpi3mr/mpi3mr.h | 13 +- + drivers/scsi/mpi3mr/mpi3mr_app.c | 87 +- + drivers/scsi/mpi3mr/mpi3mr_fw.c | 223 ++- + drivers/scsi/mpi3mr/mpi3mr_os.c | 320 ++-- + drivers/scsi/mpi3mr/mpi3mr_transport.c | 106 +- + drivers/scsi/mpt3sas/mpt3sas_scsih.c | 69 +- + drivers/scsi/mvumi.c | 28 +- + drivers/scsi/myrb.c | 67 +- + drivers/scsi/myrs.c | 23 +- + drivers/scsi/nsp32.c | 8 +- + drivers/scsi/pcmcia/sym53c500_cs.c | 8 +- + drivers/scsi/pm8001/pm8001_defs.h | 98 ++ + drivers/scsi/pm8001/pm8001_hwi.c | 2 +- + drivers/scsi/pm8001/pm8001_hwi.h | 4 +- + drivers/scsi/pm8001/pm80xx_hwi.c | 2 +- + drivers/scsi/pm8001/pm80xx_hwi.h | 100 +- + drivers/scsi/pmcraid.c | 84 +- + drivers/scsi/ps3rom.c | 5 +- + drivers/scsi/qla1280.c | 44 +- + drivers/scsi/qla2xxx/qla_attr.c | 4 +- + drivers/scsi/qla2xxx/qla_init.c | 4 +- + drivers/scsi/qla2xxx/qla_isr.c | 9 +- + drivers/scsi/qla2xxx/qla_target.c | 99 +- + drivers/scsi/qla2xxx/qla_target.h | 3 - + drivers/scsi/qlogicfas408.c | 8 +- + drivers/scsi/qlogicpti.c | 20 +- + drivers/scsi/scsi.c | 15 +- + drivers/scsi/scsi_common.c | 24 +- + drivers/scsi/scsi_debug.c | 514 ++++-- + drivers/scsi/scsi_debugfs.c | 2 +- + drivers/scsi/scsi_devinfo.c | 5 +- + drivers/scsi/scsi_error.c | 164 +- + drivers/scsi/scsi_ioctl.c | 3 +- + drivers/scsi/scsi_lib.c | 179 +- + drivers/scsi/scsi_lib_test.c | 95 +- + drivers/scsi/scsi_logging.c | 22 +- + drivers/scsi/scsi_proc.c | 9 +- + drivers/scsi/scsi_scan.c | 38 +- + drivers/scsi/scsi_sysfs.c | 28 +- + drivers/scsi/scsi_transport_fc.c | 136 +- + drivers/scsi/scsi_transport_spi.c | 12 +- + drivers/scsi/sd.c | 126 +- + drivers/scsi/sd_zbc.c | 2 +- + drivers/scsi/sense_codes.h | 2352 +++++++++++++++++--------- + drivers/scsi/ses.c | 23 +- + drivers/scsi/sgiwd93.c | 4 +- + drivers/scsi/smartpqi/smartpqi_init.c | 31 +- + drivers/scsi/snic/snic_disc.c | 16 +- + drivers/scsi/sr.c | 3 +- + drivers/scsi/sr_ioctl.c | 29 +- + drivers/scsi/st.c | 35 +- + drivers/scsi/stex.c | 51 +- + drivers/scsi/storvsc_drv.c | 15 +- + drivers/scsi/sym53c8xx_2/sym_glue.c | 56 +- + drivers/scsi/sym53c8xx_2/sym_hipd.h | 4 +- + drivers/scsi/wd33c93.c | 4 +- + drivers/scsi/wd719x.c | 24 +- + drivers/scsi/xen-scsifront.c | 40 +- + drivers/scsi/zorro7xx.c | 31 +- + drivers/target/target_core_file.c | 4 +- + drivers/target/target_core_pscsi.c | 16 +- + drivers/target/target_core_spc.c | 11 +- + drivers/target/target_core_transport.c | 11 +- + drivers/target/target_core_ua.c | 28 +- + drivers/target/target_core_ua.h | 7 +- + drivers/ufs/Kconfig | 1 + + drivers/ufs/core/ufs-debugfs.c | 4 +- + drivers/ufs/core/ufs-rpmb.c | 29 +- + drivers/ufs/core/ufs-sysfs.c | 12 +- + drivers/ufs/core/ufshcd.c | 189 ++- + drivers/ufs/host/ufs-mediatek.c | 4 +- + drivers/ufs/host/ufs-qcom.c | 47 +- + drivers/ufs/host/ufs-qcom.h | 5 +- + drivers/ufs/host/ufs-sprd.c | 4 +- + drivers/usb/gadget/function/f_mass_storage.c | 34 +- + drivers/usb/gadget/function/storage_common.h | 69 +- + drivers/usb/storage/debug.c | 11 +- + drivers/usb/storage/debug.h | 4 +- + drivers/usb/storage/transport.c | 7 +- + drivers/usb/storage/uas.c | 12 +- + drivers/usb/storage/usb.h | 4 +- + include/scsi/scsi_cmnd.h | 3 +- + include/scsi/scsi_common.h | 16 +- + include/scsi/scsi_dbg.h | 6 +- + include/scsi/scsi_device.h | 26 +- + include/scsi/scsi_host.h | 9 +- + include/scsi/scsi_proto.h | 71 +- + include/scsi/scsi_sense.h | 950 +++++++++++ + include/trace/events/scsi.h | 4 +- + 162 files changed, 5925 insertions(+), 3412 deletions(-) + create mode 100644 include/scsi/scsi_sense.h +Merging scsi-mkp/for-next (f09d2c7485b32 scsi: target: core: Use assign_bit() where applicable) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mkp/scsi.git scsi-mkp/for-next +Auto-merging drivers/ata/libata-scsi.c +Auto-merging drivers/scsi/mpi3mr/mpi3mr_transport.c +CONFLICT (content): Merge conflict in drivers/scsi/mpi3mr/mpi3mr_transport.c +Resolved 'drivers/scsi/mpi3mr/mpi3mr_transport.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master b7cb7af529a8e] Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mkp/scsi.git +$ git diff -M --stat --summary HEAD^.. + drivers/scsi/mpi3mr/mpi3mr_transport.c | 2 +- + drivers/target/target_core_user.c | 5 +- + include/scsi/scsi_sense.h | 236 ++++++++++++++++----------------- + 3 files changed, 120 insertions(+), 123 deletions(-) +$ git am -3 ../patches/device-id-zorro +Applying: foofof +Using index info to reconstruct a base tree... +M include/linux/device-id/zorro.h +Falling back to patching base and 3-way merge... +No changes -- Patch already applied. +$ git am -3 ../patches/0001-libata-Fix-up-semantic-conflict-in-ata_scsi_set_sens.patch +Applying: libata: Fix up semantic conflict in ata_scsi_set_sense() +$ git reset HEAD^ +Unstaged changes after reset: +M drivers/ata/libata-scsi.c +$ git add -A . +$ git commit -v -a --amend +warning: notes ref refs/notes/commits is invalid +[master 357471b00a525] Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mkp/scsi.git + Date: Sat Oct 3 00:55:23 2026 +0200 +Merging vhost/linux-next (8f2c2fb94a013 vhost/vsock: add VHOST_RESET_OWNER ioctl) +$ git merge -m Merge branch 'linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mst/vhost.git vhost/linux-next +Auto-merging arch/um/drivers/virtio_uml.c +Auto-merging drivers/platform/mellanox/mlxbf-tmfifo.c +Merge made by the 'ort' strategy. + arch/um/drivers/virtio_uml.c | 10 ++++ + drivers/platform/mellanox/mlxbf-tmfifo.c | 15 ++++++ + drivers/remoteproc/remoteproc_core.c | 12 +++++ + drivers/remoteproc/remoteproc_virtio.c | 39 +++++++++++--- + drivers/s390/virtio/virtio_ccw.c | 6 +-- + drivers/vhost/net.c | 4 +- + drivers/vhost/scsi.c | 2 +- + drivers/vhost/test.c | 4 +- + drivers/vhost/vdpa.c | 4 +- + drivers/vhost/vhost.c | 26 ++++++++-- + drivers/vhost/vhost.h | 5 +- + drivers/vhost/vsock.c | 89 +++++++++++++++++++++++++------- + drivers/virtio/virtio.c | 2 + + drivers/virtio/virtio_pci_legacy.c | 2 - + drivers/virtio/virtio_pci_modern.c | 15 +++--- + drivers/virtio/virtio_vdpa.c | 34 ++++++++++-- + include/linux/remoteproc.h | 3 ++ + 17 files changed, 213 insertions(+), 59 deletions(-) +Merging rpmsg/for-next (5f7775e2a3462 Merge branches 'rproc-next', 'rproc-fixes' and 'rpmsg-next' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/remoteproc/linux.git rpmsg/for-next +Auto-merging MAINTAINERS +Auto-merging arch/arm64/boot/dts/qcom/sdx75.dtsi +Auto-merging drivers/remoteproc/remoteproc_core.c +Auto-merging drivers/soc/qcom/Kconfig +Merge made by the 'ort' strategy. + .../devicetree/bindings/remoteproc/mtk,scp.yaml | 22 +- + .../bindings/remoteproc/qcom,pas-common.yaml | 9 + + .../bindings/remoteproc/qcom,sm8550-pas.yaml | 31 +- + .../bindings/remoteproc/ti,am3352-wkup-m3.yaml | 1 - + Documentation/staging/rpmsg.rst | 17 + + MAINTAINERS | 8 + + arch/arm64/boot/dts/qcom/sdx75.dtsi | 3 +- + drivers/remoteproc/Kconfig | 11 +- + drivers/remoteproc/imx_dsp_rproc.c | 7 +- + drivers/remoteproc/imx_rproc.c | 51 +- + drivers/remoteproc/omap_remoteproc.c | 12 +- + drivers/remoteproc/qcom_q6v5_adsp.c | 2 +- + drivers/remoteproc/qcom_q6v5_pas.c | 166 +++++- + drivers/remoteproc/remoteproc_core.c | 3 + + drivers/remoteproc/remoteproc_debugfs.c | 6 +- + drivers/remoteproc/remoteproc_elf_loader.c | 23 +- + drivers/remoteproc/remoteproc_sysfs.c | 7 +- + drivers/remoteproc/stm32_rproc.c | 2 +- + drivers/remoteproc/ti_k3_common.c | 10 +- + drivers/remoteproc/ti_k3_common.h | 4 + + drivers/remoteproc/ti_k3_dsp_remoteproc.c | 7 +- + drivers/remoteproc/ti_k3_m4_remoteproc.c | 4 +- + drivers/remoteproc/ti_k3_r5_remoteproc.c | 9 +- + drivers/remoteproc/xlnx_r5_remoteproc.c | 23 +- + drivers/rpmsg/virtio_rpmsg_bus.c | 156 ++++-- + drivers/soc/qcom/Kconfig | 12 + + drivers/soc/qcom/Makefile | 1 + + drivers/soc/qcom/qmi_tmd.c | 595 +++++++++++++++++++++ + include/dt-bindings/thermal/qcom,pas.h | 20 + + include/linux/omap-mailbox.h | 5 +- + include/linux/rpmsg/virtio_rpmsg.h | 41 ++ + include/linux/soc/qcom/qmi.h | 1 + + include/linux/soc/qcom/qmi_tmd.h | 36 ++ + include/uapi/linux/rpmsg.h | 15 +- + samples/rpmsg/rpmsg_client_sample.c | 20 +- + 35 files changed, 1178 insertions(+), 162 deletions(-) + create mode 100644 drivers/soc/qcom/qmi_tmd.c + create mode 100644 include/dt-bindings/thermal/qcom,pas.h + create mode 100644 include/linux/rpmsg/virtio_rpmsg.h + create mode 100644 include/linux/soc/qcom/qmi_tmd.h +Merging gpio-brgl/gpio/for-next (c4e74a7058b57 gpio: pxa: stop reading gpio_chip::base in direction callbacks) +$ git merge -m Merge branch 'gpio/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git gpio-brgl/gpio/for-next +Auto-merging MAINTAINERS +CONFLICT (content): Merge conflict in MAINTAINERS +Auto-merging drivers/gpio/Kconfig +Auto-merging drivers/gpio/Makefile +Auto-merging drivers/gpio/gpio-mvebu.c +Auto-merging drivers/gpio/gpiolib-acpi-quirks.c +CONFLICT (content): Merge conflict in drivers/gpio/gpiolib-acpi-quirks.c +Auto-merging drivers/gpio/gpiolib-cdev.c +Auto-merging drivers/gpio/gpiolib.c +Resolved 'MAINTAINERS' using previous resolution. +Resolved 'drivers/gpio/gpiolib-acpi-quirks.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 216f4acdad14a] Merge branch 'gpio/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git +$ git diff -M --stat --summary HEAD^.. + .../bindings/gpio/axiado,ax3005-sgpio.yaml | 112 ++++ + .../devicetree/bindings/gpio/fsl,qoriq-gpio.yaml | 7 + + .../devicetree/bindings/gpio/gpio-zynq.yaml | 6 + + .../bindings/gpio/nvidia,tegra186-gpio.yaml | 9 + + .../bindings/gpio/realtek,otto-gpio.yaml | 2 + + .../bindings/gpio/renesas,rcar-gpio.yaml | 5 + + Documentation/driver-api/gpio/board.rst | 2 +- + Documentation/driver-api/gpio/driver.rst | 7 + + MAINTAINERS | 9 + + drivers/gpio/Kconfig | 33 + + drivers/gpio/Makefile | 2 + + drivers/gpio/gpio-ad7768.c | 121 ++++ + drivers/gpio/gpio-adp5585.c | 20 +- + drivers/gpio/gpio-altera.c | 5 +- + drivers/gpio/gpio-axiado-sgpio.c | 736 +++++++++++++++++++++ + drivers/gpio/gpio-dln2.c | 6 +- + drivers/gpio/gpio-eic-sprd.c | 17 +- + drivers/gpio/gpio-it87.c | 2 +- + drivers/gpio/gpio-mlxbf2.c | 4 +- + drivers/gpio/gpio-mlxbf3.c | 4 +- + drivers/gpio/gpio-mmio.c | 65 +- + drivers/gpio/gpio-mvebu.c | 19 +- + drivers/gpio/gpio-omap.c | 11 +- + drivers/gpio/gpio-pcf857x.c | 16 +- + drivers/gpio/gpio-pxa.c | 4 +- + drivers/gpio/gpio-rcar.c | 86 ++- + drivers/gpio/gpio-realtek-otto.c | 21 +- + drivers/gpio/gpio-regmap.c | 98 ++- + drivers/gpio/gpio-rtd1625.c | 26 +- + drivers/gpio/gpio-siox.c | 4 +- + drivers/gpio/gpio-sloppy-logic-analyzer.c | 4 +- + drivers/gpio/gpio-tegra186.c | 74 +++ + drivers/gpio/gpiolib-acpi-quirks.c | 13 + + drivers/gpio/gpiolib-cdev.c | 2 + + drivers/gpio/gpiolib-kunit.c | 14 +- + drivers/gpio/gpiolib.c | 21 + + drivers/pinctrl/freescale/pinctrl-imx.c | 54 +- + drivers/pinctrl/freescale/pinctrl-imx.h | 4 + + drivers/pinctrl/freescale/pinctrl-vf610.c | 2 + + include/linux/gpio/driver.h | 9 + + include/linux/gpio/regmap.h | 2 + + include/linux/pinctrl/consumer.h | 4 +- + tools/gpio/gpio-hammer.c | 2 +- + 43 files changed, 1532 insertions(+), 132 deletions(-) + create mode 100644 Documentation/devicetree/bindings/gpio/axiado,ax3005-sgpio.yaml + create mode 100644 drivers/gpio/gpio-ad7768.c + create mode 100644 drivers/gpio/gpio-axiado-sgpio.c +Merging gpio-intel/for-next (0fc424b6a8ef4 gpio: wcove: use regmap_assign_bits() for conditional set/clear) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/andy/linux-gpio-intel.git gpio-intel/for-next +Merge made by the 'ort' strategy. + drivers/gpio/gpio-wcove.c | 5 +---- + 1 file changed, 1 insertion(+), 4 deletions(-) +Merging pinctrl/for-next (3b9c0102861ab Merge branch 'devel' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/linusw/linux-pinctrl.git pinctrl/for-next +Auto-merging Documentation/devicetree/bindings/pinctrl/apple,pinctrl.yaml +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + .../pinctrl/allwinner,sun55i-a523-pinctrl.yaml | 6 +- + .../bindings/pinctrl/ambarella,cv75-pinctrl.yaml | 237 ++++ + .../devicetree/bindings/pinctrl/apple,pinctrl.yaml | 12 + + .../bindings/pinctrl/awinic,aw9523-pinctrl.yaml | 50 +- + .../bindings/pinctrl/cix,sky1-pinctrl.yaml | 2 +- + .../bindings/pinctrl/econet,en7528-pinctrl.yaml | 190 +++ + .../bindings/pinctrl/fsl,imx35-pinctrl.yaml | 4 +- + .../bindings/pinctrl/fsl,imx7d-pinctrl.yaml | 12 +- + .../bindings/pinctrl/fsl,imx8m-pinctrl.yaml | 4 +- + .../bindings/pinctrl/fsl,imx8ulp-pinctrl.yaml | 4 +- + .../bindings/pinctrl/fsl,imx9-pinctrl.yaml | 4 +- + .../devicetree/bindings/pinctrl/fsl,imxrt1050.yaml | 4 +- + .../devicetree/bindings/pinctrl/fsl,imxrt1170.yaml | 4 +- + .../devicetree/bindings/pinctrl/intel,lgm-io.yaml | 14 +- + .../bindings/pinctrl/mediatek,mt8183-pinctrl.yaml | 4 +- + .../bindings/pinctrl/mediatek,mt8195-pinctrl.yaml | 2 +- + .../devicetree/bindings/pinctrl/pincfg-node.yaml | 11 + + .../devicetree/bindings/pinctrl/pinctrl-palmas.txt | 105 -- + .../bindings/pinctrl/qcom,hawi-tlmm.yaml | 4 +- + .../bindings/pinctrl/qcom,kaanapali-tlmm.yaml | 4 +- + .../bindings/pinctrl/qcom,maili-tlmm.yaml | 4 +- + .../bindings/pinctrl/renesas,rzg2l-pinctrl.yaml | 4 +- + .../bindings/pinctrl/semtech,sx1501q.yaml | 4 +- + .../pinctrl/starfive,jhb100-per0-pinctrl.yaml | 179 +++ + .../pinctrl/starfive,jhb100-per1-pinctrl.yaml | 178 +++ + .../pinctrl/starfive,jhb100-per2-pinctrl.yaml | 168 +++ + .../pinctrl/starfive,jhb100-per2pok-pinctrl.yaml | 162 +++ + .../pinctrl/starfive,jhb100-per3-pinctrl.yaml | 166 +++ + .../pinctrl/starfive,jhb100-sys0-pinctrl.yaml | 164 +++ + .../pinctrl/starfive,jhb100-sys0h-pinctrl.yaml | 162 +++ + .../pinctrl/starfive,jhb100-sys1-pinctrl.yaml | 164 +++ + .../pinctrl/starfive,jhb100-sys2-pinctrl.yaml | 165 +++ + .../bindings/pinctrl/ti,palmas-pinctrl.yaml | 170 +++ + .../bindings/pinctrl/xlnx,zynqmp-pinctrl.yaml | 46 +- + MAINTAINERS | 9 + + drivers/pinctrl/Kconfig | 18 + + drivers/pinctrl/Makefile | 2 + + drivers/pinctrl/airoha/Kconfig | 15 +- + drivers/pinctrl/airoha/Makefile | 1 + + drivers/pinctrl/airoha/airoha-common.h | 3 + + drivers/pinctrl/airoha/pinctrl-airoha.c | 33 +- + drivers/pinctrl/airoha/pinctrl-an7563.c | 1 + + drivers/pinctrl/airoha/pinctrl-an7581.c | 1 + + drivers/pinctrl/airoha/pinctrl-an7583.c | 1 + + drivers/pinctrl/airoha/pinctrl-en7523.c | 1 + + drivers/pinctrl/airoha/pinctrl-en7528.c | 1204 ++++++++++++++++ + drivers/pinctrl/core.c | 4 +- + drivers/pinctrl/freescale/pinctrl-imx.c | 2 +- + drivers/pinctrl/mediatek/pinctrl-mt7629.c | 2 +- + drivers/pinctrl/mediatek/pinctrl-mtk-common.c | 3 +- + drivers/pinctrl/microchip/pinctrl-mpfs-mssio.c | 2 +- + drivers/pinctrl/nuvoton/pinctrl-ma35.c | 8 +- + drivers/pinctrl/nxp/pinctrl-s32cc.c | 2 +- + drivers/pinctrl/pinconf-generic.c | 2 + + drivers/pinctrl/pinconf.c | 2 +- + drivers/pinctrl/pinctrl-ambarella-cv75.c | 587 ++++++++ + drivers/pinctrl/pinctrl-ambarella.c | 533 ++++++++ + drivers/pinctrl/pinctrl-ambarella.h | 40 + + drivers/pinctrl/pinctrl-amd.c | 243 +++- + drivers/pinctrl/pinctrl-amd.h | 174 +++ + drivers/pinctrl/pinctrl-apple-gpio.c | 21 +- + drivers/pinctrl/pinctrl-axp209.c | 3 - + drivers/pinctrl/pinctrl-equilibrium.c | 2 +- + drivers/pinctrl/pinctrl-gemini.c | 3 +- + drivers/pinctrl/pinctrl-generic-mux.c | 76 +- + drivers/pinctrl/pinctrl-st.c | 2 +- + drivers/pinctrl/pinmux.c | 2 +- + drivers/pinctrl/renesas/pinctrl-rzg2l.c | 412 ++++-- + drivers/pinctrl/renesas/pinctrl-rzt2h.c | 182 ++- + drivers/pinctrl/starfive/Kconfig | 105 ++ + drivers/pinctrl/starfive/Makefile | 11 + + .../starfive/pinctrl-starfive-jhb100-per0.c | 182 +++ + .../starfive/pinctrl-starfive-jhb100-per1.c | 193 +++ + .../starfive/pinctrl-starfive-jhb100-per2.c | 136 ++ + .../starfive/pinctrl-starfive-jhb100-per2pok.c | 99 ++ + .../starfive/pinctrl-starfive-jhb100-per3.c | 129 ++ + .../starfive/pinctrl-starfive-jhb100-sys0.c | 121 ++ + .../starfive/pinctrl-starfive-jhb100-sys0h.c | 99 ++ + .../starfive/pinctrl-starfive-jhb100-sys1.c | 94 ++ + .../starfive/pinctrl-starfive-jhb100-sys2.c | 140 ++ + drivers/pinctrl/starfive/pinctrl-starfive-jhb100.c | 1446 ++++++++++++++++++++ + drivers/pinctrl/starfive/pinctrl-starfive-jhb100.h | 153 +++ + drivers/pinctrl/stm32/Kconfig | 2 +- + drivers/pinctrl/stm32/pinctrl-stm32.c | 3 + + drivers/pinctrl/sunplus/Kconfig | 8 +- + drivers/pinctrl/sunplus/sppctl.c | 45 +- + drivers/pinctrl/sunxi/Kconfig | 5 + + drivers/pinctrl/sunxi/Makefile | 1 + + drivers/pinctrl/sunxi/pinctrl-sun20i-d1.c | 2 +- + drivers/pinctrl/sunxi/pinctrl-sun55i-a523-r.c | 3 +- + drivers/pinctrl/sunxi/pinctrl-sun55i-a523.c | 2 +- + drivers/pinctrl/sunxi/pinctrl-sun60i-a733.c | 50 + + drivers/pinctrl/sunxi/pinctrl-sunxi.c | 41 +- + drivers/pinctrl/sunxi/pinctrl-sunxi.h | 71 +- + include/linux/pinctrl/consumer.h | 1 + + include/linux/pinctrl/pinconf-generic.h | 5 + + 96 files changed, 8601 insertions(+), 555 deletions(-) + create mode 100644 Documentation/devicetree/bindings/pinctrl/ambarella,cv75-pinctrl.yaml + create mode 100644 Documentation/devicetree/bindings/pinctrl/econet,en7528-pinctrl.yaml + delete mode 100644 Documentation/devicetree/bindings/pinctrl/pinctrl-palmas.txt + create mode 100644 Documentation/devicetree/bindings/pinctrl/starfive,jhb100-per0-pinctrl.yaml + create mode 100644 Documentation/devicetree/bindings/pinctrl/starfive,jhb100-per1-pinctrl.yaml + create mode 100644 Documentation/devicetree/bindings/pinctrl/starfive,jhb100-per2-pinctrl.yaml + create mode 100644 Documentation/devicetree/bindings/pinctrl/starfive,jhb100-per2pok-pinctrl.yaml + create mode 100644 Documentation/devicetree/bindings/pinctrl/starfive,jhb100-per3-pinctrl.yaml + create mode 100644 Documentation/devicetree/bindings/pinctrl/starfive,jhb100-sys0-pinctrl.yaml + create mode 100644 Documentation/devicetree/bindings/pinctrl/starfive,jhb100-sys0h-pinctrl.yaml + create mode 100644 Documentation/devicetree/bindings/pinctrl/starfive,jhb100-sys1-pinctrl.yaml + create mode 100644 Documentation/devicetree/bindings/pinctrl/starfive,jhb100-sys2-pinctrl.yaml + create mode 100644 Documentation/devicetree/bindings/pinctrl/ti,palmas-pinctrl.yaml + create mode 100644 drivers/pinctrl/airoha/pinctrl-en7528.c + create mode 100644 drivers/pinctrl/pinctrl-ambarella-cv75.c + create mode 100644 drivers/pinctrl/pinctrl-ambarella.c + create mode 100644 drivers/pinctrl/pinctrl-ambarella.h + create mode 100644 drivers/pinctrl/starfive/pinctrl-starfive-jhb100-per0.c + create mode 100644 drivers/pinctrl/starfive/pinctrl-starfive-jhb100-per1.c + create mode 100644 drivers/pinctrl/starfive/pinctrl-starfive-jhb100-per2.c + create mode 100644 drivers/pinctrl/starfive/pinctrl-starfive-jhb100-per2pok.c + create mode 100644 drivers/pinctrl/starfive/pinctrl-starfive-jhb100-per3.c + create mode 100644 drivers/pinctrl/starfive/pinctrl-starfive-jhb100-sys0.c + create mode 100644 drivers/pinctrl/starfive/pinctrl-starfive-jhb100-sys0h.c + create mode 100644 drivers/pinctrl/starfive/pinctrl-starfive-jhb100-sys1.c + create mode 100644 drivers/pinctrl/starfive/pinctrl-starfive-jhb100-sys2.c + create mode 100644 drivers/pinctrl/starfive/pinctrl-starfive-jhb100.c + create mode 100644 drivers/pinctrl/starfive/pinctrl-starfive-jhb100.h + create mode 100644 drivers/pinctrl/sunxi/pinctrl-sun60i-a733.c +Merging pinctrl-intel/for-next (c016587866e57 pinctrl: intel: Try to retrieve driver data for pure platform drivers) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/pinctrl/intel.git pinctrl-intel/for-next +Merge made by the 'ort' strategy. + drivers/pinctrl/intel/pinctrl-intel.c | 51 ++++++++++++++++++++++++----------- + drivers/pinctrl/intel/pinctrl-intel.h | 2 +- + 2 files changed, 36 insertions(+), 17 deletions(-) +Merging pinctrl-renesas/renesas-pinctrl (0cd4a7b3a4883 pinctrl: renesas: rzt2h: Add a helper for reading PFC) +$ git merge -m Merge branch 'renesas-pinctrl' of https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-drivers.git pinctrl-renesas/renesas-pinctrl +Already up to date. +Merging pinctrl-samsung/for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/pinctrl/samsung.git pinctrl-samsung/for-next +Already up to date. +Merging pinctrl-qcom/pinctrl-qcom/for-next (4d7c9430a26ae pinctrl: qcom: tlmm-test: Add const to reg_names allocation type) +$ git merge -m Merge branch 'pinctrl-qcom/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git pinctrl-qcom/pinctrl-qcom/for-next +Auto-merging drivers/pinctrl/qcom/pinctrl-spmi-gpio.c +Merge made by the 'ort' strategy. + .../pinctrl/qcom,hawi-lpass-lpi-pinctrl.yaml | 109 ++ + .../bindings/pinctrl/qcom,kuno-tlmm.yaml | 110 ++ + .../bindings/pinctrl/qcom,msm8952-pinctrl.yaml | 146 +++ + .../bindings/pinctrl/qcom,pmic-gpio.yaml | 3 + + .../bindings/pinctrl/qcom,sc8280xp-tlmm.yaml | 3 + + drivers/pinctrl/qcom/Kconfig | 10 + + drivers/pinctrl/qcom/Kconfig.msm | 19 + + drivers/pinctrl/qcom/Makefile | 3 + + drivers/pinctrl/qcom/pinctrl-apq8064.c | 188 +-- + drivers/pinctrl/qcom/pinctrl-apq8084.c | 502 ++++---- + drivers/pinctrl/qcom/pinctrl-hawi-lpass-lpi.c | 244 ++++ + drivers/pinctrl/qcom/pinctrl-ipq4019.c | 208 ++-- + drivers/pinctrl/qcom/pinctrl-ipq8064.c | 208 ++-- + drivers/pinctrl/qcom/pinctrl-kuno.c | 801 +++++++++++++ + drivers/pinctrl/qcom/pinctrl-lpass-lpi.c | 6 +- + drivers/pinctrl/qcom/pinctrl-lpass-lpi.h | 17 + + drivers/pinctrl/qcom/pinctrl-msm.h | 25 - + drivers/pinctrl/qcom/pinctrl-msm8952.c | 1257 ++++++++++++++++++++ + drivers/pinctrl/qcom/pinctrl-spmi-gpio.c | 1 + + drivers/pinctrl/qcom/tlmm-test.c | 2 +- + 20 files changed, 3282 insertions(+), 580 deletions(-) + create mode 100644 Documentation/devicetree/bindings/pinctrl/qcom,hawi-lpass-lpi-pinctrl.yaml + create mode 100644 Documentation/devicetree/bindings/pinctrl/qcom,kuno-tlmm.yaml + create mode 100644 Documentation/devicetree/bindings/pinctrl/qcom,msm8952-pinctrl.yaml + create mode 100644 drivers/pinctrl/qcom/pinctrl-hawi-lpass-lpi.c + create mode 100644 drivers/pinctrl/qcom/pinctrl-kuno.c + create mode 100644 drivers/pinctrl/qcom/pinctrl-msm8952.c +Merging pwm/pwm/for-next (e74b9a7ee5007 pwm: tiehrpwm: Clear period on disable to allow reconfiguration) +$ git merge -m Merge branch 'pwm/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/ukleinek/linux.git pwm/pwm/for-next +Merge made by the 'ort' strategy. + .../devicetree/bindings/pwm/pwm-tipwmss.txt | 58 --------------- + .../devicetree/bindings/pwm/ti,am33xx-pwmss.yaml | 82 ++++++++++++++++++++++ + Documentation/driver-api/pwm.rst | 2 +- + drivers/pwm/core.c | 1 + + drivers/pwm/pwm-brcmstb.c | 2 +- + drivers/pwm/pwm-ipq.c | 2 +- + drivers/pwm/pwm-iqs620a.c | 23 +----- + drivers/pwm/pwm-loongson.c | 2 + + drivers/pwm/pwm-lp3943.c | 3 + + drivers/pwm/pwm-mtk-disp.c | 4 +- + drivers/pwm/pwm-pca9685.c | 2 +- + drivers/pwm/pwm-renesas-tpu.c | 23 ++++-- + drivers/pwm/pwm-tiehrpwm.c | 3 + + 13 files changed, 118 insertions(+), 89 deletions(-) + delete mode 100644 Documentation/devicetree/bindings/pwm/pwm-tipwmss.txt + create mode 100644 Documentation/devicetree/bindings/pwm/ti,am33xx-pwmss.yaml +Merging ktest/for-next (932cdaf3e273a ktest: Add logfile to failure directory) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/rostedt/linux-ktest.git ktest/for-next +Already up to date. +Merging kselftest/next (30af56a227e27 selftests/ftrace: skip gcov symbols when picking a function to probe) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git kselftest/next +Merge made by the 'ort' strategy. + .../ftrace/test.d/kprobe/kprobe_eventname.tc | 2 +- + tools/testing/selftests/resctrl/cat_test.c | 23 ++- + tools/testing/selftests/resctrl/mba_test.c | 1 + + tools/testing/selftests/resctrl/mbm_test.c | 1 + + tools/testing/selftests/resctrl/resctrl.h | 2 + + tools/testing/selftests/resctrl/resctrl_val.c | 169 +++++++++++---------- + 6 files changed, 115 insertions(+), 83 deletions(-) +Merging kunit/test (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'test' of https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git kunit/test +Already up to date. +Merging kunit-next/kunit (e38f53f048246 kunit: Return void from kunit_run_all_tests()) +$ git merge -m Merge branch 'kunit' of https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git kunit-next/kunit +Merge made by the 'ort' strategy. + include/kunit/test.h | 5 ++--- + lib/kunit/executor.c | 5 ++--- + 2 files changed, 4 insertions(+), 6 deletions(-) +Merging livepatching/for-next (5d791d3396ca4 selftests/livepatch: filter debug messages in check_result()) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/livepatching/livepatching.git livepatching/for-next +Merge made by the 'ort' strategy. + include/linux/livepatch.h | 4 ++++ + kernel/livepatch/core.c | 10 ++++++++++ + tools/testing/selftests/livepatch/functions.sh | 12 +++++++++--- + 3 files changed, 23 insertions(+), 3 deletions(-) +Merging rtc/rtc-next (8ef8e9839a23c rtc: ftrtc010: fix integer overflow in time calculation) +$ git merge -m Merge branch 'rtc-next' of https://git.kernel.org/pub/scm/linux/kernel/git/abelloni/linux.git rtc/rtc-next +Auto-merging drivers/rtc/rtc-bd70528.c +Merge made by the 'ort' strategy. + .../devicetree/bindings/mfd/mediatek,mt6397.yaml | 9 +++ + .../devicetree/bindings/rtc/renesas,rzn1-rtc.yaml | 30 +++++-- + drivers/rtc/rtc-bd70528.c | 5 ++ + drivers/rtc/rtc-ftrtc010.c | 74 ++++++------------ + drivers/rtc/rtc-mt6397.c | 91 +++++++++++++++++++++- + drivers/rtc/rtc-omap.c | 5 +- + drivers/rtc/rtc-rzn1.c | 52 +++++++++---- + drivers/soc/ti/pm33xx.c | 2 +- + include/linux/mfd/mt6397/rtc.h | 7 ++ + include/linux/rtc/rtc-omap.h | 2 +- + 10 files changed, 194 insertions(+), 83 deletions(-) +Merging nvdimm/libnvdimm-for-next (855a46681e0ba nvdimm/pmem: Release gendisk on probe failure) +$ git merge -m Merge branch 'libnvdimm-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/nvdimm/nvdimm.git nvdimm/libnvdimm-for-next +Auto-merging drivers/nvdimm/pmem.c +Merge made by the 'ort' strategy. + drivers/nvdimm/pmem.c | 6 ++++-- + 1 file changed, 4 insertions(+), 2 deletions(-) +Merging at24/at24/for-next (dc59e4fea9d83 Linux 7.2-rc1) +$ git merge -m Merge branch 'at24/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git at24/at24/for-next +Already up to date. +Merging ntb/ntb-next (dc59e4fea9d83 Linux 7.2-rc1) +$ git merge -m Merge branch 'ntb-next' of https://github.com/jonmason/ntb.git ntb/ntb-next +Already up to date. +Merging seccomp/for-next/seccomp (832b9b176be06 seccomp: restore knotif->state when SECCOMP_ADDFD_FLAG_SEND is interrupted) +$ git merge -m Merge branch 'for-next/seccomp' of https://git.kernel.org/pub/scm/linux/kernel/git/kees/linux.git seccomp/for-next/seccomp +Merge made by the 'ort' strategy. + kernel/seccomp.c | 7 +- + tools/testing/selftests/seccomp/seccomp_bpf.c | 179 ++++++++++++++++++++++++++ + 2 files changed, 184 insertions(+), 2 deletions(-) +Merging slimbus/for-next (4350d70546699 Merge branch 'slimbus-for-v7.4' into slimbus-for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/srini/slimbus.git slimbus/for-next +Merge made by the 'ort' strategy. + drivers/slimbus/messaging.c | 2 +- + drivers/slimbus/qcom-ngd-ctrl.c | 82 ++++++++++++++++++++++++++++++++++++++++- + drivers/slimbus/slimbus.h | 14 ++++++- + include/linux/slimbus.h | 2 +- + 4 files changed, 95 insertions(+), 5 deletions(-) +Merging nvmem/for-next (41f42ff4ae134 Merge branches 'nvmem-fixes' and 'nvmem-for-v7.4' into nvmem-for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/srini/nvmem.git nvmem/for-next +Auto-merging Documentation/devicetree/bindings/mfd/mediatek,mt6397.yaml +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + .../devicetree/bindings/mfd/mediatek,mt6397.yaml | 21 ++++ + .../devicetree/bindings/nvmem/qcom,qfprom.yaml | 1 + + .../bindings/nvmem/socionext,uniphier-efuse.yaml | 6 +- + MAINTAINERS | 5 + + drivers/nvmem/Kconfig | 14 ++- + drivers/nvmem/Makefile | 2 + + drivers/nvmem/core.c | 128 ++++++++++++--------- + drivers/nvmem/internals.h | 5 +- + drivers/nvmem/layouts.c | 13 ++- + drivers/nvmem/layouts/onie-tlv.c | 1 + + drivers/nvmem/layouts/sl28vpd.c | 1 + + drivers/nvmem/mt6323-efuse.c | 83 +++++++++++++ + drivers/nvmem/rockchip-otp.c | 15 ++- + 13 files changed, 236 insertions(+), 59 deletions(-) + create mode 100644 drivers/nvmem/mt6323-efuse.c +Merging hyperv/hyperv-next (be0cfab740e58 clocksource: hyper-v: Remove support for stimer interrupts in message mode) +$ git merge -m Merge branch 'hyperv-next' of https://git.kernel.org/pub/scm/linux/kernel/git/hyperv/linux.git hyperv/hyperv-next +Already up to date. +Merging auxdisplay/for-next (f63dc0eea92d0 docs: ABI: auxdisplay: use literal blocks for commands) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/andy/linux-auxdisplay.git auxdisplay/for-next +Merge made by the 'ort' strategy. + Documentation/ABI/testing/sysfs-auxdisplay-linedisp | 9 ++++++--- + drivers/auxdisplay/arm-charlcd.c | 7 ++----- + drivers/auxdisplay/panel.c | 5 +---- + 3 files changed, 9 insertions(+), 12 deletions(-) +Merging kgdb/kgdb/for-next (fdbdd0ccb30af kdb: remove redundant check for scancode 0xe0) +$ git merge -m Merge branch 'kgdb/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/danielt/linux.git kgdb/kgdb/for-next +Already up to date. +Merging hmm/hmm (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'hmm' of https://git.kernel.org/pub/scm/linux/kernel/git/rdma/rdma.git hmm/hmm +Already up to date. +Merging cfi/cfi/next (dc59e4fea9d83 Linux 7.2-rc1) +$ git merge -m Merge branch 'cfi/next' of https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git cfi/cfi/next +Already up to date. +Merging mhi/mhi-next (710bf7329abf8 bus: mhi: host: pci_generic: Add DUN channel configuration) +$ git merge -m Merge branch 'mhi-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mani/mhi.git mhi/mhi-next +Merge made by the 'ort' strategy. + drivers/bus/mhi/ep/main.c | 4 ++-- + drivers/bus/mhi/host/debugfs.c | 4 ++-- + drivers/bus/mhi/host/pci_generic.c | 9 ++++++++- + include/linux/mhi.h | 2 +- + 4 files changed, 13 insertions(+), 6 deletions(-) +$ git am -3 ../patches/0001-fix-up-for-net-qrtr-Drop-the-MHI-auto_queue-feature-.patch +Applying: fix up for "net: qrtr: Drop the MHI auto_queue feature for IPCR DL channels" +Using index info to reconstruct a base tree... +M drivers/net/wireless/ath/ath12k/wifi7/mhi.c +Falling back to patching base and 3-way merge... +No changes -- Patch already applied. +Merging cxl/next (71392a644e88c Merge branch 'for-7.4/cxl-misc' into cxl-for-next) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/cxl/cxl.git cxl/next +Auto-merging MAINTAINERS +Auto-merging drivers/cxl/core/ras.c +Auto-merging include/linux/device.h +Merge made by the 'ort' strategy. + .../driver-api/cxl/platform/acpi/dsdt.rst | 3 +- + MAINTAINERS | 1 + + drivers/cxl/acpi.c | 10 +- + drivers/cxl/core/cdat.c | 8 +- + drivers/cxl/core/core.h | 8 +- + drivers/cxl/core/edac.c | 30 ++++-- + drivers/cxl/core/features.c | 58 ++++++---- + drivers/cxl/core/hdm.c | 120 ++++++++++++++++----- + drivers/cxl/core/mbox.c | 3 + + drivers/cxl/core/mce.c | 8 +- + drivers/cxl/core/pci.c | 9 ++ + drivers/cxl/core/port.c | 39 +++++-- + drivers/cxl/core/ras.c | 6 +- + drivers/cxl/core/region.c | 100 +++++++++++------ + drivers/cxl/core/region_pmem.c | 6 +- + drivers/cxl/core/regs.c | 14 +++ + drivers/cxl/cxl.h | 12 +++ + drivers/cxl/mem.c | 14 ++- + drivers/cxl/pci.c | 5 +- + drivers/cxl/port.c | 3 + + include/linux/device.h | 1 + + tools/testing/cxl/Kbuild | 16 +-- + tools/testing/cxl/test/cxl.c | 117 ++++++++++++++++---- + 23 files changed, 443 insertions(+), 148 deletions(-) +Merging zstd/zstd-next (65d1f5507ed2c zstd: Import upstream v1.5.7) +$ git merge -m Merge branch 'zstd-next' of https://github.com/terrelln/linux.git zstd/zstd-next +Already up to date. +Merging efi/next (7eef16311a234 Merge remote-tracking branch 'linux-efi/bootloader-info' into next) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/efi/efi.git efi/next +Auto-merging drivers/firmware/efi/libstub/Makefile +Merge made by the 'ort' strategy. + drivers/firmware/efi/capsule-loader.c | 6 +- + drivers/firmware/efi/capsule.c | 4 +- + drivers/firmware/efi/efi.c | 33 +++++- + drivers/firmware/efi/libstub/Makefile | 3 +- + drivers/firmware/efi/libstub/bli.c | 87 ++++++++++++++ + drivers/firmware/efi/libstub/efi-stub-entry.c | 2 +- + drivers/firmware/efi/libstub/efi-stub-helper.c | 108 +++++++---------- + drivers/firmware/efi/libstub/efi-stub.c | 3 +- + drivers/firmware/efi/libstub/efistub.h | 12 +- + drivers/firmware/efi/libstub/file.c | 8 +- + drivers/firmware/efi/libstub/gop.c | 23 ++-- + drivers/firmware/efi/libstub/kaslr.c | 19 ++- + drivers/firmware/efi/libstub/mem.c | 2 +- + drivers/firmware/efi/libstub/pci.c | 2 +- + drivers/firmware/efi/libstub/printk.c | 95 ++------------- + drivers/firmware/efi/libstub/random.c | 8 +- + drivers/firmware/efi/libstub/riscv.c | 2 +- + drivers/firmware/efi/libstub/smbios.c | 3 +- + drivers/firmware/efi/libstub/tpm.c | 8 +- + drivers/firmware/efi/libstub/unaccepted_memory.c | 2 +- + drivers/firmware/efi/libstub/vsprintf.c | 140 ++++++++--------------- + drivers/firmware/efi/libstub/x86-stub.c | 18 +-- + drivers/firmware/efi/libstub/zboot.c | 3 +- + drivers/firmware/efi/runtime-wrappers.c | 28 +++++ + include/linux/efi.h | 23 ++++ + include/linux/ucs2_string.h | 11 +- + lib/ucs2_string.c | 19 +-- + 27 files changed, 360 insertions(+), 312 deletions(-) + create mode 100644 drivers/firmware/efi/libstub/bli.c +Merging unicode/for-next (a511442085c14 unicode: Properly reject invalid encoding version strings) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/krisman/unicode.git unicode/for-next +Already up to date. +Merging random/master (eb13a1ff271b0 random: vDSO: avoid call to memset() when zeroing reserved parameter) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/crng/random.git random/master +Already up to date. +$ git am -3 ../patches/random-memset-check +Patch format detection failed. +$ git reset HEAD^ +Unstaged changes after reset: +M scripts/Kconfig.toolchain +$ git add -A . +$ git commit -v -a --amend +warning: notes ref refs/notes/commits is invalid +[master 24d4b7cba12be] Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/efi/efi.git + Date: Sat Oct 3 00:58:14 2026 +0200 +Merging landlock/next (1cd82903f6d7f landlock: Add documentation for capability and namespace restrictions) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/mic/linux.git landlock/next +Auto-merging security/landlock/syscalls.c +CONFLICT (content): Merge conflict in security/landlock/syscalls.c +Auto-merging tools/testing/selftests/landlock/fs_test.c +Recorded preimage for 'security/landlock/syscalls.c' +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +Recorded resolution for 'security/landlock/syscalls.c'. +[master 7ac9587c27398] Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/mic/linux.git +$ git diff -M --stat --summary HEAD^.. + Documentation/admin-guide/LSM/landlock.rst | 46 +- + Documentation/security/landlock.rst | 178 +- + Documentation/trace/events-landlock.rst | 66 +- + Documentation/userspace-api/landlock.rst | 243 ++- + include/linux/landlock.h | 29 +- + include/trace/events/landlock.h | 244 ++- + include/uapi/linux/landlock.h | 139 +- + samples/Kconfig | 6 +- + samples/landlock/Makefile | 1 + + samples/landlock/sandboxer.c | 230 ++- + security/landlock/Makefile | 4 +- + security/landlock/access.h | 61 +- + security/landlock/audit.c | 32 +- + security/landlock/cap.c | 163 ++ + security/landlock/cap.h | 48 + + security/landlock/cred.h | 2 +- + security/landlock/domain.c | 13 +- + security/landlock/domain.h | 98 +- + security/landlock/fs.c | 2 +- + security/landlock/limits.h | 9 + + security/landlock/log.c | 53 +- + security/landlock/log.h | 10 +- + security/landlock/net.c | 2 +- + security/landlock/ns.c | 241 +++ + security/landlock/ns.h | 18 + + security/landlock/ruleset.c | 27 +- + security/landlock/ruleset.h | 33 +- + security/landlock/setup.c | 4 + + security/landlock/syscalls.c | 217 +- + security/landlock/trace.c | 27 +- + tools/testing/selftests/landlock/base_test.c | 21 +- + tools/testing/selftests/landlock/cap_test.c | 1364 +++++++++++++ + tools/testing/selftests/landlock/common.h | 23 + + tools/testing/selftests/landlock/config | 5 + + tools/testing/selftests/landlock/fs_test.c | 13 +- + tools/testing/selftests/landlock/ns_test.c | 2643 +++++++++++++++++++++++++ + tools/testing/selftests/landlock/trace.h | 52 +- + tools/testing/selftests/landlock/trace_test.c | 8 +- + tools/testing/selftests/landlock/wrappers.h | 29 + + 39 files changed, 6225 insertions(+), 179 deletions(-) + create mode 100644 security/landlock/cap.c + create mode 100644 security/landlock/cap.h + create mode 100644 security/landlock/ns.c + create mode 100644 security/landlock/ns.h + create mode 100644 tools/testing/selftests/landlock/cap_test.c + create mode 100644 tools/testing/selftests/landlock/ns_test.c +$ git reset --hard HEAD^ +HEAD is now at 24d4b7cba12be Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/efi/efi.git +Merging next-20261001 version of landlock +$ git merge -m next-20261001/landlock 02619311dbfc4351e78e5dc6652532cbe7603661 +Merge made by the 'ort' strategy. +Merging sysctl/sysctl-next (4991c8b72b564 time/jiffies: Saturate in mult_hz() instead of wrapping) +$ git merge -m Merge branch 'sysctl-next' of https://git.kernel.org/pub/scm/linux/kernel/git/sysctl/sysctl.git sysctl/sysctl-next +Auto-merging fs/dcache.c +Auto-merging fs/proc/proc_sysctl.c +Auto-merging kernel/sysctl.c +Auto-merging kernel/time/jiffies.c +Merge made by the 'ort' strategy. + drivers/parport/procfs.c | 6 +- + fs/dcache.c | 2 +- + fs/file_table.c | 2 +- + fs/proc/proc_sysctl.c | 2 +- + kernel/sysctl.c | 427 +++++---- + lib/test_sysctl.c | 33 +- + tools/testing/selftests/sysctl/sysctl.sh | 1546 ++++++++++++------------------ + 7 files changed, 900 insertions(+), 1118 deletions(-) +Merging execve/for-next/execve (df2908090cda3 Linux 7.3-rc2) +$ git merge -m Merge branch 'for-next/execve' of https://git.kernel.org/pub/scm/linux/kernel/git/kees/linux.git execve/for-next/execve +Already up to date. +Merging bitmap/bitmap-for-next (452a6d5b5e055 bitmap: test bitmap_parselist() with a group size close to UINT_MAX) +$ git merge -m Merge branch 'bitmap-for-next' of https://github.com/norov/linux.git bitmap/bitmap-for-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 2 + + include/linux/bitfield-fix-width.h | 241 +++++++++++++++++++++++++++++++ + include/linux/bitfield.h | 49 +------ + lib/bitmap-str.c | 24 ++- + lib/test_bitmap.c | 9 ++ + tools/include/linux/bitfield-fix-width.h | 236 ++++++++++++++++++++++++++++++ + tools/include/linux/bitfield.h | 49 +------ + 7 files changed, 509 insertions(+), 101 deletions(-) + create mode 100644 include/linux/bitfield-fix-width.h + create mode 100644 tools/include/linux/bitfield-fix-width.h +Merging hte/for-next (30167fadbf87f hte: tegra194-test: shut down timer before GPIO release) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/pateldipen1984/linux.git hte/for-next +Merge made by the 'ort' strategy. + drivers/hte/hte-tegra194-test.c | 2 +- + drivers/hte/hte-tegra194.c | 14 ++++++-------- + 2 files changed, 7 insertions(+), 9 deletions(-) +Merging kspp/for-next/kspp (760f96b7f54be um: fix CONFIG_GCOV for built-in code) +$ git merge -m Merge branch 'for-next/kspp' of https://git.kernel.org/pub/scm/linux/kernel/git/kees/linux.git kspp/for-next/kspp +Auto-merging arch/um/include/asm/common.lds.S +Merge made by the 'ort' strategy. + arch/um/include/asm/common.lds.S | 1 + + drivers/misc/lkdtm/core.c | 16 +-- + fs/signalfd.c | 28 ++++- + include/linux/fortify-string.h | 2 - + scripts/coccinelle/api/kmalloc_objs.cocci | 161 +++++++++++++++++++++----- + scripts/gcc-plugins/randomize_layout_plugin.c | 63 +++++++++- + 6 files changed, 219 insertions(+), 52 deletions(-) +Merging nolibc/for-next (5f81ffdf26072 tools/nolibc: fix the function argument order of calloc()) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/nolibc/linux-nolibc.git nolibc/for-next +Auto-merging tools/testing/selftests/nolibc/Makefile.nolibc +Merge made by the 'ort' strategy. + tools/include/nolibc/Makefile | 19 ++- + tools/include/nolibc/arch-arm.h | 7 +- + tools/include/nolibc/arch-hexagon.h | 159 +++++++++++++++++++++++++ + tools/include/nolibc/arch-mips.h | 8 +- + tools/include/nolibc/arch-parisc.h | 28 ++++- + tools/include/nolibc/arch-powerpc.h | 7 +- + tools/include/nolibc/arch-sh.h | 18 +++ + tools/include/nolibc/arch-x86.h | 2 + + tools/include/nolibc/arch.h | 7 ++ + tools/include/nolibc/compiler.h | 9 ++ + tools/include/nolibc/dirent.h | 25 +++- + tools/include/nolibc/err.h | 2 +- + tools/include/nolibc/nolibc.h | 1 + + tools/include/nolibc/stdlib.h | 12 +- + tools/include/nolibc/sys.h | 60 ++++++++-- + tools/include/nolibc/sys/select.h | 6 +- + tools/include/nolibc/sys/sendfile.h | 41 +++++++ + tools/include/nolibc/types.h | 2 +- + tools/include/nolibc/unistd.h | 4 + + tools/testing/selftests/nolibc/Makefile.nolibc | 15 ++- + tools/testing/selftests/nolibc/nolibc-test.c | 159 +++++++++++++++++++++++-- + tools/testing/selftests/nolibc/run-tests.sh | 30 ++++- + 22 files changed, 560 insertions(+), 61 deletions(-) + create mode 100644 tools/include/nolibc/arch-hexagon.h + create mode 100644 tools/include/nolibc/sys/sendfile.h +Merging iommufd/for-next (54dadb030c7e2 iommufd: Fix iommufd_hw_capabilities reference in kernel-doc) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/jgg/iommufd.git iommufd/for-next +Auto-merging drivers/iommu/intel/nested.c +Merge made by the 'ort' strategy. + .../iommu/arm/arm-smmu-v3/arm-smmu-v3-iommufd.c | 199 ++++++++++++++++----- + drivers/iommu/intel/nested.c | 50 +++--- + drivers/iommu/iommufd/hw_pagetable.c | 25 +-- + drivers/iommu/iommufd/selftest.c | 163 ++++++++--------- + include/linux/iommu.h | 6 +- + include/linux/iommufd.h | 2 + + include/uapi/linux/iommufd.h | 6 +- + tools/testing/selftests/iommu/iommufd.c | 14 +- + tools/testing/selftests/iommu/iommufd_fail_nth.c | 14 +- + 9 files changed, 315 insertions(+), 164 deletions(-) +Merging turbostat/next (ccdfcb7e7ab84 turbostat 2026.09.28) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/lenb/linux.git turbostat/next +Merge made by the 'ort' strategy. + tools/power/x86/turbostat/turbostat.c | 36 ++++++++++++++++++++++------------- + 1 file changed, 23 insertions(+), 13 deletions(-) +Merging pwrseq/pwrseq/for-next (09baa2f4caab0 Merge tag 'pwrseq-is-controllable-for-v7.4' of git://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux into pwrseq/for-next) +$ git merge -m Merge branch 'pwrseq/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git pwrseq/pwrseq/for-next +Auto-merging drivers/power/sequencing/pwrseq-pcie-m2.c +Merge made by the 'ort' strategy. + drivers/power/sequencing/Kconfig | 19 + + drivers/power/sequencing/Makefile | 2 + + drivers/power/sequencing/core.c | 49 + + drivers/power/sequencing/pwrseq-kunit.c | 1497 ++++++++++++++++++++++ + drivers/power/sequencing/pwrseq-pcie-m2.c | 45 +- + drivers/power/sequencing/pwrseq-qcom-wcn.c | 30 + + drivers/power/sequencing/pwrseq-renesas-pwrrdy.c | 142 ++ + include/linux/pwrseq/consumer.h | 7 + + include/linux/pwrseq/provider.h | 8 + + 9 files changed, 1797 insertions(+), 2 deletions(-) + create mode 100644 drivers/power/sequencing/pwrseq-kunit.c + create mode 100644 drivers/power/sequencing/pwrseq-renesas-pwrrdy.c +Merging capabilities-next/caps-next (507adb6448378 security: commoncap: clarify CAP_FS_SET comment in cap_task_fix_setuid()) +$ git merge -m Merge branch 'caps-next' of https://git.kernel.org/pub/scm/linux/kernel/git/sergeh/linux.git capabilities-next/caps-next +Auto-merging security/commoncap.c +Merge made by the 'ort' strategy. + security/commoncap.c | 10 ++++++---- + 1 file changed, 6 insertions(+), 4 deletions(-) +$ git am -3 ../patches/0001-sign-file-Fix-up-merge-issue.patch +Applying: sign-file: Fix up merge issue +Using index info to reconstruct a base tree... +M scripts/sign-file.c +Falling back to patching base and 3-way merge... +Auto-merging scripts/sign-file.c +No changes -- Patch already applied. +Merging ipe/next (bcaa4d1d69368 ipe: fix invalid sgid value in audit event documentation) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/wufan/ipe.git ipe/next +Merge made by the 'ort' strategy. + Documentation/admin-guide/LSM/ipe.rst | 4 ++-- + 1 file changed, 2 insertions(+), 2 deletions(-) +Merging kcsan/next (a8488ecbd7ba4 kcsan: avoid unintended access checking in NMIs) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/melver/linux.git kcsan/next +Already up to date. +Merging crc/crc-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'crc-next' of https://git.kernel.org/pub/scm/linux/kernel/git/ebiggers/linux.git crc/crc-next +Already up to date. +Merging keys-next/keys-next (965e9a2cf23b0 pkcs7: Change a pr_warn() to pr_warn_once()) +$ git merge -m Merge branch 'keys-next' of https://git.kernel.org/pub/scm/linux/kernel/git/dhowells/linux-fs.git keys-next/keys-next +Already up to date. +Merging fwctl/for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/fwctl/fwctl.git fwctl/for-next +Already up to date. +Merging devsec-tsm/next (3177779ae17db virt: coco: change tsm_class to a const struct) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/devsec/tsm.git devsec-tsm/next +Already up to date. +Merging hisilicon/for-next (a5db65458a911 Merge branch 'next/dt64' into for-next) +$ git merge -m Merge branch 'for-next' of https://github.com/hisilicon/linux-hisi.git hisilicon/for-next +Merge made by the 'ort' strategy. +Merging device-id/device-id-rework (995832b2cebe6 Replace by more specific (c files)) +$ git merge -m Merge branch 'device-id-rework' of https://git.kernel.org/pub/scm/linux/kernel/git/ukleinek/linux.git device-id/device-id-rework +Already up to date. +Merging kthread/for-next (fa39ec4f89f26 doc: Add housekeeping documentation) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/frederic/linux-dynticks.git kthread/for-next +Already up to date. +Merging pagemap-headers/headers (e02cb91d4644d ksm: Remove pagemap.h include) +$ git merge -m Merge branch 'headers' of git://git.infradead.org/users/willy/pagecache.git pagemap-headers/headers +Auto-merging arch/alpha/include/asm/pgtable.h +Auto-merging arch/arc/include/asm/pgtable-levels.h +Auto-merging arch/arm/mach-pxa/pxa3xx.c +Auto-merging arch/microblaze/include/asm/pgtable.h +Auto-merging arch/sparc/include/asm/pgtable_64.h +Auto-merging arch/sparc/include/asm/tlb_64.h +Auto-merging arch/x86/virt/vmx/tdx/tdx.c +Auto-merging block/fops.c +Auto-merging drivers/dma/ste_dma40.c +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gpuvm.c +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_cs.c +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_gem.c +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_ttm.c +Auto-merging drivers/gpu/drm/amd/amdkfd/kfd_migrate.c +CONFLICT (content): Merge conflict in drivers/gpu/drm/amd/amdkfd/kfd_migrate.c +Auto-merging drivers/gpu/drm/amd/amdkfd/kfd_priv.h +Auto-merging drivers/gpu/drm/i915/display/intel_display_power.c +Auto-merging drivers/gpu/drm/msm/msm_fb.c +Auto-merging drivers/gpu/drm/msm/msm_gem_shrinker.c +Auto-merging drivers/gpu/drm/nouveau/nouveau_dmem.c +Auto-merging drivers/gpu/drm/omapdrm/dss/hdmi4.c +Auto-merging drivers/gpu/drm/omapdrm/dss/hdmi5.c +Auto-merging drivers/gpu/drm/panel/panel-ilitek-ili9806e-core.c +Auto-merging drivers/gpu/drm/panthor/panthor_gem.c +Auto-merging drivers/hwmon/pmbus/pmbus_core.c +Auto-merging drivers/i2c/busses/i2c-amd-asf-plat.c +Auto-merging drivers/i2c/busses/i2c-gxp.c +Auto-merging drivers/i2c/busses/i2c-k1.c +Auto-merging drivers/iio/accel/fxls8962af-core.c +Auto-merging drivers/iio/adc/ti-ads1298.c +Auto-merging drivers/iio/light/rohm-bu27034.c +Auto-merging drivers/iio/pressure/bmp280-core.c +Auto-merging drivers/leds/leds-turris-omnia.c +Auto-merging drivers/media/platform/st/stm32/stm32-csi.c +Auto-merging drivers/media/platform/ti/omap3isp/ispvideo.c +Auto-merging drivers/media/usb/go7007/go7007-v4l2.c +Auto-merging drivers/media/usb/gspca/gspca.c +Auto-merging drivers/mfd/88pm886.c +Auto-merging drivers/mfd/cs42l43.c +Auto-merging drivers/mfd/twl-core.c +Auto-merging drivers/mmc/core/core.c +Auto-merging drivers/mmc/core/host.c +Auto-merging drivers/mmc/host/sh_mmcif.c +Auto-merging drivers/net/ethernet/atheros/atl1e/atl1e.h +Auto-merging drivers/pinctrl/pinctrl-sx150x.c +Auto-merging drivers/platform/arm64/huawei-gaokun-ec.c +Auto-merging drivers/platform/arm64/lenovo-yoga-c630.c +Auto-merging drivers/power/supply/bq25630_charger.c +Auto-merging drivers/regulator/fixed.c +Auto-merging drivers/regulator/tps65185.c +Auto-merging drivers/regulator/tps6594-regulator.c +Auto-merging drivers/ufs/host/ufs-mediatek.c +Auto-merging drivers/ufs/host/ufs-qcom.c +Auto-merging drivers/usb/typec/mux/it5205.c +Auto-merging fs/aio.c +Auto-merging fs/buffer.c +Auto-merging fs/file_table.c +Auto-merging fs/inode.c +Auto-merging fs/nfs/blocklayout/blocklayout.c +Auto-merging fs/nfs/nfstrace.h +Auto-merging fs/nfs/pnfs.c +Auto-merging fs/nfsd/nfs4proc.c +Auto-merging include/drm/drm_print.h +Auto-merging include/linux/ceph/libceph.h +Auto-merging include/linux/i3c/master.h +Auto-merging include/linux/nfs_fs.h +Auto-merging include/linux/nfs_page.h +Auto-merging include/linux/suspend.h +Auto-merging include/linux/swap.h +Auto-merging kernel/power/snapshot.c +Auto-merging net/ceph/osd_client.c +CONFLICT (content): Merge conflict in net/ceph/osd_client.c +Auto-merging net/ceph/pagevec.c +Auto-merging net/sunrpc/auth_gss/gss_krb5_crypto.c +Auto-merging net/sunrpc/xdr.c +Auto-merging net/sunrpc/xprtsock.c +Auto-merging security/commoncap.c +Auto-merging security/selinux/hooks.c +Auto-merging security/selinux/selinuxfs.c +Auto-merging security/smack/smack_lsm.c +Resolved 'drivers/gpu/drm/amd/amdkfd/kfd_migrate.c' using previous resolution. +Resolved 'net/ceph/osd_client.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master d28412a35a639] Merge branch 'headers' of git://git.infradead.org/users/willy/pagecache.git +$ git diff -M --stat --summary HEAD^.. + arch/alpha/include/asm/pgtable.h | 2 +- + arch/arc/include/asm/Kbuild | 1 + + arch/arc/include/asm/pgtable-levels.h | 2 ++ + arch/arc/include/asm/tlb.h | 12 ---------- + arch/arm/include/asm/pgalloc.h | 2 -- + arch/arm/include/asm/tlb.h | 2 -- + arch/arm/mach-pxa/pxa3xx.c | 1 + + arch/arm64/include/asm/tlb.h | 3 --- + arch/hexagon/include/asm/tlb.h | 1 - + arch/microblaze/include/asm/pgtable.h | 2 +- + arch/nios2/include/asm/tlb.h | 1 - + arch/openrisc/include/asm/Kbuild | 1 + + arch/openrisc/include/asm/tlb.h | 26 ---------------------- + arch/powerpc/include/asm/tlb.h | 2 -- + arch/sh/include/asm/pgtable.h | 2 +- + arch/sh/include/asm/tlb.h | 1 - + arch/sparc/include/asm/pgtable_64.h | 2 +- + arch/sparc/include/asm/tlb_64.h | 1 - + arch/x86/include/asm/pgalloc.h | 1 - + arch/x86/virt/vmx/tdx/tdx.c | 1 + + arch/xtensa/kernel/hibernate.c | 1 + + block/fops.c | 1 + + drivers/char/agp/backend.c | 1 - + drivers/char/agp/generic.c | 1 - + drivers/char/agp/intel-agp.c | 1 - + drivers/char/agp/intel-gtt.c | 1 - + drivers/char/agp/uninorth-agp.c | 3 ++- + drivers/dma/ste_dma40.c | 1 + + drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gpuvm.c | 1 - + drivers/gpu/drm/amd/amdgpu/amdgpu_cs.c | 1 - + drivers/gpu/drm/amd/amdgpu/amdgpu_gem.c | 1 - + drivers/gpu/drm/amd/amdgpu/amdgpu_ttm.c | 3 +-- + drivers/gpu/drm/amd/amdkfd/kfd_migrate.c | 1 + + drivers/gpu/drm/amd/amdkfd/kfd_priv.h | 1 - + drivers/gpu/drm/display/drm_dp_cec.c | 1 + + drivers/gpu/drm/etnaviv/etnaviv_gpu.c | 1 + + drivers/gpu/drm/i915/display/intel_display_power.c | 1 + + drivers/gpu/drm/i915/gem/i915_gem_userptr.c | 1 + + drivers/gpu/drm/msm/msm_fb.c | 2 ++ + drivers/gpu/drm/msm/msm_gem_shrinker.c | 2 ++ + drivers/gpu/drm/nouveau/nouveau_dmem.c | 1 + + drivers/gpu/drm/omapdrm/dss/hdmi4.c | 1 + + drivers/gpu/drm/omapdrm/dss/hdmi5.c | 1 + + drivers/gpu/drm/panel/panel-ilitek-ili9806e-core.c | 1 + + drivers/gpu/drm/panfrost/panfrost_gem.c | 2 ++ + drivers/gpu/drm/panthor/panthor_gem.c | 3 +++ + drivers/hwmon/pmbus/pmbus_core.c | 1 + + drivers/i2c/busses/i2c-amd-asf-plat.c | 1 + + drivers/i2c/busses/i2c-gxp.c | 1 + + drivers/i2c/busses/i2c-k1.c | 1 + + drivers/i2c/busses/i2c-pasemi-platform.c | 1 + + drivers/iio/accel/fxls8962af-core.c | 1 + + drivers/iio/adc/nct7201.c | 1 + + drivers/iio/adc/ti-ads1298.c | 1 + + drivers/iio/light/rohm-bu27034.c | 1 + + drivers/iio/pressure/bmp280-core.c | 1 + + drivers/iio/pressure/bmp280.h | 1 + + drivers/input/misc/aw86927.c | 1 + + drivers/input/touchscreen/goodix_berlin_core.c | 1 + + drivers/input/touchscreen/hynitron_cstxxx.c | 1 + + drivers/input/touchscreen/imagis.c | 1 + + drivers/leds/leds-turris-omnia.c | 1 + + drivers/media/common/videobuf2/frame_vector.c | 1 - + drivers/media/pci/cx18/cx18-driver.h | 1 - + drivers/media/pci/ivtv/ivtv-driver.h | 2 +- + drivers/media/platform/st/stm32/stm32-csi.c | 1 + + drivers/media/platform/ti/omap3isp/ispvideo.c | 1 - + drivers/media/usb/go7007/go7007-v4l2.c | 1 - + drivers/media/usb/gspca/gspca.c | 1 - + drivers/mfd/88pm886.c | 1 + + drivers/mfd/abx500-core.c | 1 + + drivers/mfd/adp5585.c | 1 + + drivers/mfd/cs40l50-core.c | 1 + + drivers/mfd/cs42l43.c | 1 + + drivers/mfd/rt5120.c | 1 + + drivers/mfd/tps65219.c | 1 + + drivers/mfd/twl-core.c | 2 +- + drivers/mmc/core/core.c | 1 - + drivers/mmc/core/host.c | 1 - + drivers/mmc/host/renesas_sdhi_internal_dmac.c | 1 - + drivers/mmc/host/renesas_sdhi_sys_dmac.c | 1 - + drivers/mmc/host/sh_mmcif.c | 1 - + drivers/mmc/host/tmio_mmc.h | 1 - + drivers/mmc/host/tmio_mmc_core.c | 1 - + drivers/mmc/host/usdhi6rol0.c | 1 - + drivers/net/ethernet/atheros/atl1c/atl1c.h | 1 - + drivers/net/ethernet/atheros/atl1e/atl1e.h | 1 - + drivers/net/ethernet/intel/e1000/e1000.h | 1 - + drivers/net/ethernet/intel/e1000e/netdev.c | 1 - + drivers/net/ethernet/intel/igb/igb_main.c | 1 - + drivers/net/ethernet/intel/igbvf/netdev.c | 1 - + drivers/phy/st/phy-stm32-combophy.c | 1 + + drivers/pinctrl/pinctrl-sx150x.c | 1 + + drivers/platform/arm64/huawei-gaokun-ec.c | 1 + + drivers/platform/arm64/lenovo-yoga-c630.c | 2 +- + .../platform/x86/intel/int3472/clk_and_regulator.c | 1 + + drivers/power/supply/bq25630_charger.c | 1 + + drivers/power/supply/rk817_charger.c | 2 +- + drivers/regulator/fixed.c | 1 + + drivers/regulator/fp9931.c | 1 + + drivers/regulator/mt6360-regulator.c | 1 + + drivers/regulator/qcom-labibb-regulator.c | 1 + + drivers/regulator/rtq2208-regulator.c | 1 + + drivers/regulator/tps65185.c | 1 + + drivers/regulator/tps65219-regulator.c | 1 + + drivers/regulator/tps6594-regulator.c | 1 + + drivers/soc/xilinx/zynqmp_power.c | 1 + + drivers/ufs/host/ufs-mediatek.c | 1 + + drivers/ufs/host/ufs-qcom.c | 1 + + drivers/usb/core/hcd.c | 1 + + drivers/usb/typec/mux/it5205.c | 1 + + drivers/usb/typec/wusb3801.c | 1 + + drivers/video/fbdev/core/fb_procfs.c | 1 + + drivers/xen/balloon.c | 1 + + fs/aio.c | 1 + + fs/buffer.c | 1 + + fs/file_table.c | 1 + + fs/inode.c | 1 + + fs/nfs/blocklayout/blocklayout.c | 1 + + fs/nfs/nfs42proc.c | 1 + + fs/nfs/nfstrace.h | 1 + + fs/nfs/pnfs.c | 1 + + fs/nfsd/nfs4proc.c | 1 + + include/drm/drm_print.h | 1 + + include/drm/ttm/ttm_tt.h | 1 - + include/linux/balloon.h | 1 - + include/linux/ceph/libceph.h | 1 - + include/linux/i3c/master.h | 1 + + include/linux/ksm.h | 1 - + include/linux/mempolicy.h | 1 - + include/linux/nfs_fs.h | 1 - + include/linux/nfs_page.h | 1 - + include/linux/suspend.h | 1 - + include/linux/swap.h | 1 - + include/sound/cs35l41.h | 1 + + kernel/power/snapshot.c | 1 + + net/ceph/osd_client.c | 1 - + net/ceph/pagevec.c | 1 + + net/core/datagram.c | 1 - + net/rds/rdma.c | 1 - + net/sunrpc/auth_gss/auth_gss.c | 1 - + net/sunrpc/auth_gss/gss_krb5_crypto.c | 1 - + net/sunrpc/auth_gss/gss_krb5_wrap.c | 1 - + net/sunrpc/auth_gss/svcauth_gss.c | 1 - + net/sunrpc/cache.c | 1 - + net/sunrpc/rpc_pipe.c | 1 - + net/sunrpc/socklib.c | 1 - + net/sunrpc/xdr.c | 1 - + net/sunrpc/xprtsock.c | 1 - + security/commoncap.c | 3 --- + security/inode.c | 1 - + security/selinux/hooks.c | 1 - + security/selinux/selinuxfs.c | 1 - + security/smack/smack_lsm.c | 1 - + sound/soc/codecs/cs35l41-lib.c | 1 + + 155 files changed, 98 insertions(+), 118 deletions(-) + delete mode 100644 arch/arc/include/asm/tlb.h + delete mode 100644 arch/openrisc/include/asm/tlb.h diff --git a/localversion-next b/localversion-next new file mode 100644 index 00000000000000..461243b5184b01 --- /dev/null +++ b/localversion-next @@ -0,0 +1 @@ +-next-20261002 From 871d1d47bcf2cb3a45529a00e5f70813231897d5 Mon Sep 17 00:00:00 2001 From: Denis Benato Date: Tue, 18 Aug 2026 20:14:06 +0000 Subject: [PATCH 1334/1352] ogc: linux-unstable: first commit -- 2.47.3 (cherry picked from commit 536526d1b587ace66924207d8dcbfd0a50279aec) (cherry picked from commit 0ebc551a74d0ee721a6f59870a5a88ff88e42268) (cherry picked from commit 2e5980a43d81039ee5a4b02fc8ff6637fead084d) -- 2.47.3 -- 2.47.3 (cherry picked from commit 82fa59412110c4bc4906a7f0e76c12fb533a7ec2) -- 2.47.3 (cherry picked from commit d62da27d35f725a5e3aa04a18c3f2387893ad262) (cherry picked from commit 9f7995156eb955d05d0fe39e76c1f33d40db26e8) (cherry picked from commit fc175a9a34c517a967c218f74f2d3768ff18d543) (cherry picked from commit 340590505d17fce619d2dff2388fba7ac655d568) (cherry picked from commit 5059972b4faeaa769567deb0e6e367aa70850fa2) From dcc24ff7de806375729f5f4008a15de8c45eefd1 Mon Sep 17 00:00:00 2001 From: Denis Benato Date: Tue, 1 Sep 2026 19:15:25 +0000 Subject: [PATCH 1335/1352] [NOT_FOR-UPSTREAM] ogc: linux-unstable: add github workflow (cherry picked from commit 64a16187d10090ade4d9134f4047f343e9856346) (cherry picked from commit a2c9baf1d2b5c60a0126f9e226187545905993cd) (cherry picked from commit 0424d14eecfff8eae20e0b78f9a62950f024f459) (cherry picked from commit 1cbd534c2db8c6fb3b3153708c9cbcff4975cf35) (cherry picked from commit dd4b6df2d44cbbc9faa4be94077ba78ddd5d7921) (cherry picked from commit bd679ff9d02f2e19afc36652e76aa681d312a49b) --- .github/packaging/PKGBUILD | 262 ++++++++++++++++++ .github/packaging/boot-smoke-test.sh | 234 ++++++++++++++++ .github/packaging/config.fragment | 124 +++++++++ .github/packaging/fedora/kernel.spec | 279 +++++++++++++++++++ .github/packaging/merge-fragments.sh | 62 +++++ .github/workflows/build-arch-packages.yml | 163 +++++++++++ .github/workflows/build-debian-packages.yml | 292 ++++++++++++++++++++ .github/workflows/build-fedora-packages.yml | 207 ++++++++++++++ .github/workflows/build-kernel.yml | 129 +++++++++ .github/workflows/publish-release.yml | 157 +++++++++++ .github/workflows/sync-linux-next.yml | 115 ++++++++ .github/workflows/test-pr.yml | 227 +++++++++++++++ 12 files changed, 2251 insertions(+) create mode 100644 .github/packaging/PKGBUILD create mode 100755 .github/packaging/boot-smoke-test.sh create mode 100644 .github/packaging/config.fragment create mode 100644 .github/packaging/fedora/kernel.spec create mode 100644 .github/packaging/merge-fragments.sh create mode 100644 .github/workflows/build-arch-packages.yml create mode 100644 .github/workflows/build-debian-packages.yml create mode 100644 .github/workflows/build-fedora-packages.yml create mode 100644 .github/workflows/build-kernel.yml create mode 100644 .github/workflows/publish-release.yml create mode 100644 .github/workflows/sync-linux-next.yml create mode 100644 .github/workflows/test-pr.yml diff --git a/.github/packaging/PKGBUILD b/.github/packaging/PKGBUILD new file mode 100644 index 00000000000000..85dcd59ff2abe7 --- /dev/null +++ b/.github/packaging/PKGBUILD @@ -0,0 +1,262 @@ +# SPDX-License-Identifier: GPL-2.0-only +# Maintainer: OpenGamingCollective +# +# PKGBUILD for the linux-unstable-ogc kernel. +# +# It is designed to be built by the CI job defined in +# .github/workflows/build-kernel.yml, which stages the following next to this +# file before invoking makepkg: +# - linux.tar.gz -> tarball of the checked-out kernel tree (no VCS data) +# - config -> Arch linux-headers .config with the OGC +# kernel-packages config fragments already applied +# - config.fragment -> repo-local overrides merged on top of it +# +# Based on the official Arch Linux kernel PKGBUILD and on the (known to work +# with linux-next) https://github.com/NeroReflex/linux-bisector PKGBUILD. + +pkgbase=linux-unstable-ogc +# Placeholder: makepkg refuses an empty pkgver before pkgver() runs; the real +# version is computed dynamically by pkgver() once sources are extracted. +pkgver=0.0.0 +pkgrel=1 +pkgdesc='linux-next kernel for the Open Gaming Collective' +url='https://github.com/OpenGamingCollective/linux-unstable' +arch=(x86_64) +license=(GPL2) +makedepends=( + bc + cpio + gettext + libelf + pahole + perl + python + tar + xz + gcc + ccache + git + + llvm + clang + lld + + # CONFIG_RUST=y in the base config + rust + rust-bindgen +) +options=('!strip') +source=( + 'linux.tar.gz' + 'config' + 'config.fragment' +) +b2sums=( + 'SKIP' + 'SKIP' + 'SKIP' +) + +export KBUILD_BUILD_HOST=archlinux +export KBUILD_BUILD_USER=$pkgbase +export KBUILD_BUILD_TIMESTAMP="" +export CC="ccache clang" +export MAKEFLAGS="-j$(nproc)" + +_make() { + test -s version + LLVM=1 LLVM_IAS=1 WERROR=0 KBUILD_BUILD_TIMESTAMP="" make CC="$CC" KERNELRELEASE="$(.. + $(cat localversion-next) + # + the git short sha recorded by CI in .build_commit + # e.g. "7.2.0" + "-next-20260818" + ".g4ca2fc86" -> "7.2.0-next-20260818.g4ca2fc86". + # Pacman does not allow hyphens in pkgver, so they become underscores: + # 7.2.0-next-20260818.g4ca2fc86 -> 7.2.0_next_20260818.g4ca2fc86 + cd "$srcdir/linux" + local ver + ver="$(make -s kernelversion)$(cat localversion-next 2>/dev/null || true)" + if [[ -s .build_commit ]]; then + ver+=".g$(<.build_commit)" + fi + printf '%s\n' "${ver//-/_}" +} + +prepare() { + cd "$srcdir/linux" + + # glibc >= 2.42 made strstr/strchr const-correct; some trees hardcode + # -Werror in tools/lib/bpf which then fails to compile. Keep builds green. + if [[ -f tools/lib/bpf/Makefile ]]; then + sed -i 's/ -Werror -Wall/ -Wall/' tools/lib/bpf/Makefile || true + fi + + echo "Setting version..." + # localversion* files are picked up sorted by name, so the final kernel + # release becomes e.g.: 7.2.0-next-20260818-unstable-ogc-g4ca2fc86-1 + # The -g suffix comes from the .build_commit file the CI writes into + # the tarball (same trick linux-bisector uses with .bisector_commit). + local _commit="" + if [[ -s .build_commit ]]; then + _commit="-g$(<.build_commit)" + fi + echo "${pkgbase#linux}${_commit}" > localversion.10-pkgname + echo "-$pkgrel" > localversion.20-pkgrel + LLVM=1 LLVM_IAS=1 make defconfig + LLVM=1 LLVM_IAS=1 make -s kernelrelease > version + LLVM=1 LLVM_IAS=1 make mrproper + + echo "Setting config..." + cp ../config .config + _make olddefconfig + + echo "Applying OGC config fragment..." + scripts/kconfig/merge_config.sh -m .config ../config.fragment + _make olddefconfig + + diff -u ../config .config || : + + echo "Prepared $pkgbase version $( +# Env: KREL expected kernel release string; when set, the boot banner +# "Linux version " must appear on every console log. +# BOOT_SMOKE_TIMEOUT override the per-boot timeout in seconds +# (default: 300 under KVM, 1200 under TCG). +# Requires (installed by the calling CI job): qemu-system-x86_64, cpio, +# gzip, a C compiler (gcc or clang), an OVMF build for the UEFI +# leg (packages: ovmf / edk2-ovmf), coreutils (timeout, find). + +set -euo pipefail + +die() { echo "::error::$*" >&2; exit 1; } + +IMAGE="${1:?usage: boot-smoke-test.sh }" +[ -s "$IMAGE" ] || die "kernel image '$IMAGE' not found or empty" + +for tool in qemu-system-x86_64 cpio gzip timeout sha256sum find truncate; do + command -v "$tool" >/dev/null || die "required tool '$tool' not installed" +done +CC_BIN="$(command -v gcc || command -v clang || true)" +[ -n "$CC_BIN" ] || die "no C compiler (gcc or clang) found" + +WORK="$(mktemp -d /tmp/boot-smoke.XXXXXX)" +trap 'rm -rf "$WORK"' EXIT + +# --------------------------------------------------------------------------- +# PID 1 for the test initramfs: freestanding, no libc, raw x86_64 syscalls, +# so it compiles with gcc or clang on any distro without extra static-libc +# packages (the Fedora build env has no glibc-static, for example). +# --------------------------------------------------------------------------- +cat > "$WORK/init.c" <<'EOF' +/* + * Tiny freestanding PID 1 for the QEMU boot smoke test: print BOOT_OK on + * the serial console, then power the VM off so QEMU exits on its own. + */ +static long sys3(long nr, long a, long b, long c) +{ + long ret; + + __asm__ volatile ("syscall" + : "=a"(ret) + : "a"(nr), "D"(a), "S"(b), "d"(c) + : "rcx", "r11", "memory"); + return ret; +} + +void _start(void) +{ + static const char msg[] = + "BOOT_OK: linux-unstable-ogc booted to userspace init\n"; + long fd; + + /* CONFIG_DEVTMPFS_MOUNT does not apply to initramfs, so mount + * devtmpfs ourselves to get /dev/ttyS0. */ + sys3(83, (long)"/dev", 0755, 0); /* mkdir */ + sys3(165, (long)"devtmpfs", (long)"/dev", (long)"devtmpfs"); /* mount */ + + /* Announce success on the serial port and on stdio (fd 1 exists + * only when the kernel could open /dev/console from the initramfs). */ + fd = sys3(2, (long)"/dev/ttyS0", 1, 0); /* open */ + if (fd >= 0) + sys3(1, fd, (long)msg, sizeof(msg) - 1); /* write */ + sys3(1, 1, (long)msg, sizeof(msg) - 1); /* write */ + + /* reboot(LINUX_REBOOT_CMD_POWER_OFF) -> ACPI S5 -> QEMU exits. */ + sys3(169, 0xfee1deadL, 672274793L, 0x4321fedcL); + for (;;) + sys3(34, 0, 0, 0); /* pause */ +} +EOF + +mkdir -p "$WORK/initramfs/dev" +"$CC_BIN" -Os -g0 -static -no-pie -nostdlib -ffreestanding \ + -fno-stack-protector -fno-asynchronous-unwind-tables \ + -fno-unwind-tables -Wl,-z,noexecstack \ + -o "$WORK/initramfs/init" "$WORK/init.c" +# /dev/console lets the kernel wire init's stdio to the console; creating it +# needs mknod privileges and may fail for unprivileged callers. Harmless: the +# init above then opens /dev/ttyS0 itself after mounting devtmpfs. +mknod -m 600 "$WORK/initramfs/dev/console" c 5 1 2>/dev/null || true + +(cd "$WORK/initramfs" && find . -print0 | cpio --null -o -H newc --quiet) \ + | gzip -1 > "$WORK/initrd.img" + +# Use KVM when the runner exposes it, software emulation (TCG) otherwise. +# TCG boots a distro-config kernel in a couple of minutes; the generous +# timeout also covers hangs, which are exactly what this test must catch. +if [ -w /dev/kvm ]; then + ACCEL="kvm" + ACCEL_ARGS=(-accel kvm -cpu host) + TMO=300 +else + ACCEL="tcg" + ACCEL_ARGS=(-accel tcg -cpu max) + TMO=1200 +fi + +# --------------------------------------------------------------------------- +# Locate an OVMF build for the UEFI leg. Layouts, by distro: +# Debian: /usr/share/OVMF/OVMF_CODE(_4M).fd (package: ovmf) +# Fedora: /usr/share/edk2/ovmf/OVMF_CODE.fd (package: edk2-ovmf) +# Arch: /usr/share/edk2/x64/OVMF_CODE.4m.fd (package: edk2-ovmf) +# qemu: /usr/share/qemu/edk2-x86_64-code.fd (qemu-system-data) +# CODE and VARS must be the same flash size, hence the paired candidates. +# --------------------------------------------------------------------------- +OVMF_PAIR="" +for pair in \ + "/usr/share/OVMF/OVMF_CODE_4M.fd:/usr/share/OVMF/OVMF_VARS_4M.fd" \ + "/usr/share/OVMF/OVMF_CODE.fd:/usr/share/OVMF/OVMF_VARS.fd" \ + "/usr/share/edk2/x64/OVMF_CODE.4m.fd:/usr/share/edk2/x64/OVMF_VARS.4m.fd" \ + "/usr/share/edk2/x64/OVMF_CODE.fd:/usr/share/edk2/x64/OVMF_VARS.fd" \ + "/usr/share/edk2/ovmf/OVMF_CODE.fd:/usr/share/edk2/ovmf/OVMF_VARS.fd" \ + "/usr/share/ovmf/x64/OVMF_CODE.fd:/usr/share/ovmf/x64/OVMF_VARS.fd" \ + "/usr/share/qemu/edk2-x86_64-code.fd:/usr/share/qemu/edk2-x86_64-vars.fd" \ + ; do + if [ -r "${pair%%:*}" ] && [ -r "${pair##*:}" ]; then + OVMF_PAIR="$pair" + break + fi +done +[ -n "$OVMF_PAIR" ] || die "no OVMF (UEFI) firmware found; install the 'ovmf' (Debian) or 'edk2-ovmf' (Arch/Fedora) package" +OVMF_CODE="${OVMF_PAIR%%:*}" +cp "${OVMF_PAIR##*:}" "$WORK/OVMF_VARS.fd" # pflash needs a writable copy +echo "UEFI firmware: $OVMF_CODE" + +# Blank disks so the storage controllers have something to enumerate. +for d in nvme virtio scsi; do + truncate -s 64M "$WORK/disk-$d.img" +done + +# Device spread common to every boot: covers the storage/USB/net drivers a +# gaming kernel is most likely to boot from. Attached whether or not the +# kernel contains them; drivers that are modules are simply not probed, +# which costs nothing. +DRIVERS_ARGS=( + -drive if=none,id=dsk-nvme,format=raw,file="$WORK/disk-nvme.img" + -device nvme,drive=dsk-nvme,serial=ogcsmoke + -drive if=none,id=dsk-virtio,format=raw,file="$WORK/disk-virtio.img" + -device virtio-blk-pci,drive=dsk-virtio + -drive if=none,id=dsk-scsi,format=raw,file="$WORK/disk-scsi.img" + -device virtio-scsi-pci,id=scsi0 + -device scsi-hd,drive=dsk-scsi + -device e1000e,netdev=net0 -netdev user,id=net0,restrict=on + -device qemu-xhci -device usb-tablet +) + +# run_boot — boot, wait for QEMU to exit or the +# timeout to fire, then fail hard if the marker (or the expected release +# banner) is missing from the serial console log. +run_boot() { + local name="$1"; shift + local serial="$WORK/serial-$name.log" + : > "$serial" + + echo "[$name] Booting $(basename "$IMAGE") (sha256 $(sha256sum "$IMAGE" | cut -c1-16)...) with QEMU ($ACCEL, timeout ${TMO}s)" + timeout "$TMO" qemu-system-x86_64 "${ACCEL_ARGS[@]}" "$@" \ + "${DRIVERS_ARGS[@]}" \ + -m 2048 -smp 2 -nodefaults \ + -display none -monitor none -no-reboot \ + -serial "file:$serial" \ + -kernel "$IMAGE" -initrd "$WORK/initrd.img" \ + -append "console=ttyS0,115200n8 rdinit=/init panic=-1 nokaslr" \ + || true + # A panic with panic=-1 reboots instantly and -no-reboot makes QEMU + # exit, so both "qemu exited by itself" and "timeout killed it" end up + # in the marker check below. + + if ! grep -q "BOOT_OK" "$serial"; then + echo "::error::QEMU boot smoke test FAILED in the '$name' configuration: no BOOT_OK marker on the serial console (accel=$ACCEL, timeout=${TMO}s). The kernel is unbootable." + echo "----- last 250 lines of the $name guest serial console -----" + tail -n 250 "$serial" || true + echo "-------------------------------------------------------------" + exit 1 + fi + if [ -n "${KREL:-}" ] && ! grep -qF "Linux version ${KREL} " "$serial"; then + echo "::error::QEMU boot smoke test FAILED in the '$name' configuration: the booted kernel banner does not advertise release '${KREL}'." + grep -m1 "^Linux version" "$serial" || true + exit 1 + fi + echo "[$name] Boot smoke test PASSED: kernel booted to userspace init and powered off. Serial console tail:" + tail -n 10 "$serial" || true +} + +# 1. Classic BIOS boot: SeaBIOS firmware, i440fx (PIIX IDE built in). +run_boot bios-pc -machine pc + +# 2. UEFI boot: OVMF firmware, q35 (ICH9 AHCI + PCIe built in), kernel +# launched through the EFI stub — the path handhelds actually use. +run_boot uefi-q35 \ + -machine q35 \ + -drive if=pflash,format=raw,readonly=on,file="$OVMF_CODE" \ + -drive if=pflash,format=raw,file="$WORK/OVMF_VARS.fd" + +if [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + { + echo '### QEMU boot smoke test' + echo + echo "- Image: $(basename "$IMAGE") (sha256 $(sha256sum "$IMAGE" | cut -c1-12)...)" + echo "- Accelerator: ${ACCEL}, timeout ${TMO}s per boot" + echo "- Matrix: bios-pc (SeaBIOS) and uefi-q35 (OVMF EFI stub), NVMe + virtio-blk + virtio-scsi + e1000e + xHCI attached" + echo '- Result: BOOT_OK in both configurations - kernel decompressed, booted and reached userspace init' + } >> "$GITHUB_STEP_SUMMARY" +fi diff --git a/.github/packaging/config.fragment b/.github/packaging/config.fragment new file mode 100644 index 00000000000000..8d455d78cc2417 --- /dev/null +++ b/.github/packaging/config.fragment @@ -0,0 +1,124 @@ +# OGC config overrides for linux-unstable-ogc +# +# This fragment is merged LAST, on top of the OGC config (Arch linux-headers +# base + OpenGamingCollective/kernel-packages config/*.{set,unset} fragments) +# via scripts/kconfig/merge_config.sh + olddefconfig in the PKGBUILD. +# +# Target hardware: modern gaming desktops and handhelds (Steam Deck, ROG Ally, +# Legion Go, MSI Claw, GPD, Ayaneo, OneXPlayer...). Keep IMU/gyro/accel/IIO +# drivers, Bluetooth, UVC webcams and all OGC handheld drivers (those live in +# drivers/hid, drivers/platform/x86 — NOT in staging, so STAGING can go). + +# --- Kernel identity -------------------------------------------------------- +# Distro base configs carry their own release suffix (Arch "-arch1", Debian +# "-1-amd64", ...). It must be empty so that uname -r is identical across +# the Arch/Fedora/Debian packages: -unstable-ogc-g-1 (the suffix +# itself comes from the localversion.10/20 files written by the packaging). +CONFIG_LOCALVERSION="" +# CONFIG_LOCALVERSION_AUTO is not set + +# --- Build fixes ------------------------------------------------------------ +# drivers/net/ethernet/cadence/macb_main.c fails to compile in linux-next +# (implicit declarations of macb_alloc_tieoff/macb_free_tieoff). Cadence +# MACB/GEM is an ARM SoC NIC, never present on x86_64 gaming hardware. +# MACB_PCI and MACB_USE_HWSTAMP depend on MACB and drop out automatically. +# CONFIG_MACB is not set + +# --- Build speed ------------------------------------------------------------ +# Keep DEBUG_INFO/DEBUG_INFO_BTF=y (OGC requires them for BPF/sched_ext) but +# skip running pahole on every module (thousands of invocations saved). +# CONFIG_DEBUG_INFO_BTF_MODULES is not set + +# --- Xen guest support (not running inside a Xen VM) ------------------------ +# CONFIG_XEN is not set + +# --- Ancient graphics / buses ------------------------------------------------ +# AGP is a pre-PCIe bus; radeon covers pre-GCN AMD GPUs (12+ years old). +# CONFIG_AGP is not set +# CONFIG_DRM_RADEON is not set +# PCMCIA/CardBus slots (2000s laptops) +# CONFIG_PCCARD is not set +# FireWire (IEEE 1394) ports +# CONFIG_FIREWIRE is not set +# Parallel port (printers/scanners from the 90s) +# CONFIG_PARPORT is not set +# IndustryPack carrier boards +# CONFIG_IPACK_BUS is not set +# 1-Wire bus +# CONFIG_W1 is not set +# Analog gameport joysticks (pre-USB) +# CONFIG_GAMEPORT is not set +# Floppy disk controller +# CONFIG_BLK_DEV_FD is not set +# Server BMC management +# CONFIG_IPMI_HANDLER is not set + +# --- Enterprise / datacenter storage ---------------------------------------- +# Fibre Channel HBAs and enterprise RAID controllers +# CONFIG_SCSI_LPFC is not set +# CONFIG_SCSI_QLA_FC is not set +# CONFIG_QEDF is not set +# CONFIG_FCOE is not set +# CONFIG_LIBFC is not set +# CONFIG_MEGARAID_SAS is not set +# CONFIG_MEGARAID_LEGACY is not set +# CONFIG_MEGARAID_MAILBOX is not set +# CONFIG_FUSION is not set +# CONFIG_SCSI_HPSA is not set +# CONFIG_SCSI_SMARTPQI is not set + +# --- Ancient Ethernet NICs --------------------------------------------------- +# Vendor menus for 10/100 hardware from the 90s/00s (3Com, Tulip, NatSemi, +# NE2000-era, SiS 900, VIA Rhine/Velocity, Adaptec starfire). Modern NICs +# (Intel/Realtek/Marvell/Broadcom/Aquantia) are untouched. +# CONFIG_NET_VENDOR_3COM is not set +# CONFIG_NET_VENDOR_ADAPTEC is not set +# CONFIG_NET_TULIP is not set +# CONFIG_NET_VENDOR_NATSEMI is not set +# CONFIG_NET_VENDOR_8390 is not set +# CONFIG_NET_VENDOR_SIS is not set +# CONFIG_NET_VENDOR_VIA is not set + +# --- Specialized/legacy networking ------------------------------------------ +# CONFIG_ATM is not set +# CONFIG_RDS is not set +# CONFIG_TIPC is not set +# CONFIG_PHONET is not set +# CONFIG_IEEE802154 is not set +# CONFIG_CAN is not set +# CONFIG_NFC is not set +# CONFIG_INFINIBAND is not set + +# --- Analog/digital TV, radio, legacy webcams -------------------------------- +# UVC (USB_VIDEO_CLASS) webcams stay enabled — handhelds use them. +# CONFIG_MEDIA_ANALOG_TV_SUPPORT is not set +# CONFIG_MEDIA_DIGITAL_TV_SUPPORT is not set +# CONFIG_MEDIA_RADIO_SUPPORT is not set +# CONFIG_USB_GSPCA is not set + +# --- Ancient/cluster filesystems ---------------------------------------------- +# Desktop FS (ext4/btrfs/xfs/f2fs/exfat/ntfs3/vfat/nfs/cifs) untouched. +# CONFIG_MINIX_FS is not set +# CONFIG_UFS_FS is not set +# CONFIG_BFS_FS is not set +# CONFIG_GFS2_FS is not set +# CONFIG_OCFS2_FS is not set +# CONFIG_AFS_FS is not set + +# --- Chemical / gas / air-quality sensors ------------------------------------ +# IMU/gyro/accelerometer/pressure/temp/humidity IIO drivers are KEPT +# (handheld controllers need them); only gas/VOC/CO2/PM sensors removed. +# CONFIG_CCS811 is not set +# CONFIG_SPS30 is not set +# CONFIG_PMS7003 is not set +# CONFIG_SENSIRION_SGP30 is not set +# CONFIG_SENSIRION_SGP40 is not set +# CONFIG_SCD30_CORE is not set +# CONFIG_SCD4X is not set +# CONFIG_VZ89X is not set +# CONFIG_BME680 is not set + +# --- Staging drivers ---------------------------------------------------------- +# All OGC handheld drivers live in mainline trees (drivers/hid, +# drivers/platform/x86), not staging. +# CONFIG_STAGING is not set \ No newline at end of file diff --git a/.github/packaging/fedora/kernel.spec b/.github/packaging/fedora/kernel.spec new file mode 100644 index 00000000000000..2165fefce635a9 --- /dev/null +++ b/.github/packaging/fedora/kernel.spec @@ -0,0 +1,279 @@ +# SPDX-License-Identifier: GPL-2.0-only +# +# RPM spec for the linux-unstable-ogc kernel (linux-next based). +# +# Built by .github/workflows/build-kernel.yml ("fedora" job) which, before +# invoking rpmbuild: +# - substitutes @@KBASEVER@@ / @@KVERDOTTED@@ / @@SHA8@@ placeholders below +# - stages SOURCES/linux.tar.gz (kernel tree contents, root dir "linux/", +# no VCS data, localversion-next included) +# - stages SOURCES/config (Fedora kernel-core .config with the OGC +# kernel-packages fragments and the local +# config.fragment already applied) +# +# Derived from the OpenGamingCollective/kernel-packages fedora/kernel.spec +# (itself based on CachyOS/Nobara), trimmed down to core/modules/devel only. +# The kernel is compiled with clang (LLVM=1) exactly like the Arch packages. + +%global _default_patch_fuzz 2 + +# See https://fedoraproject.org/wiki/Changes/SetBuildFlagsBuildCheck +%if 0%{?fedora} >= 37 +%undefine _auto_set_build_flags +%endif + +%define _build_id_links none +%define _disable_source_fetch 1 +# no debuginfo generation, no brp strip/mangle of kernel binaries +%define debug_package %{nil} +%define __spec_install_post /usr/lib/rpm/brp-compress || : + +# ---- substituted by CI ------------------------------------------------------ +%define kbasever @@KBASEVER@@ +%define sha8 @@SHA8@@ +# ---------------------------------------------------------------------------- + +Version: @@KVERDOTTED@@ +Release: 1.g%{sha8}%{?dist} + +%define rpmver %{version}-%{release} +# Kernel release string (uname -r). Identical to what the localversion* files +# below produce during the build, and identical to the Arch packages built +# from the same commit: +# -unstable-ogc-g-1 +%define kverstr %{kbasever}-unstable-ogc-g%{sha8}-1 +# RPM dependency versions may contain at most ONE hyphen (V-R separator), so +# the full kernel release string cannot be used as a Provides: version. Use a +# 1:1 dotted translation for the *-uname-r provides instead. +%define kverdot %(echo "%{kverstr}" | sed -e "s/-/./g") + +Name: kernel-unstable-ogc +Summary: The linux-next kernel for the Open Gaming Collective +License: GPLv2 +URL: https://github.com/OpenGamingCollective/linux-unstable +Group: System Environment/Kernel +ExclusiveArch: x86_64 +Source0: linux.tar.gz +Source1: config + +BuildRequires: bash, coreutils, make, tar, findutils, gawk, diffutils, m4 +BuildRequires: bc, bison, flex, perl-interpreter, perl-Carp, binutils +BuildRequires: xz, zstd, kmod, python3 +BuildRequires: elfutils-libelf-devel, elfutils-devel +BuildRequires: openssl, openssl-devel +BuildRequires: dwarves, hmaccalc +# clang toolchain (like the Arch packages) +BuildRequires: clang, lld, llvm, ccache +# needed when the base config enables CONFIG_RUST. The bindgen binary +# package was renamed from rust-bindgen to bindgen in newer Fedora releases; +# accept either so this spec works across distro versions. +BuildRequires: rust, rust-src +BuildRequires: (bindgen or rust-bindgen) + +# All kernel make invocations: clang via ccache, deterministic version strings +%define kmake make CC="ccache clang" LLVM=1 LLVM_IAS=1 WERROR=0 KBUILD_BUILD_HOST=ogc-ci KBUILD_BUILD_USER=kernel-unstable-ogc KBUILD_BUILD_TIMESTAMP="" + +%description +This package is a meta package that pulls in the linux-unstable-ogc kernel +(a linux-next snapshot for the Open Gaming Collective) and its matching +modules. + +%package core +Summary: The linux-unstable-ogc kernel (vmlinuz and core files) +Group: System Environment/Kernel +Provides: installonlypkg(kernel) +Provides: %{name}-core-uname-r = %{kverdot} +Requires: bash, coreutils, kmod +Requires: /usr/bin/kernel-install +Requires: %{name}-modules = %{rpmver} +Recommends: linux-firmware +%description core +This package contains the linux-unstable-ogc kernel image (vmlinuz), +System.map, the build configuration and the module symbol version file. + +%package modules +Summary: Kernel modules to match the linux-unstable-ogc core kernel +Group: System Environment/Kernel +Provides: installonlypkg(kernel-module) +Provides: %{name}-modules-uname-r = %{kverdot} +Supplements: %{name}-core = %{rpmver} +# kmod needed for depmod in %%post +Requires: kmod +%description modules +This package provides the kernel modules for the linux-unstable-ogc kernel. + +%package devel +Summary: Development files for building external modules +Group: Development/System +AutoReqProv: no +Requires: findutils, make, perl-interpreter, flex, bison +Requires: elfutils-libelf-devel, openssl-devel, gcc +Requires: clang, llvm, lld +Provides: %{name}-devel-uname-r = %{kverdot} +Enhances: akmods +Enhances: dkms +%description devel +This package provides the headers, scripts and tooling (objtool, +resolve_btfids) needed to build out-of-tree kernel modules against the +linux-unstable-ogc kernel (%{kverstr}). + +%prep +%setup -q -n linux + +cp %{SOURCE1} .config + +# The Fedora distro config references Fedora-only certificate files that do +# not exist in this tree; use the ephemeral in-tree key instead. +scripts/config --set-str SYSTEM_TRUSTED_KEYS "" +scripts/config --set-str SYSTEM_REVOCATION_KEYS "" +# CONFIG_MODULE_SIG_KEY points at the Red Hat signing cert in the distro +# config, which does not exist in this tree. Reset it to the kbuild default +# ("certs/signing_key.pem"): an ephemeral self-signed key generated during +# the build and trusted by the kernel itself. The Fedora config sets +# CONFIG_MODULE_SIG_ALL=y, and with an empty key string scripts/Makefile.modinst +# resolves sig-key to "./", making sign-file read a directory as the private +# key (SSL DECODER error) on every module. +scripts/config --set-str MODULE_SIG_KEY "certs/signing_key.pem" + +# Deterministic / distro-agnostic build identity +scripts/config -u DEFAULT_HOSTNAME +scripts/config --set-str BUILD_SALT "%{kverstr}" + +# Kernel release suffix, same scheme as the Arch packages: +# -unstable-ogc-g-1 +echo "-unstable-ogc-g%{sha8}" > localversion.10-pkgname +echo "-1" > localversion.20-pkgrel + +# glibc >= 2.42 const-correctness vs -Werror in tools/lib/bpf (used by +# resolve_btfids when DEBUG_INFO_BTF=y); keep the build green. +if [ -f tools/lib/bpf/Makefile ]; then + sed -i 's/ -Werror -Wall/ -Wall/' tools/lib/bpf/Makefile || true +fi + +%{kmake} olddefconfig + +# Fail fast if the release string ever drifts from the spec +REL="$(make -s kernelrelease)" +if [ "$REL" != "%{kverstr}" ]; then + echo "kernelrelease '$REL' does not match spec kverstr '%{kverstr}'" >&2 + exit 1 +fi +cp .config config-linux-unstable-ogc + +%build +%{kmake} %{?_smp_mflags} all + +%install +MODDIR="%{buildroot}/lib/modules/%{kverstr}" +DEVEL="%{buildroot}%{_prefix}/src/kernels/%{kverstr}" + +mkdir -p "%{buildroot}/boot" "$MODDIR" + +echo "Installing boot image..." +ImageName="$(make -s image_name | tail -n 1)" +install -m 0644 "$ImageName" "$MODDIR/vmlinuz" +chmod 0755 "$MODDIR/vmlinuz" + +echo "Installing modules..." +# Modules are compressed by modules_install itself (CONFIG_MODULE_COMPRESS_*). +# depmod runs from %%post at install time. +%{kmake} %{?_smp_mflags} KERNELRELEASE=%{kverstr} \ + INSTALL_MOD_PATH=%{buildroot} INSTALL_MOD_STRIP=1 \ + DEPMOD=/doesnt/exist modules_install + +echo "Installing core files..." +cp System.map "$MODDIR/System.map" +cp .config "$MODDIR/config" +gzip -c9 < Module.symvers > "$MODDIR/symvers.gz" +(cd "$MODDIR" && sha512hmac vmlinuz > .vmlinuz.hmac) + +# ---- kernel-devel ----------------------------------------------------------- +echo "Preparing kernel-devel..." +rm -f "$MODDIR"/build "$MODDIR"/source +ln -s "%{_prefix}/src/kernels/%{kverstr}" "$MODDIR/build" +(cd "$MODDIR" && ln -s build source) +mkdir -p "$MODDIR"/updates "$MODDIR"/weak-updates "$DEVEL" + +find . -type f \( -name 'Makefile*' -o -name 'Kconfig*' \) -print0 \ + | xargs -0 cp --parents -t "$DEVEL" +cp -a include "$DEVEL"/ +cp -a arch/x86/include "$DEVEL"/arch/x86/ +if [ -f arch/x86/kernel/module.lds ]; then + cp -a --parents arch/x86/kernel/module.lds "$DEVEL"/ +fi +cp -a scripts "$DEVEL"/ +rm -rf "$DEVEL"/scripts/tracing +rm -f "$DEVEL"/scripts/spdxcheck.py +cp Module.symvers System.map .config "$DEVEL"/ + +mkdir -p "$DEVEL"/tools/{objtool,bpf/resolve_btfids,lib,build} +cp -a tools/objtool/objtool "$DEVEL"/tools/objtool/ || : +cp -a tools/bpf/resolve_btfids/resolve_btfids "$DEVEL"/tools/bpf/resolve_btfids/ || : +cp -a tools/include "$DEVEL"/tools/ +cp -a tools/lib/subcmd "$DEVEL"/tools/lib/ +cp -a tools/lib/bpf "$DEVEL"/tools/lib/ +cp -a tools/build/Build.include tools/build/fixdep.c "$DEVEL"/tools/build/ +cp -a tools/scripts/utilities.mak "$DEVEL"/tools/scripts/ 2>/dev/null || : +cp -a --parents arch/x86/entry/syscalls/syscall_32.tbl "$DEVEL"/ +cp -a --parents arch/x86/entry/syscalls/syscall_64.tbl "$DEVEL"/ +cp -a arch/x86/tools "$DEVEL"/arch/x86/ + +# Drop intermediate build artifacts from devel tree +find "$DEVEL" \( -name '*.o' -o -name '*.cmd' -o -name '.*.cmd' \) -delete + +# Timestamps must line up so external module builds do not rerun kconfig +touch -r "$DEVEL"/Makefile \ + "$DEVEL"/include/generated/uapi/linux/version.h \ + "$DEVEL"/include/config/auto.conf + +%post core +# nothing to do at this point + +%posttrans core +# Runs after ALL packages of this transaction have been installed and their +# %%post scriptlets (incl. depmod from -modules) have run. +if [ -x /usr/bin/kernel-install ]; then + /usr/bin/kernel-install add %{kverstr} /lib/modules/%{kverstr}/vmlinuz || exit $? +fi +if [ -x /usr/sbin/grubby ]; then + grubby --set-default="/boot/vmlinuz-%{kverstr}" || : +fi + +%preun core +if [ "$1" = "0" ] && [ -x /usr/bin/kernel-install ]; then + /usr/bin/kernel-install remove %{kverstr} /lib/modules/%{kverstr}/vmlinuz || exit $? +fi + +%post modules +/sbin/depmod -a %{kverstr} + +%files +# meta package: everything lives in the subpackages + +%files core +%ghost /boot/vmlinuz-%{kverstr} +%ghost /boot/initramfs-%{kverstr}.img +/lib/modules/%{kverstr}/vmlinuz +/lib/modules/%{kverstr}/.vmlinuz.hmac +/lib/modules/%{kverstr}/System.map +/lib/modules/%{kverstr}/config +/lib/modules/%{kverstr}/symvers.gz + +%files modules +/lib/modules/%{kverstr}/ +%exclude /lib/modules/%{kverstr}/vmlinuz +%exclude /lib/modules/%{kverstr}/.vmlinuz.hmac +%exclude /lib/modules/%{kverstr}/System.map +%exclude /lib/modules/%{kverstr}/config +%exclude /lib/modules/%{kverstr}/symvers.gz +%exclude /lib/modules/%{kverstr}/build +%exclude /lib/modules/%{kverstr}/source + +%files devel +/usr/src/kernels/%{kverstr} +/lib/modules/%{kverstr}/build +/lib/modules/%{kverstr}/source + +%changelog +* Thu Aug 20 2026 OpenGamingCollective CI +- Initial linux-unstable-ogc spec, generated by CI from linux-next. \ No newline at end of file diff --git a/.github/packaging/merge-fragments.sh b/.github/packaging/merge-fragments.sh new file mode 100644 index 00000000000000..e8525b2d86ce8e --- /dev/null +++ b/.github/packaging/merge-fragments.sh @@ -0,0 +1,62 @@ +#!/usr/bin/env bash +# Textually merge kernel config fragments into a base .config file. +# +# Usage: merge-fragments.sh [...] +# +# Recognized fragment line formats (fragments are applied in order, later +# fragments win): +# CONFIG_X=value set X to value +# CONFIG_X unset X (kernel-configurator *.unset format) +# "# CONFIG_X is not set" unset X (kconfig fragment format) +# Blank lines and other comments are ignored. +# +# This replicates the semantics of the OpenGamingCollective +# kernel-configurator action (*.config.set / *.config.unset fragments) and +# additionally accepts regular kconfig fragment files, so both the OGC +# fragments and the repo-local config.fragment can be handled uniformly. +# The final `make olddefconfig` (run by the PKGBUILD / RPM spec) turns the +# textual result into a consistent kconfig. + +set -euo pipefail + +if [ "$#" -lt 2 ]; then + echo "usage: $0 ..." >&2 + exit 2 +fi + +CFG="$(realpath "$1")" +shift + +set_key() { + local key="$1" value="$2" + if grep -q "^${key}=" "$CFG"; then + sed -i "s|^${key}=.*|${key}=${value}|" "$CFG" + elif grep -q "^# ${key} is not set$" "$CFG"; then + sed -i "s|^# ${key} is not set\$|${key}=${value}|" "$CFG" + else + printf '%s=%s\n' "${key}" "${value}" >> "$CFG" + fi +} + +unset_key() { + local key="$1" + if grep -q "^${key}=" "$CFG"; then + sed -i "s|^${key}=.*|# ${key} is not set|" "$CFG" + fi +} + +for frag in "$@"; do + frag="$(realpath "$frag")" + while IFS= read -r line || [ -n "$line" ]; do + line="${line#"${line%%[![:space:]]*}"}" + line="${line%"${line##*[![:space:]]}"}" + [ -z "$line" ] && continue + if [[ "$line" =~ ^#\ (CONFIG_[A-Za-z0-9_]+)\ is\ not\ set$ ]]; then + unset_key "${BASH_REMATCH[1]}" + elif [[ "$line" =~ ^(CONFIG_[A-Za-z0-9_]+)=(.*)$ ]]; then + set_key "${BASH_REMATCH[1]}" "${BASH_REMATCH[2]}" + elif [[ "$line" =~ ^(CONFIG_[A-Za-z0-9_]+)$ ]]; then + unset_key "${BASH_REMATCH[1]}" + fi + done < "$frag" +done \ No newline at end of file diff --git a/.github/workflows/build-arch-packages.yml b/.github/workflows/build-arch-packages.yml new file mode 100644 index 00000000000000..748a422d732aa4 --- /dev/null +++ b/.github/workflows/build-arch-packages.yml @@ -0,0 +1,163 @@ +# Arch Linux packaging, split out of build-kernel.yml. +# Called (via `uses:`) by build-kernel.yml; runs inside the same workflow +# run, so the artifacts uploaded here are seen by publish-release.yml. + +name: Build Arch Linux packages + +on: + workflow_call: + inputs: + krel: + description: Full kernel release string (uname -r), e.g. 6.12.0-next-20250101-unstable-ogc-g12345678-1 + type: string + required: true + +permissions: + contents: read + +env: + # Config fragments maintained by the OGC kernel-packages repository. + # The distro base config is extracted from the official Arch + # `linux-headers` package — same approach as the kernel-packages + # arch.yaml workflow. + KERNEL_PACKAGES_RAW: https://raw.githubusercontent.com/OpenGamingCollective/kernel-packages/main + +jobs: + build: + name: Build Arch Linux packages + runs-on: ubuntu-latest + timeout-minutes: 330 + container: + image: docker.io/archlinux:base-devel + env: + PKGDIR: /tmp/pkgbuild + DISTDIR: /tmp/dist + CCACHE_DIR: /ccache + CCACHE_MAXSIZE: 10G + KREL: ${{ inputs.krel }} + steps: + - name: Show disk space + run: df -h / + + - name: Bootstrap Arch Linux build environment + run: | + set -euxo pipefail + # Refresh keyring first to avoid signature failures on stale images + pacman -Sy --needed --noconfirm archlinux-keyring + pacman -Su --needed --noconfirm + # Everything makepkg needs (mirrors the makedepends of the PKGBUILD). + # rust + rust-bindgen are needed because the Arch config enables + # CONFIG_RUST=y. + pacman -S --needed --noconfirm \ + bc cpio gettext libelf pahole perl python tar xz zstd \ + gcc clang llvm lld ccache pigz file curl git \ + rust rust-bindgen + # makepkg refuses to run as root + useradd -m build + install -d -o build -g build -m 0777 /ccache + + - name: Download staged sources + uses: actions/download-artifact@v8 + with: + name: kernel-sources + path: /tmp/stage + + - name: Restore compiler cache + uses: actions/cache@v6 + with: + path: /ccache + key: ccache-arch-${{ github.sha }} + restore-keys: | + ccache-arch- + + - name: Prepare ccache for the build user + run: chown -R build:build /ccache && su build -c 'ccache -s' || true + + - name: Assemble Arch kernel config + working-directory: /tmp/stage + run: | + set -euxo pipefail + # ------------------------------------------------------------------ + # Base config: extract the official Arch Linux kernel .config from + # the distro's `linux-headers` package (same method as the OGC + # kernel-packages arch.yaml workflow). + # ------------------------------------------------------------------ + CACHE="/tmp/pkgcache" + mkdir -p "$CACHE" + pacman -Sw --noconfirm --cachedir "$CACHE" linux-headers + PKG="$(find "$CACHE" -maxdepth 1 -name 'linux-headers-*.pkg.tar.zst' -type f)" + if [ -z "$PKG" ]; then + echo "::error::linux-headers package not found in cache"; exit 1 + fi + CFG="$(tar --zstd -tf "$PKG" | grep -E '^usr/lib/modules/[^/]+/build/\.config$' || true)" + if [ -z "$CFG" ]; then + echo "::error::kernel .config not found inside $PKG"; exit 1 + fi + tar --zstd -xOf "$PKG" "$CFG" > config + test -s config + rm -rf "$CACHE" + + # ------------------------------------------------------------------ + # Layer the OGC config fragments from kernel-packages on top, then + # the repo-local config.fragment (which wins). Order is critical: + # unsets apply first (removing things we explicitly don't want), + # then sets apply (enabling things we do want), so explicit enables + # can override explicit disables. + # ------------------------------------------------------------------ + for f in arch.config.set ogc.config.set arch.config.unset ogc.config.unset; do + curl -fsSL "${KERNEL_PACKAGES_RAW}/config/${f}" -o "${f}" + done + bash ./merge-fragments.sh config \ + arch.config.unset ogc.config.unset \ + arch.config.set ogc.config.set \ + config.fragment + echo "Config after fragment merge (head):" + head -n 3 config + + - name: Stage PKGBUILD directory + run: | + set -euxo pipefail + mkdir -p "${PKGDIR}" "${DISTDIR}" + cd /tmp/stage + cp PKGBUILD config.fragment linux.tar.gz config "${PKGDIR}/" + chown -R build:build "${PKGDIR}" + + - name: Build packages with makepkg + run: | + set -euxo pipefail + cd "${PKGDIR}" + runuser -u build -- env HOME=/home/build CCACHE_DIR=/ccache \ + makepkg -f --noconfirm --noprogressbar + + ls -lh ./*.pkg.tar.zst + cp -v ./*.pkg.tar.zst "${DISTDIR}/" + # Ship the exact .config used for the build as well + cp -v "src/linux/.config" "${DISTDIR}/config-arch-${KREL}" + + - name: Boot smoke test in QEMU (release gate) + run: | + set -euxo pipefail + pacman -S --needed --noconfirm qemu-system-x86 edk2-ovmf + # ---------------------------------------------------------------- + # Boot the exact kernel image that ships in the package. This + # step fails the job (and with it the release) if the kernel + # panics, hangs or never reaches userspace, so an unbootable + # kernel is never uploaded or published. + # ---------------------------------------------------------------- + WORK=/tmp/bootsmoke + rm -rf "$WORK"; mkdir -p "$WORK" + PKG="$(find /tmp/dist -name 'linux-unstable-ogc-*.pkg.tar.zst' ! -name '*-headers-*' | head -n1)" + test -n "$PKG" + tar --zstd -xf "$PKG" -C "$WORK" "usr/lib/modules/${KREL}/vmlinuz" + test -s "$WORK/usr/lib/modules/${KREL}/vmlinuz" + bash /tmp/stage/boot-smoke-test.sh "$WORK/usr/lib/modules/${KREL}/vmlinuz" + + - name: Clean build tree (free disk before cache save) + run: rm -rf "${PKGDIR}/src" "${PKGDIR}/pkg" ; df -h / + + - name: Upload Arch packages + uses: actions/upload-artifact@v7 + with: + name: arch-packages + path: /tmp/dist + retention-days: 3 diff --git a/.github/workflows/build-debian-packages.yml b/.github/workflows/build-debian-packages.yml new file mode 100644 index 00000000000000..6eef922dc09eb4 --- /dev/null +++ b/.github/workflows/build-debian-packages.yml @@ -0,0 +1,292 @@ +# Debian .deb packaging, split out of build-kernel.yml. +# Called (via `uses:`) by build-kernel.yml; runs inside the same workflow +# run, so the artifacts uploaded here are seen by publish-release.yml. + +name: Build Debian packages + +on: + workflow_call: + inputs: + krel: + description: Full kernel release string (uname -r), e.g. 6.12.0-next-20250101-unstable-ogc-g12345678-1 + type: string + required: true + kverdot: + description: Kernel version with hyphens replaced by dots, for KDEB_PKGVERSION + type: string + required: true + sha8: + description: Short (8 char) source commit the packages are built from + type: string + required: true + +permissions: + contents: read + +env: + # Config fragments maintained by the OGC kernel-packages repository. + # The distro base config comes from the official Debian + # linux-config/linux-image packages. + KERNEL_PACKAGES_RAW: https://raw.githubusercontent.com/OpenGamingCollective/kernel-packages/main + +jobs: + build: + name: Build Debian packages + runs-on: ubuntu-latest + timeout-minutes: 330 + container: + image: docker.io/library/debian:trixie + env: + CCACHE_DIR: /ccache + CCACHE_MAXSIZE: 10G + KREL: ${{ inputs.krel }} + KVERDOT: ${{ inputs.kverdot }} + SHA8: ${{ inputs.sha8 }} + steps: + - name: Show disk space + run: df -h / + + - name: Install build tools + run: | + set -euxo pipefail + apt-get update + # Satisfies the Build-Depends generated by scripts/package/mkdebian + # (debhelper-compat, bc, bison, flex, kmod, libdw/libelf/libssl-dev, + # python3, rsync) plus the clang toolchain used by all OGC builds. + apt-get install -y --no-install-recommends \ + build-essential debhelper rsync \ + bc bison flex python3 kmod \ + libelf-dev libdw-dev libssl-dev zlib1g-dev \ + dwarves zstd xz-utils \ + clang llvm lld ccache curl ca-certificates git \ + openssl + # dpkg-buildpackage refuses to run as root + useradd -m builder + install -d -o builder -g builder -m 0777 /ccache + + - name: Download staged sources + uses: actions/download-artifact@v8 + with: + name: kernel-sources + path: /tmp/stage + + - name: Restore compiler cache + uses: actions/cache@v6 + with: + path: /ccache + key: ccache-debian-${{ github.sha }} + restore-keys: | + ccache-debian- + + - name: Assemble Debian kernel config + working-directory: /tmp/stage + run: | + set -euxo pipefail + # ------------------------------------------------------------------ + # Base config: the official Debian kernel configuration. It ships in + # the small linux-config- package; fall back to /boot/config-* + # of the linux-image- package if that is unavailable. + # ------------------------------------------------------------------ + ABI="$(apt-cache depends linux-image-amd64 \ + | awk '/^ *Depends: *linux-image-[0-9]/ {print $2; exit}' \ + | sed 's/^linux-image-//')" + test -n "${ABI}" + echo "Debian kernel ABI: ${ABI}" + # linux-config is versioned by major.minor only (e.g. 6.12) and + # stores the per-flavour config xz-compressed, e.g. + # /usr/src/linux-config-6.12/config.amd64_none_amd64.xz + KMAJMIN="$(printf '%s' "${ABI}" | cut -d. -f1,2)" + mkdir -p /tmp/pkgcfg + if apt-get download "linux-config-${KMAJMIN}"; then + dpkg-deb -x linux-config-"${KMAJMIN}"_*.deb /tmp/pkgcfg + BASE="$(find /tmp/pkgcfg/usr/src -name 'config.amd64_none_amd64.xz' | head -n1)" + test -s "${BASE}" + xz -dc "${BASE}" > config + else + apt-get download "linux-image-${ABI}" + dpkg-deb -x linux-image-"${ABI}"_*.deb /tmp/pkgcfg + BASE="$(find /tmp/pkgcfg/boot -maxdepth 1 -name 'config-*' | head -n1)" + test -s "${BASE}" + cp "${BASE}" config + fi + test -s config + rm -rf /tmp/pkgcfg + + # ------------------------------------------------------------------ + # Layer the OGC config fragments and the repo-local config.fragment + # (which wins). kernel-packages ships no debian-specific fragments. + # Order is critical: unsets apply first, then sets, so explicit + # enables can override explicit disables. + # ------------------------------------------------------------------ + for f in ogc.config.set ogc.config.unset; do + curl -fsSL "${KERNEL_PACKAGES_RAW}/config/${f}" -o "${f}" + done + bash ./merge-fragments.sh config \ + ogc.config.unset \ + ogc.config.set \ + config.fragment + echo "Config after fragment merge (head):" + head -n 3 config + + - name: Ensure pahole >= 1.31 (sched_ext/BTF) + run: | + set -euxo pipefail + VER="$(pahole --version | tr -d 'v')" + MAJ="${VER%%.*}"; MIN="${VER#*.}"; MIN="${MIN%%.*}" + echo "Installed pahole: ${VER}" + if [ "$MAJ" -lt 1 ] || { [ "$MAJ" -eq 1 ] && [ "$MIN" -lt 31 ]; }; then + echo "pahole < 1.31 breaks sched_ext; building dwarves 1.31 from source" + apt-get install -y --no-install-recommends cmake pkg-config + cd /tmp + curl -fsSLO https://fedorapeople.org/~acme/dwarves/dwarves-1.31.tar.xz + echo "0a7f255ccacf8cc7f8cd119099eb327179b4b3c67cb015af646af6d0cb03054d dwarves-1.31.tar.xz" | sha256sum -c + tar -xf dwarves-1.31.tar.xz + cmake -B dwarves-1.31/build -S dwarves-1.31 \ + -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr + make -C dwarves-1.31/build -j"$(nproc)" install + fi + pahole --version + + - name: Build .deb packages with the in-tree packaging + run: | + set -euxo pipefail + mkdir -p /build + tar -xzf /tmp/stage/linux.tar.gz -C /build + cd /build/linux + + # Config adjustments + kernel release suffix, identical scheme to + # the Arch and Fedora packages (uname -r == ${KREL}). + cp /tmp/stage/config .config + # Debian's config references distro-only certificate files that do + # not exist in this tree; use the ephemeral in-tree key instead. + scripts/config --set-str SYSTEM_TRUSTED_KEYS "" + scripts/config --set-str SYSTEM_REVOCATION_KEYS "" + # CONFIG_MODULE_SIG_KEY points at the Debian signing cert, which + # does not exist in this tree. Reset it to the kbuild default + # ("certs/signing_key.pem"): an ephemeral self-signed key generated + # during the build and trusted by the kernel itself. The Debian + # config sets CONFIG_MODULE_SIG_ALL=y, and with an empty key string + # scripts/Makefile.modinst resolves sig-key to "./", making + # sign-file read a directory as the private key (SSL DECODER error) + # on every module. + scripts/config --set-str MODULE_SIG_KEY "certs/signing_key.pem" + scripts/config -u DEFAULT_HOSTNAME + scripts/config --set-str BUILD_SALT "${KREL}" + echo "-unstable-ogc-g${SHA8}" > localversion.10-pkgname + echo "-1" > localversion.20-pkgrel + # The staged tarball carries no VCS data, but mkdebian (via + # gen-diff-patch) runs "git diff HEAD" and aborts on failure. + # A throwaway repo with everything committed makes that a no-op. + git init -q + git config user.email "ci@opengamingcollective.org" + git config user.name "OpenGamingCollective CI" + git add -A + git commit -qm "linux-unstable-ogc ${KREL}" + chown -R builder:builder /build + + # scripts/setlocalversion appends a "+" whenever a git repo exists + # and LOCALVERSION is unset and the HEAD is not at an annotated + # version tag. The throwaway repo below would therefore corrupt the + # release string. Setting LOCALVERSION to the empty string (as + # documented in the script) suppresses that suffix everywhere. + runuser -u builder -- env HOME=/home/builder CCACHE_DIR=/ccache \ + LOCALVERSION= \ + make CC="ccache clang" LLVM=1 LLVM_IAS=1 WERROR=0 \ + KBUILD_BUILD_HOST=ogc-ci KBUILD_BUILD_USER=kernel-unstable-ogc \ + olddefconfig + + # Fail fast if the release string ever drifts from the other distros + REL="$(runuser -u builder -- env HOME=/home/builder LOCALVERSION= make -s kernelrelease)" + if [ "$REL" != "${KREL}" ]; then + echo "kernelrelease '$REL' does not match expected '${KREL}'" >&2 + exit 1 + fi + + # Generate the debian/ directory with the tree's own packaging. + # mkdebian is invoked directly as a script: a bare "make debian" is + # NOT a top-level make target in this tree (only *-pkg patterns are + # delegated to scripts/Makefile.package), and going through make + # without the exact same CC flags as the olddefconfig above made + # kbuild re-sync the config interactively. As a plain sh script it + # touches nothing kbuild-related. + # KDEB_PKGVERSION must not contain hyphens except the final revision + # separator (Debian policy), hence the dotted translation. + KDEBVER="${KVERDOT}.unstable.ogc.g${SHA8}-1" + runuser -u builder -- env HOME=/home/builder \ + srctree="$PWD" \ + ARCH=x86_64 SRCARCH=x86 UTS_MACHINE=x86_64 \ + KERNELRELEASE="${KREL}" \ + KCONFIG_CONFIG=.config \ + KDEB_SOURCENAME=linux-unstable-ogc \ + KDEB_PKGVERSION="${KDEBVER}" \ + KDEB_CHANGELOG_DIST=trixie \ + DEBFULLNAME="OpenGamingCollective CI" \ + DEBEMAIL="ci@opengamingcollective.org" \ + sh scripts/package/mkdebian + echo "debian arch: $(cat debian/arch)" + + # Kbuild only honours command-line variables, so inject the + # clang/ccache toolchain into the generated debian/rules (this is + # what "make bindeb-pkg" would otherwise lose). LOCALVERSION= (set, + # empty) keeps setlocalversion from appending "+" inside the build. + sed -i 's|^make-opts = |make-opts = CC="ccache clang" LLVM=1 LLVM_IAS=1 WERROR=0 LOCALVERSION= KBUILD_BUILD_HOST=ogc-ci KBUILD_BUILD_USER=kernel-unstable-ogc |' debian/rules + grep -n '^make-opts' debian/rules + + # Same invocation as "make bindeb-pkg" (scripts/Makefile.package), + # but with parallel jobs, --no-check-builddeps (the build deps are + # preinstalled above; the generated Build-Depends-Arch also names a + # distro-only cross-gcc package that need not exist), and without + # the multi-GB debug-symbol package (all other distro jobs ship no + # debug packages either). + runuser -u builder -- env HOME=/home/builder CCACHE_DIR=/ccache \ + LOCALVERSION= \ + DEB_BUILD_PROFILES="pkg.linux-unstable-ogc.nokerneldbg" \ + dpkg-buildpackage --build=binary --no-pre-clean --unsigned-changes \ + --no-check-builddeps \ + -R'make -f debian/rules' -j"$(nproc)" -a"$(cat debian/arch)" + + ls -lh /build/*.deb + + - name: Collect Debian artifacts + run: | + set -euxo pipefail + DISTDIR=/tmp/dist + mkdir -p "${DISTDIR}" + cp -v "/build/linux-image-${KREL}"_*.deb "${DISTDIR}/" + cp -v "/build/linux-headers-${KREL}"_*.deb "${DISTDIR}/" + cp -v /build/linux/.config "${DISTDIR}/config-debian-${KREL}" + # linux-libc-dev is intentionally NOT shipped: installing it would + # replace the distribution's own linux-libc-dev package. + ls -lh "${DISTDIR}" + + - name: Boot smoke test in QEMU (release gate) + run: | + set -euxo pipefail + apt-get update -qq + # qemu boots the kernel (OVMF provides the UEFI leg); cpio builds + # the test initramfs. + apt-get install -y --no-install-recommends qemu-system-x86 cpio ovmf + # ---------------------------------------------------------------- + # Boot the exact kernel image that ships in the linux-image .deb. + # This step fails the job (and with it the release) if the kernel + # panics, hangs or never reaches userspace, so an unbootable + # kernel is never uploaded or published. + # ---------------------------------------------------------------- + WORK=/tmp/bootsmoke + rm -rf "$WORK"; mkdir -p "$WORK" + DEB="$(find /tmp/dist -name "linux-image-${KREL}_*.deb" | head -n1)" + test -n "$DEB" + dpkg-deb -x "$DEB" "$WORK" + IMG="$(find "$WORK" -name 'vmlinuz-*' -type f | head -n1)" + test -n "$IMG" + bash /tmp/stage/boot-smoke-test.sh "$IMG" + + - name: Clean build tree (free disk before cache save) + run: rm -rf /build ; df -h / + + - name: Upload Debian packages + uses: actions/upload-artifact@v7 + with: + name: debian-packages + path: /tmp/dist + retention-days: 3 diff --git a/.github/workflows/build-fedora-packages.yml b/.github/workflows/build-fedora-packages.yml new file mode 100644 index 00000000000000..0a95d181f8ab77 --- /dev/null +++ b/.github/workflows/build-fedora-packages.yml @@ -0,0 +1,207 @@ +# Fedora RPM packaging, split out of build-kernel.yml. +# Called (via `uses:`) by build-kernel.yml; runs inside the same workflow +# run, so the artifacts uploaded here are seen by publish-release.yml. + +name: Build Fedora RPM packages + +on: + workflow_call: + inputs: + krel: + description: Full kernel release string (uname -r), e.g. 6.12.0-next-20250101-unstable-ogc-g12345678-1 + type: string + required: true + kbase: + description: Base kernel version incl. prerelease suffix but no packaging suffix (e.g. 6.12.0-next-20250101) + type: string + required: true + kverdot: + description: Kernel version with hyphens replaced by dots, for the RPM Version field + type: string + required: true + sha8: + description: Short (8 char) source commit the packages are built from + type: string + required: true + +permissions: + contents: read + +env: + # Config fragments maintained by the OGC kernel-packages repository. + # The distro base config is extracted from the official Fedora + # `kernel-core` package — same approach as the kernel-packages + # fedora.yaml workflow. + KERNEL_PACKAGES_RAW: https://raw.githubusercontent.com/OpenGamingCollective/kernel-packages/main + +jobs: + build: + name: Build Fedora RPM packages + runs-on: ubuntu-latest + timeout-minutes: 330 + container: + image: docker.io/library/fedora:43 + env: + CCACHE_DIR: /ccache + CCACHE_MAXSIZE: 10G + KREL: ${{ inputs.krel }} + KBASE: ${{ inputs.kbase }} + KVERDOT: ${{ inputs.kverdot }} + SHA8: ${{ inputs.sha8 }} + steps: + - name: Show disk space + run: df -h / + + - name: Install build tools + run: | + set -euxo pipefail + dnf -y install dnf5-plugins rpm-build + dnf -y install cpio curl findutils tar gzip + mkdir -p /ccache + + - name: Download staged sources + uses: actions/download-artifact@v8 + with: + name: kernel-sources + path: /tmp/stage + + - name: Restore compiler cache + uses: actions/cache@v6 + with: + path: /ccache + key: ccache-fedora-${{ github.sha }} + restore-keys: | + ccache-fedora- + + - name: Assemble Fedora kernel config + working-directory: /tmp/stage + run: | + set -euxo pipefail + # ------------------------------------------------------------------ + # Base config: extract the official Fedora kernel .config from the + # distro's `kernel-core` package (same method as the OGC + # kernel-packages fedora.yaml workflow). + # ------------------------------------------------------------------ + CACHE="/tmp/pkgcache" + mkdir -p "$CACHE" + dnf download --destdir "$CACHE" kernel-core + RPM="$(find "$CACHE" -maxdepth 1 -name 'kernel-core-*.rpm' -type f)" + if [ -z "$RPM" ]; then + echo "::error::kernel-core package not found in cache"; exit 1 + fi + CFG="$(rpm -qlp "$RPM" | grep -E '^/lib/modules/[^/]+/config$' | head -n1 || true)" + if [ -z "$CFG" ]; then + echo "::error::kernel config not found inside $RPM"; exit 1 + fi + rpm2cpio "$RPM" | cpio -i --to-stdout ".${CFG}" > config + test -s config + rm -rf "$CACHE" + + # ------------------------------------------------------------------ + # Layer the OGC config fragments on top, then the repo-local + # config.fragment (which wins). Order is critical: unsets apply + # first (removing things we explicitly don't want), then sets apply + # (enabling things we do want), so explicit enables can override + # explicit disables. + # ------------------------------------------------------------------ + for f in fedora.config.set ogc.config.set fedora.config.unset ogc.config.unset; do + curl -fsSL "${KERNEL_PACKAGES_RAW}/config/${f}" -o "${f}" + done + bash ./merge-fragments.sh config \ + fedora.config.unset ogc.config.unset \ + fedora.config.set ogc.config.set \ + config.fragment + echo "Config after fragment merge (head):" + head -n 3 config + + - name: Finalize spec and install build dependencies + working-directory: /tmp/stage + run: | + set -euxo pipefail + # Substitute the CI placeholders (must happen before `dnf builddep` + # so RPM can parse Version:/Release:). + sed -i \ + -e "s/@@KBASEVER@@/${KBASE}/" \ + -e "s/@@KVERDOTTED@@/${KVERDOT}/" \ + -e "s/@@SHA8@@/${SHA8}/" \ + kernel.spec + grep -n '^Version:\|^Release:\|%define kbasever\|%define sha8' kernel.spec + ! grep -q '@@' kernel.spec + + # Installs the toolchain from the spec BuildRequires, including the + # distro dwarves/pahole. Run BEFORE the pahole check below so a + # source-built pahole is not overwritten by builddep. + dnf -y builddep kernel.spec + + - name: Ensure pahole >= 1.31 (sched_ext/BTF) + working-directory: /tmp + run: | + set -euxo pipefail + VER="$(pahole --version | tr -d 'v')" + MAJ="${VER%%.*}"; MIN="${VER#*.}"; MIN="${MIN%%.*}" + echo "Installed pahole: ${VER}" + if [ "$MAJ" -lt 1 ] || { [ "$MAJ" -eq 1 ] && [ "$MIN" -lt 31 ]; }; then + echo "pahole < 1.31 breaks sched_ext; building dwarves 1.31 from source" + dnf -y install cmake make gcc elfutils-devel zlib-devel + curl -fsSLO https://fedorapeople.org/~acme/dwarves/dwarves-1.31.tar.xz + echo "0a7f255ccacf8cc7f8cd119099eb327179b4b3c67cb015af646af6d0cb03054d dwarves-1.31.tar.xz" | sha256sum -c + tar -xf dwarves-1.31.tar.xz + cmake -B dwarves-1.31/build -S dwarves-1.31 \ + -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -D__LIB=lib + make -C dwarves-1.31/build -j"$(nproc)" install + fi + pahole --version + + - name: Build RPMs with rpmbuild + working-directory: /tmp + run: | + set -euxo pipefail + TOPDIR=/tmp/rpmbuild + mkdir -p "${TOPDIR}"/{BUILD,BUILDROOT,RPMS,SOURCES,SPECS,SRPMS} + cp /tmp/stage/linux.tar.gz "${TOPDIR}/SOURCES/" + cp /tmp/stage/config "${TOPDIR}/SOURCES/config" + cp /tmp/stage/kernel.spec "${TOPDIR}/SPECS/kernel.spec" + + rpmbuild --define "_topdir ${TOPDIR}" -ba "${TOPDIR}/SPECS/kernel.spec" + + ls -lh "${TOPDIR}"/RPMS/x86_64/ + + - name: Collect Fedora artifacts + run: | + set -euxo pipefail + DISTDIR=/tmp/dist + mkdir -p "${DISTDIR}" + cp -v /tmp/rpmbuild/RPMS/x86_64/*.rpm "${DISTDIR}/" + cp -v /tmp/stage/config "${DISTDIR}/config-fedora-${KREL}" + ls -lh "${DISTDIR}" + + - name: Boot smoke test in QEMU (release gate) + run: | + set -euxo pipefail + dnf -y install qemu-system-x86 edk2-ovmf + # ---------------------------------------------------------------- + # Boot the exact kernel image that ships in the kernel-core RPM. + # This step fails the job (and with it the release) if the kernel + # panics, hangs or never reaches userspace, so an unbootable + # kernel is never uploaded or published. + # ---------------------------------------------------------------- + WORK=/tmp/bootsmoke + rm -rf "$WORK"; mkdir -p "$WORK" + RPM="$(find /tmp/dist -name 'kernel-unstable-ogc-core-*.rpm' | head -n1)" + test -n "$RPM" + # rpm cpio entries are stored with a "./" prefix (same trick as + # the config extraction above). + rpm2cpio "$RPM" | cpio -idm --quiet -D "$WORK" "./lib/modules/${KREL}/vmlinuz" || true + IMG="$WORK/lib/modules/${KREL}/vmlinuz" + test -s "$IMG" || { echo "::error::vmlinuz not found inside ${RPM}"; exit 1; } + bash /tmp/stage/boot-smoke-test.sh "$IMG" + + - name: Clean build tree (free disk before cache save) + run: rm -rf /tmp/rpmbuild/BUILD /tmp/rpmbuild/BUILDROOT ; df -h / + + - name: Upload Fedora packages + uses: actions/upload-artifact@v7 + with: + name: fedora-packages + path: /tmp/dist + retention-days: 3 diff --git a/.github/workflows/build-kernel.yml b/.github/workflows/build-kernel.yml new file mode 100644 index 00000000000000..8ca98fa697d2d6 --- /dev/null +++ b/.github/workflows/build-kernel.yml @@ -0,0 +1,129 @@ +# Entry point of the linux-unstable-ogc build pipeline. The heavy lifting +# lives in reusable workflows next to this file: +# +# build-arch-packages.yml - Arch Linux .pkg.tar.zst builds +# build-fedora-packages.yml - Fedora RPM builds +# build-debian-packages.yml - Debian .deb builds +# publish-release.yml - collects artifacts and cuts the GitHub release +# +# They are called with `uses:` so everything stays a single workflow run: +# artifacts uploaded by the build jobs are downloaded by publish-release.yml, +# and the needs: chain below keeps the release gated on the QEMU boot smoke +# tests that run inside each build workflow. + +name: Build & release linux-unstable-ogc + +on: + push: + branches: [master] + workflow_dispatch: {} + +permissions: + contents: write + +concurrency: + group: kernel-build-${{ github.ref }} + cancel-in-progress: true + +jobs: + prepare: + name: Prepare sources and version + runs-on: ubuntu-latest + outputs: + kver: ${{ steps.kernel.outputs.kver }} + next: ${{ steps.kernel.outputs.next }} + sha8: ${{ steps.kernel.outputs.sha8 }} + krel: ${{ steps.kernel.outputs.krel }} + tag: ${{ steps.kernel.outputs.tag }} + kbase: ${{ steps.kernel.outputs.kbase }} + kverdot: ${{ steps.kernel.outputs.kverdot }} + steps: + - name: Checkout kernel sources + uses: actions/checkout@v7 + with: + fetch-depth: 1 + + - name: Compute kernel version + id: kernel + run: | + set -euxo pipefail + KVER="$(make -s kernelversion)" + NEXT="$(cat localversion-next 2>/dev/null || true)" + SHA8="$(git rev-parse --short=8 HEAD)" + KREL="${KVER}${NEXT}-unstable-ogc-g${SHA8}-1" + TAG="v${KVER}${NEXT}-g${SHA8}" + # RPM Version: field must not contain hyphens + KVERDOT="$(printf '%s%s' "${KVER}" "${NEXT}" | tr '-' '.')" + echo "kver=${KVER}" >> "$GITHUB_OUTPUT" + echo "next=${NEXT}" >> "$GITHUB_OUTPUT" + echo "sha8=${SHA8}" >> "$GITHUB_OUTPUT" + echo "krel=${KREL}" >> "$GITHUB_OUTPUT" + echo "tag=${TAG}" >> "$GITHUB_OUTPUT" + echo "kbase=${KVER}${NEXT}" >> "$GITHUB_OUTPUT" + echo "kverdot=${KVERDOT}" >> "$GITHUB_OUTPUT" + echo "Kernel release: ${KREL}" + echo "Release tag: ${TAG}" + + - name: Stage sources and packaging files + run: | + set -euxo pipefail + # Commit marker consumed by the Arch PKGBUILD (same trick as + # linux-bisector). Written before the tarball is created so it is + # included in it. + git rev-parse --short=8 HEAD > .build_commit + + STAGE="/tmp/stage" + mkdir -p "${STAGE}" + cp .github/packaging/PKGBUILD "${STAGE}/PKGBUILD" + cp .github/packaging/merge-fragments.sh "${STAGE}/merge-fragments.sh" + cp .github/packaging/config.fragment "${STAGE}/config.fragment" + cp .github/packaging/fedora/kernel.spec "${STAGE}/kernel.spec" + cp .github/packaging/boot-smoke-test.sh "${STAGE}/boot-smoke-test.sh" + + # Source tarball: contents of the repo as ./linux/, without VCS data. + tar --exclude-vcs -I 'gzip -1' -cf "${STAGE}/linux.tar.gz" \ + --transform 's|^\./|linux/|' -C "$GITHUB_WORKSPACE" . + ls -lh "${STAGE}" + + - name: Upload staged sources + uses: actions/upload-artifact@v7 + with: + name: kernel-sources + path: /tmp/stage + retention-days: 3 + + arch: + name: Arch packages + needs: prepare + uses: ./.github/workflows/build-arch-packages.yml + with: + krel: ${{ needs.prepare.outputs.krel }} + + fedora: + name: Fedora packages + needs: prepare + uses: ./.github/workflows/build-fedora-packages.yml + with: + krel: ${{ needs.prepare.outputs.krel }} + kbase: ${{ needs.prepare.outputs.kbase }} + kverdot: ${{ needs.prepare.outputs.kverdot }} + sha8: ${{ needs.prepare.outputs.sha8 }} + + debian: + name: Debian packages + needs: prepare + uses: ./.github/workflows/build-debian-packages.yml + with: + krel: ${{ needs.prepare.outputs.krel }} + kverdot: ${{ needs.prepare.outputs.kverdot }} + sha8: ${{ needs.prepare.outputs.sha8 }} + + release: + name: Publish GitHub release + needs: [prepare, arch, fedora, debian] + uses: ./.github/workflows/publish-release.yml + with: + tag: ${{ needs.prepare.outputs.tag }} + krel: ${{ needs.prepare.outputs.krel }} + kver: ${{ needs.prepare.outputs.kver }} + next: ${{ needs.prepare.outputs.next }} diff --git a/.github/workflows/publish-release.yml b/.github/workflows/publish-release.yml new file mode 100644 index 00000000000000..e9f641a6b36644 --- /dev/null +++ b/.github/workflows/publish-release.yml @@ -0,0 +1,157 @@ +# Release publisher, split out of build-kernel.yml. +# Called (via `uses:`) by build-kernel.yml after all three distro builds +# (each gated on its QEMU boot smoke test) have succeeded. + +name: Publish linux-unstable-ogc release + +on: + workflow_call: + inputs: + tag: + description: Release tag, e.g. v6.12.0-next-20250101-g12345678 + type: string + required: true + krel: + description: Full kernel release string (uname -r) + type: string + required: true + kver: + description: Base kernel version from `make kernelversion` + type: string + required: true + next: + description: localversion-next suffix (may be empty) + type: string + required: true + +permissions: + contents: write + +env: + # Config fragments maintained by the OGC kernel-packages repository + # (referenced in the release notes). + KERNEL_PACKAGES_RAW: https://raw.githubusercontent.com/OpenGamingCollective/kernel-packages/main + +jobs: + release: + name: Create GitHub release + runs-on: ubuntu-latest + env: + TAG: ${{ inputs.tag }} + KREL: ${{ inputs.krel }} + KVER: ${{ inputs.kver }} + NEXT: ${{ inputs.next }} + steps: + - name: Download package artifacts + uses: actions/download-artifact@v8 + with: + path: /tmp/dist + pattern: "*-packages" + + - name: Flatten artifact directory + run: | + set -euxo pipefail + cd /tmp/dist + find . -mindepth 2 -maxdepth 2 -type f -exec mv -t . {} + + find . -mindepth 1 -type d -delete + ls -lh + + - name: Generate checksums + run: | + set -euxo pipefail + cd /tmp/dist + sha256sum * > SHA256SUMS + cat SHA256SUMS + + - name: Create GitHub release and upload packages + env: + GH_TOKEN: ${{ github.token }} + run: | + set -euxo pipefail + + # Idempotency: if a stale release exists for this tag (e.g. re-run + # of the same commit), remove it so this run can recreate it. + if gh release view "$TAG" --repo "$GITHUB_REPOSITORY" 2>/dev/null; then + echo "Deleting existing release $TAG, will recreate it" + gh release delete "$TAG" --repo "$GITHUB_REPOSITORY" --yes --cleanup-tag + fi + # In case a tag without a release is left over, drop it too. + gh api -X DELETE "repos/${GITHUB_REPOSITORY}/git/refs/tags/${TAG}" >/dev/null 2>&1 || true + + NOTES="$(mktemp)" + { + echo "Automated build of linux-unstable-ogc." + echo + echo "- Source commit: https://github.com/${GITHUB_REPOSITORY}/commit/${GITHUB_SHA}" + echo "- Kernel release: ${KREL}" + echo "- Upstream version: ${KVER}${NEXT}" + echo "- Base configs: Arch \`linux-headers\` + Fedora \`kernel-core\`, plus [OGC kernel-packages fragments](${KERNEL_PACKAGES_RAW}/config)" + echo "- Compiler: clang / LLVM=1 (with ccache)" + echo "- Boot smoke test: every packaged kernel (Arch, Fedora, Debian) was booted in QEMU and reached userspace init before publishing" + echo + echo "### Artifacts" + echo + echo "#### Arch Linux" + echo + echo '- `linux-unstable-ogc` — kernel image and modules' + echo '- `linux-unstable-ogc-headers` — headers for building external modules' + echo "- \`config-arch-${KREL}\` — the exact .config used for this build" + echo + echo "#### Fedora" + echo + echo '- `kernel-unstable-ogc-core` — kernel image (vmlinuz) and core files' + echo '- `kernel-unstable-ogc-modules` — kernel modules' + echo '- `kernel-unstable-ogc-devel` — headers for building external modules' + echo "- \`config-fedora-${KREL}\` — the exact .config used for this build" + echo + echo "#### Debian (and derivatives)" + echo + echo "- \`linux-image-${KREL}\` — kernel image and modules" + echo "- \`linux-headers-${KREL}\` — headers for building external modules" + echo "- \`config-debian-${KREL}\` — the exact .config used for this build" + echo + echo '- `SHA256SUMS` — checksums of all artifacts' + echo + echo "### Install" + echo + echo "Arch Linux:" + echo + echo '```sh' + echo 'sudo pacman -U linux-unstable-ogc-headers-*.pkg.tar.zst linux-unstable-ogc-*.pkg.tar.zst' + echo '```' + echo + echo "Fedora:" + echo + echo '```sh' + echo 'sudo dnf install ./kernel-unstable-ogc-core-*.rpm ./kernel-unstable-ogc-modules-*.rpm' + echo '```' + echo + echo "Debian and derivatives:" + echo + echo '```sh' + echo 'sudo apt install ./linux-image-*.deb ./linux-headers-*.deb' + echo '```' + echo + echo "> The initramfs is generated automatically on install (mkinitcpio hooks" + echo "> on Arch, kernel-install/dracut on Fedora, initramfs-tools hooks on Debian)." + } > "$NOTES" + + cd /tmp/dist + # Upload every artifact produced by the distro jobs (package files + # plus the config-* files). SHA256SUMS itself is included by ./*. + gh release create "$TAG" \ + --repo "$GITHUB_REPOSITORY" \ + --target "$GITHUB_SHA" \ + --title "linux-unstable-ogc ${KREL}" \ + --notes-file "$NOTES" \ + ./* + + - name: Summary + run: | + { + echo "## linux-unstable-ogc build" + echo + echo "- Kernel release: \`${KREL}\`" + echo "- Release: https://github.com/${GITHUB_REPOSITORY}/releases/tag/${TAG}" + echo "- QEMU boot smoke test: passed (Arch, Fedora and Debian kernel images booted to userspace init)" + } >> "$GITHUB_STEP_SUMMARY" diff --git a/.github/workflows/sync-linux-next.yml b/.github/workflows/sync-linux-next.yml new file mode 100644 index 00000000000000..45b236c3f24cb6 --- /dev/null +++ b/.github/workflows/sync-linux-next.yml @@ -0,0 +1,115 @@ +name: Sync linux-next -> master (replay fork commits) + +on: + schedule: + - cron: "0 2 * * *" # every day at 02:00 UTC + workflow_dispatch: {} + +permissions: + contents: write + +concurrency: + group: sync-linux-next + cancel-in-progress: false + +jobs: + sync: + runs-on: ubuntu-latest + steps: + - name: Checkout fork + uses: actions/checkout@v7 + with: + ref: master + fetch-depth: 0 + + - name: Configure upstream + run: | + git remote remove upstream || true + git remote add upstream https://git.kernel.org/pub/scm/linux/kernel/git/next/linux-next.git + git fetch --no-tags upstream --prune + + - name: Replay fork commits on top of upstream/master + run: | + set -euo pipefail + + git checkout master + + # REQUIRED for cherry-pick commit creation (runner has no identity by default) + git config user.name "github-actions[bot]" + git config user.email "github-actions[bot]@users.noreply.github.com" + + UP_BASE="upstream/master" + git show -s --oneline "$UP_BASE" >/dev/null + + # ------------------------------------------------------------------ + # Find the upstream snapshot that master's fork commits sit on. + # + # Anchor: the empty marker commit "ogc: linux-unstable: first + # commit" is always the first fork commit ever made. Its parent is + # therefore the upstream snapshot the fork was built on. + # ------------------------------------------------------------------ + MARKER="$(git log --format="%H" --grep="^ogc: linux-unstable: first commit$" master)" + if [ -z "$MARKER" ]; then + echo "::error::Could not find the 'ogc: linux-unstable: first commit' marker; refusing to touch master." + exit 1 + fi + BASE="$(git rev-parse "${MARKER}^")" + echo "Upstream base: $(git show -s --oneline "$BASE")" + + # If master already sits on the current upstream tip there is + # nothing to sync: replaying would only churn the fork commits' + # hashes for identical content. + if [ "$(git rev-parse "$BASE")" = "$(git rev-parse "$UP_BASE")" ]; then + echo "master is already based on the current linux-next tip; nothing to do." + exit 0 + fi + + # Everything on master on top of the upstream base = the fork + # commits (including non-"ogc:"-prefixed ones, e.g. squash-merged + # PRs). --no-merges matches the cherry-pick loop below and makes + # merge commits replay as their constituent commits. + MY_COMMITS="$(git rev-list --reverse --no-merges "${BASE}..master")" + if [ -z "${MY_COMMITS// }" ]; then + echo "::error::No commits found on top of the upstream base; refusing to reset master (that would delete the fork)." + exit 1 + fi + + # Safety net: a mis-detected base would replay days of upstream + # history. The fork accumulates its own commits over time (and PR + # squash-merges add to that), so allow a generous 150; anything + # beyond that is far more likely a mis-detected base than real fork + # commits, so bail out and ask for a manual look instead of + # corrupting master. + N="$(echo "$MY_COMMITS" | wc -l)" + if [ "$N" -gt 150 ]; then + echo "::error::Refusing to replay ${N} commits (expected only fork commits). Check master's history (all fork commits must carry the 'ogc:' subject prefix)." + exit 1 + fi + + echo "Replaying ${N} commit(s):" + for c in $MY_COMMITS; do + echo " $c $(git show -s --format=%s "$c")" + done + echo + + # Reset master to the new upstream tip, then replay the fork commits. + git reset --hard "$UP_BASE" + + for c in $MY_COMMITS; do + echo "Cherry-picking: $c $(git show -s --format=%s "$c")" + + if ! git cherry-pick -x --allow-empty "$c"; then + git status || true + # If we're left mid-cherry-pick, abort to avoid a broken + # working state. Nothing was pushed, so master on origin is + # still intact. + if [ -f .git/CHERRY_PICK_HEAD ]; then + git cherry-pick --abort || true + fi + echo "::error::Cherry-pick failed for $c; nothing was pushed." + exit 1 + fi + done + + echo "Done. Pushing updated master." + git push --force-with-lease origin master diff --git a/.github/workflows/test-pr.yml b/.github/workflows/test-pr.yml new file mode 100644 index 00000000000000..29c4e2fb8f2dbe --- /dev/null +++ b/.github/workflows/test-pr.yml @@ -0,0 +1,227 @@ +name: Test PR (checkpatch + config gate + gcc build) + +on: + pull_request: + branches: [master] + +permissions: + contents: read + +concurrency: + group: pr-test-${{ github.event.pull_request.number }} + cancel-in-progress: true + +env: + KERNEL_PACKAGES_RAW: https://raw.githubusercontent.com/OpenGamingCollective/kernel-packages/main + +jobs: + checks: + name: Static checks (checkpatch, new-driver symbols) + runs-on: ubuntu-latest + outputs: + new_symbols: ${{ steps.syms.outputs.syms }} + steps: + - name: Checkout PR merge result + uses: actions/checkout@v7 + with: + # Default ref for pull_request is the merge commit refs/pull/N/merge. + # depth 2 fetches the merge commit AND its parents, so HEAD^1 + # (current master tip) is present and the PR diff is exactly + # "what landing this PR changes on master". + fetch-depth: 2 + + - name: Compute PR patch and new driver symbols + id: syms + run: | + set -euo pipefail + # For pull_request, checkout gets the merge commit refs/pull/N/merge: + # parent 1 = master tip, parent 2 = PR head. Fall back to + # origin/master if HEAD is not a merge commit for any reason. + BASE="$(git rev-parse --verify -q HEAD^1 || true)" + if [ -z "$BASE" ]; then + echo "HEAD^1 unresolvable; falling back to origin/master" + git fetch --depth=1 origin master + BASE="$(git rev-parse --verify -q FETCH_HEAD || true)" + fi + if [ -z "$BASE" ]; then + echo "::error::Could not determine the base commit for the PR diff." + exit 1 + fi + git diff "$BASE" HEAD > /tmp/pr.patch + + if [ ! -s /tmp/pr.patch ]; then + echo "Empty patch; nothing to check." + echo "syms=" >> "$GITHUB_OUTPUT" + exit 0 + fi + + # Selectable symbols introduced under drivers/ (added + # "config FOO" / "menuconfig FOO" lines in any Kconfig file). + # NOTE: grep exits 1 on zero matches; under `set -euo pipefail` + # that would abort the step, so it is neutralised here only + # (git diff / awk failures still propagate). + SYMS="$(git diff "$BASE" HEAD -- drivers/ \ + | { grep -E '^\+[[:space:]]*(menu)?config[[:space:]]+[A-Z0-9_]+' || true; } \ + | awk '{print $NF}' | sort -u | tr '\n' ' ')" + echo "syms=${SYMS}" >> "$GITHUB_OUTPUT" + { + echo "### PR checks" + echo + echo "- Changed files: $(git diff --name-only "$BASE" HEAD | wc -l)" + echo "- New driver CONFIG symbols: ${SYMS:-none}" + } >> "$GITHUB_STEP_SUMMARY" + + - name: checkpatch (fail on errors) + run: | + set -uo pipefail + if [ ! -s /tmp/pr.patch ]; then + echo "Empty patch; skipping checkpatch." + exit 0 + fi + # --no-signoff: internal fork PRs do not require Signed-off-by. + perl scripts/checkpatch.pl --no-signoff /tmp/pr.patch \ + > /tmp/checkpatch.out 2>&1 || true + cat /tmp/checkpatch.out + + if grep -q '^ERROR:' /tmp/checkpatch.out; then + NERRS="$(grep -c '^ERROR:' /tmp/checkpatch.out)" + echo "::error::checkpatch reported ${NERRS} error(s); fix them before merge (warnings do not block)." + exit 1 + fi + echo "checkpatch: no errors." + + build: + name: Build with GCC (Arch config + fragments) + needs: checks + runs-on: ubuntu-latest + timeout-minutes: 330 + container: + image: docker.io/archlinux:base-devel + env: + CCACHE_DIR: /ccache + CCACHE_MAXSIZE: 10G + NEW_SYMBOLS: ${{ needs.checks.outputs.new_symbols }} + steps: + - name: Show disk space + run: df -h / + + - name: Bootstrap build environment + run: | + set -euxo pipefail + pacman -Sy --needed --noconfirm archlinux-keyring + pacman -Su --needed --noconfirm + # base-devel provides gcc/make/perl; the rest mirrors the makedepends + # of the release PKGBUILD minus clang and the rust toolchain. + pacman -S --needed --noconfirm \ + bc cpio gettext libelf pahole perl python tar xz zstd \ + file curl git ccache + + - name: Checkout PR merge result + uses: actions/checkout@v7 + with: + fetch-depth: 1 + + - name: Restore compiler cache + uses: actions/cache@v6 + with: + path: /ccache + key: ccache-pr-gcc-${{ github.sha }} + restore-keys: | + ccache-pr-gcc- + + - name: Assemble test config (same as the Arch release build) + run: | + set -euxo pipefail + # ------------------------------------------------------------------ + # Base config: official Arch Linux kernel .config extracted from the + # distro's linux-headers package (same method as the release build). + # ------------------------------------------------------------------ + CACHE="/tmp/pkgcache" + mkdir -p "$CACHE" + pacman -Sw --noconfirm --cachedir "$CACHE" linux-headers + PKG="$(find "$CACHE" -maxdepth 1 -name 'linux-headers-*.pkg.tar.zst' -type f)" + if [ -z "$PKG" ]; then + echo "::error::linux-headers package not found in cache"; exit 1 + fi + CFG="$(tar --zstd -tf "$PKG" | grep -E '^usr/lib/modules/[^/]+/build/\.config$' || true)" + if [ -z "$CFG" ]; then + echo "::error::kernel .config not found inside $PKG"; exit 1 + fi + tar --zstd -xOf "$PKG" "$CFG" > .config + test -s .config + rm -rf "$CACHE" + + # ------------------------------------------------------------------ + # OGC kernel-packages fragments + the repo-local config.fragment + # (merged last, wins). Same order as the release build. + # ------------------------------------------------------------------ + for f in arch.config.set ogc.config.set arch.config.unset ogc.config.unset; do + curl -fsSL "${KERNEL_PACKAGES_RAW}/config/${f}" -o "${f}" + done + bash .github/packaging/merge-fragments.sh .config \ + arch.config.set ogc.config.set \ + arch.config.unset ogc.config.unset \ + .github/packaging/config.fragment + + # Same fixups as the release packaging... + scripts/config --set-str SYSTEM_TRUSTED_KEYS "" + scripts/config --set-str SYSTEM_REVOCATION_KEYS "" + scripts/config --set-str MODULE_SIG_KEY "certs/signing_key.pem" + scripts/config --set-str CONFIG_LOCALVERSION "" || true + # ...plus test-only tweaks: + # No rust toolchain is installed in this job; the release builds + # (clang) compile the rust bits, this job focuses on the C parts. + scripts/config -d RUST + make olddefconfig + + - name: "Gate: new driver symbols must be enabled in the config" + run: | + set -euo pipefail + if [ -z "${NEW_SYMBOLS// }" ]; then + echo "No new driver CONFIG symbols in this PR; skipping gate." + exit 0 + fi + echo "Gate symbols: ${NEW_SYMBOLS}" + + FAIL=0 + for s in ${NEW_SYMBOLS}; do + if grep -qE "^CONFIG_${s}=[ym]$" .config; then + echo "OK: CONFIG_${s} is enabled." + elif grep -q "^# CONFIG_${s} is not set$" .config; then + echo "::error::CONFIG_${s} was added by this PR but is disabled in the merged config." + echo "::error::Enable it in .github/packaging/config.fragment (CONFIG_${s}=y or =m) if it should ship." + FAIL=1 + elif grep -q "^CONFIG_${s}=" .config; then + echo "::error::CONFIG_${s} is set but not to y/m: $(grep "^CONFIG_${s}=" .config)" + FAIL=1 + else + echo "::error::CONFIG_${s} is absent from the final .config (its dependencies or its vendor menu are off in the merged config)." + echo "::error::If this driver should ship, enable it (and its dependencies) in .github/packaging/config.fragment." + FAIL=1 + fi + done + if [ "$FAIL" -ne 0 ]; then + echo "::error::config gate failed: new drivers must be enabled in the OGC config." + exit 1 + fi + echo "config gate passed." + + - name: Build kernel with GCC + run: | + set -euxo pipefail + ccache -s || true + make CC="ccache gcc" WERROR=0 -j"$(nproc)" all + echo "Kernel release: $(make -s kernelrelease)" + ccache -s + df -h / + + - name: Summary + if: always() + run: | + { + echo "### GCC compile test" + echo + echo "- Compiler: gcc (Arch Linux) via ccache" + echo "- Config: Arch linux-headers base + OGC fragments + repo config.fragment (Rust disabled for this test)" + echo "- New driver symbols checked: ${NEW_SYMBOLS:-none}" + } >> "$GITHUB_STEP_SUMMARY" \ No newline at end of file From d57dba4e132510aeb5e715ef1f185c58eb710a27 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Tomasz=20Paku=C5=82a?= Date: Tue, 3 Feb 2026 18:56:14 +0000 Subject: [PATCH 1336/1352] drm/amd/display: Add CH7218 PCON ID MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit [Why] Chrontel CH7218 found in Ugreen DP -> HDMI 2.1 adapter (model 85564) works perfectly with VRR after testing. VRR and FreeSync compatibility is explicitly advertised as a feature so it's addition is a formality. Support FreeSync info packet passthrough and "generic" HDMI VRR. [How] Add CH7218's ID to dm_helpers_is_vrr_pcon_allowed() Closes: https://gitlab.freedesktop.org/drm/amd/-/issues/4773 Signed-off-by: Tomasz Pakuła (cherry picked from commit 7b2436287ed953496ad1c9eb5820f00b75db597f) (cherry picked from commit a9c75486e9c655c604e05d55e1546f4e44b1bfd2) (cherry picked from commit 63b562c3902fe1324bc0e81d3ccf78f95a811e1a) (cherry picked from commit e20d4609955d599e1cb81a5277a4dbb308c649e1) (cherry picked from commit 39c6e1f4320b1b3416c8c64a7ef9de6d733289ab) (cherry picked from commit f31c35d096451bf274236ae09b305395be16b881) (cherry picked from commit 5a29609e68c3270a675ffdc5d15a95eef929638e) --- drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_helpers.c | 1 + drivers/gpu/drm/amd/display/include/ddc_service_types.h | 1 + 2 files changed, 2 insertions(+) diff --git a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_helpers.c b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_helpers.c index 3217aa82bea248..ec312eba80caba 100644 --- a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_helpers.c +++ b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_helpers.c @@ -1681,6 +1681,7 @@ STATIC_IFN_KUNIT const uint32_t dm_freesync_pcon_whitelist[] = { DP_BRANCH_DEVICE_ID_90CC24, DP_BRANCH_DEVICE_ID_001CF8, DP_BRANCH_DEVICE_ID_001FF2, + DP_BRANCH_DEVICE_ID_2B02F0, }; EXPORT_IF_KUNIT(dm_freesync_pcon_whitelist); diff --git a/drivers/gpu/drm/amd/display/include/ddc_service_types.h b/drivers/gpu/drm/amd/display/include/ddc_service_types.h index 827e9bd7c5cff3..d2a8e712d4a712 100644 --- a/drivers/gpu/drm/amd/display/include/ddc_service_types.h +++ b/drivers/gpu/drm/amd/display/include/ddc_service_types.h @@ -37,6 +37,7 @@ #define DP_BRANCH_DEVICE_ID_001CF8 0x001CF8 #define DP_BRANCH_DEVICE_ID_0060AD 0x0060AD #define DP_BRANCH_DEVICE_ID_001FF2 0x001FF2 +#define DP_BRANCH_DEVICE_ID_2B02F0 0x2B02F0 /* Chrontel CH7218 */ #define DP_BRANCH_HW_REV_10 0x10 #define DP_BRANCH_HW_REV_20 0x20 From 4d67b52a0ee31b5243e9733f3efa0390c81869bc Mon Sep 17 00:00:00 2001 From: Ahmed Yaseen Date: Tue, 18 Aug 2026 06:53:12 +0500 Subject: [PATCH 1337/1352] [FOR-UPSTREAM] HID: asus: add ROG Zephyrus Duo GX651AR keyboard The detachable keyboard shipped with the ROG Zephyrus Duo GX651AR (0b05:1ce6) is a ROG N-Key keyboard, but it is not listed in asus_devices[], so its interfaces are left to hid-generic and its vendor usages are never mapped by asus_input_mapping(). Add it with QUIRK_USE_KBD_BACKLIGHT | QUIRK_ROG_NKEY_KEYBOARD, matching the other ROG N-Key keyboards. Tested-by: Cymirk Signed-off-by: Ahmed Yaseen (cherry picked from commit 8d70b5f90b41b98f9fa6293e03be78c31bac8c36) (cherry picked from commit 9582bd441f1fefbe164037ec5b587b9db39d1167) (cherry picked from commit 29ee8b6c10c5d31601444dd69bf04a10c66c8288) (cherry picked from commit 9bd6541370e7ca2553bdd8d4f8465b80a0a325f7) (cherry picked from commit 1d4e5afac4427067d36bf2b0bfa9108ffa99da09) (cherry picked from commit bde69e2cdb3ae9f6fd2c18a78222f12fbca21401) --- drivers/hid/hid-asus.c | 3 +++ drivers/hid/hid-ids.h | 1 + 2 files changed, 4 insertions(+) diff --git a/drivers/hid/hid-asus.c b/drivers/hid/hid-asus.c index 7dc6417fe2622a..a3584d819c696f 100644 --- a/drivers/hid/hid-asus.c +++ b/drivers/hid/hid-asus.c @@ -1695,6 +1695,9 @@ static const struct hid_device_id asus_devices[] = { { HID_I2C_DEVICE(USB_VENDOR_ID_ASUSTEK, USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD2), QUIRK_USE_KBD_BACKLIGHT | QUIRK_ROG_NKEY_KEYBOARD | QUIRK_HID_FN_LOCK }, + { HID_USB_DEVICE(USB_VENDOR_ID_ASUSTEK, + USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD3), + QUIRK_USE_KBD_BACKLIGHT | QUIRK_ROG_NKEY_KEYBOARD }, { HID_USB_DEVICE(USB_VENDOR_ID_ASUSTEK, USB_DEVICE_ID_ASUSTEK_ROG_Z13_LIGHTBAR), QUIRK_USE_KBD_BACKLIGHT | QUIRK_ROG_NKEY_KEYBOARD }, diff --git a/drivers/hid/hid-ids.h b/drivers/hid/hid-ids.h index a9537b7bb03d12..53502440403437 100644 --- a/drivers/hid/hid-ids.h +++ b/drivers/hid/hid-ids.h @@ -227,6 +227,7 @@ #define USB_DEVICE_ID_ASUSTEK_ROG_KEYBOARD3 0x1822 #define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD 0x1866 #define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD2 0x19b6 +#define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD3 0x1ce6 #define USB_DEVICE_ID_ASUSTEK_ROG_Z13_FOLIO 0x1a30 #define USB_DEVICE_ID_ASUSTEK_ROG_Z13_LIGHTBAR 0x18c6 #define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_ALLY 0x1abe From b5ca2016420b7e0cb74fcc0b75af13143853f207 Mon Sep 17 00:00:00 2001 From: Ahmed Yaseen Date: Tue, 18 Aug 2026 09:46:22 +0500 Subject: [PATCH 1338/1352] [FOR-UPSTREAM] HID: asus: force input connection on vendor-only N-Key interfaces On the ROG Zephyrus Duo GX651AR (0b05:1ce6) the hotkeys live on report 0x5a on an interface whose descriptor holds nothing but two ASUS vendor collections. Neither satisfies IS_INPUT_APPLICATION(), so hidinput_connect() creates no input device, asus_input_mapping() never runs and every hotkey is dropped by asus_event() as unmapped. Set HID_QUIRK_HIDINPUT_FORCE on ROG N-Key interfaces that carry an ASUS vendor input report so those usages get mapped. Interfaces left with no mapped usage are still discarded by hidinput_has_been_populated(). The vendor check reads report_enum[HID_INPUT_REPORT], so interfaces with no input reports, such as the RGB control interface, are unaffected. Tested-by: Cymirk Signed-off-by: Ahmed Yaseen (cherry picked from commit 69887e3b53350a792c34272d0b101a21131a7ed7) (cherry picked from commit 4e707839a861d371b367c285665fd77cbfd1a113) (cherry picked from commit b3ceeae53756d8b73e99bf00a412136cdd130853) (cherry picked from commit c492209aed4a61e14efa274aa3cf153d55895d2e) (cherry picked from commit 26d962c1a76bba672d49d56fae0fd6b54ca9b86a) (cherry picked from commit 79bac16bfa443c4a263505da7b70243a771eb600) --- drivers/hid/hid-asus.c | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/drivers/hid/hid-asus.c b/drivers/hid/hid-asus.c index a3584d819c696f..502cd54fad793a 100644 --- a/drivers/hid/hid-asus.c +++ b/drivers/hid/hid-asus.c @@ -1480,6 +1480,14 @@ static int asus_probe(struct hid_device *hdev, const struct hid_device_id *id) is_vendor = true; } + /* + * A vendor collection may be the only application collection on the + * interface, which hidinput_connect() otherwise skips, leaving the + * hotkey usages unmapped. Unpopulated inputs are dropped later. + */ + if (is_vendor && (drvdata->quirks & QUIRK_ROG_NKEY_KEYBOARD)) + hdev->quirks |= HID_QUIRK_HIDINPUT_FORCE; + ret = asus_worker_create(hdev, drvdata); if (ret) { hid_warn(hdev, "Failed to initialize worker: %d\n", ret); From 5f585c1e2825b35dd0f6d62169b58d4e1dfd7044 Mon Sep 17 00:00:00 2001 From: Ahmed Yaseen Date: Sat, 22 Aug 2026 16:28:13 +0500 Subject: [PATCH 1339/1352] [FOR-UPSTREAM] HID: asus: add ROG Zephyrus Duo GX651AR keyboard over Bluetooth The GX651AR keyboard enumerates as 0b05:1ce6 over USB but pairs as 0b05:1ce7 in Bluetooth mode, where the keyboard, consumer and both ASUS vendor collections (reports 0x5a and 0x5d) sit on a single HID device. Add it with the same quirks as the USB entry. Bind to HID_GROUP_GENERIC so that hid-multitouch keeps the digitizer. Tested-by: Cymirk Signed-off-by: Ahmed Yaseen (cherry picked from commit b0dbc09e46a5d146c25081943f16fc0b1d0c0492) (cherry picked from commit f460429e28dc70ff7572c384a7a40ee66868e4c9) (cherry picked from commit af0d8926a206e3c4a6fa7544ba6a5aa0d63da03a) (cherry picked from commit 98993b7ccce3d580574b0bc6c5c04e9b7c266967) (cherry picked from commit d8a5f656c56d014e1e91a3737f310fd06e4e82bb) (cherry picked from commit 31636fd50a4ac27e9248928fad6f01d2b64b5e76) --- drivers/hid/hid-asus.c | 3 +++ drivers/hid/hid-ids.h | 1 + 2 files changed, 4 insertions(+) diff --git a/drivers/hid/hid-asus.c b/drivers/hid/hid-asus.c index 502cd54fad793a..c63e03a876a32f 100644 --- a/drivers/hid/hid-asus.c +++ b/drivers/hid/hid-asus.c @@ -1744,6 +1744,9 @@ static const struct hid_device_id asus_devices[] = { { HID_DEVICE(BUS_USB, HID_GROUP_GENERIC, USB_VENDOR_ID_ASUSTEK, USB_DEVICE_ID_ASUSTEK_ROG_Z13_FOLIO), QUIRK_USE_KBD_BACKLIGHT | QUIRK_ROG_NKEY_KEYBOARD }, + { HID_DEVICE(BUS_BLUETOOTH, HID_GROUP_GENERIC, + USB_VENDOR_ID_ASUSTEK, USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD3_BT), + QUIRK_USE_KBD_BACKLIGHT | QUIRK_ROG_NKEY_KEYBOARD }, { HID_DEVICE(BUS_USB, HID_GROUP_GENERIC, USB_VENDOR_ID_ASUSTEK, USB_DEVICE_ID_ASUSTEK_T101HA_KEYBOARD) }, { } diff --git a/drivers/hid/hid-ids.h b/drivers/hid/hid-ids.h index 53502440403437..c9b2c7782d9a1d 100644 --- a/drivers/hid/hid-ids.h +++ b/drivers/hid/hid-ids.h @@ -228,6 +228,7 @@ #define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD 0x1866 #define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD2 0x19b6 #define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD3 0x1ce6 +#define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD3_BT 0x1ce7 #define USB_DEVICE_ID_ASUSTEK_ROG_Z13_FOLIO 0x1a30 #define USB_DEVICE_ID_ASUSTEK_ROG_Z13_LIGHTBAR 0x18c6 #define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_ALLY 0x1abe From da06a918d1dc2f602ee919634b46609984c3096e Mon Sep 17 00:00:00 2001 From: Ahmed Yaseen Date: Sat, 22 Aug 2026 16:35:21 +0500 Subject: [PATCH 1340/1352] [FOR-UPSTREAM] HID: asus: map the tent mode key on ROG Zephyrus Duo GX651AR Fn+F12 on the GX651AR keyboard emits ASUS vendor code 0x9c, which asus_input_mapping() does not know about, so asus_event() drops it as unmapped. Map it to KEY_F19. F13 to F18 are already used for ASUS toggles that have no generic keycode. Tested-by: Cymirk Signed-off-by: Ahmed Yaseen (cherry picked from commit 337a811e8f0536e8ff649c533d00bfa973e5618b) (cherry picked from commit a980051bc52de3cea2bf95e56cf6de936da24b57) (cherry picked from commit aafe6db9d256a66ed1fd70aa9a1e6e36d643dac3) (cherry picked from commit 6bc535e4cf27505b1dbd89507d938703a4b20f9e) (cherry picked from commit 397e1aa78b270d2d1a1a43198dc3d54c19140d31) (cherry picked from commit 711123e8f04105fd9e3291258cc3a8d8352b4a9c) --- drivers/hid/hid-asus.c | 1 + 1 file changed, 1 insertion(+) diff --git a/drivers/hid/hid-asus.c b/drivers/hid/hid-asus.c index c63e03a876a32f..4112c0fc7aaf97 100644 --- a/drivers/hid/hid-asus.c +++ b/drivers/hid/hid-asus.c @@ -1267,6 +1267,7 @@ static int asus_input_mapping(struct hid_device *hdev, case 0xa6: asus_map_key_clear(KEY_F16); break; /* ROG Ally QAM button */ case 0xa7: asus_map_key_clear(KEY_F17); break; /* ROG Ally ROG long-press */ case 0xa8: asus_map_key_clear(KEY_F18); break; /* ROG Ally ROG long-press-release */ + case 0x9c: asus_map_key_clear(KEY_F19); break; /* Zephyrus Duo tent mode */ default: /* ASUS lazily declares 256 usages, ignore the rest, From a9cf7c74393ee03611a56139a4d8d72b0b8e93fe Mon Sep 17 00:00:00 2001 From: Denis Benato Date: Sat, 5 Sep 2026 13:39:38 +0000 Subject: [PATCH 1341/1352] [NOT-FOR-UPSTREAM] ogc: linux-unstable: boot smoke test: pin loglevel=7 and dump serial log head on banner failure The smoke test greps the captured serial console log for the kernel banner ("Linux version "), but the banner is printed at KERN_NOTICE while some distro configs default the console to a quieter level: Arch ships CONFIG_CONSOLE_LOGLEVEL_DEFAULT=4, which suppresses notice (and info) messages entirely. The result was a confusing failure mode: the kernel booted to userspace just fine (BOOT_OK marker, printed by init directly on /dev/ttyS0), yet the banner check failed because the console log only carried a few high-priority lines. Pin loglevel=7 on the test kernel command line so every distro kernel logs verbosely enough for the banner to be captured, and replace the useless "^Linux version" grep in the failure path (banner lines are prefixed with a "[ 0.000000] " timestamp, so that grep could never match) with a dump of the first serial console lines. Verified locally against a kernel built with the exact Arch config pipeline: the run failed identically to CI before the change and passes both the bios-pc and uefi-q35 legs after it. (cherry picked from commit af2b9020fdbc50ab19fb22a82ca861adac7dc7dd) (cherry picked from commit 98d39d0b1f5fd301c87b7b67f504a73ba3fbc269) (cherry picked from commit adfb8ee52243077a3c71798a874dbc75b6687e5b) (cherry picked from commit cf9935674fc584c44b6efa4c5eb4c3cb486d3a9c) (cherry picked from commit f3282c8514d4c60e4ccad5719ee3ecd054991a84) --- .github/packaging/boot-smoke-test.sh | 13 +++++++++++-- 1 file changed, 11 insertions(+), 2 deletions(-) diff --git a/.github/packaging/boot-smoke-test.sh b/.github/packaging/boot-smoke-test.sh index 021e8e596998f3..f315029982fd72 100755 --- a/.github/packaging/boot-smoke-test.sh +++ b/.github/packaging/boot-smoke-test.sh @@ -34,6 +34,10 @@ # "Linux version " must appear on every console log. # BOOT_SMOKE_TIMEOUT override the per-boot timeout in seconds # (default: 300 under KVM, 1200 under TCG). +# The kernel command line pins loglevel=7: some distro configs default the +# console to a quieter level (Arch ships CONSOLE_LOGLEVEL_DEFAULT=4), which +# would suppress the KERN_NOTICE boot banner and make the banner check below +# fail on an otherwise perfectly bootable kernel. # Requires (installed by the calling CI job): qemu-system-x86_64, cpio, # gzip, a C compiler (gcc or clang), an OVMF build for the UEFI # leg (packages: ovmf / edk2-ovmf), coreutils (timeout, find). @@ -190,7 +194,7 @@ run_boot() { -display none -monitor none -no-reboot \ -serial "file:$serial" \ -kernel "$IMAGE" -initrd "$WORK/initrd.img" \ - -append "console=ttyS0,115200n8 rdinit=/init panic=-1 nokaslr" \ + -append "console=ttyS0,115200n8 rdinit=/init panic=-1 nokaslr loglevel=7" \ || true # A panic with panic=-1 reboots instantly and -no-reboot makes QEMU # exit, so both "qemu exited by itself" and "timeout killed it" end up @@ -205,7 +209,12 @@ run_boot() { fi if [ -n "${KREL:-}" ] && ! grep -qF "Linux version ${KREL} " "$serial"; then echo "::error::QEMU boot smoke test FAILED in the '$name' configuration: the booted kernel banner does not advertise release '${KREL}'." - grep -m1 "^Linux version" "$serial" || true + # Show what the console actually carried: log lines carry a + # "[ 0.000000] " timestamp prefix, so anchor-free context of + # the early console output is what makes this diagnosable. + echo "----- first 25 lines of the $name guest serial console -----" + head -n 25 "$serial" | sed 's/\r$//' + echo "-------------------------------------------------------------" exit 1 fi echo "[$name] Boot smoke test PASSED: kernel booted to userspace init and powered off. Serial console tail:" From bf2f32cf7ef7b76277a4a6b20d1b56334c4abbf5 Mon Sep 17 00:00:00 2001 From: "Mario Limonciello (AMD)" Date: Wed, 30 Sep 2026 16:39:22 -0500 Subject: [PATCH 1342/1352] [FROM-ML] PCI/PM: Split out code from pci_pm_suspend_noirq() into helper In order to unify suspend and hibernate code paths without code duplication the common code should be in common helpers. Move the common code from pci_pm_suspend_noirq() into a pci_pm_suspend_noirq_common() helper. No intended functional changes. Signed-off-by: Mario Limonciello (AMD) Signed-off-by: Bjorn Helgaas Tested-by: Eric Naim Reviewed-by: Rafael J. Wysocki (Intel) Link: https://patch.msgid.link/20260930213923.566846-2-superm1@kernel.org --- drivers/pci/pci-driver.c | 77 +++++++++++++++++++++++++--------------- 1 file changed, 49 insertions(+), 28 deletions(-) diff --git a/drivers/pci/pci-driver.c b/drivers/pci/pci-driver.c index c1ac2ef025e233..fe8d773129e14f 100644 --- a/drivers/pci/pci-driver.c +++ b/drivers/pci/pci-driver.c @@ -823,6 +823,52 @@ static void pci_pm_complete(struct device *dev) #endif /* !CONFIG_PM_SLEEP */ +#if defined(CONFIG_SUSPEND) +/** + * pci_pm_suspend_noirq_common - prepare a device to enter a low-power state + * @pci_dev: pci device + * + * Save the device state and decide whether bus-level power management should + * skipped. Returns true if bus-level power management should be skipped, + * false otherwise. + */ +static bool pci_pm_suspend_noirq_common(struct pci_dev *pci_dev) +{ + if (!pci_dev->state_saved) { + pci_save_state(pci_dev); + + /* + * If the device is a bridge with a child in D0 below it, + * it needs to stay in D0, so check skip_bus_pm to avoid + * putting it into a low-power state in that case. + */ + if (!pci_dev->skip_bus_pm && pci_power_manageable(pci_dev)) + pci_prepare_to_sleep(pci_dev); + } + + pci_dbg(pci_dev, "PCI PM: Sleep power state: %s\n", + pci_power_name(pci_dev->current_state)); + + if (pci_dev->current_state == PCI_D0) { + pci_dev->skip_bus_pm = true; + /* + * Per PCI PM r1.2, table 6-1, a bridge must be in D0 if any + * downstream device is in D0, so avoid changing the power state + * of the parent bridge by setting the skip_bus_pm flag for it. + */ + if (pci_dev->bus->self) + pci_dev->bus->self->skip_bus_pm = true; + } + + if (pci_dev->skip_bus_pm && pm_suspend_no_platform()) { + pci_dbg(pci_dev, "PCI PM: Skipped\n"); + return true; + } + + return false; +} +#endif /* CONFIG_SUSPEND */ + #ifdef CONFIG_SUSPEND static void pcie_pme_root_status_cleanup(struct pci_dev *pci_dev) { @@ -912,6 +958,7 @@ static int pci_pm_suspend_noirq(struct device *dev) { struct pci_dev *pci_dev = to_pci_dev(dev); const struct dev_pm_ops *pm = dev->driver ? dev->driver->pm : NULL; + bool skip_bus_pm; if (dev_pm_skip_suspend(dev)) return 0; @@ -942,36 +989,10 @@ static int pci_pm_suspend_noirq(struct device *dev) } } - if (!pci_dev->state_saved) { - pci_save_state(pci_dev); - - /* - * If the device is a bridge with a child in D0 below it, - * it needs to stay in D0, so check skip_bus_pm to avoid - * putting it into a low-power state in that case. - */ - if (!pci_dev->skip_bus_pm && pci_power_manageable(pci_dev)) - pci_prepare_to_sleep(pci_dev); - } + skip_bus_pm = pci_pm_suspend_noirq_common(pci_dev); - pci_dbg(pci_dev, "PCI PM: Suspend power state: %s\n", - pci_power_name(pci_dev->current_state)); - - if (pci_dev->current_state == PCI_D0) { - pci_dev->skip_bus_pm = true; - /* - * Per PCI PM r1.2, table 6-1, a bridge must be in D0 if any - * downstream device is in D0, so avoid changing the power state - * of the parent bridge by setting the skip_bus_pm flag for it. - */ - if (pci_dev->bus->self) - pci_dev->bus->self->skip_bus_pm = true; - } - - if (pci_dev->skip_bus_pm && pm_suspend_no_platform()) { - pci_dbg(pci_dev, "PCI PM: Skipped\n"); + if (skip_bus_pm) goto Fixup; - } set_unknown: pci_pm_set_unknown_state(pci_dev); From 15cfb94a83aa0c677c7782755e43ff16016c6bd3 Mon Sep 17 00:00:00 2001 From: "Mario Limonciello (AMD)" Date: Wed, 30 Sep 2026 16:39:23 -0500 Subject: [PATCH 1343/1352] [FROM-ML] PCI: Align hibernate poweroff flow with suspend flow for bridges MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit During S3 suspend, pci_pm_suspend_noirq() puts PCIe bridges with downstream devices in a low-power state (D3hot or D3cold) when the platform allows it. The hibernate poweroff_noirq path never did this: it only called pci_prepare_to_sleep() for devices with no subordinate, so bridges with active children were left in D0. On many designs the platform firmware leaves bridges alone when the system enters S4. This prevents being able to meet various energy certification criteria for different parts of the world, particularly in designs with a dGPU. Align the hibernate flow with suspend by making pci_pm_poweroff_noirq() use pci_pm_suspend_noirq_common() instead of the open-coded pci_prepare_to_sleep() call. This reuses exactly the logic the S3 suspend path uses, including the skip_bus_pm handling that keeps a bridge in D0 when a downstream device must stay in D0 (e.g. a configured wakeup source) and the pm_suspend_no_platform() bus-PM skip. This alone is safe for the normal power-off-and-reboot path, since the hibernation image is snapshotted before poweroff_noirq runs and restoring it always goes through a full boot that re-enumerates and retrains the PCIe links. It isn't enough if hibernation_platform_enter() aborts after dpm_suspend_end(PMSG_HIBERNATE) but before power-off: that recovers via dpm_resume_start(PMSG_RESTORE) without rebooting, which calls pci_pm_restore_noirq() instead of pci_pm_resume_noirq(). Give pci_pm_restore_noirq() the same D3cold handling pci_pm_resume_noirq() already has, so a bridge coming out of D3cold gets pci_pm_bridge_power_up_actions() before its children's restore_noirq callbacks run against a possibly untrained link. Mirror the suspend_noirq guard for drivers as well: if a driver's poweroff_noirq callback already left the device in a low-power state without saving its configuration, skip pci_pm_suspend_noirq_common() and go straight to the fixups, exactly as pci_pm_suspend_noirq() does. This avoids having the core call pci_save_state() on a device the driver has already powered down, whose configuration space may no longer be readable, which could otherwise corrupt the saved state used on a poweroff abort. Because the poweroff_noirq path now mirrors the already-shipping suspend_noirq path and is guarded identically, bridges that must remain in D0 are unaffected; only bridges that S3 suspend would have powered down are now also powered down at hibernate. Signed-off-by: Mario Limonciello (AMD) Signed-off-by: Bjorn Helgaas Tested-by: Eric Naim Acked-by: Rafael J. Wysocki (Intel) Cc: AceLan Kao Cc: Kai-Heng Feng Cc: Mark Pearson Cc: Denis Benato Cc: Merthan Karakaş Link: https://patch.msgid.link/20260930213923.566846-3-superm1@kernel.org --- drivers/pci/pci-driver.c | 27 +++++++++++++++++++++++---- 1 file changed, 23 insertions(+), 4 deletions(-) diff --git a/drivers/pci/pci-driver.c b/drivers/pci/pci-driver.c index fe8d773129e14f..2398f03a77c124 100644 --- a/drivers/pci/pci-driver.c +++ b/drivers/pci/pci-driver.c @@ -823,7 +823,7 @@ static void pci_pm_complete(struct device *dev) #endif /* !CONFIG_PM_SLEEP */ -#if defined(CONFIG_SUSPEND) +#if defined(CONFIG_SUSPEND) || defined(CONFIG_HIBERNATE_CALLBACKS) /** * pci_pm_suspend_noirq_common - prepare a device to enter a low-power state * @pci_dev: pci device @@ -867,7 +867,7 @@ static bool pci_pm_suspend_noirq_common(struct pci_dev *pci_dev) return false; } -#endif /* CONFIG_SUSPEND */ +#endif /* CONFIG_SUSPEND || CONFIG_HIBERNATE_CALLBACKS */ #ifdef CONFIG_SUSPEND static void pcie_pme_root_status_cleanup(struct pci_dev *pci_dev) @@ -1222,6 +1222,8 @@ static int pci_pm_poweroff(struct device *dev) struct pci_dev *pci_dev = to_pci_dev(dev); const struct dev_pm_ops *pm = dev->driver ? dev->driver->pm : NULL; + pci_dev->skip_bus_pm = false; + if (pci_has_legacy_pm_support(pci_dev)) return pci_legacy_suspend(dev, PMSG_HIBERNATE); @@ -1264,6 +1266,7 @@ static int pci_pm_poweroff_noirq(struct device *dev) { struct pci_dev *pci_dev = to_pci_dev(dev); const struct dev_pm_ops *pm = dev->driver ? dev->driver->pm : NULL; + bool skip_bus_pm; if (dev_pm_skip_suspend(dev)) return 0; @@ -1277,16 +1280,26 @@ static int pci_pm_poweroff_noirq(struct device *dev) } if (pm->poweroff_noirq) { + pci_power_t prev = pci_dev->current_state; int error; error = pm->poweroff_noirq(dev); suspend_report_result(dev, pm->poweroff_noirq, error); if (error) return error; + + if (!pci_dev->state_saved && pci_dev->current_state != PCI_D0 + && pci_dev->current_state != PCI_UNKNOWN) { + pci_WARN_ONCE(pci_dev, pci_dev->current_state != prev, + "PCI PM: State of device not saved by %pS\n", + pm->poweroff_noirq); + goto Fixup; + } } - if (!pci_dev->state_saved && !pci_has_subordinate(pci_dev)) - pci_prepare_to_sleep(pci_dev); + skip_bus_pm = pci_pm_suspend_noirq_common(pci_dev); + if (skip_bus_pm) + goto Fixup; /* * The reason for doing this here is the same as for the analogous code @@ -1295,6 +1308,7 @@ static int pci_pm_poweroff_noirq(struct device *dev) if (pci_dev->class == PCI_CLASS_SERIAL_USB_EHCI) pci_write_config_word(pci_dev, PCI_COMMAND, 0); +Fixup: pci_fixup_device(pci_fixup_suspend_late, pci_dev); return 0; @@ -1304,10 +1318,15 @@ static int pci_pm_restore_noirq(struct device *dev) { struct pci_dev *pci_dev = to_pci_dev(dev); const struct dev_pm_ops *pm = dev->driver ? dev->driver->pm : NULL; + pci_power_t prev_state = pci_dev->current_state; + bool skip_bus_pm = pci_dev->skip_bus_pm; pci_pm_default_resume_early(pci_dev); pci_fixup_device(pci_fixup_resume_early, pci_dev); + if (!skip_bus_pm && prev_state == PCI_D3cold) + pci_pm_bridge_power_up_actions(pci_dev); + if (pci_has_legacy_pm_support(pci_dev)) return 0; From a9fca3a8f859b2a916ace23afba38e3ae08cde97 Mon Sep 17 00:00:00 2001 From: "Mario Limonciello (AMD)" Date: Mon, 11 Aug 2025 12:00:06 -0500 Subject: [PATCH 1344/1352] [FROM-ML] PM: Use hibernate flows for system power off MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit When the system is powered off the kernel will call device_shutdown() which will issue callbacks into PCI core to wake up a device and call it's shutdown() callback. This will leave devices in ACPI D0 which can cause some devices to misbehave with spurious wakeups and also leave some devices on which will consume power needlessly. The issue won't happen if the device is in D3 before system shutdown, so putting device to low power state before shutdown solves the issue. ACPI Spec 6.5, "7.4.2.5 System \_S4 State" says "Devices states are compatible with the current Power Resource states. In other words, all devices are in the D3 state when the system state is S4." The following "7.4.2.6 System \_S5 State (Soft Off)" states "The S5 state is similar to the S4 state except that OSPM does not save any context." so it's safe to assume devices should be at D3 for S5. To accomplish this, use the PMSG_POWEROFF event to call all the device hibernate callbacks when the kernel is compiled with hibernate support. If compiled without hibernate support or hibernate fails fall back into the previous shutdown flow. Cc: AceLan Kao Cc: Kai-Heng Feng Cc: Mark Pearson Cc: Merthan Karakaş Tested-by: Eric Naim Tested-by: Denis Benato Link: https://lore.kernel.org/linux-pci/20231213182656.6165-1-mario.limonciello@amd.com/ Link: https://lore.kernel.org/linux-pci/20250506041934.1409302-1-superm1@kernel.org/ Signed-off-by: Mario Limonciello (AMD) --- kernel/reboot.c | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/kernel/reboot.c b/kernel/reboot.c index 535d9d5b2d220b..775b1057d57e6d 100644 --- a/kernel/reboot.c +++ b/kernel/reboot.c @@ -13,6 +13,7 @@ #include #include #include +#include #include #include #include @@ -306,6 +307,13 @@ static void kernel_shutdown_prepare(enum system_states state) (state == SYSTEM_HALT) ? SYS_HALT : SYS_POWER_OFF, NULL); system_state = state; usermodehelper_disable(); +#ifdef CONFIG_HIBERNATE_CALLBACKS + if (state == SYSTEM_POWER_OFF) { + if (!dpm_suspend_start(PMSG_POWEROFF) && !dpm_suspend_end(PMSG_POWEROFF)) + return; + pr_emerg("Failed to power off devices, using shutdown instead.\n"); + } +#endif device_shutdown(); } /** From 603c5e069acc0b92ed35b9c4be8532d0ea2d43d6 Mon Sep 17 00:00:00 2001 From: Denis Benato Date: Mon, 5 Oct 2026 19:13:50 +0000 Subject: [PATCH 1345/1352] [FOR-UPSTREAM] kbuild: install-extmod-build: install kernel/kallsyms_internal.h Commit 4fafd1165b33 ("kallsyms: increase marker density to 16:1 to accelerate lookups") made scripts/kallsyms.c include ../kernel/kallsyms_internal.h, but the tree staged by install-extmod-build contains no kernel/ files. install-extmod-build rebuilds the host programs inside the staged tree whenever CC differs from HOSTCC (cross builds, or a ccache-wrapped CC). scripts/kallsyms is hostprogs-always-y, so that rebuild fails with: scripts/kallsyms.c:39:10: fatal error: '../kernel/kallsyms_internal.h' file not found Stage the header so the linux-headers-* and kernel-devel style packages can rebuild scripts/kallsyms. Assisted-by: ZCode:glm-5.3 Signed-off-by: Denis Benato --- scripts/package/install-extmod-build | 2 ++ 1 file changed, 2 insertions(+) diff --git a/scripts/package/install-extmod-build b/scripts/package/install-extmod-build index f12e1ffe409eb0..aad2cd59a37ba4 100755 --- a/scripts/package/install-extmod-build +++ b/scripts/package/install-extmod-build @@ -20,6 +20,8 @@ mkdir -p "${destdir}" ( cd "${srctree}" echo Makefile + # scripts/kallsyms.c includes this header + echo kernel/kallsyms_internal.h find "arch/${SRCARCH}" -maxdepth 1 -name 'Makefile*' find "arch/${SRCARCH}" -name generated -prune -o -name include -type d -print find "arch/${SRCARCH}" -name Kbuild.platforms -o -name Platform From 6dadeae0dda87017d8230658f36d6df6dafa7572 Mon Sep 17 00:00:00 2001 From: Mario Limonciello Date: Tue, 6 Oct 2026 13:13:54 -0500 Subject: [PATCH 1346/1352] PM: Add shutdown=legacy command line option Powering off now uses the hibernation callback flow when it is available. Add an escape valve for systems that fail with this flow. The shutdown=legacy option restores the device_shutdown() path. Document it in the kernel parameter reference and the shutdown debugging guide. Signed-off-by: Mario Limonciello --- Documentation/admin-guide/kernel-parameters.txt | 4 ++++ Documentation/power/shutdown-debugging.rst | 7 ++++++- kernel/reboot.c | 10 +++++++++- 3 files changed, 19 insertions(+), 2 deletions(-) diff --git a/Documentation/admin-guide/kernel-parameters.txt b/Documentation/admin-guide/kernel-parameters.txt index 6a3f43dfbce741..6779223f1ac4c4 100644 --- a/Documentation/admin-guide/kernel-parameters.txt +++ b/Documentation/admin-guide/kernel-parameters.txt @@ -7060,6 +7060,10 @@ Kernel parameters shapers= [NET] Maximal number of shapers. + shutdown=legacy [KNL] + Use the legacy device shutdown callbacks when powering off + the system instead of the hibernation power-off callbacks. + show_lapic= [APIC,X86] Advanced Programmable Interrupt Controller Limit apic dumping. The parameter defines the maximal number of local apics being dumped. Also it is possible diff --git a/Documentation/power/shutdown-debugging.rst b/Documentation/power/shutdown-debugging.rst index c510122e0bbc25..254cd3e0b5907c 100644 --- a/Documentation/power/shutdown-debugging.rst +++ b/Documentation/power/shutdown-debugging.rst @@ -33,7 +33,7 @@ some potential options include: Kernel Command-line Parameters ============================== -Add these parameters to your kernel command line: +Add these parameters to your kernel command line to capture shutdown logs: * ``printk.always_kmsg_dump=Y`` * Forces the kernel to dump the entire message buffer to pstore during @@ -41,6 +41,11 @@ Add these parameters to your kernel command line: * ``efi_pstore.pstore_disable=N`` * For EFI-based systems, ensures the EFI backend is active +If the system fails while running the hibernation power-off callbacks, add +``shutdown=legacy`` to use the legacy device shutdown callbacks instead. This +can be used to work around the failure and confirm which callback flow caused +it. + Userspace Interaction and Log Retrieval ======================================= On the next boot after a hang, pstore logs will be available in the pstore diff --git a/kernel/reboot.c b/kernel/reboot.c index 775b1057d57e6d..f5b7cfebc0a436 100644 --- a/kernel/reboot.c +++ b/kernel/reboot.c @@ -69,6 +69,7 @@ struct sys_off_handler { * of that. */ static bool poweroff_fallback_to_halt; +static bool shutdown_legacy; /* * Temporary stub that prevents linkage failure while we're in process @@ -308,7 +309,7 @@ static void kernel_shutdown_prepare(enum system_states state) system_state = state; usermodehelper_disable(); #ifdef CONFIG_HIBERNATE_CALLBACKS - if (state == SYSTEM_POWER_OFF) { + if (state == SYSTEM_POWER_OFF && !shutdown_legacy) { if (!dpm_suspend_start(PMSG_POWEROFF) && !dpm_suspend_end(PMSG_POWEROFF)) return; pr_emerg("Failed to power off devices, using shutdown instead.\n"); @@ -1103,6 +1104,13 @@ static ssize_t hw_protection_store(struct kobject *kobj, static struct kobj_attribute hw_protection_attr = __ATTR_RW(hw_protection); #endif +static int __init shutdown_setup(char *str) +{ + shutdown_legacy = true; + return 1; +} +__setup("shutdown=legacy", shutdown_setup); + static int __init reboot_setup(char *str) { for (;;) { From d5b9c86a48c8493ae9ec22457fbb83ba1480fc5f Mon Sep 17 00:00:00 2001 From: Marco Scardovi Date: Fri, 18 Sep 2026 15:44:14 +0200 Subject: [PATCH 1347/1352] leds: Add Dynamic Lighting class interface Add a dedicated Dynamic Lighting LED class for devices that expose multi-LED effects, palette programming, direct RGB streaming or lighting state persistence through sysfs. Define LED_DYNAMIC_LIGHTING on struct led_classdev and an optional led_dynamic back-pointer so the class can wrap a new LED or attach to an already registered one without replacing brightness or multi_intensity. Drivers supply their own effect name table and optional ops. Sysfs exposes only implemented attributes: effect and effect_index, optional enabled and enabled_index, speed and speed_range, direction, palette, power states, and a binary direct_buffer sink. Lighting off is enabled=false, not a dedicated off effect. Directions are left, right, up and down. Registration validates exported capabilities and serializes writes under led_access and the class-private lock so drivers can coexist with LED triggers. A KUnit test covers palette parsing, unknown-effect rejection and attribute visibility without hardware. This provides a common kernel ABI for complex lighting devices without requiring each driver to invent its own sysfs layout or rewrite an existing LED registration. Signed-off-by: Marco Scardovi --- MAINTAINERS | 9 + drivers/leds/.kunitconfig | 1 + drivers/leds/Kconfig | 9 + drivers/leds/Makefile | 14 + drivers/leds/led-class-dynamic-test.c | 203 ++++++ drivers/leds/led-class-dynamic.c | 928 ++++++++++++++++++++++++++ include/linux/led-dynamic-lighting.h | 321 +++++++++ include/linux/leds.h | 9 + 8 files changed, 1494 insertions(+) create mode 100644 drivers/leds/led-class-dynamic-test.c create mode 100644 drivers/leds/led-class-dynamic.c create mode 100644 include/linux/led-dynamic-lighting.h diff --git a/MAINTAINERS b/MAINTAINERS index 5c57cbb68060af..98cc15c4e334c1 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -15004,6 +15004,15 @@ S: Supported F: Documentation/scsi/leapraid.rst F: drivers/scsi/leapraid/ +LED DYNAMIC LIGHTING CLASS +M: Marco Scardovi +M: Denis Benato +L: linux-leds@vger.kernel.org +S: Maintained +F: drivers/leds/led-class-dynamic-test.c +F: drivers/leds/led-class-dynamic.c +F: include/linux/led-dynamic-lighting.h + LED SUBSYSTEM M: Lee Jones M: Pavel Machek diff --git a/drivers/leds/.kunitconfig b/drivers/leds/.kunitconfig index 5180f77910a115..4b5e48470c157e 100644 --- a/drivers/leds/.kunitconfig +++ b/drivers/leds/.kunitconfig @@ -1,4 +1,5 @@ CONFIG_KUNIT=y CONFIG_NEW_LEDS=y CONFIG_LEDS_CLASS=y +CONFIG_LEDS_CLASS_DYNAMIC=y CONFIG_LEDS_KUNIT_TEST=y diff --git a/drivers/leds/Kconfig b/drivers/leds/Kconfig index 7ec2c8d7854261..0dfd58a5dcd5af 100644 --- a/drivers/leds/Kconfig +++ b/drivers/leds/Kconfig @@ -46,6 +46,15 @@ config LEDS_CLASS_MULTICOLOR for multicolor LEDs that are grouped together. This class is not intended for single color LEDs. It can be built as a module. +config LEDS_CLASS_DYNAMIC + tristate "LED Dynamic Lighting Class Support" + depends on LEDS_CLASS + help + This option enables support for the Dynamic Lighting LED class in + /sys/class/leds. It wraps the LED class and adds dynamic lighting + attributes (effect, palette, per-key RGB streaming, and power + state persistence). + config LEDS_BRIGHTNESS_HW_CHANGED bool "LED Class brightness_hw_changed attribute support" depends on LEDS_CLASS diff --git a/drivers/leds/Makefile b/drivers/leds/Makefile index 4d4b089156e24a..efbe744a088d45 100644 --- a/drivers/leds/Makefile +++ b/drivers/leds/Makefile @@ -5,6 +5,20 @@ obj-$(CONFIG_NEW_LEDS) += led-core.o obj-$(CONFIG_LEDS_CLASS) += led-class.o obj-$(CONFIG_LEDS_CLASS_FLASH) += led-class-flash.o obj-$(CONFIG_LEDS_CLASS_MULTICOLOR) += led-class-multicolor.o +obj-$(CONFIG_LEDS_CLASS_DYNAMIC) += led-class-dynamic.o +# Follow LEDS_KUNIT_TEST. Skip the built-in test when the class is modular, +# so vmlinux does not reference a module. +ifeq ($(CONFIG_LEDS_CLASS_DYNAMIC),y) +ifeq ($(CONFIG_LEDS_KUNIT_TEST),y) +obj-y += led-class-dynamic-test.o +endif +ifeq ($(CONFIG_LEDS_KUNIT_TEST),m) +obj-m += led-class-dynamic-test.o +endif +endif +ifeq ($(CONFIG_LEDS_CLASS_DYNAMIC)$(CONFIG_LEDS_KUNIT_TEST),mm) +obj-m += led-class-dynamic-test.o +endif obj-$(CONFIG_LEDS_TRIGGERS) += led-triggers.o obj-$(CONFIG_LEDS_KUNIT_TEST) += led-test.o diff --git a/drivers/leds/led-class-dynamic-test.c b/drivers/leds/led-class-dynamic-test.c new file mode 100644 index 00000000000000..63cb04416e467d --- /dev/null +++ b/drivers/leds/led-class-dynamic-test.c @@ -0,0 +1,203 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * KUnit tests for the Dynamic Lighting LED class. + * + * Copyright (C) 2026 Open Gaming Collective + * Author: Marco Scardovi + * Author: Denis Benato + */ + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +static const char * const dl_test_effects[] = { + "static", + "breathe", +}; + +struct dl_test_priv { + struct led_classdev_dynamic ldev; + struct device *dev; +}; + +static int dl_test_set_effect(struct led_classdev_dynamic *ldev, unsigned int index) +{ + return 0; +} + +static int dl_test_set_palette(struct led_classdev_dynamic *ldev, + const struct dl_rgb *palette, + unsigned int num_entries) +{ + return 0; +} + +static const struct led_dynamic_ops dl_test_ops = { + .set_effect = dl_test_set_effect, + .set_palette = dl_test_set_palette, +}; + +static int dl_test_init(struct kunit *test) +{ + struct dl_test_priv *priv; + struct device *dev; + int ret; + + priv = kunit_kzalloc(test, sizeof(*priv), GFP_KERNEL); + if (!priv) + return -ENOMEM; + + dev = kunit_device_register(test, "dl_test"); + if (IS_ERR(dev)) + return PTR_ERR(dev); + + priv->dev = get_device(dev); + priv->ldev.cdev.name = "dl-test"; + priv->ldev.ops = &dl_test_ops; + priv->ldev.zone_type = "keyboard"; + priv->ldev.effects = dl_test_effects; + priv->ldev.num_effects = ARRAY_SIZE(dl_test_effects); + priv->ldev.max_palette_entries = 2; + priv->ldev.speed_min = 0; + priv->ldev.speed_max = 2; + + ret = devm_led_classdev_dynamic_register(priv->dev, &priv->ldev); + if (ret) { + put_device(priv->dev); + return ret; + } + + test->priv = priv; + return 0; +} + +static void dl_test_exit(struct kunit *test) +{ + struct dl_test_priv *priv = test->priv; + + if (priv && priv->dev) + put_device(priv->dev); +} + +static char *dl_test_attr_path(struct kunit *test, const char *attr) +{ + struct dl_test_priv *priv = test->priv; + char *kobj_path, *path; + + kobj_path = kobject_get_path(&priv->ldev.cdev.dev->kobj, GFP_KERNEL); + KUNIT_ASSERT_NOT_ERR_OR_NULL(test, kobj_path); + + path = kasprintf(GFP_KERNEL, "/sys/%s/%s", kobj_path, attr); + kfree(kobj_path); + KUNIT_ASSERT_NOT_ERR_OR_NULL(test, path); + return path; +} + +static bool dl_test_attr_visible(struct kunit *test, const char *attr) +{ + struct dl_test_priv *priv = test->priv; + struct kernfs_node *kn; + + kn = kernfs_find_and_get(priv->ldev.cdev.dev->kobj.sd, attr); + if (!kn) + return false; + + kernfs_put(kn); + return true; +} + +static int dl_test_write(struct kunit *test, const char *attr, const char *text) +{ + char *path __free(kfree) = dl_test_attr_path(test, attr); + struct file *file; + loff_t pos = 0; + ssize_t n; + + file = filp_open(path, O_WRONLY, 0); + if (IS_ERR(file)) + return PTR_ERR(file); + + n = kernel_write(file, text, strlen(text), &pos); + filp_close(file, NULL); + if (n < 0) + return n; + + return 0; +} + +static void dl_test_read(struct kunit *test, const char *attr, char *buf, size_t len) +{ + char *path __free(kfree) = dl_test_attr_path(test, attr); + struct file *file; + loff_t pos = 0; + ssize_t n; + + file = filp_open(path, O_RDONLY, 0); + KUNIT_ASSERT_FALSE(test, IS_ERR(file)); + + n = kernel_read(file, buf, len - 1, &pos); + filp_close(file, NULL); + KUNIT_ASSERT_GE(test, n, 0); + buf[n] = '\0'; +} + +static void dl_test_unknown_effect_rejected(struct kunit *test) +{ + int ret; + + KUNIT_EXPECT_TRUE(test, dl_test_attr_visible(test, "effect")); + KUNIT_EXPECT_TRUE(test, dl_test_attr_visible(test, "effect_index")); + KUNIT_EXPECT_FALSE(test, dl_test_attr_visible(test, "speed")); + KUNIT_EXPECT_FALSE(test, dl_test_attr_visible(test, "direct_buffer")); + + ret = dl_test_write(test, "effect", "no-such-effect"); + KUNIT_EXPECT_EQ(test, ret, -EINVAL); +} + +static void dl_test_effect_and_palette(struct kunit *test) +{ + char buf[64]; + int ret; + + ret = dl_test_write(test, "effect", "breathe"); + KUNIT_ASSERT_EQ(test, ret, 0); + dl_test_read(test, "effect", buf, sizeof(buf)); + KUNIT_EXPECT_STREQ(test, buf, "breathe\n"); + + ret = dl_test_write(test, "effects_palette", "#ff0000 #00ff00"); + KUNIT_ASSERT_EQ(test, ret, 0); + dl_test_read(test, "effects_palette", buf, sizeof(buf)); + KUNIT_EXPECT_STREQ(test, buf, "#ff0000 #00ff00\n"); + + ret = dl_test_write(test, "effects_palette", "#ff0000 #00ff00 #0000ff"); + KUNIT_EXPECT_EQ(test, ret, -EINVAL); + + ret = dl_test_write(test, "effects_palette", "ff0000"); + KUNIT_EXPECT_EQ(test, ret, -EINVAL); +} + +static struct kunit_case dl_test_cases[] = { + KUNIT_CASE(dl_test_unknown_effect_rejected), + KUNIT_CASE(dl_test_effect_and_palette), + { } +}; + +static struct kunit_suite dl_test_suite = { + .name = "led_class_dynamic", + .init = dl_test_init, + .exit = dl_test_exit, + .test_cases = dl_test_cases, +}; +kunit_test_suite(dl_test_suite); + +MODULE_AUTHOR("Marco Scardovi "); +MODULE_AUTHOR("Denis Benato "); +MODULE_DESCRIPTION("KUnit tests for the Dynamic Lighting LED class"); +MODULE_LICENSE("GPL"); diff --git a/drivers/leds/led-class-dynamic.c b/drivers/leds/led-class-dynamic.c new file mode 100644 index 00000000000000..6e3372cb42a5f8 --- /dev/null +++ b/drivers/leds/led-class-dynamic.c @@ -0,0 +1,928 @@ +// SPDX-License-Identifier: GPL-2.0-or-later +/* + * LED Dynamic Lighting Class Interface + * + * Copyright (C) 2026 Open Gaming Collective + * Author: Marco Scardovi + * Author: Denis Benato + */ + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +static const char * const dl_direction_names[] = { + [DL_DIRECTION_LEFT] = "left", + [DL_DIRECTION_RIGHT] = "right", + [DL_DIRECTION_UP] = "up", + [DL_DIRECTION_DOWN] = "down", +}; + +static const char * const dl_power_state_names[] = { + "boot", + "awake", + "sleep", + "shutdown", +}; + +static const char * const dl_enabled_names[] = { + "false", + "true", +}; + +static ssize_t dl_emit_indexed_names(char *buf, const char * const *names, + unsigned int count, u32 mask, bool filter) +{ + ssize_t len = 0; + unsigned int i; + + for (i = 0; i < count; i++) { + if (!names[i]) + continue; + if (filter && !(mask & BIT(i))) + continue; + + len += sysfs_emit_at(buf, len, "%s ", names[i]); + } + + if (len > 0) + buf[len - 1] = '\n'; + else + len = sysfs_emit(buf, "\n"); + + return len; +} + +static ssize_t zone_type_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct led_classdev *cdev = dev_get_drvdata(dev); + struct led_classdev_dynamic *ldev = lcdev_to_dldev(cdev); + + guard(mutex)(&ldev->lock); + + if (!ldev->zone_type || !*ldev->zone_type) + return sysfs_emit(buf, "unknown\n"); + + return sysfs_emit(buf, "%s\n", ldev->zone_type); +} +static DEVICE_ATTR_RO(zone_type); + +static ssize_t led_count_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct led_classdev *cdev = dev_get_drvdata(dev); + struct led_classdev_dynamic *ldev = lcdev_to_dldev(cdev); + + guard(mutex)(&ldev->lock); + + return sysfs_emit(buf, "%u\n", ldev->led_count); +} +static DEVICE_ATTR_RO(led_count); + +static ssize_t effect_index_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct led_classdev *cdev = dev_get_drvdata(dev); + struct led_classdev_dynamic *ldev = lcdev_to_dldev(cdev); + + guard(mutex)(&ldev->lock); + + return dl_emit_indexed_names(buf, ldev->effects, ldev->num_effects, 0, false); +} +static DEVICE_ATTR_RO(effect_index); + +static ssize_t effect_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct led_classdev *cdev = dev_get_drvdata(dev); + struct led_classdev_dynamic *ldev = lcdev_to_dldev(cdev); + const char *name; + + guard(mutex)(&ldev->lock); + + name = led_dynamic_effect_name(ldev); + if (!name) + return sysfs_emit(buf, "unknown\n"); + + return sysfs_emit(buf, "%s\n", name); +} + +static ssize_t effect_store(struct device *dev, + struct device_attribute *attr, + const char *buf, size_t count) +{ + struct led_classdev *cdev = dev_get_drvdata(dev); + struct led_classdev_dynamic *ldev = lcdev_to_dldev(cdev); + int match, ret; + + if (!ldev->ops->set_effect) + return -EOPNOTSUPP; + + match = __sysfs_match_string(ldev->effects, ldev->num_effects, buf); + if (match < 0) + return -EINVAL; + + guard(mutex)(&cdev->led_access); + if (led_sysfs_is_disabled(cdev)) + return -EBUSY; + led_trigger_remove(cdev); + guard(mutex)(&ldev->lock); + + ret = ldev->ops->set_effect(ldev, match); + if (ret < 0) + return ret; + + ldev->current_effect = match; + return count; +} +static DEVICE_ATTR_RW(effect); + +static ssize_t enabled_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct led_classdev *cdev = dev_get_drvdata(dev); + struct led_classdev_dynamic *ldev = lcdev_to_dldev(cdev); + + guard(mutex)(&ldev->lock); + + return sysfs_emit(buf, "%s\n", ldev->enabled ? "true" : "false"); +} + +static ssize_t enabled_store(struct device *dev, + struct device_attribute *attr, + const char *buf, size_t count) +{ + struct led_classdev *cdev = dev_get_drvdata(dev); + struct led_classdev_dynamic *ldev = lcdev_to_dldev(cdev); + int match, ret; + + if (!ldev->ops->set_enabled) + return -EOPNOTSUPP; + + match = sysfs_match_string(dl_enabled_names, buf); + if (match < 0) + return -EINVAL; + + guard(mutex)(&cdev->led_access); + if (led_sysfs_is_disabled(cdev)) + return -EBUSY; + led_trigger_remove(cdev); + guard(mutex)(&ldev->lock); + + ret = ldev->ops->set_enabled(ldev, match); + if (ret < 0) + return ret; + + ldev->enabled = !!match; + return count; +} +static DEVICE_ATTR_RW(enabled); + +static ssize_t enabled_index_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + return sysfs_emit(buf, "false true\n"); +} +static DEVICE_ATTR_RO(enabled_index); + +static ssize_t speed_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct led_classdev *cdev = dev_get_drvdata(dev); + struct led_classdev_dynamic *ldev = lcdev_to_dldev(cdev); + + guard(mutex)(&ldev->lock); + + return sysfs_emit(buf, "%u\n", ldev->speed); +} + +static ssize_t speed_range_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct led_classdev *cdev = dev_get_drvdata(dev); + struct led_classdev_dynamic *ldev = lcdev_to_dldev(cdev); + + guard(mutex)(&ldev->lock); + + return sysfs_emit(buf, "%u-%u\n", ldev->speed_min, ldev->speed_max); +} +static DEVICE_ATTR_RO(speed_range); + +static ssize_t speed_store(struct device *dev, + struct device_attribute *attr, + const char *buf, size_t count) +{ + struct led_classdev *cdev = dev_get_drvdata(dev); + struct led_classdev_dynamic *ldev = lcdev_to_dldev(cdev); + unsigned int speed; + int ret; + + if (!ldev->ops->set_speed) + return -EOPNOTSUPP; + + ret = kstrtouint(buf, 10, &speed); + if (ret) + return ret; + + if (speed < ldev->speed_min || speed > ldev->speed_max) + return -EINVAL; + + guard(mutex)(&cdev->led_access); + if (led_sysfs_is_disabled(cdev)) + return -EBUSY; + guard(mutex)(&ldev->lock); + + ret = ldev->ops->set_speed(ldev, speed); + if (ret < 0) + return ret; + + ldev->speed = speed; + return count; +} +static DEVICE_ATTR_RW(speed); + +static ssize_t direction_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct led_classdev *cdev = dev_get_drvdata(dev); + struct led_classdev_dynamic *ldev = lcdev_to_dldev(cdev); + + guard(mutex)(&ldev->lock); + + if (ldev->direction >= ARRAY_SIZE(dl_direction_names) || + !dl_direction_names[ldev->direction]) + return sysfs_emit(buf, "unknown\n"); + + return sysfs_emit(buf, "%s\n", dl_direction_names[ldev->direction]); +} + +static ssize_t direction_index_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct led_classdev *cdev = dev_get_drvdata(dev); + struct led_classdev_dynamic *ldev = lcdev_to_dldev(cdev); + + guard(mutex)(&ldev->lock); + + return dl_emit_indexed_names(buf, dl_direction_names, + ARRAY_SIZE(dl_direction_names), + ldev->supported_directions, true); +} +static DEVICE_ATTR_RO(direction_index); + +static ssize_t direction_store(struct device *dev, + struct device_attribute *attr, + const char *buf, size_t count) +{ + struct led_classdev *cdev = dev_get_drvdata(dev); + struct led_classdev_dynamic *ldev = lcdev_to_dldev(cdev); + int match, ret; + + if (!ldev->ops->set_direction || !ldev->supported_directions) + return -EOPNOTSUPP; + + match = sysfs_match_string(dl_direction_names, buf); + if (match < 0 || !(ldev->supported_directions & BIT(match))) + return -EINVAL; + + guard(mutex)(&cdev->led_access); + if (led_sysfs_is_disabled(cdev)) + return -EBUSY; + guard(mutex)(&ldev->lock); + + ret = ldev->ops->set_direction(ldev, match); + if (ret < 0) + return ret; + + ldev->direction = match; + return count; +} +static DEVICE_ATTR_RW(direction); + +static ssize_t effects_palette_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct led_classdev *cdev = dev_get_drvdata(dev); + struct led_classdev_dynamic *ldev = lcdev_to_dldev(cdev); + int len = 0; + unsigned int i; + + guard(mutex)(&ldev->lock); + + for (i = 0; i < ldev->num_palette_entries; i++) { + len += sysfs_emit_at(buf, len, "#%02x%02x%02x%c", + ldev->palette[i].r, + ldev->palette[i].g, + ldev->palette[i].b, + (i == ldev->num_palette_entries - 1) ? '\n' : ' '); + } + + if (!len) + len = sysfs_emit(buf, "\n"); + + return len; +} + +static ssize_t max_palette_entries_show(struct device *dev, + struct device_attribute *attr, + char *buf) +{ + struct led_classdev *cdev = dev_get_drvdata(dev); + struct led_classdev_dynamic *ldev = lcdev_to_dldev(cdev); + + guard(mutex)(&ldev->lock); + + return sysfs_emit(buf, "%u\n", ldev->max_palette_entries); +} +static DEVICE_ATTR_RO(max_palette_entries); + +static ssize_t effects_palette_store(struct device *dev, + struct device_attribute *attr, + const char *buf, size_t count) +{ + struct led_classdev *cdev = dev_get_drvdata(dev); + struct led_classdev_dynamic *ldev = lcdev_to_dldev(cdev); + const char *cur = buf; + unsigned int num_parsed = 0; + int ret; + + if (!ldev->ops->set_palette || !ldev->max_palette_entries) + return -EOPNOTSUPP; + + struct dl_rgb *temp_palette __free(kfree) = kmalloc_array(ldev->max_palette_entries, + sizeof(*temp_palette), + GFP_KERNEL); + if (!temp_palette) + return -ENOMEM; + + while (*cur) { + cur = skip_spaces(cur); + if (!*cur) + break; + + if (num_parsed >= ldev->max_palette_entries) + return -EINVAL; + + if (*cur != '#') + return -EINVAL; + cur++; + + if (strnlen(cur, 6) < 6) + return -EINVAL; + + if (hex2bin((u8 *)&temp_palette[num_parsed], cur, 3) < 0) + return -EINVAL; + cur += 6; + if (*cur && !isspace(*cur)) + return -EINVAL; + num_parsed++; + } + + if (!num_parsed) + return -EINVAL; + + guard(mutex)(&cdev->led_access); + if (led_sysfs_is_disabled(cdev)) + return -EBUSY; + led_trigger_remove(cdev); + guard(mutex)(&ldev->lock); + + ret = ldev->ops->set_palette(ldev, temp_palette, num_parsed); + if (ret < 0) + return ret; + + memcpy(ldev->palette, temp_palette, num_parsed * sizeof(*temp_palette)); + ldev->num_palette_entries = num_parsed; + + return count; +} +static DEVICE_ATTR_RW(effects_palette); + +static ssize_t power_states_index_show(struct device *dev, + struct device_attribute *attr, + char *buf) +{ + struct led_classdev *cdev = dev_get_drvdata(dev); + struct led_classdev_dynamic *ldev = lcdev_to_dldev(cdev); + + guard(mutex)(&ldev->lock); + + return dl_emit_indexed_names(buf, dl_power_state_names, + ARRAY_SIZE(dl_power_state_names), + ldev->supported_power_states, true); +} +static DEVICE_ATTR_RO(power_states_index); + +static ssize_t power_states_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct led_classdev *cdev = dev_get_drvdata(dev); + struct led_classdev_dynamic *ldev = lcdev_to_dldev(cdev); + + guard(mutex)(&ldev->lock); + + return dl_emit_indexed_names(buf, dl_power_state_names, + ARRAY_SIZE(dl_power_state_names), + ldev->active_power_states, true); +} + +static ssize_t power_states_store(struct device *dev, + struct device_attribute *attr, + const char *buf, size_t count) +{ + struct led_classdev *cdev = dev_get_drvdata(dev); + struct led_classdev_dynamic *ldev = lcdev_to_dldev(cdev); + char state_name[16]; + const char *cur = buf; + u32 target_states = 0; + int ret, match; + size_t tok_len; + + if (!ldev->ops->set_power_states || !ldev->supported_power_states) + return -EOPNOTSUPP; + + while (*cur) { + cur = skip_spaces(cur); + if (!*cur || *cur == '\n') + break; + + tok_len = strcspn(cur, " \t\n"); + if (!tok_len || tok_len >= sizeof(state_name)) + return -EINVAL; + + memcpy(state_name, cur, tok_len); + state_name[tok_len] = '\0'; + cur += tok_len; + + match = sysfs_match_string(dl_power_state_names, state_name); + if (match < 0 || !(ldev->supported_power_states & BIT(match))) + return -EINVAL; + + target_states |= BIT(match); + } + + guard(mutex)(&cdev->led_access); + if (led_sysfs_is_disabled(cdev)) + return -EBUSY; + guard(mutex)(&ldev->lock); + + ret = ldev->ops->set_power_states(ldev, target_states); + if (ret < 0) + return ret; + + ldev->active_power_states = target_states; + return count; +} +static DEVICE_ATTR_RW(power_states); + +/* + * kernfs delivers bin-attribute writes in at most PAGE_SIZE chunks. Stage + * partial writes and invoke the driver only when the full direct_buffer + * payload (led_count * 3) has arrived. A short write that cannot grow into + * that size is rejected. Any other non-zero offset is accepted only as the + * next contiguous chunk. + */ +static ssize_t led_dynamic_stage_bin_write(struct led_classdev *cdev, + struct led_classdev_dynamic *ldev, + char *buf, loff_t off, size_t count, + size_t size, + int (*commit)(struct led_classdev_dynamic *ldev, + const u8 *data, size_t len)) +{ + size_t end; + int ret; + + if (!size || !count || off < 0) + return -EINVAL; + + if (check_add_overflow((size_t)off, count, &end) || end > size) + return -EINVAL; + + if (off == 0 && count < PAGE_SIZE && count != size) + return -EINVAL; + + guard(mutex)(&cdev->led_access); + if (led_sysfs_is_disabled(cdev)) + return -EBUSY; + guard(mutex)(&ldev->lock); + + if (off == 0) { + ldev->write_filled = 0; + if (!ldev->write_staging || ldev->write_staging_size != size) { + kfree(ldev->write_staging); + ldev->write_staging = kmalloc(size, GFP_KERNEL); + if (!ldev->write_staging) { + ldev->write_staging_size = 0; + return -ENOMEM; + } + ldev->write_staging_size = size; + } + } else if (!ldev->write_staging || ldev->write_staging_size != size || + (size_t)off != ldev->write_filled) { + return -EINVAL; + } + + memcpy(ldev->write_staging + off, buf, count); + ldev->write_filled = end; + + if (end != size) + return count; + + led_trigger_remove(cdev); + ret = commit(ldev, ldev->write_staging, size); + ldev->write_filled = 0; + if (ret < 0) + return ret; + + return count; +} + +static int led_dynamic_commit_direct(struct led_classdev_dynamic *ldev, + const u8 *data, size_t len) +{ + int ret, direct_idx; + + ret = ldev->ops->direct_write(ldev, data, len); + if (ret < 0) + return ret; + + direct_idx = led_dynamic_effect_index(ldev, "direct"); + if (direct_idx >= 0) + ldev->current_effect = direct_idx; + + return 0; +} + +static ssize_t direct_buffer_write(struct file *filp, struct kobject *kobj, + const struct bin_attribute *bin_attr, + char *buf, loff_t off, size_t count) +{ + struct device *dev = kobj_to_dev(kobj); + struct led_classdev *cdev = dev_get_drvdata(dev); + struct led_classdev_dynamic *ldev = lcdev_to_dldev(cdev); + size_t expected_size; + + if (!ldev->ops->direct_write) + return -EOPNOTSUPP; + + if (check_mul_overflow((size_t)ldev->led_count, 3, &expected_size)) + return -EOVERFLOW; + + return led_dynamic_stage_bin_write(cdev, ldev, buf, off, count, + expected_size, + led_dynamic_commit_direct); +} + +static int led_dynamic_validate(struct led_classdev_dynamic *ldev, size_t *direct_buffer_size) +{ + unsigned int i; + + if (check_mul_overflow((size_t)ldev->led_count, 3, direct_buffer_size)) + return -EOVERFLOW; + + if (ldev->ops->set_effect) { + if (!ldev->effects || !ldev->num_effects) + return -EINVAL; + + for (i = 0; i < ldev->num_effects; i++) { + if (!ldev->effects[i] || !*ldev->effects[i]) + return -EINVAL; + } + + if (ldev->current_effect >= ldev->num_effects) + return -EINVAL; + } + + if (ldev->speed_min > ldev->speed_max) + return -EINVAL; + + if (ldev->ops->set_speed && + (ldev->speed < ldev->speed_min || ldev->speed > ldev->speed_max)) + return -EINVAL; + + if (ldev->direction >= DL_DIRECTION_MAX) + return -EINVAL; + + if (ldev->supported_directions && + !(ldev->supported_directions & BIT(ldev->direction))) + return -EINVAL; + + if (ldev->num_palette_entries > ldev->max_palette_entries) + return -EINVAL; + + if (ldev->num_palette_entries && !ldev->palette) + return -EINVAL; + + if (ldev->active_power_states & ~ldev->supported_power_states) + return -EINVAL; + + if (ldev->host) { + if (!ldev->host->dev) + return -EINVAL; + if (ldev->host->led_dynamic || + (ldev->host->flags & LED_DYNAMIC_LIGHTING)) + return -EBUSY; + } + + return 0; +} + +static umode_t dl_attr_is_visible(struct kobject *kobj, struct attribute *attr, int n) +{ + struct device *dev = kobj_to_dev(kobj); + struct led_classdev *cdev = dev_get_drvdata(dev); + struct led_classdev_dynamic *ldev = lcdev_to_dldev(cdev); + + if (attr == &dev_attr_power_states_index.attr || + attr == &dev_attr_power_states.attr) { + if (!ldev->supported_power_states || !ldev->ops->set_power_states) + return 0; + } + + if (attr == &dev_attr_speed_range.attr || + attr == &dev_attr_speed.attr) { + if (!ldev->ops->set_speed) + return 0; + } + + if (attr == &dev_attr_direction_index.attr || + attr == &dev_attr_direction.attr) { + if (!ldev->supported_directions || !ldev->ops->set_direction) + return 0; + } + + if (attr == &dev_attr_max_palette_entries.attr || + attr == &dev_attr_effects_palette.attr) { + if (!ldev->max_palette_entries || !ldev->ops->set_palette) + return 0; + } + + if (attr == &dev_attr_effect.attr || attr == &dev_attr_effect_index.attr) { + if (!ldev->num_effects || !ldev->ops->set_effect) + return 0; + } + + if (attr == &dev_attr_enabled.attr || + attr == &dev_attr_enabled_index.attr) { + if (!ldev->ops->set_enabled) + return 0; + } + + return attr->mode; +} + +static umode_t dl_bin_attr_is_visible(struct kobject *kobj, + const struct bin_attribute *attr, int n) +{ + struct device *dev = kobj_to_dev(kobj); + struct led_classdev *cdev = dev_get_drvdata(dev); + struct led_classdev_dynamic *ldev = lcdev_to_dldev(cdev); + + if (attr == &ldev->bin_attr_direct) { + if (!ldev->ops->direct_write || !ldev->led_count) + return 0; + } + + return attr->attr.mode; +} + +static struct attribute *led_dynamic_attrs[] = { + &dev_attr_zone_type.attr, + &dev_attr_led_count.attr, + &dev_attr_effect_index.attr, + &dev_attr_effect.attr, + &dev_attr_enabled.attr, + &dev_attr_enabled_index.attr, + &dev_attr_speed_range.attr, + &dev_attr_speed.attr, + &dev_attr_direction_index.attr, + &dev_attr_direction.attr, + &dev_attr_max_palette_entries.attr, + &dev_attr_effects_palette.attr, + &dev_attr_power_states_index.attr, + &dev_attr_power_states.attr, + NULL, +}; + +static void led_dynamic_release_resources(struct led_classdev_dynamic *ldev, + struct led_classdev *cdev) +{ + if (ldev->palette_allocated) { + kfree(ldev->palette); + ldev->palette = NULL; + ldev->palette_allocated = false; + } + kfree(ldev->write_staging); + ldev->write_staging = NULL; + ldev->write_staging_size = 0; + ldev->write_filled = 0; + kfree(ldev->merged_groups); + ldev->merged_groups = NULL; + if (cdev) + cdev->groups = ldev->driver_groups; + mutex_destroy(&ldev->lock); +} + +int led_classdev_dynamic_register_ext(struct device *parent, + struct led_classdev_dynamic *ldev, + struct led_init_data *init_data) +{ + struct led_classdev *cdev; + size_t direct_buffer_size; + unsigned int num_driver_groups = 0; + int ret; + + if (!ldev || !ldev->ops) + return -EINVAL; + + ret = led_dynamic_validate(ldev, &direct_buffer_size); + if (ret) + return ret; + + mutex_init(&ldev->lock); + ldev->enabled = true; + ldev->attached = false; + ldev->palette_allocated = false; + ldev->merged_groups = NULL; + ldev->write_staging = NULL; + ldev->write_staging_size = 0; + ldev->write_filled = 0; + + sysfs_bin_attr_init(&ldev->bin_attr_direct); + ldev->bin_attr_direct.attr.name = "direct_buffer"; + ldev->bin_attr_direct.attr.mode = 0200; + ldev->bin_attr_direct.write = direct_buffer_write; + ldev->bin_attr_direct.size = direct_buffer_size; + + ldev->bin_attrs[0] = &ldev->bin_attr_direct; + ldev->bin_attrs[1] = NULL; + + ldev->group.attrs = led_dynamic_attrs; + ldev->group.bin_attrs = ldev->bin_attrs; + ldev->group.is_visible = dl_attr_is_visible; + ldev->group.is_bin_visible = dl_bin_attr_is_visible; + + if (ldev->max_palette_entries > 0 && !ldev->palette) { + ldev->palette = + kcalloc(ldev->max_palette_entries, sizeof(*ldev->palette), + GFP_KERNEL); + if (!ldev->palette) { + mutex_destroy(&ldev->lock); + return -ENOMEM; + } + ldev->palette_allocated = true; + } + + if (ldev->host) { + cdev = ldev->host; + cdev->led_dynamic = ldev; + cdev->flags |= LED_DYNAMIC_LIGHTING; + ldev->attached = true; + + ret = device_add_group(cdev->dev, &ldev->group); + if (ret) { + cdev->led_dynamic = NULL; + cdev->flags &= ~LED_DYNAMIC_LIGHTING; + led_dynamic_release_resources(ldev, NULL); + return ret; + } + + return 0; + } + + cdev = &ldev->cdev; + cdev->flags |= LED_DYNAMIC_LIGHTING; + cdev->led_dynamic = ldev; + + ldev->groups[0] = &ldev->group; + ldev->groups[1] = NULL; + ldev->driver_groups = cdev->groups; + + while (cdev->groups && cdev->groups[num_driver_groups]) + num_driver_groups++; + + if (num_driver_groups) { + unsigned int i; + + ldev->merged_groups = kcalloc(num_driver_groups + 2, + sizeof(*ldev->merged_groups), + GFP_KERNEL); + if (!ldev->merged_groups) { + led_dynamic_release_resources(ldev, cdev); + cdev->led_dynamic = NULL; + cdev->flags &= ~LED_DYNAMIC_LIGHTING; + return -ENOMEM; + } + + for (i = 0; i < num_driver_groups; i++) + ldev->merged_groups[i] = cdev->groups[i]; + ldev->merged_groups[num_driver_groups] = &ldev->group; + ldev->merged_groups[num_driver_groups + 1] = NULL; + cdev->groups = ldev->merged_groups; + } else { + cdev->groups = ldev->groups; + } + + ret = led_classdev_register_ext(parent, cdev, init_data); + if (ret) { + led_dynamic_release_resources(ldev, cdev); + cdev->led_dynamic = NULL; + cdev->flags &= ~LED_DYNAMIC_LIGHTING; + } + + return ret; +} +EXPORT_SYMBOL_GPL(led_classdev_dynamic_register_ext); + +void led_classdev_dynamic_unregister(struct led_classdev_dynamic *ldev) +{ + struct led_classdev *cdev; + + if (!ldev) + return; + + if (ldev->attached) { + cdev = ldev->host; + if (cdev && cdev->dev) + device_remove_group(cdev->dev, &ldev->group); + if (cdev) { + cdev->led_dynamic = NULL; + cdev->flags &= ~LED_DYNAMIC_LIGHTING; + } + led_dynamic_release_resources(ldev, NULL); + ldev->attached = false; + return; + } + + cdev = &ldev->cdev; + led_classdev_unregister(cdev); + cdev->led_dynamic = NULL; + led_dynamic_release_resources(ldev, cdev); +} +EXPORT_SYMBOL_GPL(led_classdev_dynamic_unregister); + +static void devm_led_classdev_dynamic_release(struct device *dev, void *res) +{ + led_classdev_dynamic_unregister(*(struct led_classdev_dynamic **)res); +} + +int devm_led_classdev_dynamic_register_ext(struct device *parent, + struct led_classdev_dynamic *ldev, + struct led_init_data *init_data) +{ + struct led_classdev_dynamic **dr; + int ret; + + dr = devres_alloc(devm_led_classdev_dynamic_release, + sizeof(*dr), GFP_KERNEL); + if (!dr) + return -ENOMEM; + + ret = led_classdev_dynamic_register_ext(parent, ldev, init_data); + if (ret) { + devres_free(dr); + return ret; + } + + *dr = ldev; + devres_add(parent, dr); + + return 0; +} +EXPORT_SYMBOL_GPL(devm_led_classdev_dynamic_register_ext); + +static int devm_led_classdev_dynamic_match(struct device *dev, + void *res, void *data) +{ + struct led_classdev_dynamic **p = res; + + if (WARN_ON(!p || !*p)) + return 0; + + return *p == data; +} + +void devm_led_classdev_dynamic_unregister(struct device *dev, + struct led_classdev_dynamic *ldev) +{ + WARN_ON(devres_release(dev, + devm_led_classdev_dynamic_release, + devm_led_classdev_dynamic_match, ldev)); +} +EXPORT_SYMBOL_GPL(devm_led_classdev_dynamic_unregister); + +MODULE_AUTHOR("Marco Scardovi "); +MODULE_AUTHOR("Denis Benato "); +MODULE_DESCRIPTION("LED Dynamic Lighting Class Interface"); +MODULE_LICENSE("GPL"); diff --git a/include/linux/led-dynamic-lighting.h b/include/linux/led-dynamic-lighting.h new file mode 100644 index 00000000000000..aa042b8741ce33 --- /dev/null +++ b/include/linux/led-dynamic-lighting.h @@ -0,0 +1,321 @@ +/* SPDX-License-Identifier: GPL-2.0-or-later */ +/* + * LED Dynamic Lighting Class Interface + * + * Copyright (C) 2026 Open Gaming Collective + * Author: Marco Scardovi + * Author: Denis Benato + */ + +#ifndef _LINUX_LED_DYNAMIC_LIGHTING_H +#define _LINUX_LED_DYNAMIC_LIGHTING_H + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +/** + * enum dl_direction - Effect animation propagation direction + * @DL_DIRECTION_LEFT: Animation moves toward the left + * @DL_DIRECTION_RIGHT: Animation moves toward the right + * @DL_DIRECTION_UP: Animation moves upward + * @DL_DIRECTION_DOWN: Animation moves downward + * @DL_DIRECTION_MAX: Number of directions + */ +enum dl_direction { + DL_DIRECTION_LEFT = 0, + DL_DIRECTION_RIGHT, + DL_DIRECTION_UP, + DL_DIRECTION_DOWN, + DL_DIRECTION_MAX, +}; + +/* Power state bitmask flags */ +#define DL_POWER_STATE_BOOT BIT(0) +#define DL_POWER_STATE_AWAKE BIT(1) +#define DL_POWER_STATE_SLEEP BIT(2) +#define DL_POWER_STATE_SHUTDOWN BIT(3) +#define DL_POWER_STATE_ALL (DL_POWER_STATE_BOOT | \ + DL_POWER_STATE_AWAKE | \ + DL_POWER_STATE_SLEEP | \ + DL_POWER_STATE_SHUTDOWN) + +/** + * struct dl_rgb - 24-bit RGB color representation + * @r: Red component (0-255) + * @g: Green component (0-255) + * @b: Blue component (0-255) + */ +struct dl_rgb { + u8 r; + u8 g; + u8 b; +}; + +struct led_classdev_dynamic; + +/** + * struct led_dynamic_ops - Hardware driver callback vector + * @set_effect: Select the effect at @index in ldev->effects + * @set_speed: Configure effect speed (within speed_min..speed_max) + * @set_direction: Configure effect propagation direction + * @set_palette: Apply multi-color stacked palette + * @set_enabled: Turn the lighting engine on or off without changing effect + * @direct_write: Stream packed RGB buffer (size must equal led_count * 3) + * @set_power_states: Update active power state persistence bitmask + * + * Every callback is optional. Sysfs files are created only for implemented + * ops (and the matching capability fields). Existing LED drivers can keep + * their current brightness / multi_intensity path and implement only the + * ops they already have hardware for. + */ +struct led_dynamic_ops { + int (*set_effect)(struct led_classdev_dynamic *ldev, + unsigned int index); + int (*set_speed)(struct led_classdev_dynamic *ldev, + unsigned int speed); + int (*set_direction)(struct led_classdev_dynamic *ldev, + enum dl_direction direction); + int (*set_palette)(struct led_classdev_dynamic *ldev, + const struct dl_rgb *palette, + unsigned int num_entries); + int (*set_enabled)(struct led_classdev_dynamic *ldev, + bool enabled); + int (*direct_write)(struct led_classdev_dynamic *ldev, + const u8 *buffer, size_t size); + int (*set_power_states)(struct led_classdev_dynamic *ldev, + u32 active_states); +}; + +/** + * struct led_classdev_dynamic - Dynamic Lighting LED class device + * @cdev: Embedded standard LED classdev (unused when @host is set) + * @host: Optional already-registered LED to attach to (drop-in) + * @ops: Hardware callback dispatch table + * @lock: Internal mutex protecting ldev state and serialization + * @zone_type: Driver-defined physical topology string for the lighting zone + * @led_count: Total individual LEDs in this zone + * @effects: Driver-owned table of effect names (not a global enum) + * @num_effects: Number of entries in @effects + * @current_effect: Index into @effects of the currently selected effect + * @speed: Current effect animation speed + * @speed_min: Minimum supported speed level + * @speed_max: Maximum supported speed level + * @direction: Current effect animation direction + * @supported_directions: Bitmask of supported enum dl_direction values + * @enabled: Lighting engine on/off; independent from @current_effect + * @palette: Allocated array of stacked palette color entries + * @num_palette_entries: Current number of valid palette entries + * @max_palette_entries: Maximum allowable palette entries + * @palette_allocated: True if @palette was allocated by the Dynamic Lighting core + * @supported_power_states: Bitmask of DL_POWER_STATE_* supported by hardware + * @active_power_states: Bitmask of currently active DL_POWER_STATE_* states + * @driver_data: Private driver reference pointer + * @bin_attr_direct: Per-instance direct RGB binary attribute + * @bin_attrs: Array of binary attribute pointers for group + * @write_staging: Scratch buffer for kernfs PAGE_SIZE-chunked bin writes + * @write_staging_size: Allocated size of @write_staging + * @write_filled: Bytes accepted so far in the current chunked bin write + * @group: Per-instance sysfs attribute group + * @groups: Inline sysfs attribute groups pointer array for cdev + * @driver_groups: Original driver-provided sysfs groups saved during registration + * @merged_groups: Optional dynamically allocated merge of driver and Dynamic Lighting groups + * @attached: True when sysfs was added to @host instead of registering @cdev + * + * Drop-in on an existing LED class device + * --------------------------------------- + * Drivers that already register a LED (including LED_MULTI_COLOR) can attach + * Dynamic Lighting without replacing that registration or rewriting color + * handling. Color stays on brightness / multi_intensity; this class adds + * optional effect, speed, enabled, palette, and direct RGB nodes: + * + * struct led_classdev_dynamic ldev = { + * .host = existing_led_cdev, + * .ops = &my_ops, + * .effects = my_effects, + * .num_effects = ARRAY_SIZE(my_effects), + * }; + * led_classdev_dynamic_register(dev, &ldev); + * + * Effect names are defined by the driver. Userspace must read effect_index. + * Lighting off is enabled=false, not a dedicated off effect. + */ +struct led_classdev_dynamic { + struct led_classdev cdev; + struct led_classdev *host; + const struct led_dynamic_ops *ops; + struct mutex lock; /* Protects ldev state serialization */ + + const char *zone_type; + unsigned int led_count; + + const char * const *effects; + unsigned int num_effects; + unsigned int current_effect; + + unsigned int speed; + unsigned int speed_min; + unsigned int speed_max; + + enum dl_direction direction; + unsigned int supported_directions; + + bool enabled; + + struct dl_rgb *palette; + unsigned int num_palette_entries; + unsigned int max_palette_entries; + bool palette_allocated; + + u32 supported_power_states; + u32 active_power_states; + + void *driver_data; + + struct bin_attribute bin_attr_direct __aligned(__alignof__(const struct bin_attribute)); + const struct bin_attribute *bin_attrs[2]; + u8 *write_staging; + size_t write_staging_size; + size_t write_filled; + struct attribute_group group; + const struct attribute_group *groups[2]; + const struct attribute_group **driver_groups; + const struct attribute_group **merged_groups; + bool attached; +}; + +/** + * led_dynamic_cdev - LED classdev backing a Dynamic Lighting instance + * + * Returns @host when this instance was attached to an existing LED, + * otherwise the embedded @cdev. + */ +static inline struct led_classdev *led_dynamic_cdev(struct led_classdev_dynamic *ldev) +{ + return ldev->host ? ldev->host : &ldev->cdev; +} + +static inline struct led_classdev_dynamic *lcdev_to_dldev(struct led_classdev *lcdev) +{ + if (lcdev->led_dynamic) + return lcdev->led_dynamic; + + return container_of(lcdev, struct led_classdev_dynamic, cdev); +} + +static inline bool is_dynamic_lighting_led(struct led_classdev *lcdev) +{ + return !!(lcdev->flags & LED_DYNAMIC_LIGHTING); +} + +static inline const char * +led_dynamic_effect_name(const struct led_classdev_dynamic *ldev) +{ + if (!ldev->effects || ldev->current_effect >= ldev->num_effects) + return NULL; + + return ldev->effects[ldev->current_effect]; +} + +static inline bool led_dynamic_effect_is(const struct led_classdev_dynamic *ldev, + const char *name) +{ + const char *cur = led_dynamic_effect_name(ldev); + + return cur && name && !strcmp(cur, name); +} + +static inline int led_dynamic_effect_index(const struct led_classdev_dynamic *ldev, + const char *name) +{ + if (!ldev->effects || !name) + return -EINVAL; + + return __sysfs_match_string(ldev->effects, ldev->num_effects, name); +} + +/** + * led_dynamic_fill_effects - compact a name pool through a bitmask + * @dst: caller-owned array with at least hweight32(@mask) slots + * @dst_n: number of slots in @dst + * @pool: driver-owned names indexed by bit number + * @pool_n: number of entries in @pool + * @mask: bits selecting which names to include + * + * Helper for drivers that already track effects as capability bits and + * want to publish a Dynamic Lighting table without rewriting that logic. + * + * Return: number of names written, or -EINVAL. + */ +static inline int led_dynamic_fill_effects(const char **dst, unsigned int dst_n, + const char * const *pool, + unsigned int pool_n, u32 mask) +{ + unsigned int i, n = 0; + + if (!dst || !pool) + return -EINVAL; + + for (i = 0; i < pool_n && n < dst_n; i++) { + if (!(mask & BIT(i)) || !pool[i]) + continue; + dst[n++] = pool[i]; + } + + return n; +} + +#if IS_REACHABLE(CONFIG_LEDS_CLASS_DYNAMIC) + +int led_classdev_dynamic_register_ext(struct device *parent, + struct led_classdev_dynamic *ldev, + struct led_init_data *init_data); +void led_classdev_dynamic_unregister(struct led_classdev_dynamic *ldev); +int devm_led_classdev_dynamic_register_ext(struct device *parent, + struct led_classdev_dynamic *ldev, + struct led_init_data *init_data); +void devm_led_classdev_dynamic_unregister(struct device *parent, + struct led_classdev_dynamic *ldev); + +#else + +static inline int led_classdev_dynamic_register_ext(struct device *parent, + struct led_classdev_dynamic *ldev, + struct led_init_data *init_data) +{ + return -EOPNOTSUPP; +} + +static inline void led_classdev_dynamic_unregister(struct led_classdev_dynamic *ldev) {} + +static inline int devm_led_classdev_dynamic_register_ext(struct device *parent, + struct led_classdev_dynamic *ldev, + struct led_init_data *init_data) +{ + return -EOPNOTSUPP; +} + +static inline void devm_led_classdev_dynamic_unregister(struct device *parent, + struct led_classdev_dynamic *ldev) {} + +#endif /* IS_REACHABLE(CONFIG_LEDS_CLASS_DYNAMIC) */ + +static inline int devm_led_classdev_dynamic_register(struct device *parent, + struct led_classdev_dynamic *ldev) +{ + return devm_led_classdev_dynamic_register_ext(parent, ldev, NULL); +} + +static inline int led_classdev_dynamic_register(struct device *parent, + struct led_classdev_dynamic *ldev) +{ + return led_classdev_dynamic_register_ext(parent, ldev, NULL); +} + +#endif /* _LINUX_LED_DYNAMIC_LIGHTING_H */ diff --git a/include/linux/leds.h b/include/linux/leds.h index beaf2399306393..e1e0bfcb7b82b8 100644 --- a/include/linux/leds.h +++ b/include/linux/leds.h @@ -110,6 +110,7 @@ struct led_classdev { #define LED_REJECT_NAME_CONFLICT BIT(24) #define LED_MULTI_COLOR BIT(25) #define LED_TRIG_HW_CHANGED BIT(26) +#define LED_DYNAMIC_LIGHTING BIT(27) /* set_brightness_work / blink_timer flags, atomic, private. */ unsigned long work_flags; @@ -163,6 +164,14 @@ struct led_classdev { struct device *dev; const struct attribute_group **groups; + /* + * Back-pointer to a Dynamic Lighting extension, if any. Set by + * led_classdev_dynamic_register() for both standalone devices and + * drop-in attach onto an existing LED. Typed as void * so leds.h + * does not depend on led-dynamic-lighting.h. + */ + void *led_dynamic; + struct list_head node; /* LED Device list */ const char *default_trigger; /* Trigger to use */ From 83b749261077aa86e6c3fafdc36358957399cc9a Mon Sep 17 00:00:00 2001 From: Marco Scardovi Date: Fri, 18 Sep 2026 15:44:14 +0200 Subject: [PATCH 1348/1352] docs: leds: Document the Dynamic Lighting class ABI Document the Dynamic Lighting LED class ABI and user-facing sysfs interface. Describe the common attributes, visibility rules for optional controls, and how a vendor driver can attach the class to an existing LED. Effect names are defined by the driver and discovered through effect_index. enabled/enabled_index turn lighting off without changing the selected effect. Writing power_states replaces the active bitmask (an empty list clears all enabled states). Direction values are left, right, up and down. Also add the new document to the LED documentation index and register it in MAINTAINERS. Signed-off-by: Marco Scardovi --- .../ABI/testing/sysfs-class-leds-dynamic | 143 ++++++++++++ Documentation/leds/index.rst | 1 + Documentation/leds/leds-class-dynamic.rst | 205 ++++++++++++++++++ MAINTAINERS | 2 + 4 files changed, 351 insertions(+) create mode 100644 Documentation/ABI/testing/sysfs-class-leds-dynamic create mode 100644 Documentation/leds/leds-class-dynamic.rst diff --git a/Documentation/ABI/testing/sysfs-class-leds-dynamic b/Documentation/ABI/testing/sysfs-class-leds-dynamic new file mode 100644 index 00000000000000..b1cc6c0ec4e3b1 --- /dev/null +++ b/Documentation/ABI/testing/sysfs-class-leds-dynamic @@ -0,0 +1,143 @@ +What: /sys/class/leds//direction +Date: September 2026 +Contact: Marco Scardovi +Contact: Denis Benato +Description: read/write + Animation propagation direction. Outputs or accepts one of + ``left``, ``right``, ``up``, or ``down``. Only visible when the driver supports + directional animations. + +What: /sys/class/leds//direction_index +Date: September 2026 +Contact: Marco Scardovi +Contact: Denis Benato +Description: read + Space-separated list of supported animation directions. + Only visible when the driver supports directional animations. + +What: /sys/class/leds//direct_buffer +Date: September 2026 +Contact: Marco Scardovi +Contact: Denis Benato +Description: write-only (binary) + Raw packed RGB stream (3 bytes per LED: R, G, B in sequence). + Writing to this node transmits direct per-key or matrix frame + data bypassing hardware effect generators. The complete write + must total exactly (led_count * 3) bytes; kernfs may deliver + that payload in PAGE_SIZE chunks starting at offset 0. Only + visible when the driver implements direct RGB streaming. + +What: /sys/class/leds//effect +Date: September 2026 +Contact: Marco Scardovi +Contact: Denis Benato +Description: read/write + Currently selected hardware or driver-synthesized animation + effect. Reading outputs the effect name. Writing a name from + ``effect_index`` selects that effect. Names are defined by the + driver, not by a global enum; userspace must read + ``effect_index``. Any active trigger is automatically detached + upon switching effects. + +What: /sys/class/leds//effect_index +Date: September 2026 +Contact: Marco Scardovi +Contact: Denis Benato +Description: read + Space-separated list of animation effect names supported by + this LED. The vocabulary is defined by the driver. Lighting + off is not an effect; use ``enabled``. + +What: /sys/class/leds//effects_palette +Date: September 2026 +Contact: Marco Scardovi +Contact: Denis Benato +Description: read/write + Multi-color stacked palette used by multi-color animation + effects. Reading outputs space-separated 24-bit hex colors + (``#RRGGBB``). Writing accepts a space-separated sequence of + hex triplets. The number of entries must not exceed + ``max_palette_entries``. Color of a single-color LED remains + on the standard LED ``brightness`` / ``multi_intensity`` nodes. + Only visible when the driver supports programmable palettes. + +What: /sys/class/leds//enabled +Date: September 2026 +Contact: Marco Scardovi +Contact: Denis Benato +Description: read/write + Lighting engine on/off as ``true`` or ``false``. Independent + from ``effect``: disabling turns the lights off without + forgetting the selected effect. Only visible when the driver + implements a dedicated enable callback. + +What: /sys/class/leds//enabled_index +Date: September 2026 +Contact: Marco Scardovi +Contact: Denis Benato +Description: read + Space-separated list of values accepted by ``enabled`` + (``false true``). Only visible when the driver implements a + dedicated enable callback. + +What: /sys/class/leds//led_count +Date: September 2026 +Contact: Marco Scardovi +Contact: Denis Benato +Description: read + Total number of individual, addressable LEDs in this zone. + +What: /sys/class/leds//max_palette_entries +Date: September 2026 +Contact: Marco Scardovi +Contact: Denis Benato +Description: read + Maximum number of palette entries accepted by + ``effects_palette``. Only visible when the driver supports + programmable palettes. + +What: /sys/class/leds//power_states +Date: September 2026 +Contact: Marco Scardovi +Contact: Denis Benato +Description: read/write + Space-separated list of currently enabled persistence power + states. Writing a space-separated list of state names replaces + the active bitmask (an empty list clears all enabled states). + Only visible on devices supporting power state configuration. + +What: /sys/class/leds//power_states_index +Date: September 2026 +Contact: Marco Scardovi +Contact: Denis Benato +Description: read + Space-separated list of system power states supported for + lighting persistence (``boot``, ``awake``, ``sleep``, + ``shutdown``). Only visible on devices supporting power state + configuration. + +What: /sys/class/leds//speed +Date: September 2026 +Contact: Marco Scardovi +Contact: Denis Benato +Description: read/write + Current animation speed level (integer within ``speed_range``). + Only visible when the driver supports adjustable speed. + +What: /sys/class/leds//speed_range +Date: September 2026 +Contact: Marco Scardovi +Contact: Denis Benato +Description: read + Minimum and maximum animation speed level accepted by + ``speed``. Formatted as ``-``. Only visible when the + driver supports adjustable speed. + +What: /sys/class/leds//zone_type +Date: September 2026 +Contact: Marco Scardovi +Contact: Denis Benato +Description: read + Driver-defined string describing the physical topology of + this lighting zone. Example values include ``generic``, + ``keyboard``, ``keyboard_per_key``, ``logo``, and ``lightbar``. diff --git a/Documentation/leds/index.rst b/Documentation/leds/index.rst index 23fa9ff7aaf4b0..39d93f1f9842d1 100644 --- a/Documentation/leds/index.rst +++ b/Documentation/leds/index.rst @@ -10,6 +10,7 @@ LEDs leds-class leds-class-flash leds-class-multicolor + leds-class-dynamic ledtrig-oneshot ledtrig-transient ledtrig-usbport diff --git a/Documentation/leds/leds-class-dynamic.rst b/Documentation/leds/leds-class-dynamic.rst new file mode 100644 index 00000000000000..7ef2d23d9b9422 --- /dev/null +++ b/Documentation/leds/leds-class-dynamic.rst @@ -0,0 +1,205 @@ +.. SPDX-License-Identifier: GPL-2.0 + +====================================== +Dynamic Lighting LED class under Linux +====================================== + +Author: Marco Scardovi +Author: Denis Benato + +Description +=========== +The Dynamic Lighting LED class provides a standardized sysfs interface for +complex, addressable illumination hardware such as per-key RGB keyboard +matrices, 2D LED matrix displays, addressable segment strips, and chassis +lightbars. + +The class can either wrap a new LED class device or attach onto an LED that +the vendor driver already registered (including ``LED_MULTI_COLOR``). Color +stays on the standard ``brightness`` / ``multi_intensity`` nodes. Dynamic +Lighting adds optional effect, speed, enable, palette, and direct RGB +attributes without requiring hidraw or a rewrite of the existing LED +registration. + +Directory Layout Example +======================== +The following examples use ```` as a placeholder for a Dynamic Lighting +LED class device name. Optional attributes are omitted when the driver does +not implement the matching callback. + +.. code-block:: console + + # ls -l /sys/class/leds// + -rw-r--r-- 1 root root 4096 Sep 4 17:00 brightness + -r--r--r-- 1 root root 4096 Sep 4 17:00 max_brightness + -r--r--r-- 1 root root 4096 Sep 4 17:00 zone_type + -r--r--r-- 1 root root 4096 Sep 4 17:00 led_count + -r--r--r-- 1 root root 4096 Sep 4 17:00 effect_index + -rw-r--r-- 1 root root 4096 Sep 4 17:00 effect + -rw-r--r-- 1 root root 4096 Sep 4 17:00 enabled + -r--r--r-- 1 root root 4096 Sep 4 17:00 enabled_index + -r--r--r-- 1 root root 4096 Sep 4 17:00 speed_range + -rw-r--r-- 1 root root 4096 Sep 4 17:00 speed + -r--r--r-- 1 root root 4096 Sep 4 17:00 direction_index + -rw-r--r-- 1 root root 4096 Sep 4 17:00 direction + -r--r--r-- 1 root root 4096 Sep 4 17:00 max_palette_entries + -rw-r--r-- 1 root root 4096 Sep 4 17:00 effects_palette + -r--r--r-- 1 root root 4096 Sep 4 17:00 power_states_index + -rw-r--r-- 1 root root 4096 Sep 4 17:00 power_states + --w------- 1 root root 504 Sep 4 17:00 direct_buffer + +Attaching to an existing LED +============================ +Drivers that already expose a LED can publish Dynamic Lighting as extra +attributes on that same directory. Keep the existing color path, point +``host`` at the registered LED, and supply only the ops the hardware +already implements: + +.. code-block:: c + + static const struct led_dynamic_ops my_ops = { + .set_effect = my_set_effect, + .set_speed = my_set_speed, + .set_enabled = my_set_enabled, + }; + + ldev->host = existing_led_cdev; + ldev->ops = &my_ops; + ldev->effects = my_effects; + ldev->num_effects = ARRAY_SIZE(my_effects); + led_classdev_dynamic_register(dev, ldev); + +``led_dynamic_fill_effects()`` can compact an existing capability bitmask +into a name table. Effect names are defined by the driver; userspace must +read ``effect_index``. + +Sysfs Attributes +================ + +``zone_type`` (read-only) + Driver-defined string describing the physical topology of the zone. + Example values include ``generic``, ``keyboard``, ``keyboard_per_key``, + ``logo``, or ``lightbar``. + +``led_count`` (read-only) + Total number of individually addressable LEDs in this zone. + +``effect_index`` (read-only) + Space-separated list of animation effect names supported by this LED. + Names are defined by the driver. + +``effect`` (read/write) + Currently selected animation effect. Writing a name from ``effect_index`` + switches the mode. Any active trigger is automatically detached upon + effect change. + +``enabled`` (read/write) + Lighting engine on/off (``true`` / ``false``). Independent from + ``effect``. Only visible when the driver implements ``set_enabled``. + +``enabled_index`` (read-only) + Values accepted by ``enabled`` (``false true``). Only visible when the + driver implements ``set_enabled``. + +``speed_range`` (read-only) + Minimum and maximum effect animation speed accepted by ``speed``, + formatted as ``-``. Only visible when the hardware supports + adjustable speed. + +``speed`` (read/write) + Current effect animation speed (within ``speed_range``). Only visible when + the hardware supports adjustable speed. + +``direction_index`` (read-only) + Space-separated list of directions accepted by ``direction``. Only + visible when directional effects are supported. + +``direction`` (read/write) + Animation propagation direction: ``left``, ``right``, ``up``, or ``down``. + Only visible when directional effects are supported. + +``max_palette_entries`` (read-only) + Maximum number of palette entries accepted by ``effects_palette``. Only + visible when programmable palettes are supported. + +``effects_palette`` (read/write) + Space-separated list of 24-bit RGB hex colors (e.g. ``#ff0000 #00ff00``). + Up to ``max_palette_entries`` colors can be defined. Single-LED color + continues to use ``brightness`` / ``multi_intensity``. + +``power_states_index`` (read-only) + List of platform power states supported for illumination persistence + (``boot``, ``awake``, ``sleep``, ``shutdown``). + +``power_states`` (read/write) + Currently active persistence states. Writing a space-separated list of + state names replaces the active state bitmask. + +``direct_buffer`` (write-only, binary) + Raw binary sink for streaming per-key RGB frames. Each LED requires 3 bytes + in sequence (R, G, B). The complete write must total ``led_count * 3`` + bytes; kernfs may deliver that payload in ``PAGE_SIZE`` chunks starting at + offset 0. Enables efficient high-rate streaming for visualizers and canvas + sinks. + +Locking Hierarchy & Invariants +============================== +To prevent kernel deadlocks between LED triggers, sysfs handlers, and bus +transfers, the subsystem enforces the following lock order: + +1. Acquire outer mutex: ``mutex_lock(&cdev->led_access)``. +2. If the operation replaces trigger-driven output, disengage/remove the active + LED trigger via ``led_trigger_remove(cdev)``. +3. Acquire internal mutex: ``mutex_lock(&ldev->lock)``. +4. Validate inputs, update state, and dispatch driver callbacks. +5. Release internal mutex: ``mutex_unlock(&ldev->lock)``. +6. Release outer mutex: ``mutex_unlock(&cdev->led_access)``. + +Driver callbacks must not persist class-owned fields (``current_effect``, +``speed``, ``enabled``, ``palette``, ``active_power_states``) on failure; the core +writes those fields only after a successful callback. ``brightness_set_blocking`` +is not called with ``ldev->lock`` held and must take it if it mutates the +same state. + +When Dynamic Lighting is attached to an existing LED, ``cdev`` in the lock +order is that host LED (``led_dynamic_cdev()``), not the unused embedded +``ldev->cdev``. + +``direct_buffer`` writes that are not exactly ``led_count * 3`` bytes, and that +cannot be completed by later ``PAGE_SIZE`` chunks from offset 0, are rejected. +Empty writes are rejected. + +Examples +======== + +Selecting an effect and speed advertised by the device: +------------------------------------------------------- +.. code-block:: console + + # cat /sys/class/leds//effect_index + # echo > /sys/class/leds//effect + # echo 1 > /sys/class/leds//speed + +Turning lighting off without changing the selected effect: +---------------------------------------------------------- +.. code-block:: console + + # echo false > /sys/class/leds//enabled + +Configuring a custom 3-color palette: +------------------------------------- +.. code-block:: console + + # echo "#ff0000 #00ff00 #0000ff" > /sys/class/leds//effects_palette + +Enabling illumination during boot and awake states: +--------------------------------------------------- +.. code-block:: console + + # echo "boot awake" > /sys/class/leds//power_states + +Streaming a direct RGB frame (for a 168-LED device, 504 bytes): +---------------------------------------------------------------- +.. code-block:: console + + # dd if=/dev/urandom of=/sys/class/leds//direct_buffer bs=504 count=1 diff --git a/MAINTAINERS b/MAINTAINERS index 98cc15c4e334c1..dce0dc13d48cff 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -15009,6 +15009,8 @@ M: Marco Scardovi M: Denis Benato L: linux-leds@vger.kernel.org S: Maintained +F: Documentation/ABI/testing/sysfs-class-leds-dynamic +F: Documentation/leds/leds-class-dynamic.rst F: drivers/leds/led-class-dynamic-test.c F: drivers/leds/led-class-dynamic.c F: include/linux/led-dynamic-lighting.h From 19a09b3acfa6fe98ad3af0347b8976a211b22ee7 Mon Sep 17 00:00:00 2001 From: Marco Scardovi Date: Fri, 18 Sep 2026 15:44:15 +0200 Subject: [PATCH 1349/1352] HID: asus: Discover Aura/Slash reports and skip AniMe Matrix USB ID 0x193b is shared by standalone Slash MCUs and AniMe Matrix panels. Bind it only when the interface exposes Aura/Slash LED reports (0x5d/0x5e) or a sibling HID LampArray lighting interface. Detect LampArray by report IDs on usage page 0x59 and start that interface without hidraw so lighting is not exported to userspace. Firmware animations stay on the Aura 0x5d interface; Aura 0xBC remains the fallback when LampArray is absent. Signed-off-by: Marco Scardovi --- drivers/hid/hid-asus.c | 90 ++++++++++++++++++++++++++++++++++++++++++ drivers/hid/hid-ids.h | 2 + 2 files changed, 92 insertions(+) diff --git a/drivers/hid/hid-asus.c b/drivers/hid/hid-asus.c index 4112c0fc7aaf97..9097c1cf19b55d 100644 --- a/drivers/hid/hid-asus.c +++ b/drivers/hid/hid-asus.c @@ -37,6 +37,8 @@ MODULE_AUTHOR("Yusuke Fujimaki "); MODULE_AUTHOR("Brendan McGrath "); MODULE_AUTHOR("Victor Vlasenko "); MODULE_AUTHOR("Frederik Wenigwieser "); +MODULE_AUTHOR("Marco Scardovi "); +MODULE_AUTHOR("Denis Benato "); MODULE_DESCRIPTION("Asus HID Keyboard and TouchPad"); #define T100_TPAD_INTF 2 @@ -51,6 +53,31 @@ MODULE_DESCRIPTION("Asus HID Keyboard and TouchPad"); #define FEATURE_KBD_LED_REPORT_ID1 0x5d #define FEATURE_KBD_LED_REPORT_ID2 0x5e +/* + * Microsoft HID Lighting Illumination / LampArray (Usage Page 0x59). + * + * Linux Dynamic Lighting (led-class-dynamic / aura:*) is the userspace ABI. + * On some Strix N-KEY devices (e.g. G614PR) Aura feature 0xBC cannot address + * the chassis lightbar independently; the sibling LampArray interface is the + * correct direct-RGB backend (same path Windows DL / G-Helper LampArray use). + * When LampArray is absent, callers fall back to Aura 0xBC. Firmware effects + * (0xb3) remain on the Aura report ID 0x5d interface. + * + * The LampArray USB interface is bound without hidraw so lighting stays + * exclusively under this driver. + */ +#define ASUS_LAMPARRAY_MAX_LAMPS 64 +#define ASUS_LAMPARRAY_MULTI_MAX 8 +#define ASUS_LAMPARRAY_PURPOSE_CONTROL 0x01 +#define ASUS_LAMPARRAY_FLAG_COMPLETE 0x01 +#define ASUS_LAMPARRAY_RID_ATTR 0x01 +#define ASUS_LAMPARRAY_RID_REQUEST 0x02 +#define ASUS_LAMPARRAY_RID_RESPONSE 0x03 +#define ASUS_LAMPARRAY_RID_MULTI 0x04 +#define ASUS_LAMPARRAY_RID_CONTROL 0x06 +#define AURA_ZONE_ACTIVATE_LAMPARRAY 0x03 +#define AURA_ZONE_RELEASE_LAMPARRAY 0x04 + #define ROG_ALLY_REPORT_SIZE 64 #define ROG_ALLY_X_MIN_MCU 313 #define ROG_ALLY_MIN_MCU 319 @@ -939,6 +966,34 @@ static bool asus_has_report_id(struct hid_device *hdev, u16 report_id) return false; } + +/* + * Sibling USB interface with HID Lighting (LampArray) only — no Aura 0x5d. + * Bound by this driver without hidraw so lighting stays in-kernel DL. + */ +static int asus_lamparray_detect_base(struct hid_device *hdev) +{ + u8 base; + + if (asus_has_report_id(hdev, FEATURE_KBD_LED_REPORT_ID1) || + asus_has_report_id(hdev, FEATURE_KBD_LED_REPORT_ID2)) + return -ENODEV; + + for (base = 0; base <= 0x40; base += 0x40) { + if (asus_has_report_id(hdev, base + ASUS_LAMPARRAY_RID_ATTR) && + asus_has_report_id(hdev, base + ASUS_LAMPARRAY_RID_MULTI) && + asus_has_report_id(hdev, base + ASUS_LAMPARRAY_RID_CONTROL)) + return base; + } + + return -ENODEV; +} + +static bool asus_is_lamparray_interface(struct hid_device *hdev) +{ + return asus_lamparray_detect_base(hdev) >= 0; +} + static int asus_kbd_register_leds(struct hid_device *hdev) { struct asus_drvdata *drvdata = hid_get_drvdata(hdev); @@ -1474,6 +1529,18 @@ static int asus_probe(struct hid_device *hdev, const struct hid_device_id *id) return ret; } + /* + * USB 0x193b is reused by AniMe Matrix. Bind only LampArray or + * interfaces that expose Aura/Slash LED reports. + */ + if (hdev->product == USB_DEVICE_ID_ASUSTEK_ROG_SLASH && + !asus_is_lamparray_interface(hdev) && + !asus_has_report_id(hdev, FEATURE_KBD_LED_REPORT_ID1) && + !asus_has_report_id(hdev, FEATURE_KBD_LED_REPORT_ID2)) { + hid_dbg(hdev, "Skipping 0x193b without Aura/Slash LED reports\n"); + return -ENODEV; + } + /* Check for vendor for RGB init and handle generic devices properly. */ rep_enum = &hdev->report_enum[HID_INPUT_REPORT]; list_for_each_entry(rep, &rep_enum->report_list, list) { @@ -1489,6 +1556,21 @@ static int asus_probe(struct hid_device *hdev, const struct hid_device_id *id) if (is_vendor && (drvdata->quirks & QUIRK_ROG_NKEY_KEYBOARD)) hdev->quirks |= HID_QUIRK_HIDINPUT_FORCE; + /* + * LampArray is a Dynamic Lighting backend owned in-kernel: do not + * export hidraw for that interface. Aura 0xBC remains the fallback + * when LampArray is absent. + */ + if (asus_is_lamparray_interface(hdev)) { + ret = hid_hw_start(hdev, 0); + if (ret) { + hid_err(hdev, "Asus LampArray hw start failed: %d\n", ret); + return ret; + } + hid_info(hdev, "Bound ASUS LampArray interface (no hidraw)\n"); + return 0; + } + ret = asus_worker_create(hdev, drvdata); if (ret) { hid_warn(hdev, "Failed to initialize worker: %d\n", ret); @@ -1562,6 +1644,11 @@ static void asus_remove(struct hid_device *hdev) if (drvdata->listener.brightness_set) asus_hid_unregister_listener(&drvdata->listener); + if (asus_is_lamparray_interface(hdev)) { + hid_hw_stop(hdev); + return; + } + asus_worker_stop(drvdata->worker); hid_hw_stop(hdev); } @@ -1698,6 +1785,9 @@ static const struct hid_device_id asus_devices[] = { { HID_USB_DEVICE(USB_VENDOR_ID_ASUSTEK, USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD), QUIRK_USE_KBD_BACKLIGHT | QUIRK_ROG_NKEY_KEYBOARD }, + { HID_USB_DEVICE(USB_VENDOR_ID_ASUSTEK, + USB_DEVICE_ID_ASUSTEK_ROG_SLASH), + QUIRK_USE_KBD_BACKLIGHT | QUIRK_ROG_NKEY_KEYBOARD | QUIRK_HID_FN_LOCK }, { HID_USB_DEVICE(USB_VENDOR_ID_ASUSTEK, USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD2), QUIRK_USE_KBD_BACKLIGHT | QUIRK_ROG_NKEY_KEYBOARD | QUIRK_HID_FN_LOCK }, diff --git a/drivers/hid/hid-ids.h b/drivers/hid/hid-ids.h index c9b2c7782d9a1d..5972cbf310a38c 100644 --- a/drivers/hid/hid-ids.h +++ b/drivers/hid/hid-ids.h @@ -226,6 +226,8 @@ #define USB_DEVICE_ID_ASUSTEK_ROG_KEYBOARD2 0x1837 #define USB_DEVICE_ID_ASUSTEK_ROG_KEYBOARD3 0x1822 #define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD 0x1866 +/* 0x193b is also AniMe Matrix; hid-asus binds it only when Aura/Slash LED reports exist. */ +#define USB_DEVICE_ID_ASUSTEK_ROG_SLASH 0x193b #define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD2 0x19b6 #define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD3 0x1ce6 #define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD3_BT 0x1ce7 From ffbccba2e5e0ec6247701edcab959946f20660aa Mon Sep 17 00:00:00 2001 From: Marco Scardovi Date: Fri, 18 Sep 2026 15:44:15 +0200 Subject: [PATCH 1350/1352] HID: asus: Add Aura protocol and Dynamic Lighting nodes Add Dynamic Lighting class support to hid-asus for Aura-capable ROG keyboards and chassis lightbars. Discover Aura layout, lightbar, and per-key/direct RGB from HID feature reports rather than DMI board lists. Register aura:keyboard and, when the chassis lightbar is present, aura:lightbar. Each node keeps its own effect and palette. Publish the firmware effect list from the 0x9e capability mask (static, breathe, rainbow_cycle, rainbow_wave, star, rain, highlight, laser, ripple, pulse, comet, flash) and advertise direct when the keyboard path supports packed RGB. Lighting off uses enabled rather than a dedicated off effect. Commit firmware effects, including static color, with Aura 0xb3/0xb5/0xb4 so the controller keeps them across reboot. Direct frames use Aura 0xBC. Before the keyboard handshake, read the feature report: adopt a programmed effect, apply static red only when the report is blank, and leave the controller alone when the payload cannot be decoded. Resume reapplies only an effect this driver has adopted or written. Map boot/awake/sleep/shutdown via power_states to AURA_CMD_POWER (0xbd). Keep asus::kbd_backlight brightness behaviour unchanged. Signed-off-by: Marco Scardovi --- .../testing/sysfs-class-led-driver-hid-asus | 13 + MAINTAINERS | 9 + drivers/hid/Kconfig | 1 + drivers/hid/hid-asus.c | 1818 +++++++++++++++-- 4 files changed, 1644 insertions(+), 197 deletions(-) create mode 100644 Documentation/ABI/testing/sysfs-class-led-driver-hid-asus diff --git a/Documentation/ABI/testing/sysfs-class-led-driver-hid-asus b/Documentation/ABI/testing/sysfs-class-led-driver-hid-asus new file mode 100644 index 00000000000000..5c32302aebfeb7 --- /dev/null +++ b/Documentation/ABI/testing/sysfs-class-led-driver-hid-asus @@ -0,0 +1,13 @@ +What: /sys/class/leds/aura:keyboard/ +What: /sys/class/leds/aura:lightbar/ +Date: September 2026 +Contact: Marco Scardovi +Contact: Denis Benato +Description: ASUS Aura keyboard and chassis lightbar Dynamic Lighting nodes. + Each node accepts its own effect, palette, and direct RGB + writes. Firmware effects, including static color, are + committed with Aura 0xb3/0xb5/0xb4 so the controller keeps + them across reboot. Direct RGB may be backed by HID LampArray + (Usage Page 0x59) when Aura 0xBC cannot address the lightbar + independently. The lightbar node is absent when the device + has no chassis lightbar. diff --git a/MAINTAINERS b/MAINTAINERS index dce0dc13d48cff..d76f13cbe72607 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -11765,6 +11765,15 @@ F: include/linux/pm.h F: include/linux/suspend.h F: kernel/power/ +HID ASUS DRIVERS +M: Marco Scardovi +M: Denis Benato +L: linux-input@vger.kernel.org +S: Maintained +W: https://asus-linux.org/ +F: Documentation/ABI/testing/sysfs-class-led-driver-hid-asus +F: drivers/hid/hid-asus.c + HID CORE LAYER M: Jiri Kosina M: Benjamin Tissoires diff --git a/drivers/hid/Kconfig b/drivers/hid/Kconfig index 0e3a0ccd6901e8..88e9a152f8440f 100644 --- a/drivers/hid/Kconfig +++ b/drivers/hid/Kconfig @@ -189,6 +189,7 @@ config HID_ASUS depends on USB_HID depends on LEDS_CLASS depends on ASUS_WMI || ASUS_WMI=n + imply LEDS_CLASS_DYNAMIC select POWER_SUPPLY help Support for Asus notebook built-in keyboard and touchpad via i2c, and diff --git a/drivers/hid/hid-asus.c b/drivers/hid/hid-asus.c index 9097c1cf19b55d..e3ea58f4f2d779 100644 --- a/drivers/hid/hid-asus.c +++ b/drivers/hid/hid-asus.c @@ -30,6 +30,10 @@ #include /* For to_usb_interface for T100 touchpad intf check */ #include #include +#include +#include +#include +#include #include "hid-ids.h" @@ -53,6 +57,113 @@ MODULE_DESCRIPTION("Asus HID Keyboard and TouchPad"); #define FEATURE_KBD_LED_REPORT_ID1 0x5d #define FEATURE_KBD_LED_REPORT_ID2 0x5e +#define AURA_FEATURE_REPORT_SIZE 64 +#define ASUS_AURA_MAX_EFFECTS 13 + +#define AURA_CMD_PROBE 0x05 +#define AURA_CMD_RUN_MODE 0x9e +#define AURA_CMD_SET_EFFECT 0xb3 +#define AURA_CMD_COMMIT 0xb4 +#define AURA_CMD_SET 0xb5 +#define AURA_CMD_ZONE_ENABLE 0xc0 +#define AURA_CMD_DIRECT 0xbc +#define AURA_CMD_POWER 0xbd + +#define AURA_ZONE_ACTIVATE_KEYBOARD 0x00 +#define AURA_ZONE_ACTIVATE_LIGHTBAR 0x01 + +#define AURA_POWER_CMD_ENABLE 0x01 + +/* Power Byte 0: Logo (even bits) & Keyboard (odd bits) */ +#define AURA_POWER_LOGO_BOOT BIT(0) +#define AURA_POWER_KBD_BOOT BIT(1) +#define AURA_POWER_LOGO_AWAKE BIT(2) +#define AURA_POWER_KBD_AWAKE BIT(3) +#define AURA_POWER_LOGO_SLEEP BIT(4) +#define AURA_POWER_KBD_SLEEP BIT(5) +#define AURA_POWER_LOGO_SHUTDOWN BIT(6) +#define AURA_POWER_KBD_SHUTDOWN BIT(7) +#define AURA_POWER_MASK_LOGO_ALL (AURA_POWER_LOGO_BOOT | \ + AURA_POWER_LOGO_AWAKE | \ + AURA_POWER_LOGO_SLEEP | \ + AURA_POWER_LOGO_SHUTDOWN) +#define AURA_POWER_MASK_KBD_LOGO_ALL 0xff + +/* Power Byte 1: Chassis Lightbar */ +#define AURA_POWER_LB_AUX BIT(0) +#define AURA_POWER_LB_BOOT BIT(1) +#define AURA_POWER_LB_AWAKE BIT(2) +#define AURA_POWER_LB_SLEEP BIT(3) +#define AURA_POWER_LB_SHUTDOWN BIT(4) +#define AURA_POWER_MASK_LIGHTBAR_ALL 0x1f + +/* Power Byte 2: Lid Display Bezel / Lid segments */ +#define AURA_POWER_LID_BOOT BIT(0) +#define AURA_POWER_LID_AWAKE BIT(1) +#define AURA_POWER_LID_SLEEP BIT(2) +#define AURA_POWER_LID_SHUTDOWN BIT(3) +#define AURA_POWER_LID_PERSISTENCE 0xd0 +#define AURA_POWER_MASK_LID_ALL (AURA_POWER_LID_PERSISTENCE | 0x0f) + +/* Power Byte 3: Rear Glow */ +#define AURA_POWER_REAR_BOOT BIT(0) +#define AURA_POWER_REAR_AWAKE BIT(1) +#define AURA_POWER_REAR_SLEEP BIT(2) +#define AURA_POWER_REAR_SHUTDOWN BIT(3) +#define AURA_POWER_MASK_REAR_ALL 0x0f + +#define AURA_ZONE_ALL 0x00 +#define AURA_ZONE_KEY1 0x01 +#define AURA_ZONE_KEY2 0x02 +#define AURA_ZONE_KEY3 0x03 +#define AURA_ZONE_KEY4 0x04 +#define AURA_ZONE_LOGO 0x05 +#define AURA_ZONE_BAR_LEFT 0x06 +#define AURA_ZONE_BAR_RIGHT 0x07 +#define AURA_ZONE_KEYBOARD_CHANNEL 0x01 +#define AURA_ZONE_LIGHTBAR_CHANNEL 0x04 + +#define AURA_EFFECT_STATIC 0x00 +#define AURA_EFFECT_BREATHING 0x01 +#define AURA_EFFECT_SPECTRUM_CYCLE 0x02 +#define AURA_EFFECT_RAINBOW 0x03 +#define AURA_EFFECT_STARS 0x04 +#define AURA_EFFECT_RAIN 0x05 +#define AURA_EFFECT_REACTIVE 0x06 +#define AURA_EFFECT_LASER 0x07 +#define AURA_EFFECT_RIPPLE 0x08 +#define AURA_EFFECT_PULSE 0x0a +#define AURA_EFFECT_COMET 0x0b +#define AURA_EFFECT_FLASH 0x0c + +#define AURA_SPEED_SLOW 0xe1 +#define AURA_SPEED_MED 0xeb +#define AURA_SPEED_FAST 0xf5 + +#define ROG_STRIX_LEDS_PER_PKT 16 +#define ROG_STRIX_PERKEY_FULL_PKTS 10 +#define ROG_STRIX_PERKEY_FINAL_PKT_LEDS 8 +#define ROG_STRIX_PERKEY_PACKETS (ROG_STRIX_PERKEY_FULL_PKTS + 1) +#define ROG_STRIX_DIRECT_LEDS \ + ((ROG_STRIX_PERKEY_FULL_PKTS * ROG_STRIX_LEDS_PER_PKT) + \ + ROG_STRIX_PERKEY_FINAL_PKT_LEDS) +#define ROG_STRIX_DIRECT_BUF_SIZE (ROG_STRIX_DIRECT_LEDS * 3) +#define ROG_STRIX_LIGHTBAR_LEDS 12 +#define ROG_STRIX_LIGHTBAR_BUF_SIZE (ROG_STRIX_LIGHTBAR_LEDS * 3) +#define ROG_STRIX_4ZONE_KBD_LEDS 4 +#define ROG_STRIX_4ZONE_KBD_BUF_SIZE (ROG_STRIX_4ZONE_KBD_LEDS * 3) +#define ROG_STRIX_4ZONE_LIGHTBAR_LEDS 6 +#define ROG_STRIX_4ZONE_LIGHTBAR_BUF_SIZE \ + (ROG_STRIX_4ZONE_LIGHTBAR_LEDS * 3) +#define ROG_STRIX_4ZONE_DIRECT_KBD_OFFSET 9 +#define ROG_STRIX_4ZONE_DIRECT_LB_OFFSET 27 +#define ROG_STRIX_PERKEY_DIRECT_PAYLOAD_OFFSET 9 + +#define AURA_DIRECT_FRAME_PERKEY 0x00 +#define AURA_DIRECT_FRAME_ZONED 0x01 +#define AURA_DIRECT_ROUTING_DEFAULT 0x01 +#define AURA_DIRECT_CHUNK_FLAG 0x01 + /* * Microsoft HID Lighting Illumination / LampArray (Usage Page 0x59). * @@ -78,10 +189,6 @@ MODULE_DESCRIPTION("Asus HID Keyboard and TouchPad"); #define AURA_ZONE_ACTIVATE_LAMPARRAY 0x03 #define AURA_ZONE_RELEASE_LAMPARRAY 0x04 -#define ROG_ALLY_REPORT_SIZE 64 -#define ROG_ALLY_X_MIN_MCU 313 -#define ROG_ALLY_MIN_MCU 319 - /* Spurious HID codes sent by QUIRK_ROG_NKEY_KEYBOARD devices */ #define ASUS_SPURIOUS_CODE_0XEA 0xea #define ASUS_SPURIOUS_CODE_0XEC 0xec @@ -197,6 +304,34 @@ struct asus_drvdata { unsigned long battery_next_query; struct asus_hid_listener listener; bool fn_lock; +#if IS_REACHABLE(CONFIG_LEDS_CLASS_DYNAMIC) + struct mutex aura_lock; /* Serializes Aura HID reports and buffers */ + u8 *aura_buf; + struct led_classdev_dynamic dldev_kbd; + struct led_classdev_dynamic dldev_lightbar; + const char *effects_kbd[ASUS_AURA_MAX_EFFECTS]; + const char *effects_lightbar[ASUS_AURA_MAX_EFFECTS]; + bool has_dldev_kbd; + bool has_dldev_lightbar; + bool kbd_effect_set; + bool lightbar_effect_set; + bool boot_effect_valid; + bool boot_effect_blank; + u8 boot_zone; + u8 boot_mode; + u8 boot_r; + u8 boot_g; + u8 boot_b; + u8 boot_r2; + u8 boot_g2; + u8 boot_b2; + u8 boot_speed; + u8 boot_direction; + bool is_strix_4zone; + bool has_lightbar; + u8 kbd_direct_buf[ROG_STRIX_4ZONE_KBD_BUF_SIZE]; + u8 lb_direct_buf[ROG_STRIX_LIGHTBAR_BUF_SIZE]; +#endif }; static int asus_report_battery(struct asus_drvdata *, u8 *, int); @@ -604,15 +739,37 @@ static int asus_raw_event(struct hid_device *hdev, static int asus_kbd_set_report(struct hid_device *hdev, const u8 *buf, size_t buf_size) { u8 *dmabuf __free(kfree) = kmemdup(buf, buf_size, GFP_KERNEL); + int ret; + if (!dmabuf) return -ENOMEM; + /* + * For Aura/Slash LED reports (0x5d/0x5e), match asus_aura_set_feature: + * interrupt OUT, then raw OUTPUT, then raw FEATURE. + */ + if (buf[0] == FEATURE_KBD_LED_REPORT_ID1 || + buf[0] == FEATURE_KBD_LED_REPORT_ID2) { + ret = hid_hw_output_report(hdev, dmabuf, buf_size); + if (ret >= 0) + return 0; + + ret = hid_hw_raw_request(hdev, buf[0], dmabuf, buf_size, + HID_OUTPUT_REPORT, HID_REQ_SET_REPORT); + if (ret >= 0) + return 0; + } + /* * The report ID should be set from the incoming buffer due to LED and key * interfaces having different pages */ - return hid_hw_raw_request(hdev, buf[0], dmabuf, buf_size, HID_FEATURE_REPORT, - HID_REQ_SET_REPORT); + ret = hid_hw_raw_request(hdev, buf[0], dmabuf, buf_size, HID_FEATURE_REPORT, + HID_REQ_SET_REPORT); + if (ret < 0) + return ret; + + return 0; } static int asus_kbd_init(struct hid_device *hdev, u8 report_id) @@ -895,7 +1052,7 @@ static int mcu_parse_version_string(const u8 *response, size_t response_size) static int mcu_request_version(struct hid_device *hdev) { - u8 *response __free(kfree) = kzalloc(ROG_ALLY_REPORT_SIZE, GFP_KERNEL); + u8 *response __free(kfree) = kzalloc(FEATURE_KBD_REPORT_SIZE, GFP_KERNEL); const u8 request[] = { 0x5a, 0x05, 0x03, 0x31, 0x00, 0x20 }; int ret; @@ -907,39 +1064,53 @@ static int mcu_request_version(struct hid_device *hdev) return ret; ret = hid_hw_raw_request(hdev, FEATURE_REPORT_ID, response, - ROG_ALLY_REPORT_SIZE, HID_FEATURE_REPORT, + FEATURE_KBD_REPORT_SIZE, HID_FEATURE_REPORT, HID_REQ_GET_REPORT); if (ret < 0) return ret; - ret = mcu_parse_version_string(response, ROG_ALLY_REPORT_SIZE); + ret = mcu_parse_version_string(response, FEATURE_KBD_REPORT_SIZE); if (ret < 0) { pr_err("Failed to parse MCU version: %d\n", ret); print_hex_dump(KERN_ERR, "MCU: ", DUMP_PREFIX_NONE, - 16, 1, response, ROG_ALLY_REPORT_SIZE, false); + 16, 1, response, FEATURE_KBD_REPORT_SIZE, false); } return ret; } +/* Minimum MCU FW versions that no longer need the WMI suspend quirk. */ +static const struct { + u16 product; + int min_version; +} asus_mcu_min_versions[] = { + { USB_DEVICE_ID_ASUSTEK_ROG_NKEY_ALLY, 319 }, + { USB_DEVICE_ID_ASUSTEK_ROG_NKEY_ALLY_X, 313 }, +}; + +static int asus_mcu_min_version(u16 id_product) +{ + int i; + + for (i = 0; i < ARRAY_SIZE(asus_mcu_min_versions); i++) { + if (asus_mcu_min_versions[i].product == id_product) + return asus_mcu_min_versions[i].min_version; + } + + return 0; +} + static void validate_mcu_fw_version(struct hid_device *hdev, int idProduct) { - int min_version, version; + int min_version = asus_mcu_min_version(idProduct); + int version; version = mcu_request_version(hdev); if (version < 0) return; - switch (idProduct) { - case USB_DEVICE_ID_ASUSTEK_ROG_NKEY_ALLY: - min_version = ROG_ALLY_MIN_MCU; - break; - case USB_DEVICE_ID_ASUSTEK_ROG_NKEY_ALLY_X: - min_version = ROG_ALLY_X_MIN_MCU; - break; - default: - min_version = 0; - } + if (!min_version) + return; if (version < min_version) { hid_warn(hdev, @@ -966,7 +1137,6 @@ static bool asus_has_report_id(struct hid_device *hdev, u16 report_id) return false; } - /* * Sibling USB interface with HID Lighting (LampArray) only — no Aura 0x5d. * Bound by this driver without hidraw so lighting stays in-kernel DL. @@ -993,7 +1163,6 @@ static bool asus_is_lamparray_interface(struct hid_device *hdev) { return asus_lamparray_detect_base(hdev) >= 0; } - static int asus_kbd_register_leds(struct hid_device *hdev) { struct asus_drvdata *drvdata = hid_get_drvdata(hdev); @@ -1034,231 +1203,1451 @@ static int asus_kbd_register_leds(struct hid_device *hdev) return ret; } -/* - * [0] REPORT_ID (same value defined in report descriptor) - * [1] rest battery level. range [0..255] - * [2]..[7] Bluetooth hardware address (MAC address) - * [8] charging status - * = 0 : AC offline / discharging - * = 1 : AC online / charging - * = 2 : AC online / fully charged - */ -static int asus_parse_battery(struct asus_drvdata *drvdata, u8 *data, int size) +#if IS_REACHABLE(CONFIG_LEDS_CLASS_DYNAMIC) + +static int asus_aura_set_feature_unlocked(struct asus_drvdata *drvdata, + const u8 *buf, size_t buf_size) { - u8 sts; - u8 lvl; - int val; + int ret; - lvl = data[1]; - sts = data[8]; + if (buf_size > AURA_FEATURE_REPORT_SIZE) + return -EINVAL; - drvdata->battery_capacity = ((int)lvl * 100) / (int)BATTERY_LEVEL_MAX; + memcpy(drvdata->aura_buf, buf, buf_size); + if (buf_size < AURA_FEATURE_REPORT_SIZE) + memset(drvdata->aura_buf + buf_size, 0, + AURA_FEATURE_REPORT_SIZE - buf_size); - switch (sts) { - case BATTERY_STAT_CHARGING: - val = POWER_SUPPLY_STATUS_CHARGING; - break; - case BATTERY_STAT_FULL: - val = POWER_SUPPLY_STATUS_FULL; - break; - case BATTERY_STAT_DISCONNECT: - default: - val = POWER_SUPPLY_STATUS_DISCHARGING; - break; - } - drvdata->battery_stat = val; + /* + * Try Output Report first matching Armoury Crate / asus_kbd_set_report. + * If the device lacks an interrupt OUT endpoint, fall back to + * hid_hw_raw_request() with HID_OUTPUT_REPORT, and finally to + * HID_FEATURE_REPORT. + */ + ret = hid_hw_output_report(drvdata->hdev, drvdata->aura_buf, + AURA_FEATURE_REPORT_SIZE); + if (ret >= 0) + return 0; + + ret = hid_hw_raw_request(drvdata->hdev, drvdata->aura_buf[0], + drvdata->aura_buf, AURA_FEATURE_REPORT_SIZE, + HID_OUTPUT_REPORT, HID_REQ_SET_REPORT); + if (ret >= 0) + return 0; + + ret = hid_hw_raw_request(drvdata->hdev, drvdata->aura_buf[0], + drvdata->aura_buf, AURA_FEATURE_REPORT_SIZE, + HID_FEATURE_REPORT, HID_REQ_SET_REPORT); + if (ret < 0) + return ret; return 0; } -static int asus_report_battery(struct asus_drvdata *drvdata, u8 *data, int size) +static int asus_aura_set_feature(struct asus_drvdata *drvdata, + const u8 *buf, size_t buf_size) { - /* notify only the autonomous event by device */ - if ((drvdata->battery_in_query == false) && - (size == BATTERY_REPORT_SIZE)) - power_supply_changed(drvdata->battery); + guard(mutex)(&drvdata->aura_lock); - return 0; + return asus_aura_set_feature_unlocked(drvdata, buf, buf_size); } -static int asus_battery_query(struct asus_drvdata *drvdata) +static int asus_aura_get_feature(struct asus_drvdata *drvdata, + u8 *buf, size_t buf_size) { - u8 *buf; - int ret = 0; + int ret; - buf = kmalloc(BATTERY_REPORT_SIZE, GFP_KERNEL); - if (!buf) - return -ENOMEM; + if (buf_size > AURA_FEATURE_REPORT_SIZE) + return -EINVAL; - drvdata->battery_in_query = true; - ret = hid_hw_raw_request(drvdata->hdev, BATTERY_REPORT_ID, - buf, BATTERY_REPORT_SIZE, - HID_INPUT_REPORT, HID_REQ_GET_REPORT); - drvdata->battery_in_query = false; - if (ret == BATTERY_REPORT_SIZE) - ret = asus_parse_battery(drvdata, buf, BATTERY_REPORT_SIZE); - else - ret = -ENODATA; + guard(mutex)(&drvdata->aura_lock); - kfree(buf); + memset(drvdata->aura_buf, 0, AURA_FEATURE_REPORT_SIZE); + drvdata->aura_buf[0] = buf[0]; + + ret = hid_hw_raw_request(drvdata->hdev, buf[0], drvdata->aura_buf, + AURA_FEATURE_REPORT_SIZE, + HID_FEATURE_REPORT, HID_REQ_GET_REPORT); + if (ret < 0) + return ret; + memcpy(buf, drvdata->aura_buf, min_t(size_t, buf_size, ret)); return ret; } -static enum power_supply_property asus_battery_props[] = { - POWER_SUPPLY_PROP_STATUS, - POWER_SUPPLY_PROP_PRESENT, - POWER_SUPPLY_PROP_CAPACITY, - POWER_SUPPLY_PROP_SCOPE, - POWER_SUPPLY_PROP_MODEL_NAME, -}; - -#define QUERY_MIN_INTERVAL (60 * HZ) /* 60[sec] */ - -static int asus_battery_get_property(struct power_supply *psy, - enum power_supply_property psp, - union power_supply_propval *val) +static int asus_aura_commit(struct asus_drvdata *drvdata) { - struct asus_drvdata *drvdata = power_supply_get_drvdata(psy); - int ret = 0; + u8 buf_set[AURA_FEATURE_REPORT_SIZE] = { + FEATURE_KBD_LED_REPORT_ID1, + AURA_CMD_SET, + }; + u8 buf_apply[AURA_FEATURE_REPORT_SIZE] = { + FEATURE_KBD_LED_REPORT_ID1, + AURA_CMD_COMMIT, + }; + int ret; - switch (psp) { - case POWER_SUPPLY_PROP_STATUS: - case POWER_SUPPLY_PROP_CAPACITY: - if (time_before(drvdata->battery_next_query, jiffies)) { - drvdata->battery_next_query = - jiffies + QUERY_MIN_INTERVAL; - ret = asus_battery_query(drvdata); - if (ret) - return ret; - } - if (psp == POWER_SUPPLY_PROP_STATUS) - val->intval = drvdata->battery_stat; - else - val->intval = drvdata->battery_capacity; - break; - case POWER_SUPPLY_PROP_PRESENT: - val->intval = 1; - break; - case POWER_SUPPLY_PROP_SCOPE: - val->intval = POWER_SUPPLY_SCOPE_DEVICE; - break; - case POWER_SUPPLY_PROP_MODEL_NAME: - val->strval = drvdata->hdev->name; - break; - default: - ret = -EINVAL; - break; - } + /* + * Apply staged 0xb3 effect programming: + * First send 0xb5 (AURA_CMD_SET) to latch parameters, + * then send 0xb4 (AURA_CMD_COMMIT) to apply them to hardware, + * then send 0xb5 (AURA_CMD_SET) to settle as captured in firmware traces. + */ + ret = asus_aura_set_feature(drvdata, buf_set, sizeof(buf_set)); + if (ret < 0) + return ret; - return ret; + ret = asus_aura_set_feature(drvdata, buf_apply, sizeof(buf_apply)); + if (ret < 0) + return ret; + + return asus_aura_set_feature(drvdata, buf_set, sizeof(buf_set)); } -static int asus_battery_probe(struct hid_device *hdev) +static int asus_aura_query_run_mode(struct asus_drvdata *drvdata, u8 selector, + u8 *buf, size_t buf_size) { - struct asus_drvdata *drvdata = hid_get_drvdata(hdev); - struct power_supply_config pscfg = { .drv_data = drvdata }; - int ret = 0; - - drvdata->battery_capacity = 0; - drvdata->battery_stat = POWER_SUPPLY_STATUS_UNKNOWN; - drvdata->battery_in_query = false; + u8 req[AURA_FEATURE_REPORT_SIZE] = { + FEATURE_KBD_LED_REPORT_ID1, + AURA_CMD_RUN_MODE, + 0x01, + selector, + }; + int ret; - drvdata->battery_desc.properties = asus_battery_props; - drvdata->battery_desc.num_properties = ARRAY_SIZE(asus_battery_props); - drvdata->battery_desc.get_property = asus_battery_get_property; - drvdata->battery_desc.type = POWER_SUPPLY_TYPE_BATTERY; - drvdata->battery_desc.use_for_apm = 0; - drvdata->battery_desc.name = devm_kasprintf(&hdev->dev, GFP_KERNEL, - "asus-keyboard-%s-battery", - strlen(hdev->uniq) ? - hdev->uniq : dev_name(&hdev->dev)); - if (!drvdata->battery_desc.name) - return -ENOMEM; + ret = asus_aura_set_feature(drvdata, req, sizeof(req)); + if (ret < 0) + return ret; - drvdata->battery_next_query = jiffies; + memset(buf, 0, buf_size); + buf[0] = FEATURE_KBD_LED_REPORT_ID1; - drvdata->battery = devm_power_supply_register(&hdev->dev, - &(drvdata->battery_desc), &pscfg); - if (IS_ERR(drvdata->battery)) { - ret = PTR_ERR(drvdata->battery); - drvdata->battery = NULL; - hid_err(hdev, "Unable to register battery device\n"); + ret = asus_aura_get_feature(drvdata, buf, buf_size); + if (ret < 0) return ret; - } - power_supply_powers(drvdata->battery, &hdev->dev); + if (ret < 5 || buf[1] != AURA_CMD_RUN_MODE || buf[2] != 0x01 || + buf[3] != selector || buf[4] != 0x01) + return -ENODATA; return ret; } -static int asus_input_configured(struct hid_device *hdev, struct hid_input *hi) +static int asus_aura_get_effect_mask(struct asus_drvdata *drvdata, u8 effect_mask[2]) { - struct input_dev *input = hi->input; - struct asus_drvdata *drvdata = hid_get_drvdata(hdev); + u8 buf[AURA_FEATURE_REPORT_SIZE]; + int ret; - /* T100CHI uses MULTI_INPUT, bind the touchpad to the mouse hid_input */ - if (drvdata->quirks & QUIRK_T100CHI && - hi->report->id != T100CHI_MOUSE_REPORT_ID) - return 0; + ret = asus_aura_query_run_mode(drvdata, 0x20, buf, sizeof(buf)); + if (ret >= 22) + goto found; - /* Handle MULTI_INPUT on E1239T mouse/touchpad USB interface */ - if (drvdata->tp && (drvdata->quirks & QUIRK_MEDION_E1239T)) { - switch (hi->report->id) { - case E1239T_TP_TOGGLE_REPORT_ID: - input_set_capability(input, EV_KEY, KEY_F21); - input->name = "Asus Touchpad Keys"; - drvdata->tp_kbd_input = input; - return 0; - case INPUT_REPORT_ID: - break; /* Touchpad report, handled below */ - default: - return 0; /* Ignore other reports */ - } - } + ret = asus_aura_query_run_mode(drvdata, 0x15, buf, sizeof(buf)); + if (ret < 0) + return ret; + if (ret < 22) + return -ENODATA; - if (drvdata->tp) { - int ret; +found: + effect_mask[0] = buf[20]; + effect_mask[1] = buf[21]; - input_set_abs_params(input, ABS_MT_POSITION_X, 0, - drvdata->tp->max_x, 0, 0); - input_set_abs_params(input, ABS_MT_POSITION_Y, 0, - drvdata->tp->max_y, 0, 0); - input_abs_set_res(input, ABS_MT_POSITION_X, drvdata->tp->res_x); - input_abs_set_res(input, ABS_MT_POSITION_Y, drvdata->tp->res_y); + return 0; +} - if (drvdata->tp->contact_size >= 5) { - input_set_abs_params(input, ABS_TOOL_WIDTH, 0, - MAX_TOUCH_MAJOR, 0, 0); - input_set_abs_params(input, ABS_MT_TOUCH_MAJOR, 0, - MAX_TOUCH_MAJOR, 0, 0); - input_set_abs_params(input, ABS_MT_PRESSURE, 0, - MAX_PRESSURE, 0, 0); - } +static const struct { + const char *name; + u8 hw_mode; +} asus_aura_effect_map[] = { + { "static", AURA_EFFECT_STATIC }, + { "breathe", AURA_EFFECT_BREATHING }, + { "rainbow_cycle", AURA_EFFECT_SPECTRUM_CYCLE }, + { "rainbow_wave", AURA_EFFECT_RAINBOW }, + { "star", AURA_EFFECT_STARS }, + { "rain", AURA_EFFECT_RAIN }, + { "highlight", AURA_EFFECT_REACTIVE }, + { "laser", AURA_EFFECT_LASER }, + { "ripple", AURA_EFFECT_RIPPLE }, + { "pulse", AURA_EFFECT_PULSE }, + { "comet", AURA_EFFECT_COMET }, + { "flash", AURA_EFFECT_FLASH }, +}; - __set_bit(BTN_LEFT, input->keybit); - __set_bit(INPUT_PROP_BUTTONPAD, input->propbit); +static unsigned int asus_aura_fill_effects(const char **dst, + const u8 *effect_mask, + bool direct_capable) +{ + unsigned int i, n = 0; + u16 mask = effect_mask ? (effect_mask[0] | ((u16)effect_mask[1] << 8)) + : (u16)~0; - ret = input_mt_init_slots(input, drvdata->tp->max_contacts, - INPUT_MT_POINTER); + for (i = 0; i < ARRAY_SIZE(asus_aura_effect_map); i++) { + if (mask & BIT(asus_aura_effect_map[i].hw_mode)) + dst[n++] = asus_aura_effect_map[i].name; + } + if (direct_capable) + dst[n++] = "direct"; - if (ret) { - hid_err(hdev, "Asus input mt init slots failed: %d\n", ret); - return ret; + return n; +} + +static int asus_aura_hw_mode_from_name(const char *name, u8 *aura_mode) +{ + unsigned int i; + + if (!name) + return -EINVAL; + + for (i = 0; i < ARRAY_SIZE(asus_aura_effect_map); i++) { + if (!strcmp(name, asus_aura_effect_map[i].name)) { + *aura_mode = asus_aura_effect_map[i].hw_mode; + return 0; } } - drvdata->input = input; + return -EINVAL; +} - if ((drvdata->quirks & QUIRK_HID_FN_LOCK) && - (asus_kbd_fn_lock_set(drvdata, true))) - hid_warn(hdev, "Error while setting FN lock to ON\n"); +static bool asus_aura_hw_mode_known(u8 mode) +{ + unsigned int i; - return 0; + for (i = 0; i < ARRAY_SIZE(asus_aura_effect_map); i++) { + if (asus_aura_effect_map[i].hw_mode == mode) + return true; + } + + return false; } -#define asus_map_key_clear(c) hid_map_usage_clear(hi, usage, bit, \ - max, EV_KEY, (c)) +static bool asus_aura_speed_known(u8 speed) +{ + return speed == 0 || speed == AURA_SPEED_SLOW || + speed == AURA_SPEED_MED || speed == AURA_SPEED_FAST; +} + +/* + * Aura 0xb3 is write-only. Read the feature report before any SET in probe: + * some firmware still holds the last committed Set Effect packet. A blank + * report means nothing was programmed. Any other payload that is not that + * packet is left untouched, because failing to decode it is not evidence + * that the controller has no color. + */ +static void asus_aura_capture_boot_effect(struct hid_device *hdev) +{ + struct asus_drvdata *drvdata = hid_get_drvdata(hdev); + u8 buf[AURA_FEATURE_REPORT_SIZE]; + int i, ret; + bool blank = true; + + memset(buf, 0, sizeof(buf)); + buf[0] = FEATURE_KBD_LED_REPORT_ID1; + ret = asus_aura_get_feature(drvdata, buf, sizeof(buf)); + if (ret < 0) + return; + + for (i = 1; i < ret; i++) { + if (buf[i]) { + blank = false; + break; + } + } + if (blank) { + drvdata->boot_effect_blank = true; + return; + } + + if (ret < 13 || buf[1] != AURA_CMD_SET_EFFECT || + buf[2] > AURA_ZONE_BAR_RIGHT || + !asus_aura_hw_mode_known(buf[3]) || + !asus_aura_speed_known(buf[7])) + return; + + drvdata->boot_effect_valid = true; + drvdata->boot_zone = buf[2]; + drvdata->boot_mode = buf[3]; + drvdata->boot_r = buf[4]; + drvdata->boot_g = buf[5]; + drvdata->boot_b = buf[6]; + drvdata->boot_speed = buf[7]; + drvdata->boot_direction = buf[8]; + drvdata->boot_r2 = buf[10]; + drvdata->boot_g2 = buf[11]; + drvdata->boot_b2 = buf[12]; +} + +static int asus_aura_activate_zone_unlocked(struct asus_drvdata *drvdata, u8 zone) +{ + u8 buf[AURA_FEATURE_REPORT_SIZE] = { 0 }; + + buf[0] = FEATURE_KBD_LED_REPORT_ID1; + buf[1] = AURA_CMD_ZONE_ENABLE; + buf[2] = zone; + buf[3] = 0x01; + buf[4] = 0x01; + + return asus_aura_set_feature_unlocked(drvdata, buf, sizeof(buf)); +} + +static int asus_aura_activate_zone(struct asus_drvdata *drvdata, u8 zone) +{ + guard(mutex)(&drvdata->aura_lock); + + return asus_aura_activate_zone_unlocked(drvdata, zone); +} + +static int asus_aura_activate_zones(struct asus_drvdata *drvdata) +{ + int ret; + + ret = asus_aura_activate_zone(drvdata, AURA_ZONE_ACTIVATE_KEYBOARD); + if (ret < 0) + return ret; + if (drvdata->has_lightbar) + return asus_aura_activate_zone(drvdata, AURA_ZONE_ACTIVATE_LIGHTBAR); + return 0; +} + +static int asus_aura_wake_all_zones(struct asus_drvdata *drvdata) +{ + u8 buf_pwr[AURA_FEATURE_REPORT_SIZE] = { + FEATURE_KBD_LED_REPORT_ID1, + AURA_CMD_POWER, + AURA_POWER_CMD_ENABLE, + AURA_POWER_MASK_KBD_LOGO_ALL, + AURA_POWER_MASK_LIGHTBAR_ALL, + AURA_POWER_MASK_LID_ALL, + AURA_POWER_MASK_REAR_ALL, + 0x00, + }; + int ret; + + /* Unmute power gating across keyboard, lightbar, logo, lid, and rear-glow */ + ret = asus_aura_set_feature(drvdata, buf_pwr, sizeof(buf_pwr)); + if (ret < 0) + return ret; + + /* Activate keyboard zone */ + ret = asus_aura_activate_zone(drvdata, AURA_ZONE_ACTIVATE_KEYBOARD); + if (ret < 0) + return ret; + + /* Activate lightbar zone */ + return asus_aura_activate_zone(drvdata, AURA_ZONE_ACTIVATE_LIGHTBAR); +} + +static int asus_aura_apply_effect(struct led_classdev_dynamic *ldev, + enum led_brightness brightness); + +static void asus_aura_mark_effect_set(struct asus_drvdata *drvdata, + struct led_classdev_dynamic *ldev) +{ + if (ldev == &drvdata->dldev_kbd) + drvdata->kbd_effect_set = true; + else if (ldev == &drvdata->dldev_lightbar) + drvdata->lightbar_effect_set = true; +} + +static void asus_aura_restore(struct asus_drvdata *drvdata) +{ + int ret; + + if (!drvdata->kbd_effect_set && !drvdata->lightbar_effect_set) + return; + + ret = asus_aura_activate_zones(drvdata); + if (ret < 0) + hid_warn(drvdata->hdev, "Failed to activate Aura zones: %d\n", ret); + + if (drvdata->has_dldev_kbd && drvdata->kbd_effect_set) + asus_aura_apply_effect(&drvdata->dldev_kbd, + drvdata->dldev_kbd.cdev.brightness); + if (drvdata->has_dldev_lightbar && drvdata->lightbar_effect_set) + asus_aura_apply_effect(&drvdata->dldev_lightbar, + drvdata->dldev_lightbar.cdev.brightness); +} + +static int asus_aura_write_zone_effect(struct asus_drvdata *drvdata, u8 zone, + u8 aura_mode, u8 r, u8 g, u8 b, + u8 speed, u8 direction, + u8 r2, u8 g2, u8 b2, + bool commit) +{ + u8 buf[AURA_FEATURE_REPORT_SIZE] = { 0 }; + int ret; + + if (zone == AURA_ZONE_BAR_LEFT || zone == AURA_ZONE_BAR_RIGHT) { + ret = asus_aura_activate_zone(drvdata, AURA_ZONE_ACTIVATE_LIGHTBAR); + if (ret < 0) + return ret; + } + + buf[0] = FEATURE_KBD_LED_REPORT_ID1; + buf[1] = AURA_CMD_SET_EFFECT; + buf[2] = zone; + buf[3] = aura_mode; + buf[4] = r; + buf[5] = g; + buf[6] = b; + buf[7] = speed; + buf[8] = direction; + buf[9] = 0x00; + buf[10] = r2; + buf[11] = g2; + buf[12] = b2; + + ret = asus_aura_set_feature(drvdata, buf, sizeof(buf)); + if (ret < 0) + return ret; + + if (commit) + return asus_aura_commit(drvdata); + + return 0; +} + +static void asus_lamparray_fill_solid(u8 *buf, unsigned int nleds, + u8 r, u8 g, u8 b) +{ + unsigned int i; + + for (i = 0; i < nleds; i++) { + buf[i * 3 + 0] = r; + buf[i * 3 + 1] = g; + buf[i * 3 + 2] = b; + } +} + +static unsigned int asus_aura_lb_led_count(struct asus_drvdata *drvdata) +{ + return drvdata->is_strix_4zone ? ROG_STRIX_4ZONE_LIGHTBAR_LEDS : + ROG_STRIX_LIGHTBAR_LEDS; +} + +static void asus_aura_init_direct_bufs(struct asus_drvdata *drvdata) +{ + asus_lamparray_fill_solid(drvdata->kbd_direct_buf, ROG_STRIX_4ZONE_KBD_LEDS, + 255, 0, 0); + asus_lamparray_fill_solid(drvdata->lb_direct_buf, + asus_aura_lb_led_count(drvdata), 255, 0, 0); +} + + +static int asus_lamparray_try_apply_unlocked(struct asus_drvdata *drvdata) +{ + return -ENODEV; +} + +static void asus_lamparray_release_unlocked(struct asus_drvdata *drvdata) +{ +} + +static int asus_aura_write_4zone_direct_bc_unlocked(struct asus_drvdata *drvdata) +{ + u8 buf[AURA_FEATURE_REPORT_SIZE]; + + memset(buf, 0, sizeof(buf)); + buf[0] = FEATURE_KBD_LED_REPORT_ID1; + buf[1] = AURA_CMD_DIRECT; + buf[2] = AURA_DIRECT_FRAME_ZONED; + buf[3] = AURA_DIRECT_ROUTING_DEFAULT; + buf[4] = AURA_ZONE_LIGHTBAR_CHANNEL; + memcpy(&buf[ROG_STRIX_4ZONE_DIRECT_KBD_OFFSET], + drvdata->kbd_direct_buf, + sizeof(drvdata->kbd_direct_buf)); + memcpy(&buf[ROG_STRIX_4ZONE_DIRECT_LB_OFFSET], + drvdata->lb_direct_buf, + ROG_STRIX_4ZONE_LIGHTBAR_BUF_SIZE); + + return asus_aura_set_feature_unlocked(drvdata, buf, sizeof(buf)); +} + +static int asus_aura_strix_write_direct(struct asus_drvdata *drvdata, + const u8 *buffer, size_t size) +{ + u8 buf[AURA_FEATURE_REPORT_SIZE]; + unsigned int i; + int ret; + + guard(mutex)(&drvdata->aura_lock); + + if (drvdata->is_strix_4zone) { + unsigned int leds = min_t(size_t, size / 3, + ROG_STRIX_4ZONE_KBD_LEDS); + + if (buffer != drvdata->kbd_direct_buf) + memcpy(drvdata->kbd_direct_buf, buffer, leds * 3); + + ret = asus_lamparray_try_apply_unlocked(drvdata); + if (ret != -ENODEV) + return ret; + + return asus_aura_write_4zone_direct_bc_unlocked(drvdata); + } + + /* + * Stream ROG Strix per-key matrix in 16-LED chunks using + * Aura HID Feature Reports with opcode 0xbc: + * [0] = Report ID (0x5d) + * [1] = Direct frame command (0xbc) + * [2..5] = Routing header (0x00, 0x01, 0x01, 0x01) + * [6] = Start LED index (0, 16, 32, ..., 160) + * [7] = Number of LEDs in chunk (16 for chunks 0..9, 8 for chunk 10) + * [8] = Reserved / 0x00 + * [9..] = RGB payload (3 bytes per LED) + * + * Total LEDs: 168 (11 packets). Strictly terminate at packet 10; + * sending a 12th packet triggers a firmware defect that shuts off the + * rear and front lightbars. No 0xb4 commit command is issued for raw + * direct frames to avoid stepping hardware animation registers. + */ + for (i = 0; i < ROG_STRIX_DIRECT_LEDS; i += ROG_STRIX_LEDS_PER_PKT) { + unsigned int leds = min_t(unsigned int, ROG_STRIX_DIRECT_LEDS - i, + ROG_STRIX_LEDS_PER_PKT); + size_t payload_len = leds * 3; + + memset(buf, 0, sizeof(buf)); + buf[0] = FEATURE_KBD_LED_REPORT_ID1; + buf[1] = AURA_CMD_DIRECT; + buf[2] = AURA_DIRECT_FRAME_PERKEY; + buf[3] = AURA_DIRECT_ROUTING_DEFAULT; + buf[4] = AURA_ZONE_KEYBOARD_CHANNEL; + buf[5] = AURA_DIRECT_CHUNK_FLAG; + buf[6] = (u8)i; + buf[7] = (u8)leds; + buf[8] = 0x00; + memcpy(&buf[ROG_STRIX_PERKEY_DIRECT_PAYLOAD_OFFSET], + buffer + (i * 3), payload_len); + + ret = asus_aura_set_feature_unlocked(drvdata, buf, sizeof(buf)); + if (ret < 0) + return ret; + } + + return 0; +} + +static bool asus_aura_is_lightbar(struct asus_drvdata *drvdata, + struct led_classdev_dynamic *ldev) +{ + return ldev == &drvdata->dldev_lightbar; +} + +static int asus_aura_write_zone_range(struct asus_drvdata *drvdata, + u8 zone_first, u8 zone_last, + u8 aura_effect, u8 r, u8 g, u8 b, + u8 speed, u8 direction, + u8 r2, u8 g2, u8 b2) +{ + u8 z; + int ret; + + for (z = zone_first; z <= zone_last; z++) { + ret = asus_aura_write_zone_effect(drvdata, z, aura_effect, + r, g, b, speed, direction, + r2, g2, b2, false); + if (ret < 0) + return ret; + } + + return asus_aura_commit(drvdata); +} + +static int asus_aura_strix_set_direct(struct led_classdev_dynamic *ldev, + const u8 *buffer, size_t size) +{ + struct asus_drvdata *drvdata = ldev->driver_data; + int ret; + + if (size != ldev->led_count * 3) + return -EINVAL; + + if (!led_dynamic_effect_is(ldev, "direct")) { + ret = asus_aura_activate_zone(drvdata, AURA_ZONE_ACTIVATE_KEYBOARD); + if (ret < 0) + return ret; + } + + return asus_aura_strix_write_direct(drvdata, buffer, size); +} + +static int asus_aura_lightbar_write_packet(struct asus_drvdata *drvdata, + const u8 *buffer, size_t size) +{ + u8 buf[AURA_FEATURE_REPORT_SIZE]; + size_t payload_len; + int ret; + + guard(mutex)(&drvdata->aura_lock); + + if (drvdata->is_strix_4zone) { + unsigned int leds = min_t(size_t, size / 3, + ROG_STRIX_4ZONE_LIGHTBAR_LEDS); + + if (buffer != drvdata->lb_direct_buf) + memcpy(drvdata->lb_direct_buf, buffer, leds * 3); + + ret = asus_lamparray_try_apply_unlocked(drvdata); + if (ret != -ENODEV) + return ret; + + return asus_aura_write_4zone_direct_bc_unlocked(drvdata); + } + + payload_len = min_t(size_t, size, ROG_STRIX_LIGHTBAR_BUF_SIZE); + + /* + * Per-Key models use buf[2] = 0x00 (chunked packet), so the MCU + * expects valid chunk headers in bytes 5-7. Without them it reads + * "0 LEDs to update" and silently drops the packet. + */ + memset(buf, 0, sizeof(buf)); + buf[0] = FEATURE_KBD_LED_REPORT_ID1; + buf[1] = AURA_CMD_DIRECT; + buf[2] = AURA_DIRECT_FRAME_PERKEY; + buf[3] = AURA_DIRECT_ROUTING_DEFAULT; + buf[4] = AURA_ZONE_LIGHTBAR_CHANNEL; + buf[5] = AURA_DIRECT_CHUNK_FLAG; + buf[6] = 0x00; /* start LED index */ + buf[7] = (u8)(payload_len / 3); /* number of LEDs in this packet */ + + if (buffer != drvdata->lb_direct_buf) + memcpy(drvdata->lb_direct_buf, buffer, payload_len); + memcpy(&buf[ROG_STRIX_PERKEY_DIRECT_PAYLOAD_OFFSET], + drvdata->lb_direct_buf, payload_len); + + return asus_aura_set_feature_unlocked(drvdata, buf, sizeof(buf)); +} + +static int asus_aura_lightbar_set_direct(struct led_classdev_dynamic *ldev, + const u8 *buffer, size_t size) +{ + struct asus_drvdata *drvdata = ldev->driver_data; + int ret; + + if (size != ldev->led_count * 3) + return -EINVAL; + + if (!led_dynamic_effect_is(ldev, "direct")) { + ret = asus_aura_activate_zone(drvdata, AURA_ZONE_ACTIVATE_LIGHTBAR); + if (ret < 0) + return ret; + } + + return asus_aura_lightbar_write_packet(drvdata, buffer, size); +} + +static int asus_aura_apply_effect(struct led_classdev_dynamic *ldev, + enum led_brightness brightness) +{ + struct asus_drvdata *drvdata = ldev->driver_data; + const char *name = led_dynamic_effect_name(ldev); + bool lights_on = brightness > LED_OFF; + u8 aura_mode; + u8 speed; + u8 direction = 0; + u8 r = 0, g = 0, b = 0; + u8 r2 = 0, g2 = 0, b2 = 0; + int ret; + + if (lights_on && ldev->num_palette_entries > 0) { + r = (u8)(((unsigned int)ldev->palette[0].r * brightness) / 255); + g = (u8)(((unsigned int)ldev->palette[0].g * brightness) / 255); + b = (u8)(((unsigned int)ldev->palette[0].b * brightness) / 255); + if (ldev->num_palette_entries > 1) { + r2 = (u8)(((unsigned int)ldev->palette[1].r * brightness) / 255); + g2 = (u8)(((unsigned int)ldev->palette[1].g * brightness) / 255); + b2 = (u8)(((unsigned int)ldev->palette[1].b * brightness) / 255); + } + } + + if (led_dynamic_effect_is(ldev, "direct")) + return 0; + + switch (ldev->speed) { + case 0: + speed = AURA_SPEED_SLOW; + break; + case 2: + speed = AURA_SPEED_FAST; + break; + case 1: + default: + speed = AURA_SPEED_MED; + break; + } + + if (!lights_on) { + aura_mode = AURA_EFFECT_STATIC; + r = 0; + g = 0; + b = 0; + r2 = 0; + g2 = 0; + b2 = 0; + } else if (asus_aura_hw_mode_from_name(name, &aura_mode)) { + return -EINVAL; + } + + if (ldev->direction == DL_DIRECTION_LEFT) + direction = 1; + else if (ldev->direction == DL_DIRECTION_RIGHT) + direction = 0; + else if (ldev->direction == DL_DIRECTION_UP) + direction = 2; + else if (ldev->direction == DL_DIRECTION_DOWN) + direction = 3; + + /* + * Static and animations are firmware effects: commit them with 0xb3 + * so the controller keeps the color across reboot. Release LampArray + * first when a later commit uses it as the direct-RGB backend. + */ + scoped_guard(mutex, &drvdata->aura_lock) + asus_lamparray_release_unlocked(drvdata); + + if (asus_aura_is_lightbar(drvdata, ldev)) + ret = asus_aura_write_zone_range(drvdata, AURA_ZONE_BAR_LEFT, + AURA_ZONE_BAR_RIGHT, aura_mode, + r, g, b, speed, direction, + r2, g2, b2); + else if (drvdata->is_strix_4zone) + ret = asus_aura_write_zone_range(drvdata, AURA_ZONE_KEY1, + AURA_ZONE_KEY4, aura_mode, + r, g, b, speed, direction, + r2, g2, b2); + else + ret = asus_aura_write_zone_effect(drvdata, AURA_ZONE_ALL, aura_mode, + r, g, b, speed, direction, + r2, g2, b2, true); + if (ret < 0) + return ret; + + asus_aura_mark_effect_set(drvdata, ldev); + return 0; +} + +static int asus_aura_set_effect(struct led_classdev_dynamic *ldev, + unsigned int index) +{ + struct led_classdev *cdev = led_dynamic_cdev(ldev); + unsigned int old = ldev->current_effect; + int ret; + + if (!ldev->enabled) + return 0; + + ldev->current_effect = index; + ret = asus_aura_apply_effect(ldev, cdev->brightness); + ldev->current_effect = old; + return ret; +} + +static int asus_aura_set_enabled(struct led_classdev_dynamic *ldev, bool enabled) +{ + struct led_classdev *cdev = led_dynamic_cdev(ldev); + + return asus_aura_apply_effect(ldev, enabled ? cdev->brightness : LED_OFF); +} + +static int asus_aura_set_speed(struct led_classdev_dynamic *ldev, + unsigned int speed) +{ + struct led_classdev *cdev = led_dynamic_cdev(ldev); + unsigned int old_speed = ldev->speed; + int ret; + + ldev->speed = speed; + ret = asus_aura_apply_effect(ldev, cdev->brightness); + ldev->speed = old_speed; + return ret; +} + +static int asus_aura_set_direction(struct led_classdev_dynamic *ldev, + enum dl_direction direction) +{ + struct led_classdev *cdev = led_dynamic_cdev(ldev); + enum dl_direction old_dir = ldev->direction; + int ret; + + ldev->direction = direction; + ret = asus_aura_apply_effect(ldev, cdev->brightness); + ldev->direction = old_dir; + return ret; +} + +static int asus_aura_set_palette(struct led_classdev_dynamic *ldev, + const struct dl_rgb *palette, + unsigned int num_entries) +{ + struct led_classdev *cdev = led_dynamic_cdev(ldev); + struct dl_rgb saved[2]; + unsigned int old_n = ldev->num_palette_entries; + unsigned int copy_n; + int ret; + + if (!palette || !num_entries || num_entries > ldev->max_palette_entries) + return -EINVAL; + + copy_n = min_t(unsigned int, old_n, ARRAY_SIZE(saved)); + if (copy_n) + memcpy(saved, ldev->palette, copy_n * sizeof(*saved)); + + memcpy(ldev->palette, palette, num_entries * sizeof(*palette)); + ldev->num_palette_entries = num_entries; + + ret = asus_aura_apply_effect(ldev, cdev->brightness); + + memcpy(ldev->palette, saved, copy_n * sizeof(*saved)); + ldev->num_palette_entries = old_n; + return ret; +} + +static int asus_aura_brightness_set_blocking(struct led_classdev *cdev, + enum led_brightness brightness) +{ + struct led_classdev_dynamic *ldev = lcdev_to_dldev(cdev); + struct asus_drvdata *drvdata = ldev->driver_data; + int ret; + + guard(mutex)(&ldev->lock); + + if (brightness == LED_OFF || !ldev->enabled) + return asus_aura_apply_effect(ldev, LED_OFF); + + ret = asus_aura_activate_zones(drvdata); + if (ret < 0) + hid_warn(drvdata->hdev, "Failed to activate Aura hardware zones: %d\n", ret); + + return asus_aura_apply_effect(ldev, brightness); +} + +static int asus_aura_set_power_states(struct led_classdev_dynamic *ldev, + u32 active_states) +{ + struct asus_drvdata *drvdata = ldev->driver_data; + u8 buf[AURA_FEATURE_REPORT_SIZE] = { + FEATURE_KBD_LED_REPORT_ID1, + AURA_CMD_POWER, + AURA_POWER_CMD_ENABLE, + }; + u8 kbd = 0; + u8 lightbar = 0; + + if (active_states & DL_POWER_STATE_BOOT) { + kbd |= AURA_POWER_KBD_BOOT; + lightbar |= AURA_POWER_LB_BOOT; + } + if (active_states & DL_POWER_STATE_AWAKE) { + kbd |= AURA_POWER_KBD_AWAKE; + lightbar |= AURA_POWER_LB_AWAKE; + } + if (active_states & DL_POWER_STATE_SLEEP) { + kbd |= AURA_POWER_KBD_SLEEP; + lightbar |= AURA_POWER_LB_SLEEP; + } + if (active_states & DL_POWER_STATE_SHUTDOWN) { + kbd |= AURA_POWER_KBD_SHUTDOWN; + lightbar |= AURA_POWER_LB_SHUTDOWN; + } + + /* + * Map generic DL boot/awake/sleep/shutdown onto keyboard bits, and also + * lightbar bits when the chassis lightbar is present. Leave lid/rear + * enabled so unmanaged zones are not accidentally gated off. + */ + buf[3] = kbd | AURA_POWER_MASK_LOGO_ALL; + buf[4] = drvdata->has_lightbar ? lightbar : 0; + buf[5] = AURA_POWER_MASK_LID_ALL; + buf[6] = AURA_POWER_MASK_REAR_ALL; + + return asus_aura_set_feature(drvdata, buf, sizeof(buf)); +} + +static const struct led_dynamic_ops asus_aura_ops = { + .set_effect = asus_aura_set_effect, + .set_speed = asus_aura_set_speed, + .set_direction = asus_aura_set_direction, + .set_palette = asus_aura_set_palette, + .set_enabled = asus_aura_set_enabled, + .set_power_states = asus_aura_set_power_states, + .direct_write = asus_aura_strix_set_direct, +}; + +static const struct led_dynamic_ops asus_aura_lightbar_ops = { + .set_effect = asus_aura_set_effect, + .set_speed = asus_aura_set_speed, + .set_direction = asus_aura_set_direction, + .set_palette = asus_aura_set_palette, + .set_enabled = asus_aura_set_enabled, + .set_power_states = asus_aura_set_power_states, + .direct_write = asus_aura_lightbar_set_direct, +}; + +static int asus_aura_discover(struct asus_drvdata *drvdata, bool *has_lightbar, + bool *is_strix_4zone, bool *is_per_key) +{ + u8 buf[AURA_FEATURE_REPORT_SIZE] = { + FEATURE_KBD_LED_REPORT_ID1, + AURA_CMD_PROBE, + 0x20, + 0x31, + 0x00, + 0x20, + }; + int ret; + + *has_lightbar = false; + *is_strix_4zone = false; + *is_per_key = false; + + /* + * Query hardware configuration via Report 0x5D opcode 0x05. + * Byte 9 describes the keyboard layout class (0x02 = 4-zone, + * 0x03 = per-key) and byte 13 is the physical-region bitmap + * (bit 1 = lightbar present). + */ + ret = asus_aura_set_feature(drvdata, buf, sizeof(buf)); + if (ret < 0) + return ret; + + memset(buf, 0, sizeof(buf)); + buf[0] = FEATURE_KBD_LED_REPORT_ID1; + ret = asus_aura_get_feature(drvdata, buf, sizeof(buf)); + if (ret < 0) + return ret; + + if (ret < 14) + return -EPROTO; + + if (buf[1] != AURA_CMD_PROBE || buf[2] != 0x20 || buf[3] != 0x31) + return -ENODEV; + + *is_strix_4zone = (buf[9] == 0x02); + *is_per_key = (buf[9] == 0x03); + *has_lightbar = !!(buf[13] & 0x02); + + return 0; +} + +static void asus_aura_dldev_set_default_palette(struct led_classdev_dynamic *ldev) +{ + if (!ldev->palette) + return; + + ldev->palette[0].r = 255; + ldev->palette[0].g = 0; + ldev->palette[0].b = 0; + ldev->num_palette_entries = 1; +} + +static void asus_aura_dldev_common_init(struct led_classdev_dynamic *ldev, + struct asus_drvdata *drvdata, + const char *name, + const struct led_dynamic_ops *ops, + const char **effects, + unsigned int num_effects) +{ + unsigned int i; + + ldev->cdev.name = name; + ldev->cdev.max_brightness = 255; + ldev->cdev.brightness = 255; + ldev->cdev.brightness_set_blocking = asus_aura_brightness_set_blocking; + ldev->ops = ops; + ldev->driver_data = drvdata; + ldev->speed = 1; + ldev->speed_min = 0; + ldev->speed_max = 2; + ldev->direction = DL_DIRECTION_RIGHT; + ldev->supported_directions = BIT(DL_DIRECTION_RIGHT) | + BIT(DL_DIRECTION_LEFT); + ldev->max_palette_entries = 2; + ldev->effects = (const char * const *)effects; + ldev->num_effects = num_effects; + ldev->current_effect = 0; + for (i = 0; i < num_effects; i++) { + if (!strcmp(effects[i], "static")) { + ldev->current_effect = i; + break; + } + } + ldev->supported_power_states = DL_POWER_STATE_ALL; + ldev->active_power_states = DL_POWER_STATE_ALL; +} + +static int asus_aura_effect_index(struct led_classdev_dynamic *ldev, u8 hw_mode) +{ + const char *name = NULL; + unsigned int i; + + for (i = 0; i < ARRAY_SIZE(asus_aura_effect_map); i++) { + if (asus_aura_effect_map[i].hw_mode == hw_mode) { + name = asus_aura_effect_map[i].name; + break; + } + } + if (!name) + return -1; + + for (i = 0; i < ldev->num_effects; i++) { + if (!strcmp(ldev->effects[i], name)) + return i; + } + + return -1; +} + +static void asus_aura_apply_boot_levels(struct led_classdev_dynamic *ldev, + struct asus_drvdata *drvdata) +{ + switch (drvdata->boot_speed) { + case AURA_SPEED_SLOW: + ldev->speed = 0; + break; + case AURA_SPEED_FAST: + ldev->speed = 2; + break; + default: + ldev->speed = 1; + break; + } + + switch (drvdata->boot_direction) { + case 1: + ldev->direction = DL_DIRECTION_LEFT; + break; + case 2: + ldev->direction = DL_DIRECTION_UP; + break; + case 3: + ldev->direction = DL_DIRECTION_DOWN; + break; + default: + ldev->direction = DL_DIRECTION_RIGHT; + break; + } + + if (ldev->supported_directions && + !(ldev->supported_directions & BIT(ldev->direction))) + ldev->direction = DL_DIRECTION_RIGHT; +} + +static bool asus_aura_adopt_node(struct asus_drvdata *drvdata, + struct led_classdev_dynamic *ldev) +{ + int idx; + + if (!ldev->palette) + return false; + + idx = asus_aura_effect_index(ldev, drvdata->boot_mode); + if (idx < 0) + return false; + + ldev->current_effect = idx; + ldev->palette[0].r = drvdata->boot_r; + ldev->palette[0].g = drvdata->boot_g; + ldev->palette[0].b = drvdata->boot_b; + ldev->num_palette_entries = 1; + if (drvdata->boot_r2 || drvdata->boot_g2 || drvdata->boot_b2) { + ldev->palette[1].r = drvdata->boot_r2; + ldev->palette[1].g = drvdata->boot_g2; + ldev->palette[1].b = drvdata->boot_b2; + ldev->num_palette_entries = 2; + } + asus_aura_apply_boot_levels(ldev, drvdata); + return true; +} + +static bool asus_aura_adopt_boot_effect(struct asus_drvdata *drvdata) +{ + u8 zone = drvdata->boot_zone; + bool kbd = zone == AURA_ZONE_ALL || + (zone >= AURA_ZONE_KEY1 && zone <= AURA_ZONE_KEY4); + bool lightbar = zone == AURA_ZONE_ALL || + zone == AURA_ZONE_BAR_LEFT || + zone == AURA_ZONE_BAR_RIGHT; + bool any = false; + + if (kbd && drvdata->has_dldev_kbd && + asus_aura_adopt_node(drvdata, &drvdata->dldev_kbd)) { + drvdata->kbd_effect_set = true; + any = true; + } + + if (lightbar && drvdata->has_dldev_lightbar && + asus_aura_adopt_node(drvdata, &drvdata->dldev_lightbar)) { + drvdata->lightbar_effect_set = true; + any = true; + } + + if (any) { + asus_lamparray_fill_solid(drvdata->kbd_direct_buf, + ROG_STRIX_4ZONE_KBD_LEDS, + drvdata->boot_r, drvdata->boot_g, + drvdata->boot_b); + asus_lamparray_fill_solid(drvdata->lb_direct_buf, + asus_aura_lb_led_count(drvdata), + drvdata->boot_r, drvdata->boot_g, + drvdata->boot_b); + } + + return any; +} + +static void asus_aura_apply_static_red(struct asus_drvdata *drvdata) +{ + if (drvdata->has_dldev_kbd) + asus_aura_apply_effect(&drvdata->dldev_kbd, + drvdata->dldev_kbd.cdev.brightness); + if (drvdata->has_dldev_lightbar) + asus_aura_apply_effect(&drvdata->dldev_lightbar, + drvdata->dldev_lightbar.cdev.brightness); +} + +static int asus_init_dynamic_lighting(struct hid_device *hdev) +{ + struct asus_drvdata *drvdata = hid_get_drvdata(hdev); + unsigned int kbd_n, lightbar_n; + u8 effect_mask[2]; + const u8 *mask = NULL; + bool is_per_key = false; + bool has_lightbar = false; + bool kbd_direct; + bool lightbar_direct; + int ret; + + ret = asus_aura_discover(drvdata, &has_lightbar, &drvdata->is_strix_4zone, + &is_per_key); + if (ret == -ENODEV) + return 0; + if (ret < 0) { + hid_warn(hdev, "Aura device discovery failed: %d\n", ret); + return 0; + } + + asus_aura_init_direct_bufs(drvdata); + + ret = asus_aura_wake_all_zones(drvdata); + if (ret < 0) + hid_warn(hdev, "Failed to wake Aura hardware zones: %d\n", ret); + + kbd_direct = is_per_key || drvdata->is_strix_4zone; + lightbar_direct = has_lightbar; + ret = asus_aura_get_effect_mask(drvdata, effect_mask); + if (ret < 0) { + hid_warn(hdev, + "Aura 0x9e capability probe failed: %d, using fallback effect list\n", + ret); + } else { + mask = effect_mask; + } + + kbd_n = asus_aura_fill_effects(drvdata->effects_kbd, mask, kbd_direct); + lightbar_n = asus_aura_fill_effects(drvdata->effects_lightbar, mask, + lightbar_direct); + + drvdata->has_lightbar = has_lightbar; + + /* Keyboard Dynamic Lighting zone: always registered on Aura models. */ + asus_aura_dldev_common_init(&drvdata->dldev_kbd, drvdata, "aura:keyboard", + &asus_aura_ops, + drvdata->effects_kbd, kbd_n); + if (kbd_direct) { + drvdata->dldev_kbd.zone_type = drvdata->is_strix_4zone ? + "keyboard" : "keyboard_per_key"; + drvdata->dldev_kbd.led_count = drvdata->is_strix_4zone ? + ROG_STRIX_4ZONE_KBD_LEDS : ROG_STRIX_DIRECT_LEDS; + } else { + drvdata->dldev_kbd.zone_type = "keyboard"; + if (drvdata->is_strix_4zone) + drvdata->dldev_kbd.led_count = ROG_STRIX_4ZONE_KBD_LEDS; + } + + ret = devm_led_classdev_dynamic_register(&hdev->dev, &drvdata->dldev_kbd); + if (ret < 0) { + hid_warn(hdev, "Failed to register kbd dynamic lighting: %d\n", ret); + return ret; + } + drvdata->has_dldev_kbd = true; + asus_aura_dldev_set_default_palette(&drvdata->dldev_kbd); + + if (has_lightbar) { + asus_aura_dldev_common_init(&drvdata->dldev_lightbar, drvdata, + "aura:lightbar", + &asus_aura_lightbar_ops, + drvdata->effects_lightbar, lightbar_n); + drvdata->dldev_lightbar.zone_type = "lightbar"; + drvdata->dldev_lightbar.led_count = asus_aura_lb_led_count(drvdata); + + ret = devm_led_classdev_dynamic_register(&hdev->dev, + &drvdata->dldev_lightbar); + if (ret < 0) { + hid_warn(hdev, "Failed to register lightbar dynamic lighting: %d\n", + ret); + } else { + drvdata->has_dldev_lightbar = true; + asus_aura_dldev_set_default_palette(&drvdata->dldev_lightbar); + } + } + + hid_info(hdev, "Registered dynamic lighting zones: kbd=%d, lightbar=%d\n", + drvdata->has_dldev_kbd, drvdata->has_dldev_lightbar); + + if (drvdata->boot_effect_valid && asus_aura_adopt_boot_effect(drvdata)) { + hid_info(hdev, "Preserving programmed Aura effect\n"); + return 0; + } + + if (!drvdata->boot_effect_blank) { + hid_info(hdev, "Aura effect not readable, leaving controller state\n"); + return 0; + } + + hid_info(hdev, "No Aura effect programmed, applying static red\n"); + asus_aura_apply_static_red(drvdata); + return 0; +} + +#else /* !IS_REACHABLE(CONFIG_LEDS_CLASS_DYNAMIC) */ + +static inline int asus_init_dynamic_lighting(struct hid_device *hdev) +{ + return 0; +} + +static inline void asus_aura_capture_boot_effect(struct hid_device *hdev) +{ +} + +#endif /* IS_REACHABLE(CONFIG_LEDS_CLASS_DYNAMIC) */ + +/* + * [0] REPORT_ID (same value defined in report descriptor) + * [1] rest battery level. range [0..255] + * [2]..[7] Bluetooth hardware address (MAC address) + * [8] charging status + * = 0 : AC offline / discharging + * = 1 : AC online / charging + * = 2 : AC online / fully charged + */ +static int asus_parse_battery(struct asus_drvdata *drvdata, u8 *data, int size) +{ + u8 sts; + u8 lvl; + int val; + + lvl = data[1]; + sts = data[8]; + + drvdata->battery_capacity = ((int)lvl * 100) / (int)BATTERY_LEVEL_MAX; + + switch (sts) { + case BATTERY_STAT_CHARGING: + val = POWER_SUPPLY_STATUS_CHARGING; + break; + case BATTERY_STAT_FULL: + val = POWER_SUPPLY_STATUS_FULL; + break; + case BATTERY_STAT_DISCONNECT: + default: + val = POWER_SUPPLY_STATUS_DISCHARGING; + break; + } + drvdata->battery_stat = val; + + return 0; +} + +static int asus_report_battery(struct asus_drvdata *drvdata, u8 *data, int size) +{ + /* notify only the autonomous event by device */ + if ((drvdata->battery_in_query == false) && + (size == BATTERY_REPORT_SIZE)) + power_supply_changed(drvdata->battery); + + return 0; +} + +static int asus_battery_query(struct asus_drvdata *drvdata) +{ + u8 *buf; + int ret = 0; + + buf = kmalloc(BATTERY_REPORT_SIZE, GFP_KERNEL); + if (!buf) + return -ENOMEM; + + drvdata->battery_in_query = true; + ret = hid_hw_raw_request(drvdata->hdev, BATTERY_REPORT_ID, + buf, BATTERY_REPORT_SIZE, + HID_INPUT_REPORT, HID_REQ_GET_REPORT); + drvdata->battery_in_query = false; + if (ret == BATTERY_REPORT_SIZE) + ret = asus_parse_battery(drvdata, buf, BATTERY_REPORT_SIZE); + else + ret = -ENODATA; + + kfree(buf); + + return ret; +} + +static enum power_supply_property asus_battery_props[] = { + POWER_SUPPLY_PROP_STATUS, + POWER_SUPPLY_PROP_PRESENT, + POWER_SUPPLY_PROP_CAPACITY, + POWER_SUPPLY_PROP_SCOPE, + POWER_SUPPLY_PROP_MODEL_NAME, +}; + +#define QUERY_MIN_INTERVAL (60 * HZ) /* 60[sec] */ + +static int asus_battery_get_property(struct power_supply *psy, + enum power_supply_property psp, + union power_supply_propval *val) +{ + struct asus_drvdata *drvdata = power_supply_get_drvdata(psy); + int ret = 0; + + switch (psp) { + case POWER_SUPPLY_PROP_STATUS: + case POWER_SUPPLY_PROP_CAPACITY: + if (time_before(drvdata->battery_next_query, jiffies)) { + drvdata->battery_next_query = + jiffies + QUERY_MIN_INTERVAL; + ret = asus_battery_query(drvdata); + if (ret) + return ret; + } + if (psp == POWER_SUPPLY_PROP_STATUS) + val->intval = drvdata->battery_stat; + else + val->intval = drvdata->battery_capacity; + break; + case POWER_SUPPLY_PROP_PRESENT: + val->intval = 1; + break; + case POWER_SUPPLY_PROP_SCOPE: + val->intval = POWER_SUPPLY_SCOPE_DEVICE; + break; + case POWER_SUPPLY_PROP_MODEL_NAME: + val->strval = drvdata->hdev->name; + break; + default: + ret = -EINVAL; + break; + } + + return ret; +} + +static int asus_battery_probe(struct hid_device *hdev) +{ + struct asus_drvdata *drvdata = hid_get_drvdata(hdev); + struct power_supply_config pscfg = { .drv_data = drvdata }; + int ret = 0; + + drvdata->battery_capacity = 0; + drvdata->battery_stat = POWER_SUPPLY_STATUS_UNKNOWN; + drvdata->battery_in_query = false; + + drvdata->battery_desc.properties = asus_battery_props; + drvdata->battery_desc.num_properties = ARRAY_SIZE(asus_battery_props); + drvdata->battery_desc.get_property = asus_battery_get_property; + drvdata->battery_desc.type = POWER_SUPPLY_TYPE_BATTERY; + drvdata->battery_desc.use_for_apm = 0; + drvdata->battery_desc.name = devm_kasprintf(&hdev->dev, GFP_KERNEL, + "asus-keyboard-%s-battery", + strlen(hdev->uniq) ? + hdev->uniq : dev_name(&hdev->dev)); + if (!drvdata->battery_desc.name) + return -ENOMEM; + + drvdata->battery_next_query = jiffies; + + drvdata->battery = devm_power_supply_register(&hdev->dev, + &(drvdata->battery_desc), &pscfg); + if (IS_ERR(drvdata->battery)) { + ret = PTR_ERR(drvdata->battery); + drvdata->battery = NULL; + hid_err(hdev, "Unable to register battery device\n"); + return ret; + } + + power_supply_powers(drvdata->battery, &hdev->dev); + + return ret; +} + +static int asus_input_configured(struct hid_device *hdev, struct hid_input *hi) +{ + struct input_dev *input = hi->input; + struct asus_drvdata *drvdata = hid_get_drvdata(hdev); + + /* T100CHI uses MULTI_INPUT, bind the touchpad to the mouse hid_input */ + if (drvdata->quirks & QUIRK_T100CHI && + hi->report->id != T100CHI_MOUSE_REPORT_ID) + return 0; + + /* Handle MULTI_INPUT on E1239T mouse/touchpad USB interface */ + if (drvdata->tp && (drvdata->quirks & QUIRK_MEDION_E1239T)) { + switch (hi->report->id) { + case E1239T_TP_TOGGLE_REPORT_ID: + input_set_capability(input, EV_KEY, KEY_F21); + input->name = "Asus Touchpad Keys"; + drvdata->tp_kbd_input = input; + return 0; + case INPUT_REPORT_ID: + break; /* Touchpad report, handled below */ + default: + return 0; /* Ignore other reports */ + } + } + + if (drvdata->tp) { + int ret; + + input_set_abs_params(input, ABS_MT_POSITION_X, 0, + drvdata->tp->max_x, 0, 0); + input_set_abs_params(input, ABS_MT_POSITION_Y, 0, + drvdata->tp->max_y, 0, 0); + input_abs_set_res(input, ABS_MT_POSITION_X, drvdata->tp->res_x); + input_abs_set_res(input, ABS_MT_POSITION_Y, drvdata->tp->res_y); + + if (drvdata->tp->contact_size >= 5) { + input_set_abs_params(input, ABS_TOOL_WIDTH, 0, + MAX_TOUCH_MAJOR, 0, 0); + input_set_abs_params(input, ABS_MT_TOUCH_MAJOR, 0, + MAX_TOUCH_MAJOR, 0, 0); + input_set_abs_params(input, ABS_MT_PRESSURE, 0, + MAX_PRESSURE, 0, 0); + } + + __set_bit(BTN_LEFT, input->keybit); + __set_bit(INPUT_PROP_BUTTONPAD, input->propbit); + + ret = input_mt_init_slots(input, drvdata->tp->max_contacts, + INPUT_MT_POINTER); + + if (ret) { + hid_err(hdev, "Asus input mt init slots failed: %d\n", ret); + return ret; + } + } + + drvdata->input = input; + + if ((drvdata->quirks & QUIRK_HID_FN_LOCK) && + (asus_kbd_fn_lock_set(drvdata, true))) + hid_warn(hdev, "Error while setting FN lock to ON\n"); + + return 0; +} + +#define asus_map_key_clear(c) hid_map_usage_clear(hi, usage, bit, \ + max, EV_KEY, (c)) static int asus_input_mapping(struct hid_device *hdev, struct hid_input *hi, struct hid_field *field, struct hid_usage *usage, unsigned long **bit, @@ -1420,6 +2809,11 @@ static int __maybe_unused asus_resume(struct hid_device *hdev) { struct asus_drvdata *drvdata = hid_get_drvdata(hdev); +#if IS_REACHABLE(CONFIG_LEDS_CLASS_DYNAMIC) + if (drvdata->has_dldev_kbd || drvdata->has_dldev_lightbar) + asus_aura_restore(drvdata); +#endif + /* * If we have a backlight listener registered, restore the previous state, * in case of error do not fail: most models restore the backlight @@ -1435,6 +2829,11 @@ static int __maybe_unused asus_reset_resume(struct hid_device *hdev) { struct asus_drvdata *drvdata = hid_get_drvdata(hdev); +#if IS_REACHABLE(CONFIG_LEDS_CLASS_DYNAMIC) + if (drvdata->has_dldev_kbd || drvdata->has_dldev_lightbar) + asus_aura_restore(drvdata); +#endif + if (drvdata->tp) return asus_start_multitouch(hdev); @@ -1455,6 +2854,17 @@ static int asus_probe(struct hid_device *hdev, const struct hid_device_id *id) hid_set_drvdata(hdev, drvdata); +#if IS_REACHABLE(CONFIG_LEDS_CLASS_DYNAMIC) + ret = devm_mutex_init(&hdev->dev, &drvdata->aura_lock); + if (ret) + return ret; + + drvdata->aura_buf = devm_kzalloc(&hdev->dev, AURA_FEATURE_REPORT_SIZE, + GFP_KERNEL); + if (!drvdata->aura_buf) + return -ENOMEM; +#endif + drvdata->quirks = id->driver_data; /* @@ -1584,6 +2994,13 @@ static int asus_probe(struct hid_device *hdev, const struct hid_device_id *id) return ret; } + /* + * Read any committed Aura effect before the keyboard handshake + * overwrites the feature report. + */ + if (asus_has_report_id(hdev, FEATURE_KBD_LED_REPORT_ID1)) + asus_aura_capture_boot_effect(hdev); + if (!drvdata->tp) { for (int r = 0; r < ARRAY_SIZE(asus_report_id_init); r++) { if (asus_has_report_id(hdev, asus_report_id_init[r])) { @@ -1601,6 +3018,13 @@ static int asus_probe(struct hid_device *hdev, const struct hid_device_id *id) (asus_kbd_register_leds(hdev))) hid_warn(hdev, "Failed to initialize backlight.\n"); + if (asus_has_report_id(hdev, FEATURE_KBD_LED_REPORT_ID1) || + asus_has_report_id(hdev, FEATURE_KBD_LED_REPORT_ID2)) { + ret = asus_init_dynamic_lighting(hdev); + if (ret < 0) + hid_warn(hdev, "Failed to initialize dynamic lighting: %d\n", ret); + } + /* * For ROG keyboards, skip rename for consistency and ->input check as * some devices do not have inputs. From 538615d31979161565ad8d5da19230423c414063 Mon Sep 17 00:00:00 2001 From: Marco Scardovi Date: Fri, 18 Sep 2026 15:44:15 +0200 Subject: [PATCH 1351/1352] HID: asus: Use LampArray as in-kernel direct-RGB backend On N-KEY devices where Aura 0xBC cannot drive the chassis lightbar independently, use the sibling HID LampArray interface as the in-kernel direct-RGB backend and drop the owner reference on unbind. Linux Dynamic Lighting sysfs remains the userspace ABI. Fall back to Aura 0xBC when LampArray is absent. Firmware effects, including static color, stay on Aura 0xb3; release LampArray before committing one. Signed-off-by: Marco Scardovi --- drivers/hid/hid-asus.c | 438 ++++++++++++++++++++++++++++++++++++++++- 1 file changed, 429 insertions(+), 9 deletions(-) diff --git a/drivers/hid/hid-asus.c b/drivers/hid/hid-asus.c index e3ea58f4f2d779..fb400b8a23705b 100644 --- a/drivers/hid/hid-asus.c +++ b/drivers/hid/hid-asus.c @@ -288,6 +288,14 @@ struct asus_touchpad_info { int report_size; }; +#if IS_REACHABLE(CONFIG_LEDS_CLASS_DYNAMIC) +struct asus_lamparray_lamp { + u16 id; + s32 x; + bool keyboard; +}; +#endif + struct asus_drvdata { unsigned long quirks; struct hid_device *hdev; @@ -331,6 +339,14 @@ struct asus_drvdata { bool has_lightbar; u8 kbd_direct_buf[ROG_STRIX_4ZONE_KBD_BUF_SIZE]; u8 lb_direct_buf[ROG_STRIX_LIGHTBAR_BUF_SIZE]; + struct hid_device *lamparray_hdev; + u8 *lamparray_buf; + size_t lamparray_buf_len; + u8 lamparray_rid_base; + unsigned int lamparray_count; + bool lamparray_controlled; + bool lamparray_unavailable; + struct asus_lamparray_lamp *lamparray_lamps; #endif }; @@ -1606,6 +1622,322 @@ static int asus_aura_write_zone_effect(struct asus_drvdata *drvdata, u8 zone, return 0; } +static size_t asus_lamparray_report_len(struct hid_device *hdev, u8 id) +{ + struct hid_report *report; + + report = hdev->report_enum[HID_FEATURE_REPORT].report_id_hash[id]; + if (!report) + return 0; + return hid_report_len(report); +} + +static u8 asus_lamparray_rid(struct asus_drvdata *drvdata, u8 offset) +{ + return drvdata->lamparray_rid_base + offset; +} + +static int asus_lamparray_raw(struct asus_drvdata *drvdata, u8 *buf, size_t len, + bool get) +{ + int ret; + + if (!drvdata->lamparray_hdev || !len) + return -ENODEV; + + ret = hid_hw_raw_request(drvdata->lamparray_hdev, buf[0], buf, len, + HID_FEATURE_REPORT, + get ? HID_REQ_GET_REPORT : HID_REQ_SET_REPORT); + return ret < 0 ? ret : 0; +} + +static void asus_lamparray_prepare(struct asus_drvdata *drvdata, u8 rid) +{ + memset(drvdata->lamparray_buf, 0, drvdata->lamparray_buf_len); + drvdata->lamparray_buf[0] = rid; +} + +static struct hid_device *asus_find_lamparray_sibling(struct hid_device *hdev) +{ + struct usb_interface *intf; + struct usb_device *udev; + struct usb_host_config *config; + unsigned int i; + + if (!hid_is_usb(hdev)) + return NULL; + + intf = to_usb_interface(hdev->dev.parent); + udev = interface_to_usbdev(intf); + if (!udev->actconfig) + return NULL; + + config = udev->actconfig; + for (i = 0; i < config->desc.bNumInterfaces; i++) { + struct usb_interface *other = config->interface[i]; + struct hid_device *other_hdev; + + if (!other || other == intf) + continue; + if (!other->dev.driver || other->dev.driver != intf->dev.driver) + continue; + + other_hdev = usb_get_intfdata(other); + if (!other_hdev) + continue; + if (other_hdev->vendor != hdev->vendor || + other_hdev->product != hdev->product) + continue; + if (asus_is_lamparray_interface(other_hdev)) + return other_hdev; + } + + return NULL; +} + +static void asus_lamparray_unbind_from_owners(struct hid_device *la_hdev) +{ + struct usb_interface *intf; + struct usb_device *udev; + struct usb_host_config *config; + unsigned int i; + + if (!hid_is_usb(la_hdev)) + return; + + intf = to_usb_interface(la_hdev->dev.parent); + udev = interface_to_usbdev(intf); + if (!udev->actconfig) + return; + + config = udev->actconfig; + for (i = 0; i < config->desc.bNumInterfaces; i++) { + struct usb_interface *other = config->interface[i]; + struct hid_device *other_hdev; + struct asus_drvdata *owner; + + if (!other || other == intf) + continue; + if (!other->dev.driver || other->dev.driver != intf->dev.driver) + continue; + + other_hdev = usb_get_intfdata(other); + if (!other_hdev || other_hdev->driver != la_hdev->driver) + continue; + + owner = hid_get_drvdata(other_hdev); + if (!owner) + continue; + + mutex_lock(&owner->aura_lock); + if (owner->lamparray_hdev == la_hdev) { + owner->lamparray_controlled = false; + owner->lamparray_count = 0; + owner->lamparray_lamps = NULL; + owner->lamparray_hdev = NULL; + owner->lamparray_unavailable = true; + mutex_unlock(&owner->aura_lock); + put_device(&la_hdev->dev); + return; + } + mutex_unlock(&owner->aura_lock); + } +} + +static int asus_lamparray_cmp_x(const void *a, const void *b) +{ + const struct asus_lamparray_lamp *la = a; + const struct asus_lamparray_lamp *lb = b; + + return cmp_int(la->x, lb->x); +} + +static int asus_lamparray_init_from_hdev(struct asus_drvdata *drvdata, + struct hid_device *la_hdev) +{ + u8 *buf; + size_t attr_len, req_len, resp_len, multi_len, ctrl_len, max_len; + unsigned int count, i; + int base, ret; + + base = asus_lamparray_detect_base(la_hdev); + if (base < 0) + return base; + + attr_len = asus_lamparray_report_len(la_hdev, base + ASUS_LAMPARRAY_RID_ATTR); + req_len = asus_lamparray_report_len(la_hdev, base + ASUS_LAMPARRAY_RID_REQUEST); + resp_len = asus_lamparray_report_len(la_hdev, base + ASUS_LAMPARRAY_RID_RESPONSE); + multi_len = asus_lamparray_report_len(la_hdev, base + ASUS_LAMPARRAY_RID_MULTI); + ctrl_len = asus_lamparray_report_len(la_hdev, base + ASUS_LAMPARRAY_RID_CONTROL); + max_len = max3(max(attr_len, req_len), max(resp_len, multi_len), ctrl_len); + if (attr_len < 3 || req_len < 3 || resp_len < 23 || multi_len < 51 || + ctrl_len < 2 || !max_len) + return -EPROTO; + + buf = devm_kzalloc(&drvdata->hdev->dev, max_len, GFP_KERNEL); + if (!buf) + return -ENOMEM; + + drvdata->lamparray_hdev = la_hdev; + drvdata->lamparray_buf = buf; + drvdata->lamparray_buf_len = max_len; + drvdata->lamparray_rid_base = base; + + asus_lamparray_prepare(drvdata, base + ASUS_LAMPARRAY_RID_ATTR); + ret = asus_lamparray_raw(drvdata, buf, attr_len, true); + if (ret < 0) + return ret; + + count = get_unaligned_le16(buf + 1); + if (!count || count > ASUS_LAMPARRAY_MAX_LAMPS) + return -EPROTO; + + drvdata->lamparray_lamps = devm_kcalloc(&drvdata->hdev->dev, count, + sizeof(*drvdata->lamparray_lamps), + GFP_KERNEL); + if (!drvdata->lamparray_lamps) + return -ENOMEM; + + for (i = 0; i < count; i++) { + u32 purposes; + + asus_lamparray_prepare(drvdata, base + ASUS_LAMPARRAY_RID_REQUEST); + put_unaligned_le16(i, buf + 1); + ret = asus_lamparray_raw(drvdata, buf, req_len, false); + if (ret < 0) + return ret; + + asus_lamparray_prepare(drvdata, base + ASUS_LAMPARRAY_RID_RESPONSE); + ret = asus_lamparray_raw(drvdata, buf, resp_len, true); + if (ret < 0) + return ret; + + /* LampId@1, PositionX@3, LampPurposes@19 (HID Lighting) */ + drvdata->lamparray_lamps[i].id = get_unaligned_le16(buf + 1); + drvdata->lamparray_lamps[i].x = (s32)get_unaligned_le32(buf + 3); + purposes = get_unaligned_le32(buf + 19); + drvdata->lamparray_lamps[i].keyboard = + !!(purposes & ASUS_LAMPARRAY_PURPOSE_CONTROL); + } + + drvdata->lamparray_count = count; + sort(drvdata->lamparray_lamps, count, sizeof(*drvdata->lamparray_lamps), + asus_lamparray_cmp_x, NULL); + get_device(&la_hdev->dev); + hid_info(drvdata->hdev, + "LampArray direct RGB backend (%u lamps, rid_base=0x%02x)\n", + count, base); + return 0; +} + +static void asus_lamparray_try_init_unlocked(struct asus_drvdata *drvdata) +{ + struct hid_device *sibling; + int ret; + + if (drvdata->lamparray_count || drvdata->lamparray_unavailable) + return; + + sibling = asus_find_lamparray_sibling(drvdata->hdev); + if (!sibling) + return; + + ret = asus_lamparray_init_from_hdev(drvdata, sibling); + if (ret < 0) { + hid_warn(drvdata->hdev, "LampArray init failed: %d\n", ret); + drvdata->lamparray_hdev = NULL; + drvdata->lamparray_lamps = NULL; + drvdata->lamparray_count = 0; + drvdata->lamparray_unavailable = true; + } +} + +static int asus_lamparray_set_control_unlocked(struct asus_drvdata *drvdata, + bool autonomous) +{ + u8 rid = asus_lamparray_rid(drvdata, ASUS_LAMPARRAY_RID_CONTROL); + size_t len = asus_lamparray_report_len(drvdata->lamparray_hdev, rid); + + if (len < 2) + return -EPROTO; + + asus_lamparray_prepare(drvdata, rid); + drvdata->lamparray_buf[1] = autonomous ? 0x01 : 0x00; + return asus_lamparray_raw(drvdata, drvdata->lamparray_buf, len, false); +} + +static int asus_lamparray_aura_handoff_unlocked(struct asus_drvdata *drvdata, + u8 zone, bool release) +{ + u8 aura[AURA_FEATURE_REPORT_SIZE] = { + FEATURE_KBD_LED_REPORT_ID1, + AURA_CMD_ZONE_ENABLE, + zone, + 0x01, + }; + + if (release) + aura[4] = 0x01; + + return asus_aura_set_feature_unlocked(drvdata, aura, sizeof(aura)); +} + +static int asus_lamparray_take_control_unlocked(struct asus_drvdata *drvdata) +{ + int ret; + + if (drvdata->lamparray_controlled) + return 0; + + ret = asus_lamparray_aura_handoff_unlocked(drvdata, + AURA_ZONE_ACTIVATE_LAMPARRAY, + false); + if (ret < 0) + return ret; + + /* Match G-Helper: pulse then clear AutonomousMode for host control */ + ret = asus_lamparray_set_control_unlocked(drvdata, true); + if (ret < 0) + return ret; + ret = asus_lamparray_set_control_unlocked(drvdata, false); + if (ret < 0) + return ret; + + drvdata->lamparray_controlled = true; + return 0; +} + +static void asus_lamparray_release_unlocked(struct asus_drvdata *drvdata) +{ + if (!drvdata->lamparray_count || !drvdata->lamparray_controlled) + return; + + asus_lamparray_set_control_unlocked(drvdata, true); + asus_lamparray_aura_handoff_unlocked(drvdata, + AURA_ZONE_RELEASE_LAMPARRAY, + true); + drvdata->lamparray_controlled = false; +} + +static void asus_lamparray_sample_rgb(const u8 *buf, unsigned int nleds, + unsigned int idx, unsigned int n, + u8 *r, u8 *g, u8 *b) +{ + unsigned int zi; + + if (!buf || !nleds || !n) { + *r = *g = *b = 0; + return; + } + + zi = (idx * nleds) / n; + if (zi >= nleds) + zi = nleds - 1; + *r = buf[zi * 3]; + *g = buf[zi * 3 + 1]; + *b = buf[zi * 3 + 2]; +} + static void asus_lamparray_fill_solid(u8 *buf, unsigned int nleds, u8 r, u8 g, u8 b) { @@ -1632,14 +1964,80 @@ static void asus_aura_init_direct_bufs(struct asus_drvdata *drvdata) asus_aura_lb_led_count(drvdata), 255, 0, 0); } - -static int asus_lamparray_try_apply_unlocked(struct asus_drvdata *drvdata) +static int asus_lamparray_apply_unlocked(struct asus_drvdata *drvdata) { - return -ENODEV; + unsigned int kbd_n = 0, lb_n = 0, kbd_i = 0, lb_i = 0, i; + unsigned int kbd_leds = ROG_STRIX_4ZONE_KBD_LEDS; + unsigned int lb_leds = asus_aura_lb_led_count(drvdata); + u8 *buf = drvdata->lamparray_buf; + u8 rid_multi = asus_lamparray_rid(drvdata, ASUS_LAMPARRAY_RID_MULTI); + size_t multi_len; + int ret; + + ret = asus_lamparray_take_control_unlocked(drvdata); + if (ret < 0) + return ret; + + for (i = 0; i < drvdata->lamparray_count; i++) { + if (drvdata->lamparray_lamps[i].keyboard) + kbd_n++; + else + lb_n++; + } + + multi_len = asus_lamparray_report_len(drvdata->lamparray_hdev, rid_multi); + if (multi_len < 51) + return -EPROTO; + + for (i = 0; i < drvdata->lamparray_count; i += ASUS_LAMPARRAY_MULTI_MAX) { + unsigned int n = min_t(unsigned int, ASUS_LAMPARRAY_MULTI_MAX, + drvdata->lamparray_count - i); + unsigned int j; + unsigned int id_off = 3; + unsigned int col_off = 3 + ASUS_LAMPARRAY_MULTI_MAX * 2; + + asus_lamparray_prepare(drvdata, rid_multi); + buf[1] = n; + buf[2] = (i + n >= drvdata->lamparray_count) ? + ASUS_LAMPARRAY_FLAG_COMPLETE : 0; + + for (j = 0; j < n; j++) { + unsigned int lamp = i + j; + u8 r, g, b; + u16 id = drvdata->lamparray_lamps[lamp].id; + + if (drvdata->lamparray_lamps[lamp].keyboard) { + asus_lamparray_sample_rgb(drvdata->kbd_direct_buf, + kbd_leds, kbd_i++, kbd_n, + &r, &g, &b); + } else { + asus_lamparray_sample_rgb(drvdata->lb_direct_buf, + lb_leds, lb_i++, lb_n, + &r, &g, &b); + } + + put_unaligned_le16(id, buf + id_off + j * 2); + buf[col_off + j * 4 + 0] = r; + buf[col_off + j * 4 + 1] = g; + buf[col_off + j * 4 + 2] = b; + buf[col_off + j * 4 + 3] = 0xff; + } + + ret = asus_lamparray_raw(drvdata, buf, multi_len, false); + if (ret < 0) + return ret; + } + + return 0; } -static void asus_lamparray_release_unlocked(struct asus_drvdata *drvdata) +/* Prefer LampArray; callers keep Aura 0xBC as fallback on -ENODEV. */ +static int asus_lamparray_try_apply_unlocked(struct asus_drvdata *drvdata) { + asus_lamparray_try_init_unlocked(drvdata); + if (!drvdata->lamparray_count) + return -ENODEV; + return asus_lamparray_apply_unlocked(drvdata); } static int asus_aura_write_4zone_direct_bc_unlocked(struct asus_drvdata *drvdata) @@ -1900,11 +2298,14 @@ static int asus_aura_apply_effect(struct led_classdev_dynamic *ldev, /* * Static and animations are firmware effects: commit them with 0xb3 - * so the controller keeps the color across reboot. Release LampArray - * first when a later commit uses it as the direct-RGB backend. + * so the controller keeps the color across reboot. LampArray stays + * the direct-RGB backend and must be released first. */ - scoped_guard(mutex, &drvdata->aura_lock) - asus_lamparray_release_unlocked(drvdata); + scoped_guard(mutex, &drvdata->aura_lock) { + asus_lamparray_try_init_unlocked(drvdata); + if (drvdata->lamparray_count) + asus_lamparray_release_unlocked(drvdata); + } if (asus_aura_is_lightbar(drvdata, ldev)) ret = asus_aura_write_zone_range(drvdata, AURA_ZONE_BAR_LEFT, @@ -2333,7 +2734,11 @@ static int asus_init_dynamic_lighting(struct hid_device *hdev) if (ret < 0) hid_warn(hdev, "Failed to wake Aura hardware zones: %d\n", ret); - kbd_direct = is_per_key || drvdata->is_strix_4zone; + scoped_guard(mutex, &drvdata->aura_lock) + asus_lamparray_try_init_unlocked(drvdata); + + kbd_direct = is_per_key || drvdata->is_strix_4zone || + drvdata->lamparray_hdev != NULL; lightbar_direct = has_lightbar; ret = asus_aura_get_effect_mask(drvdata, effect_mask); if (ret < 0) { @@ -2966,6 +3371,7 @@ static int asus_probe(struct hid_device *hdev, const struct hid_device_id *id) if (is_vendor && (drvdata->quirks & QUIRK_ROG_NKEY_KEYBOARD)) hdev->quirks |= HID_QUIRK_HIDINPUT_FORCE; +#if IS_REACHABLE(CONFIG_LEDS_CLASS_DYNAMIC) /* * LampArray is a Dynamic Lighting backend owned in-kernel: do not * export hidraw for that interface. Aura 0xBC remains the fallback @@ -2980,6 +3386,7 @@ static int asus_probe(struct hid_device *hdev, const struct hid_device_id *id) hid_info(hdev, "Bound ASUS LampArray interface (no hidraw)\n"); return 0; } +#endif ret = asus_worker_create(hdev, drvdata); if (ret) { @@ -3068,10 +3475,23 @@ static void asus_remove(struct hid_device *hdev) if (drvdata->listener.brightness_set) asus_hid_unregister_listener(&drvdata->listener); +#if IS_REACHABLE(CONFIG_LEDS_CLASS_DYNAMIC) if (asus_is_lamparray_interface(hdev)) { + asus_lamparray_unbind_from_owners(hdev); hid_hw_stop(hdev); return; } + if (drvdata->lamparray_hdev) { + scoped_guard(mutex, &drvdata->aura_lock) { + asus_lamparray_release_unlocked(drvdata); + if (drvdata->lamparray_hdev) { + put_device(&drvdata->lamparray_hdev->dev); + drvdata->lamparray_hdev = NULL; + } + drvdata->lamparray_count = 0; + } + } +#endif asus_worker_stop(drvdata->worker); hid_hw_stop(hdev); From dfe29bfcdd2d156886080599b10a6405b8482cc4 Mon Sep 17 00:00:00 2001 From: Marco Scardovi Date: Fri, 18 Sep 2026 15:44:15 +0200 Subject: [PATCH 1352/1352] HID: asus: Add Slash lighting support Register Slash when feature report 0x5e is present, or on USB 0x193b when Aura LED report 0x5d exists. Identify Slash from HID reports, never from DMI board lists. Expose asus::slash with mode, interval and brightness controls using the Aura feature-report path already used for keyboard lighting. Probe only runs the Slash handshake and does not overwrite a mode saved in the controller. Signed-off-by: Marco Scardovi --- drivers/hid/hid-asus.c | 281 +++++++++++++++++++++++++++++++++++++++++ 1 file changed, 281 insertions(+) diff --git a/drivers/hid/hid-asus.c b/drivers/hid/hid-asus.c index fb400b8a23705b..47c88ffc1112e6 100644 --- a/drivers/hid/hid-asus.c +++ b/drivers/hid/hid-asus.c @@ -298,6 +298,11 @@ struct asus_lamparray_lamp { struct asus_drvdata { unsigned long quirks; + struct led_classdev slash_led; + bool has_slash_led; + u8 slash_mode; + u8 slash_brightness; + u8 slash_interval; struct hid_device *hdev; struct input_dev *input; struct input_dev *tp_kbd_input; @@ -1179,6 +1184,22 @@ static bool asus_is_lamparray_interface(struct hid_device *hdev) { return asus_lamparray_detect_base(hdev) >= 0; } + +/* + * Slash is identified by HID feature reports, never DMI board lists: + * - report 0x5e is the dedicated Slash feature report on Aura keyboards + * - standalone Slash MCU reuses USB 0x193b (also AniMe) and talks over 0x5d + * when 0x5e is absent; AniMe is rejected earlier because it has neither + */ +static bool asus_device_has_slash(struct hid_device *hdev) +{ + if (asus_has_report_id(hdev, FEATURE_KBD_LED_REPORT_ID2)) + return true; + + return hdev->product == USB_DEVICE_ID_ASUSTEK_ROG_SLASH && + asus_has_report_id(hdev, FEATURE_KBD_LED_REPORT_ID1); +} + static int asus_kbd_register_leds(struct hid_device *hdev) { struct asus_drvdata *drvdata = hid_get_drvdata(hdev); @@ -2531,6 +2552,257 @@ static int asus_aura_discover(struct asus_drvdata *drvdata, bool *has_lightbar, return 0; } +static const struct asus_slash_mode { + const char *name; + u8 mode; +} asus_slash_modes[] = { + { "Static", 0x06 }, + { "Bounce", 0x10 }, + { "Slash", 0x12 }, + { "Loading", 0x13 }, + { "BitStream", 0x1d }, + { "Transmission", 0x1a }, + { "Flow", 0x19 }, + { "Flux", 0x25 }, + { "Phantom", 0x24 }, + { "Spectrum", 0x26 }, + { "Hazard", 0x32 }, + { "Interfacing", 0x33 }, + { "Ramp", 0x34 }, + { "GameOver", 0x42 }, + { "Start", 0x43 }, + { "Buzzer", 0x44 }, +}; + +static inline u8 asus_slash_report_id(struct asus_drvdata *drvdata) +{ + if (asus_has_report_id(drvdata->hdev, FEATURE_KBD_LED_REPORT_ID2)) + return FEATURE_KBD_LED_REPORT_ID2; + return FEATURE_KBD_LED_REPORT_ID1; +} + +static int asus_slash_init_unlocked(struct asus_drvdata *drvdata) +{ + u8 rpt = asus_slash_report_id(drvdata); + u8 pkt1[] = { rpt, 0xd7, 0x00, 0x00, 0x01, 0xac }; + u8 pkt2[] = { rpt, 0xd2, 0x02, 0x01, 0x08, 0xab }; + int ret; + + ret = asus_aura_set_feature_unlocked(drvdata, pkt1, sizeof(pkt1)); + if (ret < 0) + return ret; + + return asus_aura_set_feature_unlocked(drvdata, pkt2, sizeof(pkt2)); +} + +static int asus_slash_set_options_unlocked(struct asus_drvdata *drvdata, bool enabled, + u8 brightness, u8 interval) +{ + u8 rpt = asus_slash_report_id(drvdata); + u8 pkt[] = { + rpt, 0xd3, 0x03, 0x01, 0x08, 0xab, 0xff, 0x01, + enabled ? 1 : 0, 0x06, brightness, 0xff, interval + }; + + return asus_aura_set_feature_unlocked(drvdata, pkt, sizeof(pkt)); +} + +static int asus_slash_set_mode_unlocked(struct asus_drvdata *drvdata, u8 mode) +{ + u8 rpt = asus_slash_report_id(drvdata); + u8 pkt1[] = { rpt, 0xd2, 0x03, 0x00, 0x0c }; + u8 pkt2[] = { + rpt, 0xd3, 0x04, 0x00, 0x0c, 0x01, mode, 0x02, + 0x19, 0x03, 0x13, 0x04, 0x11, 0x05, 0x12, 0x06, 0x13 + }; + int ret; + + ret = asus_aura_set_feature_unlocked(drvdata, pkt1, sizeof(pkt1)); + if (ret < 0) + return ret; + + return asus_aura_set_feature_unlocked(drvdata, pkt2, sizeof(pkt2)); +} + +static int asus_slash_save_unlocked(struct asus_drvdata *drvdata) +{ + u8 rpt = asus_slash_report_id(drvdata); + u8 pkt[] = { rpt, 0xd4, 0x00, 0x00, 0x01, 0xab }; + + return asus_aura_set_feature_unlocked(drvdata, pkt, sizeof(pkt)); +} + +static int asus_slash_brightness_set_blocking(struct led_classdev *led_cdev, + enum led_brightness brightness) +{ + struct asus_drvdata *drvdata = container_of(led_cdev, struct asus_drvdata, slash_led); + int ret; + + guard(mutex)(&drvdata->aura_lock); + + drvdata->slash_brightness = brightness; + ret = asus_slash_set_options_unlocked(drvdata, brightness > 0, (u8)brightness, + drvdata->slash_interval); + if (ret < 0) + return ret; + + return asus_slash_save_unlocked(drvdata); +} + +static ssize_t slash_mode_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct led_classdev *led = dev_get_drvdata(dev); + struct asus_drvdata *drvdata = container_of(led, struct asus_drvdata, slash_led); + unsigned int i; + + for (i = 0; i < ARRAY_SIZE(asus_slash_modes); i++) { + if (asus_slash_modes[i].mode == drvdata->slash_mode) + return sysfs_emit(buf, "%s\n", asus_slash_modes[i].name); + } + + return sysfs_emit(buf, "0x%02x\n", drvdata->slash_mode); +} + +static ssize_t slash_mode_store(struct device *dev, + struct device_attribute *attr, + const char *buf, size_t count) +{ + struct led_classdev *led = dev_get_drvdata(dev); + struct asus_drvdata *drvdata = container_of(led, struct asus_drvdata, slash_led); + char mode_str[32]; + unsigned int i; + u8 mode_val = 0; + int ret; + + if (sscanf(buf, "%31s", mode_str) != 1) + return -EINVAL; + + for (i = 0; i < ARRAY_SIZE(asus_slash_modes); i++) { + if (sysfs_streq(mode_str, asus_slash_modes[i].name)) { + mode_val = asus_slash_modes[i].mode; + break; + } + } + + if (!mode_val) { + if (kstrtou8(mode_str, 0, &mode_val)) + return -EINVAL; + } + + guard(mutex)(&drvdata->aura_lock); + + ret = asus_slash_set_mode_unlocked(drvdata, mode_val); + if (ret < 0) + return ret; + + ret = asus_slash_save_unlocked(drvdata); + if (ret < 0) + return ret; + + drvdata->slash_mode = mode_val; + return count; +} +static DEVICE_ATTR_RW(slash_mode); + +static ssize_t slash_mode_index_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + return sysfs_emit(buf, + "Static Bounce Slash Loading BitStream Transmission Flow Flux Phantom Spectrum Hazard Interfacing Ramp GameOver Start Buzzer\n"); +} +static DEVICE_ATTR_RO(slash_mode_index); + +static ssize_t slash_interval_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct led_classdev *led = dev_get_drvdata(dev); + struct asus_drvdata *drvdata = container_of(led, struct asus_drvdata, slash_led); + + return sysfs_emit(buf, "%u\n", drvdata->slash_interval); +} + +static ssize_t slash_interval_store(struct device *dev, + struct device_attribute *attr, + const char *buf, size_t count) +{ + struct led_classdev *led = dev_get_drvdata(dev); + struct asus_drvdata *drvdata = container_of(led, struct asus_drvdata, slash_led); + u8 interval; + int ret; + + if (kstrtou8(buf, 0, &interval)) + return -EINVAL; + + guard(mutex)(&drvdata->aura_lock); + + drvdata->slash_interval = interval; + ret = asus_slash_set_options_unlocked(drvdata, drvdata->slash_brightness > 0, + drvdata->slash_brightness, interval); + if (ret < 0) + return ret; + + ret = asus_slash_save_unlocked(drvdata); + if (ret < 0) + return ret; + + return count; +} +static DEVICE_ATTR_RW(slash_interval); + +static struct attribute *asus_slash_attrs[] = { + &dev_attr_slash_mode.attr, + &dev_attr_slash_mode_index.attr, + &dev_attr_slash_interval.attr, + NULL, +}; + +static const struct attribute_group asus_slash_group = { + .attrs = asus_slash_attrs, +}; + +static const struct attribute_group *asus_slash_groups[] = { + &asus_slash_group, + NULL, +}; + +static int asus_init_slash(struct hid_device *hdev) +{ + struct asus_drvdata *drvdata = hid_get_drvdata(hdev); + int ret; + + if (!asus_device_has_slash(hdev)) + return 0; + + drvdata->slash_led.name = "asus::slash"; + drvdata->slash_led.max_brightness = 255; + drvdata->slash_led.brightness = 255; + drvdata->slash_led.brightness_set_blocking = asus_slash_brightness_set_blocking; + drvdata->slash_led.groups = asus_slash_groups; + + drvdata->slash_brightness = 255; + drvdata->slash_interval = 0; + + scoped_guard(mutex, &drvdata->aura_lock) { + ret = asus_slash_init_unlocked(drvdata); + if (ret < 0) { + hid_warn(hdev, "Failed to initialize Slash lighting: %d\n", ret); + return ret; + } + } + + ret = devm_led_classdev_register(&hdev->dev, &drvdata->slash_led); + if (ret < 0) { + hid_warn(hdev, "Failed to register Slash LED classdev: %d\n", ret); + return ret; + } + + drvdata->has_slash_led = true; + hid_info(hdev, "Registered Slash lighting LED: asus::slash\n"); + + return 0; +} + static void asus_aura_dldev_set_default_palette(struct led_classdev_dynamic *ldev) { if (!ldev->palette) @@ -2822,6 +3094,11 @@ static inline int asus_init_dynamic_lighting(struct hid_device *hdev) return 0; } +static inline int asus_init_slash(struct hid_device *hdev) +{ + return 0; +} + static inline void asus_aura_capture_boot_effect(struct hid_device *hdev) { } @@ -3430,6 +3707,10 @@ static int asus_probe(struct hid_device *hdev, const struct hid_device_id *id) ret = asus_init_dynamic_lighting(hdev); if (ret < 0) hid_warn(hdev, "Failed to initialize dynamic lighting: %d\n", ret); + + ret = asus_init_slash(hdev); + if (ret < 0) + hid_warn(hdev, "Failed to initialize Slash lighting: %d\n", ret); } /*